[{"slug": "aider-edit", "name": "Aider Edit", "description": "Aider Edit leaderboard tracks edit-focused coding tasks via Aider (`aider_leaderboard` source). Public YAML in the Aider repository.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://aider.chat/docs/leaderboards/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://aider.chat/docs/leaderboards/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 84, "score_count": 84, "best_success_rate": 84.2}, {"slug": "aider-polyglot", "name": "Aider Polyglot", "description": "Aider Polyglot benchmarks multi-language editing via Aider on 225 Exercism exercises. Scores ingested from public YAML (`aider_leaderboard` source, harness `aider`, metric `pass_rate_2`).\n\nMeasures Aider's edit loop \u2014 not SWE-bench issue resolution. Top public runs approach ~88% pass_rate_2 on recent frontier models.", "category": "eval", "split": null, "language": null, "task_count": 225, "homepage_url": "https://aider.chat/docs/leaderboards/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://aider.chat/docs/leaderboards/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 59, "score_count": 59, "best_success_rate": 88.0}, {"slug": "aider-refactor", "name": "Aider Refactor", "description": "Aider Refactor measures refactoring tasks in Aider's terminal pair-programming loop (`aider_leaderboard` source, harness `aider`).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://aider.chat/docs/leaderboards/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://aider.chat/docs/leaderboards/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 13, "score_count": 13, "best_success_rate": 92.1}, {"slug": "hal-assistantbench", "name": "AssistantBench (HAL)", "description": "Real-world assistant tasks on the HAL platform (AssistantBench board).", "category": "eval", "split": null, "language": null, "task_count": 30, "homepage_url": "https://hal.cs.princeton.edu/assistantbench", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/assistantbench", "is_public_leaderboard": true, "harness_count": 1, "model_count": 14, "score_count": 14, "best_success_rate": 38.81}, {"slug": "hal-corebench-hard", "name": "CORE-Bench Hard (HAL)", "description": "CORE-Bench Hard on HAL measures whether agents can reproduce published computational results. Rows include accuracy and total API cost normalized to per-task cost in the catalog (`hal_site`).\n\nUseful for research-agent and notebook-style scaffolds (Claude Code, CORE-Agent, etc.) beyond pure issue-fixing.", "category": "eval", "split": null, "language": null, "task_count": 45, "homepage_url": "https://hal.cs.princeton.edu/corebench_hard", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/corebench_hard", "is_public_leaderboard": true, "harness_count": 3, "model_count": 24, "score_count": 47, "best_success_rate": 77.78}, {"slug": "commit0", "name": "Commit0", "description": "Commit0 evaluates agents building new libraries from scratch (54\u201357 libs). OpenHands Index publishes SDK runs (`openhands_index`). Historical leaderboards also appear on commit-0.github.io.\n\nGreenfield category \u2014 scores are not comparable to issue-resolution % on Verified without context.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://commit-0.github.io/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 62.5}, {"slug": "deepswe", "name": "DeepSWE (AA component)", "description": "DeepSWE on Artificial Analysis measures long-horizon software engineering performance as part of the Coding Agent Index. AA runs vendor agents on a DeepSWE slice (`source = artificialanalysis`).\n\nThis is not the official [DataCurve DeepSWE leaderboard](https://deepswe.datacurve.ai/) (mini-SWE-agent standardized). Compare AA DeepSWE component scores only against other AA components on the same agent\u00d7model row.\n\nOfficial DeepSWE ingest remains on the catalog watchlist.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://deepswe.datacurve.ai/", "is_public_leaderboard": true, "harness_count": 10, "model_count": 40, "score_count": 45, "best_success_rate": 66.96}, {"slug": "gaia", "name": "GAIA", "description": "GAIA (165 public validation questions) tests general-purpose agent reasoning and tool use. OpenHands Index includes GAIA as the information-gathering pillar (`openhands_index`, harness `openhands-sdk`).\n\nAlso tracked on HAL (hal.cs.princeton.edu/gaia). Peripheral to pure SWE issue-fixing but part of the OH Index average_score.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/gaia", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 86.1}, {"slug": "hal-gaia", "name": "GAIA (HAL)", "description": "The Holistic Agent Leaderboard (HAL) at hal.cs.princeton.edu publishes GAIA validation results with standardized cost tracking and verified traces. Rows are harness \u00d7 model (`source = hal_site`, benchmark `hal-gaia`).\n\nDistinct from OpenHands Index GAIA (`openhands_index`, harness `openhands-sdk`). HAL leaderboard updates were paused in 2026 while the team focuses on reliability; existing rows remain authoritative snapshots.\n\nHAL also publishes multi-dimensional reliability scores on the same tasks (`hal-reliability-gaia`, `hal_reliability_site`).", "category": "eval", "split": null, "language": null, "task_count": 165, "homepage_url": "https://hal.cs.princeton.edu/gaia", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/gaia", "is_public_leaderboard": true, "harness_count": 2, "model_count": 16, "score_count": 30, "best_success_rate": 74.55}, {"slug": "hal-reliability-gaia", "name": "HAL Reliability \u2014 GAIA", "description": "HAL's reliability dashboard evaluates agents on GAIA with repeated runs and multi-dimensional metrics beyond single-shot accuracy. Catalog stores the mean of HAL's four dimension aggregates as `success_rate` (`hal_reliability_site`, harness `hal-generalist-agent`).\n\nSee hal.cs.princeton.edu/reliability/ for full consistency, predictability, robustness, and safety breakdowns.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/reliability/benchmark/gaia/", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/reliability/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 15, "score_count": 15, "best_success_rate": 96.25}, {"slug": "hal-reliability-tau-bench-airline", "name": "HAL Reliability \u2014 \u03c4-bench Airline (clean)", "description": "HAL reliability evaluation on \u03c4-bench airline (clean split).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/reliability/benchmark/taubench_airline/", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/reliability/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 15, "score_count": 15, "best_success_rate": 93.5}, {"slug": "hal-reliability-tau-bench-airline-original", "name": "HAL Reliability \u2014 \u03c4-bench Airline (original)", "description": "HAL reliability evaluation on \u03c4-bench airline (original split).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/reliability/benchmark/taubench_airline_original/", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/reliability/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 13, "score_count": 13, "best_success_rate": 87.75}, {"slug": "livecodebench-generation", "name": "LiveCodeBench (code generation)", "description": "LiveCodeBench generation track evaluates code generation on competitive programming problems with contamination-aware splits. Catalog ingests aggregated mean pass@1 per model from public JSON (`livecodebench` source).\n\nModel-level metric (implicit eval harness `livecodebench`) \u2014 not harness\u00d7model matrix like SWE-bench Verified. Leaderboard at livecodebench.github.io.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://livecodebench.github.io/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://livecodebench.github.io/leaderboard.html", "is_public_leaderboard": true, "harness_count": 1, "model_count": 26, "score_count": 26, "best_success_rate": 87.3}, {"slug": "mle-bench", "name": "MLE-bench", "description": "MLE-bench measures ML engineering agents on 75 Kaggle competitions with Low/Medium/High complexity splits. Catalog ingests the **All (%)** any-medal column from the openai/mle-bench README leaderboard table (`mle_bench_site` source).\n\nHarness\u00d7model matrix (AIDE, OpenHands, R&D-Agent, Famou-Agent, etc.) \u2014 not software issue-fixing. Canonical setup: 24h runtime, 36 vCPUs, A10 GPU (see paper). Leaderboard submissions paused Apr 2026 per maintainer notice.", "category": "eval", "split": null, "language": null, "task_count": 75, "homepage_url": "https://mlebench.com/leaderboard", "paper_url": "https://arxiv.org/abs/2410.07095", "hf_dataset_id": null, "official_leaderboard_url": "https://mlebench.com/leaderboard", "is_public_leaderboard": true, "harness_count": 19, "model_count": 15, "score_count": 25, "best_success_rate": 64.44}, {"slug": "hal-online-mind2web", "name": "Online Mind2Web (HAL)", "description": "Live web navigation benchmark (Online Mind2Web) on HAL.", "category": "eval", "split": null, "language": null, "task_count": 300, "homepage_url": "https://hal.cs.princeton.edu/online_mind2web", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/online_mind2web", "is_public_leaderboard": true, "harness_count": 2, "model_count": 11, "score_count": 20, "best_success_rate": 42.33}, {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)", "description": "SWE-Atlas-QnA tests repository understanding and question answering \u2014 a component of Artificial Analysis's Coding Agent Index (`artificialanalysis` source).\n\nDistinct from Scale's official [SWE Atlas QnA leaderboard](https://scale.com/leaderboard/sweatlas-qna). AA rows reflect AA's eval harness and task slice, not Scale's public submission pipeline.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://scale.com/leaderboard/sweatlas-qna", "is_public_leaderboard": true, "harness_count": 10, "model_count": 40, "score_count": 45, "best_success_rate": 54.84}, {"slug": "swe-bench", "name": "SWE-bench", "description": "SWE-bench is the original benchmark suite (2,294 real GitHub issues) introduced in Jimenez et al., ICLR 2024. Verified (500 tasks) is the human-curated subset used for most modern leaderboards.\n\nFull swe-bench remains relevant for training and historical comparisons. OpenHands Index publishes swe-bench component scores under `openhands-swe-bench` (OpenHands SDK runs) \u2014 not identical to swebench.com Verified submissions.", "category": "eval", "split": null, "language": "python", "task_count": 2294, "homepage_url": "https://www.swebench.com/", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/", "is_public_leaderboard": true, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}, {"slug": "openhands-swe-bench", "name": "SWE-bench (OpenHands Index)", "description": "OpenHands Index runs a standardized OpenHands SDK configuration across five benchmark families. The swe-bench component measures issue-resolution style tasks (`source = openhands_index`, harness `openhands-sdk`).\n\nScores include cost and runtime metadata from index.openhands.dev. Not interchangeable with swebench.com community submissions or bash-only mini-SWE-agent rows.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://index.openhands.dev/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 95.8}, {"slug": "swe-bench-lite", "name": "SWE-bench Lite", "description": "SWE-bench Lite is the 300-task subset of the original SWE-bench suite used for faster eval. The public leaderboard on swebench.com accepts community harness\u00d7model submissions (`source = swe_bench_site`).\n\nDistinct from Verified (500 tasks) and from training-only indices. Useful for historical comparisons and lighter eval cycles.", "category": "eval", "split": "lite", "language": "python", "task_count": 300, "homepage_url": "https://www.swebench.com/lite", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/lite", "is_public_leaderboard": true, "harness_count": 36, "model_count": 18, "score_count": 50, "best_success_rate": 60.33}, {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual", "description": "SWE-bench Multilingual evaluates issue fixing across multiple languages. The swebench.com board uses mini-SWE-agent for every model (like bash-only on Verified) so rows isolate foundation-model capability.\n\nCatalog: harness `mini-swe-agent`, benchmark `swe-bench-multilingual`, `source = swe_bench_site`.", "category": "eval", "split": null, "language": "multilingual", "task_count": 300, "homepage_url": "https://www.swebench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 13, "score_count": 13, "best_success_rate": 72.7}, {"slug": "swe-bench-multimodal", "name": "SWE-bench Multimodal (OpenHands Index)", "description": "SWE-bench Multimodal in OpenHands Index covers UI-forward software engineering with visual context. Harness is fixed to OpenHands SDK (`openhands_index` source).\n\nDistinct from the smaller Multimodal board on swebench.com leaderboards.json (not ingested here).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://index.openhands.dev/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 70.6}, {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)", "description": "The Multimodal board on swebench.com tracks harness\u00d7model scores on frontend and multimodal software engineering tasks (~517 instances). Ingested from the same leaderboards.json cache as Verified (`swe_bench_site`).\n\nNot the same as OpenHands Index `swe-bench-multimodal` component scores (`openhands_index` source, OpenHands SDK fixed harness).", "category": "eval", "split": null, "language": "python", "task_count": 517, "homepage_url": "https://www.swebench.com/multimodal", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/multimodal", "is_public_leaderboard": true, "harness_count": 7, "model_count": 8, "score_count": 15, "best_success_rate": 35.98}, {"slug": "swe-bench-test", "name": "SWE-bench Test", "description": "SWE-bench Test board on swebench.com lists community agent submissions on the Test evaluation split. Catalog ingests agentic rows from leaderboards.json (`swe_bench_site`).\n\nCompare only within this board \u2014 task coverage and scoring differ from Verified and Lite.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://www.swebench.com/", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/", "is_public_leaderboard": true, "harness_count": 5, "model_count": 7, "score_count": 10, "best_success_rate": 52.62}, {"slug": "swe-bench-verified", "name": "SWE-bench Verified", "description": "SWE-bench Verified is the standard held-out evaluation split for software engineering agents: 500 real GitHub issues from popular Python repositories, human-filtered for quality and reproducibility.\n\nThe public leaderboard at swebench.com accepts community harness submissions (`source = swe_bench_site`). Rows are harness \u00d7 model \u00d7 % resolved on the full 500 tasks. This is distinct from the bash-only board (mini-SWE-agent only) and from Artificial Analysis component scores.\n\nHarnessRL uses Verified as the primary eval benchmark in published configs. Scores tagged `is_harnessrl_measured` use fixed harness versions under the HarnessRL eval protocol.", "category": "eval", "split": "verified", "language": "python", "task_count": 500, "homepage_url": "https://www.swebench.com/", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/verified", "is_public_leaderboard": true, "harness_count": 40, "model_count": 88, "score_count": 116, "best_success_rate": 79.2}, {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)", "description": "The bash-only board on swebench.com runs every model through the same mini-SWE-agent configuration on SWE-bench Verified (500 tasks). It isolates foundation-model capability from vendor CLI scaffolding.\n\nCatalog rows use harness `mini-swe-agent` and `source = swe_bench_site`. Compare bash-only rows against product harness submissions on the main Verified board only when explaining scaffold effects \u2014 not as interchangeable rankings.\n\nRelease tags on mini-SWE-agent correspond to leaderboard version numbers in experiment folder names.", "category": "eval", "split": "verified", "language": "python", "task_count": 500, "homepage_url": "https://www.swebench.com/verified", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/verified", "is_public_leaderboard": true, "harness_count": 1, "model_count": 46, "score_count": 46, "best_success_rate": 76.8}, {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)", "description": "HAL's SWE-bench Verified Mini board evaluates community scaffolds on a 49-task subset of Verified issue-fixing tasks. Catalog ingests HTML leaderboard rows (`hal_site`).\n\nCompare against swebench.com Verified (500 tasks) and bash-only mini-SWE-agent rows only when explaining subset and scaffold effects \u2014 not as interchangeable rankings.", "category": "eval", "split": null, "language": "python", "task_count": 49, "homepage_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "is_public_leaderboard": true, "harness_count": 2, "model_count": 17, "score_count": 31, "best_success_rate": 72.0}, {"slug": "swe-rebench", "name": "SWE-rebench", "description": "Continuous SWE-rebench leaderboard with rolling task windows.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://swe-rebench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://swe-rebench.com/", "is_public_leaderboard": true, "harness_count": 5, "model_count": 107, "score_count": 110, "best_success_rate": 100.0}, {"slug": "swt-bench", "name": "SWT-Bench", "description": "SWT-Bench measures test-generation and testing quality. OpenHands Index rows use OpenHands SDK (`openhands_index`). Official site submissions use swtbench.com \u2014 distinct eval configs from OH Index runs.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://swtbench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 91.9}, {"slug": "hal-scicode", "name": "SciCode (HAL)", "description": "Scientific programming benchmark (SciCode) on HAL.", "category": "eval", "split": null, "language": null, "task_count": 80, "homepage_url": "https://hal.cs.princeton.edu/scicode", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/scicode", "is_public_leaderboard": true, "harness_count": 3, "model_count": 15, "score_count": 30, "best_success_rate": 9.23}, {"slug": "hal-scienceagentbench", "name": "ScienceAgentBench (HAL)", "description": "Scientific discovery agent benchmark on HAL.", "category": "eval", "split": null, "language": null, "task_count": 102, "homepage_url": "https://hal.cs.princeton.edu/scienceagentbench", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/scienceagentbench", "is_public_leaderboard": true, "harness_count": 2, "model_count": 15, "score_count": 21, "best_success_rate": 33.33}, {"slug": "hal-tau-bench-airline", "name": "TAU-bench Airline (HAL)", "description": "Customer-service airline booking benchmark on HAL.", "category": "eval", "split": null, "language": null, "task_count": 50, "homepage_url": "https://hal.cs.princeton.edu/taubench_airline", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/taubench_airline", "is_public_leaderboard": true, "harness_count": 2, "model_count": 13, "score_count": 24, "best_success_rate": 56.0}, {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)", "description": "Official Harbor Terminal-Bench 2.1 agent submissions.", "category": "eval", "split": null, "language": "bash", "task_count": null, "homepage_url": "https://github.com/harbor-framework/terminal-bench-2-1", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://github.com/harbor-framework/terminal-bench-2-1", "is_public_leaderboard": true, "harness_count": 6, "model_count": 11, "score_count": 16, "best_success_rate": 83.82}, {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)", "description": "Terminal-Bench v2.1 on Artificial Analysis evaluates agents in interactive terminal environments (shell workflows, CLI tools). Scores appear as AA component rows (`artificialanalysis`).\n\nNot the official [tbench.ai TB 2.1 leaderboard](https://www.tbench.ai/leaderboard/terminal-bench/2.1) matrix. Harbor Hub hosts the maintainer leaderboard \u2014 catalog ingest for official TB 2.1 is on the watchlist.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1", "is_public_leaderboard": true, "harness_count": 10, "model_count": 40, "score_count": 45, "best_success_rate": 91.01}, {"slug": "hal-usaco", "name": "USACO (HAL)", "description": "Competitive programming benchmark (USACO) on HAL.", "category": "eval", "split": null, "language": null, "task_count": 135, "homepage_url": "https://hal.cs.princeton.edu/usaco", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/usaco", "is_public_leaderboard": true, "harness_count": 2, "model_count": 11, "score_count": 12, "best_success_rate": 69.06}, {"slug": "wildclawbench", "name": "WildClawBench", "description": "WildClawBench (InternLM, 2026) evaluates agents on 60 original long-horizon tasks inside a live OpenClaw environment: productivity, coding, social interaction, search, multimodal synthesis, and safety. The official README leaderboard is a harness\u00d7model matrix \u2014 OpenClaw plus Claude Code, Codex, and Hermes Agent on the same task suite (`source = wildclawbench_site`).\n\nCatalog rows use overall score (%). OpenClaw suite totals are stored as per-task cost. Harbor-format tasks are published as internlm/WildClawBench-Harbor. Distinct from SWE-bench issue-fixing boards.", "category": "eval", "split": null, "language": "multilingual", "task_count": 60, "homepage_url": "https://internlm.github.io/WildClawBench/", "paper_url": "https://arxiv.org/abs/2605.10912", "hf_dataset_id": "internlm/WildClawBench", "official_leaderboard_url": "https://internlm.github.io/WildClawBench/", "is_public_leaderboard": true, "harness_count": 4, "model_count": 34, "score_count": 46, "best_success_rate": 67.2}, {"slug": "harnessrl-2699", "name": "HarnessRL-2699", "description": "HarnessRL-2699 is the OpenSWE-derived training index used for Harbor RL experiments (2,699 software-engineering tasks). Category: training \u2014 not a public leaderboard benchmark.\n\nDataset: Hugging Face `Lego-X/HarnessRL-2699`. Used in HarnessRL train configs for rollouts and verifier rewards.", "category": "training", "split": null, "language": "python", "task_count": 2699, "homepage_url": "https://huggingface.co/datasets/Lego-X/HarnessRL-2699", "paper_url": null, "hf_dataset_id": "Lego-X/HarnessRL-2699", "official_leaderboard_url": null, "is_public_leaderboard": false, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}, {"slug": "openswe-filtered", "name": "OpenSWE filtered", "description": "Filtered OpenSWE task subset referenced in HarnessRL training configs as `openswe_filtered`. Training category \u2014 no public harness\u00d7model leaderboard.\n\nUsed for large-scale RL data collection with Harbor verifiers.", "category": "training", "split": null, "language": "python", "task_count": 14415, "homepage_url": null, "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": null, "is_public_leaderboard": false, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}, {"slug": "swerebench-filtered", "name": "SWE-rebench filtered", "description": "Filtered SWE-rebench V2 training index (`swerebench_filtered`) integrated with Harbor verifier rewards in HarnessRL configs. Training only.\n\nPublic continuous eval lives at swe-rebench.com \u2014 separate ingest on watchlist.", "category": "training", "split": null, "language": "python", "task_count": 1800, "homepage_url": "https://swe-rebench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": null, "is_public_leaderboard": false, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}]