{"counts": {"harnesses": 118, "models": 384, "benchmarks": 38, "scores": 1201, "trending_events": 0, "sources": 4}, "hrl_scored": 45, "recent_ingests": [], "last_score_observed_at": "2026-08-27T21:54:11.189561+00:00", "top_harnesses": [{"slug": "hermes-agent", "name": "Hermes Agent", "vendor": "Nous Research", "description": "Autonomous terminal agent with tool use, skills, and multi-platform messaging gateway. Top OpenRouter coding CLI by reported token volume.", "card_summary": "Nous Research Hermes Agent coding agent harness.", "homepage_url": "https://hermes-agent.nousresearch.com/", "repo_url": "https://github.com/NousResearch/hermes-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 50.7, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 100.0, "hrl_rank": 1, "catalog_token_volume": 45629506448921, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kilo-code", "name": "Kilo Code", "vendor": "Kilo", "description": "AI coding agent in the IDE (Kilo Code / Kilo CLI). High-volume OpenRouter coding CLI listed alongside Cline-family tooling.", "card_summary": "Kilo Kilo Code coding agent harness.", "homepage_url": "https://kilocode.ai/", "repo_url": "https://github.com/Kilo-Org/kilocode", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 95.0, "hrl_rank": 2, "catalog_token_volume": 8822861340593, "openrouter_icon_url": "https://kilocode.ai/", "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "cline", "name": "Cline", "vendor": "Cline", "description": "Open-source IDE coding agent for autonomous codebase exploration, edits, terminal commands, and browser automation. Top OpenRouter coding CLI by token volume.", "card_summary": "Cline Cline coding agent harness.", "homepage_url": "https://cline.bot/", "repo_url": "https://github.com/cline/cline", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 92.0, "hrl_rank": 3, "catalog_token_volume": 6072227628651, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "openclaw", "name": "OpenClaw", "vendor": "OpenClaw", "description": "Personal AI assistant platform with terminal CLI and OpenRouter integration across messaging channels.", "card_summary": "OpenClaw OpenClaw coding agent harness.", "homepage_url": "https://openclaw.ai/", "repo_url": "https://github.com/openclaw/openclaw", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 34, "best_success_rate": 67.2, "github_stars": null, "popularity_tier": "popular", "hrl_score": 87.0, "hrl_rank": 4, "catalog_token_volume": 4592638380486, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "description": "Codex (OpenAI) is a coding agent oriented around terminal workflows: reading context, proposing edits, and executing commands against a repository. Leaderboard rows labeled Codex typically pair this harness with OpenAI frontier models on SWE-bench Verified and Artificial Analysis coding-agent benchmarks.\n\nUnlike minimal bash-only scaffolds, Codex ships vendor tooling and prompts tuned for product use. Cross-harness comparisons on the same model should use the same scaffold (for example mini-SWE-agent on the SWE-bench bash-only board) rather than mixing Codex with community SWE-agent forks.\n\nPublic scores here come from `artificialanalysis` (Coding Agent Index components) and community `swe_bench_site` submissions. Check each row's source URL for the exact leaderboard snapshot and model pairing.", "card_summary": "OpenAI's Codex agent for terminal-based coding, repo navigation, and patch application.", "homepage_url": "https://openai.com/codex", "repo_url": "https://github.com/openai/codex", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 37, "best_success_rate": 83.15, "github_stars": null, "popularity_tier": "popular", "hrl_score": 69.0, "hrl_rank": 5, "catalog_token_volume": 2011461977785, "openrouter_icon_url": "https://openai.com/favicon.ico", "leaderboard_icon_url": null, "aa_coding_index": 0.6505240024334414, "aa_mean_cost_usd": 1.5618390533742323, "aa_mean_total_tokens": 5600275}], "featured_harnesses": [{"slug": "hermes-agent", "name": "Hermes Agent", "vendor": "Nous Research", "description": "Autonomous terminal agent with tool use, skills, and multi-platform messaging gateway. Top OpenRouter coding CLI by reported token volume.", "card_summary": "Nous Research Hermes Agent coding agent harness.", "homepage_url": "https://hermes-agent.nousresearch.com/", "repo_url": "https://github.com/NousResearch/hermes-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 50.7, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 100.0, "hrl_rank": 1, "catalog_token_volume": 45629506448921, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kilo-code", "name": "Kilo Code", "vendor": "Kilo", "description": "AI coding agent in the IDE (Kilo Code / Kilo CLI). High-volume OpenRouter coding CLI listed alongside Cline-family tooling.", "card_summary": "Kilo Kilo Code coding agent harness.", "homepage_url": "https://kilocode.ai/", "repo_url": "https://github.com/Kilo-Org/kilocode", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 95.0, "hrl_rank": 2, "catalog_token_volume": 8822861340593, "openrouter_icon_url": "https://kilocode.ai/", "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "cline", "name": "Cline", "vendor": "Cline", "description": "Open-source IDE coding agent for autonomous codebase exploration, edits, terminal commands, and browser automation. Top OpenRouter coding CLI by token volume.", "card_summary": "Cline Cline coding agent harness.", "homepage_url": "https://cline.bot/", "repo_url": "https://github.com/cline/cline", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 92.0, "hrl_rank": 3, "catalog_token_volume": 6072227628651, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "openclaw", "name": "OpenClaw", "vendor": "OpenClaw", "description": "Personal AI assistant platform with terminal CLI and OpenRouter integration across messaging channels.", "card_summary": "OpenClaw OpenClaw coding agent harness.", "homepage_url": "https://openclaw.ai/", "repo_url": "https://github.com/openclaw/openclaw", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 34, "best_success_rate": 67.2, "github_stars": null, "popularity_tier": "popular", "hrl_score": 87.0, "hrl_rank": 4, "catalog_token_volume": 4592638380486, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "description": "Codex (OpenAI) is a coding agent oriented around terminal workflows: reading context, proposing edits, and executing commands against a repository. Leaderboard rows labeled Codex typically pair this harness with OpenAI frontier models on SWE-bench Verified and Artificial Analysis coding-agent benchmarks.\n\nUnlike minimal bash-only scaffolds, Codex ships vendor tooling and prompts tuned for product use. Cross-harness comparisons on the same model should use the same scaffold (for example mini-SWE-agent on the SWE-bench bash-only board) rather than mixing Codex with community SWE-agent forks.\n\nPublic scores here come from `artificialanalysis` (Coding Agent Index components) and community `swe_bench_site` submissions. Check each row's source URL for the exact leaderboard snapshot and model pairing.", "card_summary": "OpenAI's Codex agent for terminal-based coding, repo navigation, and patch application.", "homepage_url": "https://openai.com/codex", "repo_url": "https://github.com/openai/codex", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 37, "best_success_rate": 83.15, "github_stars": null, "popularity_tier": "popular", "hrl_score": 69.0, "hrl_rank": 5, "catalog_token_volume": 2011461977785, "openrouter_icon_url": "https://openai.com/favicon.ico", "leaderboard_icon_url": null, "aa_coding_index": 0.6505240024334414, "aa_mean_cost_usd": 1.5618390533742323, "aa_mean_total_tokens": 5600275}, {"slug": "claude-code", "name": "Claude Code", "vendor": "Anthropic", "description": "Claude Code is Anthropic's official agentic coding product: a terminal-first CLI that plans, searches the repository, edits files, and runs shell commands in a loop until the task is done. It is designed for day-to-day engineering work rather than a minimal benchmark scaffold.\n\nOn public leaderboards, \"Claude Code\" rows usually mean this product (or a pinned release) paired with a specific Claude model \u2014 not a stripped-down bash-only loop. Artificial Analysis publishes a composite Coding Agent Index across DeepSWE, SWE-bench Pro, and Terminal-Bench components; SWE-bench Verified community submissions use vendor-specific harness versions.\n\nHarnessRL validates Claude Code end-to-end: scaffold code `cc`, Harbor agent `BuiltinCCAgentLoop`, Anthropic Messages API protocol, and in-process proxy integration for RL rollouts. Scores tagged `is_harnessrl_measured` in this catalog are from fixed harness versions under the HarnessRL eval protocol.\n\nWhen comparing Claude Code to other vendor CLIs (Codex, Cursor CLI, Gemini CLI), look at rows with the same benchmark slug and similar model tier. When comparing models inside Anthropic's stack, prefer rows that share the same harness version and eval protocol rather than mixing AA component scores with Verified resolution %.", "card_summary": "Anthropic's agentic coding CLI \u2014 terminal edits, search, and multi-step repo workflows via the Messages API.", "homepage_url": "https://github.com/anthropics/claude-code", "repo_url": "https://github.com/anthropics/claude-code", "docs_url": "https://harness-rl.pages.dev/docs/architecture/in-process-proxy", "scaffold_code": "cc", "harbor_agent_name": "BuiltinCCAgentLoop", "api_protocol": "anthropic", "status": "validated", "supports_rl": true, "score_count": 74, "best_success_rate": 89.14, "github_stars": 18400, "popularity_tier": "popular", "hrl_score": 63.0, "hrl_rank": 6, "catalog_token_volume": 12676236218475, "openrouter_icon_url": "https://claude.ai/favicon.ico", "leaderboard_icon_url": null, "aa_coding_index": 0.6814975428587517, "aa_mean_cost_usd": 3.1718671924846644, "aa_mean_total_tokens": 7999358}], "eval_benchmarks": [{"slug": "aider-edit", "name": "Aider Edit", "description": "Aider Edit leaderboard tracks edit-focused coding tasks via Aider (`aider_leaderboard` source). Public YAML in the Aider repository.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://aider.chat/docs/leaderboards/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://aider.chat/docs/leaderboards/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 84, "score_count": 84, "best_success_rate": 84.2}, {"slug": "aider-polyglot", "name": "Aider Polyglot", "description": "Aider Polyglot benchmarks multi-language editing via Aider on 225 Exercism exercises. Scores ingested from public YAML (`aider_leaderboard` source, harness `aider`, metric `pass_rate_2`).\n\nMeasures Aider's edit loop \u2014 not SWE-bench issue resolution. Top public runs approach ~88% pass_rate_2 on recent frontier models.", "category": "eval", "split": null, "language": null, "task_count": 225, "homepage_url": "https://aider.chat/docs/leaderboards/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://aider.chat/docs/leaderboards/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 59, "score_count": 59, "best_success_rate": 88.0}, {"slug": "aider-refactor", "name": "Aider Refactor", "description": "Aider Refactor measures refactoring tasks in Aider's terminal pair-programming loop (`aider_leaderboard` source, harness `aider`).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://aider.chat/docs/leaderboards/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://aider.chat/docs/leaderboards/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 13, "score_count": 13, "best_success_rate": 92.1}, {"slug": "hal-assistantbench", "name": "AssistantBench (HAL)", "description": "Real-world assistant tasks on the HAL platform (AssistantBench board).", "category": "eval", "split": null, "language": null, "task_count": 30, "homepage_url": "https://hal.cs.princeton.edu/assistantbench", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/assistantbench", "is_public_leaderboard": true, "harness_count": 1, "model_count": 14, "score_count": 14, "best_success_rate": 38.81}, {"slug": "hal-corebench-hard", "name": "CORE-Bench Hard (HAL)", "description": "CORE-Bench Hard on HAL measures whether agents can reproduce published computational results. Rows include accuracy and total API cost normalized to per-task cost in the catalog (`hal_site`).\n\nUseful for research-agent and notebook-style scaffolds (Claude Code, CORE-Agent, etc.) beyond pure issue-fixing.", "category": "eval", "split": null, "language": null, "task_count": 45, "homepage_url": "https://hal.cs.princeton.edu/corebench_hard", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/corebench_hard", "is_public_leaderboard": true, "harness_count": 3, "model_count": 24, "score_count": 47, "best_success_rate": 77.78}, {"slug": "commit0", "name": "Commit0", "description": "Commit0 evaluates agents building new libraries from scratch (54\u201357 libs). OpenHands Index publishes SDK runs (`openhands_index`). Historical leaderboards also appear on commit-0.github.io.\n\nGreenfield category \u2014 scores are not comparable to issue-resolution % on Verified without context.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://commit-0.github.io/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 62.5}, {"slug": "deepswe", "name": "DeepSWE (AA component)", "description": "DeepSWE on Artificial Analysis measures long-horizon software engineering performance as part of the Coding Agent Index. AA runs vendor agents on a DeepSWE slice (`source = artificialanalysis`).\n\nThis is not the official [DataCurve DeepSWE leaderboard](https://deepswe.datacurve.ai/) (mini-SWE-agent standardized). Compare AA DeepSWE component scores only against other AA components on the same agent\u00d7model row.\n\nOfficial DeepSWE ingest remains on the catalog watchlist.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://deepswe.datacurve.ai/", "is_public_leaderboard": true, "harness_count": 10, "model_count": 40, "score_count": 45, "best_success_rate": 66.96}, {"slug": "gaia", "name": "GAIA", "description": "GAIA (165 public validation questions) tests general-purpose agent reasoning and tool use. OpenHands Index includes GAIA as the information-gathering pillar (`openhands_index`, harness `openhands-sdk`).\n\nAlso tracked on HAL (hal.cs.princeton.edu/gaia). Peripheral to pure SWE issue-fixing but part of the OH Index average_score.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/gaia", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 86.1}, {"slug": "hal-gaia", "name": "GAIA (HAL)", "description": "The Holistic Agent Leaderboard (HAL) at hal.cs.princeton.edu publishes GAIA validation results with standardized cost tracking and verified traces. Rows are harness \u00d7 model (`source = hal_site`, benchmark `hal-gaia`).\n\nDistinct from OpenHands Index GAIA (`openhands_index`, harness `openhands-sdk`). HAL leaderboard updates were paused in 2026 while the team focuses on reliability; existing rows remain authoritative snapshots.\n\nHAL also publishes multi-dimensional reliability scores on the same tasks (`hal-reliability-gaia`, `hal_reliability_site`).", "category": "eval", "split": null, "language": null, "task_count": 165, "homepage_url": "https://hal.cs.princeton.edu/gaia", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/gaia", "is_public_leaderboard": true, "harness_count": 2, "model_count": 16, "score_count": 30, "best_success_rate": 74.55}, {"slug": "hal-reliability-gaia", "name": "HAL Reliability \u2014 GAIA", "description": "HAL's reliability dashboard evaluates agents on GAIA with repeated runs and multi-dimensional metrics beyond single-shot accuracy. Catalog stores the mean of HAL's four dimension aggregates as `success_rate` (`hal_reliability_site`, harness `hal-generalist-agent`).\n\nSee hal.cs.princeton.edu/reliability/ for full consistency, predictability, robustness, and safety breakdowns.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/reliability/benchmark/gaia/", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/reliability/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 15, "score_count": 15, "best_success_rate": 96.25}, {"slug": "hal-reliability-tau-bench-airline", "name": "HAL Reliability \u2014 \u03c4-bench Airline (clean)", "description": "HAL reliability evaluation on \u03c4-bench airline (clean split).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/reliability/benchmark/taubench_airline/", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/reliability/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 15, "score_count": 15, "best_success_rate": 93.5}, {"slug": "hal-reliability-tau-bench-airline-original", "name": "HAL Reliability \u2014 \u03c4-bench Airline (original)", "description": "HAL reliability evaluation on \u03c4-bench airline (original split).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://hal.cs.princeton.edu/reliability/benchmark/taubench_airline_original/", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/reliability/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 13, "score_count": 13, "best_success_rate": 87.75}, {"slug": "livecodebench-generation", "name": "LiveCodeBench (code generation)", "description": "LiveCodeBench generation track evaluates code generation on competitive programming problems with contamination-aware splits. Catalog ingests aggregated mean pass@1 per model from public JSON (`livecodebench` source).\n\nModel-level metric (implicit eval harness `livecodebench`) \u2014 not harness\u00d7model matrix like SWE-bench Verified. Leaderboard at livecodebench.github.io.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://livecodebench.github.io/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://livecodebench.github.io/leaderboard.html", "is_public_leaderboard": true, "harness_count": 1, "model_count": 26, "score_count": 26, "best_success_rate": 87.3}, {"slug": "mle-bench", "name": "MLE-bench", "description": "MLE-bench measures ML engineering agents on 75 Kaggle competitions with Low/Medium/High complexity splits. Catalog ingests the **All (%)** any-medal column from the openai/mle-bench README leaderboard table (`mle_bench_site` source).\n\nHarness\u00d7model matrix (AIDE, OpenHands, R&D-Agent, Famou-Agent, etc.) \u2014 not software issue-fixing. Canonical setup: 24h runtime, 36 vCPUs, A10 GPU (see paper). Leaderboard submissions paused Apr 2026 per maintainer notice.", "category": "eval", "split": null, "language": null, "task_count": 75, "homepage_url": "https://mlebench.com/leaderboard", "paper_url": "https://arxiv.org/abs/2410.07095", "hf_dataset_id": null, "official_leaderboard_url": "https://mlebench.com/leaderboard", "is_public_leaderboard": true, "harness_count": 19, "model_count": 15, "score_count": 25, "best_success_rate": 64.44}, {"slug": "hal-online-mind2web", "name": "Online Mind2Web (HAL)", "description": "Live web navigation benchmark (Online Mind2Web) on HAL.", "category": "eval", "split": null, "language": null, "task_count": 300, "homepage_url": "https://hal.cs.princeton.edu/online_mind2web", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/online_mind2web", "is_public_leaderboard": true, "harness_count": 2, "model_count": 11, "score_count": 20, "best_success_rate": 42.33}, {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)", "description": "SWE-Atlas-QnA tests repository understanding and question answering \u2014 a component of Artificial Analysis's Coding Agent Index (`artificialanalysis` source).\n\nDistinct from Scale's official [SWE Atlas QnA leaderboard](https://scale.com/leaderboard/sweatlas-qna). AA rows reflect AA's eval harness and task slice, not Scale's public submission pipeline.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://scale.com/leaderboard/sweatlas-qna", "is_public_leaderboard": true, "harness_count": 10, "model_count": 40, "score_count": 45, "best_success_rate": 54.84}, {"slug": "swe-bench", "name": "SWE-bench", "description": "SWE-bench is the original benchmark suite (2,294 real GitHub issues) introduced in Jimenez et al., ICLR 2024. Verified (500 tasks) is the human-curated subset used for most modern leaderboards.\n\nFull swe-bench remains relevant for training and historical comparisons. OpenHands Index publishes swe-bench component scores under `openhands-swe-bench` (OpenHands SDK runs) \u2014 not identical to swebench.com Verified submissions.", "category": "eval", "split": null, "language": "python", "task_count": 2294, "homepage_url": "https://www.swebench.com/", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/", "is_public_leaderboard": true, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}, {"slug": "openhands-swe-bench", "name": "SWE-bench (OpenHands Index)", "description": "OpenHands Index runs a standardized OpenHands SDK configuration across five benchmark families. The swe-bench component measures issue-resolution style tasks (`source = openhands_index`, harness `openhands-sdk`).\n\nScores include cost and runtime metadata from index.openhands.dev. Not interchangeable with swebench.com community submissions or bash-only mini-SWE-agent rows.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://index.openhands.dev/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 95.8}, {"slug": "swe-bench-lite", "name": "SWE-bench Lite", "description": "SWE-bench Lite is the 300-task subset of the original SWE-bench suite used for faster eval. The public leaderboard on swebench.com accepts community harness\u00d7model submissions (`source = swe_bench_site`).\n\nDistinct from Verified (500 tasks) and from training-only indices. Useful for historical comparisons and lighter eval cycles.", "category": "eval", "split": "lite", "language": "python", "task_count": 300, "homepage_url": "https://www.swebench.com/lite", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/lite", "is_public_leaderboard": true, "harness_count": 36, "model_count": 18, "score_count": 50, "best_success_rate": 60.33}, {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual", "description": "SWE-bench Multilingual evaluates issue fixing across multiple languages. The swebench.com board uses mini-SWE-agent for every model (like bash-only on Verified) so rows isolate foundation-model capability.\n\nCatalog: harness `mini-swe-agent`, benchmark `swe-bench-multilingual`, `source = swe_bench_site`.", "category": "eval", "split": null, "language": "multilingual", "task_count": 300, "homepage_url": "https://www.swebench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 13, "score_count": 13, "best_success_rate": 72.7}, {"slug": "swe-bench-multimodal", "name": "SWE-bench Multimodal (OpenHands Index)", "description": "SWE-bench Multimodal in OpenHands Index covers UI-forward software engineering with visual context. Harness is fixed to OpenHands SDK (`openhands_index` source).\n\nDistinct from the smaller Multimodal board on swebench.com leaderboards.json (not ingested here).", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://index.openhands.dev/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 70.6}, {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)", "description": "The Multimodal board on swebench.com tracks harness\u00d7model scores on frontend and multimodal software engineering tasks (~517 instances). Ingested from the same leaderboards.json cache as Verified (`swe_bench_site`).\n\nNot the same as OpenHands Index `swe-bench-multimodal` component scores (`openhands_index` source, OpenHands SDK fixed harness).", "category": "eval", "split": null, "language": "python", "task_count": 517, "homepage_url": "https://www.swebench.com/multimodal", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/multimodal", "is_public_leaderboard": true, "harness_count": 7, "model_count": 8, "score_count": 15, "best_success_rate": 35.98}, {"slug": "swe-bench-test", "name": "SWE-bench Test", "description": "SWE-bench Test board on swebench.com lists community agent submissions on the Test evaluation split. Catalog ingests agentic rows from leaderboards.json (`swe_bench_site`).\n\nCompare only within this board \u2014 task coverage and scoring differ from Verified and Lite.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://www.swebench.com/", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/", "is_public_leaderboard": true, "harness_count": 5, "model_count": 7, "score_count": 10, "best_success_rate": 52.62}, {"slug": "swe-bench-verified", "name": "SWE-bench Verified", "description": "SWE-bench Verified is the standard held-out evaluation split for software engineering agents: 500 real GitHub issues from popular Python repositories, human-filtered for quality and reproducibility.\n\nThe public leaderboard at swebench.com accepts community harness submissions (`source = swe_bench_site`). Rows are harness \u00d7 model \u00d7 % resolved on the full 500 tasks. This is distinct from the bash-only board (mini-SWE-agent only) and from Artificial Analysis component scores.\n\nHarnessRL uses Verified as the primary eval benchmark in published configs. Scores tagged `is_harnessrl_measured` use fixed harness versions under the HarnessRL eval protocol.", "category": "eval", "split": "verified", "language": "python", "task_count": 500, "homepage_url": "https://www.swebench.com/", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/verified", "is_public_leaderboard": true, "harness_count": 40, "model_count": 88, "score_count": 116, "best_success_rate": 79.2}, {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)", "description": "The bash-only board on swebench.com runs every model through the same mini-SWE-agent configuration on SWE-bench Verified (500 tasks). It isolates foundation-model capability from vendor CLI scaffolding.\n\nCatalog rows use harness `mini-swe-agent` and `source = swe_bench_site`. Compare bash-only rows against product harness submissions on the main Verified board only when explaining scaffold effects \u2014 not as interchangeable rankings.\n\nRelease tags on mini-SWE-agent correspond to leaderboard version numbers in experiment folder names.", "category": "eval", "split": "verified", "language": "python", "task_count": 500, "homepage_url": "https://www.swebench.com/verified", "paper_url": "https://arxiv.org/abs/2310.06770", "hf_dataset_id": null, "official_leaderboard_url": "https://www.swebench.com/verified", "is_public_leaderboard": true, "harness_count": 1, "model_count": 46, "score_count": 46, "best_success_rate": 76.8}, {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)", "description": "HAL's SWE-bench Verified Mini board evaluates community scaffolds on a 49-task subset of Verified issue-fixing tasks. Catalog ingests HTML leaderboard rows (`hal_site`).\n\nCompare against swebench.com Verified (500 tasks) and bash-only mini-SWE-agent rows only when explaining subset and scaffold effects \u2014 not as interchangeable rankings.", "category": "eval", "split": null, "language": "python", "task_count": 49, "homepage_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "is_public_leaderboard": true, "harness_count": 2, "model_count": 17, "score_count": 31, "best_success_rate": 72.0}, {"slug": "swe-rebench", "name": "SWE-rebench", "description": "Continuous SWE-rebench leaderboard with rolling task windows.", "category": "eval", "split": null, "language": "python", "task_count": null, "homepage_url": "https://swe-rebench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://swe-rebench.com/", "is_public_leaderboard": true, "harness_count": 5, "model_count": 107, "score_count": 110, "best_success_rate": 100.0}, {"slug": "swt-bench", "name": "SWT-Bench", "description": "SWT-Bench measures test-generation and testing quality. OpenHands Index rows use OpenHands SDK (`openhands_index`). Official site submissions use swtbench.com \u2014 distinct eval configs from OH Index runs.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://swtbench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://index.openhands.dev/", "is_public_leaderboard": true, "harness_count": 1, "model_count": 33, "score_count": 33, "best_success_rate": 91.9}, {"slug": "hal-scicode", "name": "SciCode (HAL)", "description": "Scientific programming benchmark (SciCode) on HAL.", "category": "eval", "split": null, "language": null, "task_count": 80, "homepage_url": "https://hal.cs.princeton.edu/scicode", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/scicode", "is_public_leaderboard": true, "harness_count": 3, "model_count": 15, "score_count": 30, "best_success_rate": 9.23}, {"slug": "hal-scienceagentbench", "name": "ScienceAgentBench (HAL)", "description": "Scientific discovery agent benchmark on HAL.", "category": "eval", "split": null, "language": null, "task_count": 102, "homepage_url": "https://hal.cs.princeton.edu/scienceagentbench", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/scienceagentbench", "is_public_leaderboard": true, "harness_count": 2, "model_count": 15, "score_count": 21, "best_success_rate": 33.33}, {"slug": "hal-tau-bench-airline", "name": "TAU-bench Airline (HAL)", "description": "Customer-service airline booking benchmark on HAL.", "category": "eval", "split": null, "language": null, "task_count": 50, "homepage_url": "https://hal.cs.princeton.edu/taubench_airline", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/taubench_airline", "is_public_leaderboard": true, "harness_count": 2, "model_count": 13, "score_count": 24, "best_success_rate": 56.0}, {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)", "description": "Official Harbor Terminal-Bench 2.1 agent submissions.", "category": "eval", "split": null, "language": "bash", "task_count": null, "homepage_url": "https://github.com/harbor-framework/terminal-bench-2-1", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://github.com/harbor-framework/terminal-bench-2-1", "is_public_leaderboard": true, "harness_count": 6, "model_count": 11, "score_count": 16, "best_success_rate": 83.82}, {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)", "description": "Terminal-Bench v2.1 on Artificial Analysis evaluates agents in interactive terminal environments (shell workflows, CLI tools). Scores appear as AA component rows (`artificialanalysis`).\n\nNot the official [tbench.ai TB 2.1 leaderboard](https://www.tbench.ai/leaderboard/terminal-bench/2.1) matrix. Harbor Hub hosts the maintainer leaderboard \u2014 catalog ingest for official TB 2.1 is on the watchlist.", "category": "eval", "split": null, "language": null, "task_count": null, "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1", "is_public_leaderboard": true, "harness_count": 10, "model_count": 40, "score_count": 45, "best_success_rate": 91.01}, {"slug": "hal-usaco", "name": "USACO (HAL)", "description": "Competitive programming benchmark (USACO) on HAL.", "category": "eval", "split": null, "language": null, "task_count": 135, "homepage_url": "https://hal.cs.princeton.edu/usaco", "paper_url": "https://arxiv.org/abs/2510.11977", "hf_dataset_id": null, "official_leaderboard_url": "https://hal.cs.princeton.edu/usaco", "is_public_leaderboard": true, "harness_count": 2, "model_count": 11, "score_count": 12, "best_success_rate": 69.06}, {"slug": "wildclawbench", "name": "WildClawBench", "description": "WildClawBench (InternLM, 2026) evaluates agents on 60 original long-horizon tasks inside a live OpenClaw environment: productivity, coding, social interaction, search, multimodal synthesis, and safety. The official README leaderboard is a harness\u00d7model matrix \u2014 OpenClaw plus Claude Code, Codex, and Hermes Agent on the same task suite (`source = wildclawbench_site`).\n\nCatalog rows use overall score (%). OpenClaw suite totals are stored as per-task cost. Harbor-format tasks are published as internlm/WildClawBench-Harbor. Distinct from SWE-bench issue-fixing boards.", "category": "eval", "split": null, "language": "multilingual", "task_count": 60, "homepage_url": "https://internlm.github.io/WildClawBench/", "paper_url": "https://arxiv.org/abs/2605.10912", "hf_dataset_id": "internlm/WildClawBench", "official_leaderboard_url": "https://internlm.github.io/WildClawBench/", "is_public_leaderboard": true, "harness_count": 4, "model_count": 34, "score_count": 46, "best_success_rate": 67.2}], "training_benchmarks": [{"slug": "harnessrl-2699", "name": "HarnessRL-2699", "description": "HarnessRL-2699 is the OpenSWE-derived training index used for Harbor RL experiments (2,699 software-engineering tasks). Category: training \u2014 not a public leaderboard benchmark.\n\nDataset: Hugging Face `Lego-X/HarnessRL-2699`. Used in HarnessRL train configs for rollouts and verifier rewards.", "category": "training", "split": null, "language": "python", "task_count": 2699, "homepage_url": "https://huggingface.co/datasets/Lego-X/HarnessRL-2699", "paper_url": null, "hf_dataset_id": "Lego-X/HarnessRL-2699", "official_leaderboard_url": null, "is_public_leaderboard": false, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}, {"slug": "openswe-filtered", "name": "OpenSWE filtered", "description": "Filtered OpenSWE task subset referenced in HarnessRL training configs as `openswe_filtered`. Training category \u2014 no public harness\u00d7model leaderboard.\n\nUsed for large-scale RL data collection with Harbor verifiers.", "category": "training", "split": null, "language": "python", "task_count": 14415, "homepage_url": null, "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": null, "is_public_leaderboard": false, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}, {"slug": "swerebench-filtered", "name": "SWE-rebench filtered", "description": "Filtered SWE-rebench V2 training index (`swerebench_filtered`) integrated with Harbor verifier rewards in HarnessRL configs. Training only.\n\nPublic continuous eval lives at swe-rebench.com \u2014 separate ingest on watchlist.", "category": "training", "split": null, "language": "python", "task_count": 1800, "homepage_url": "https://swe-rebench.com/", "paper_url": null, "hf_dataset_id": null, "official_leaderboard_url": null, "is_public_leaderboard": false, "harness_count": 0, "model_count": 0, "score_count": 0, "best_success_rate": null}]}