[{"slug": "hermes-agent", "name": "Hermes Agent", "vendor": "Nous Research", "description": "Autonomous terminal agent with tool use, skills, and multi-platform messaging gateway. Top OpenRouter coding CLI by reported token volume.", "card_summary": "Nous Research Hermes Agent coding agent harness.", "homepage_url": "https://hermes-agent.nousresearch.com/", "repo_url": "https://github.com/NousResearch/hermes-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 50.7, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 100.0, "hrl_rank": 1, "catalog_token_volume": 45629506448921, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kilo-code", "name": "Kilo Code", "vendor": "Kilo", "description": "AI coding agent in the IDE (Kilo Code / Kilo CLI). High-volume OpenRouter coding CLI listed alongside Cline-family tooling.", "card_summary": "Kilo Kilo Code coding agent harness.", "homepage_url": "https://kilocode.ai/", "repo_url": "https://github.com/Kilo-Org/kilocode", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 95.0, "hrl_rank": 2, "catalog_token_volume": 8822861340593, "openrouter_icon_url": "https://kilocode.ai/", "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "cline", "name": "Cline", "vendor": "Cline", "description": "Open-source IDE coding agent for autonomous codebase exploration, edits, terminal commands, and browser automation. Top OpenRouter coding CLI by token volume.", "card_summary": "Cline Cline coding agent harness.", "homepage_url": "https://cline.bot/", "repo_url": "https://github.com/cline/cline", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "super_popular", "hrl_score": 92.0, "hrl_rank": 3, "catalog_token_volume": 6072227628651, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "openclaw", "name": "OpenClaw", "vendor": "OpenClaw", "description": "Personal AI assistant platform with terminal CLI and OpenRouter integration across messaging channels.", "card_summary": "OpenClaw OpenClaw coding agent harness.", "homepage_url": "https://openclaw.ai/", "repo_url": "https://github.com/openclaw/openclaw", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 34, "best_success_rate": 67.2, "github_stars": null, "popularity_tier": "popular", "hrl_score": 87.0, "hrl_rank": 4, "catalog_token_volume": 4592638380486, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "description": "Codex (OpenAI) is a coding agent oriented around terminal workflows: reading context, proposing edits, and executing commands against a repository. Leaderboard rows labeled Codex typically pair this harness with OpenAI frontier models on SWE-bench Verified and Artificial Analysis coding-agent benchmarks.\n\nUnlike minimal bash-only scaffolds, Codex ships vendor tooling and prompts tuned for product use. Cross-harness comparisons on the same model should use the same scaffold (for example mini-SWE-agent on the SWE-bench bash-only board) rather than mixing Codex with community SWE-agent forks.\n\nPublic scores here come from `artificialanalysis` (Coding Agent Index components) and community `swe_bench_site` submissions. Check each row's source URL for the exact leaderboard snapshot and model pairing.", "card_summary": "OpenAI's Codex agent for terminal-based coding, repo navigation, and patch application.", "homepage_url": "https://openai.com/codex", "repo_url": "https://github.com/openai/codex", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 37, "best_success_rate": 83.15, "github_stars": null, "popularity_tier": "popular", "hrl_score": 69.0, "hrl_rank": 5, "catalog_token_volume": 2011461977785, "openrouter_icon_url": "https://openai.com/favicon.ico", "leaderboard_icon_url": null, "aa_coding_index": 0.6505240024334414, "aa_mean_cost_usd": 1.5618390533742323, "aa_mean_total_tokens": 5600275}, {"slug": "claude-code", "name": "Claude Code", "vendor": "Anthropic", "description": "Claude Code is Anthropic's official agentic coding product: a terminal-first CLI that plans, searches the repository, edits files, and runs shell commands in a loop until the task is done. It is designed for day-to-day engineering work rather than a minimal benchmark scaffold.\n\nOn public leaderboards, \"Claude Code\" rows usually mean this product (or a pinned release) paired with a specific Claude model \u2014 not a stripped-down bash-only loop. Artificial Analysis publishes a composite Coding Agent Index across DeepSWE, SWE-bench Pro, and Terminal-Bench components; SWE-bench Verified community submissions use vendor-specific harness versions.\n\nHarnessRL validates Claude Code end-to-end: scaffold code `cc`, Harbor agent `BuiltinCCAgentLoop`, Anthropic Messages API protocol, and in-process proxy integration for RL rollouts. Scores tagged `is_harnessrl_measured` in this catalog are from fixed harness versions under the HarnessRL eval protocol.\n\nWhen comparing Claude Code to other vendor CLIs (Codex, Cursor CLI, Gemini CLI), look at rows with the same benchmark slug and similar model tier. When comparing models inside Anthropic's stack, prefer rows that share the same harness version and eval protocol rather than mixing AA component scores with Verified resolution %.", "card_summary": "Anthropic's agentic coding CLI \u2014 terminal edits, search, and multi-step repo workflows via the Messages API.", "homepage_url": "https://github.com/anthropics/claude-code", "repo_url": "https://github.com/anthropics/claude-code", "docs_url": "https://harness-rl.pages.dev/docs/architecture/in-process-proxy", "scaffold_code": "cc", "harbor_agent_name": "BuiltinCCAgentLoop", "api_protocol": "anthropic", "status": "validated", "supports_rl": true, "score_count": 74, "best_success_rate": 89.14, "github_stars": 18400, "popularity_tier": "popular", "hrl_score": 63.0, "hrl_rank": 6, "catalog_token_volume": 12676236218475, "openrouter_icon_url": "https://claude.ai/favicon.ico", "leaderboard_icon_url": null, "aa_coding_index": 0.6814975428587517, "aa_mean_cost_usd": 3.1718671924846644, "aa_mean_total_tokens": 7999358}, {"slug": "zed-editor", "name": "Zed", "vendor": "Zed Industries", "description": "High-performance code editor with built-in AI agent and assistant workflows. Listed on OpenRouter's coding CLI directory.", "card_summary": "Zed Industries Zed coding agent harness.", "homepage_url": "https://zed.dev/", "repo_url": "https://github.com/zed-industries/zed", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": "popular", "hrl_score": 62.0, "hrl_rank": 7, "catalog_token_volume": 533964221593, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "openhands-sdk", "name": "OpenHands SDK", "vendor": "All Hands AI", "description": "OpenHands SDK is the modern integration surface for the OpenHands project (All Hands AI). HarnessRL uses a thin adapter around the SDK agent loop: rollouts run inside Harbor sandboxes, and task success is scored by the Harbor verifier rather than ad-hoc string matching.\n\nThis is the default validated scaffold for many HarnessRL train/eval configs (`scaffold_code` `ohsdk`, Harbor `CustomOpenHandsSDK`). It supersedes the legacy `openhands` scaffold (`SCAFFOLD=oh`) for new work, though both share the upstream OpenHands repository.\n\nPublic scores appear under `swe_bench_site` (community submissions), `openhands_index` (OpenHands Index five-benchmark matrix), Artificial Analysis, and HarnessRL-measured rows when present.", "card_summary": "Primary OpenHands scaffold in HarnessRL \u2014 SDK agent loop with Harbor verifier reward.", "homepage_url": "https://github.com/All-Hands-AI/OpenHands", "repo_url": "https://github.com/All-Hands-AI/OpenHands", "docs_url": "https://harness-rl.pages.dev/docs/architecture/agent-loop-workers", "scaffold_code": "ohsdk", "harbor_agent_name": "CustomOpenHandsSDK", "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 177, "best_success_rate": 95.8, "github_stars": 52300, "popularity_tier": "popular", "hrl_score": 59.0, "hrl_rank": 8, "catalog_token_volume": 565593629963, "openrouter_icon_url": "https://www.all-hands.dev/favicon.ico", "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241103_OpenHands-CodeAct-2.1-sonnet-20241022.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "opencode", "name": "OpenCode", "vendor": "SST", "description": "OpenCode is an open-source agent shell from SST (makers of SST/Ion). It provides a local coding agent loop with configurable providers, tool parsing, and repository context \u2014 a general-purpose scaffold rather than a single-vendor CLI.\n\nHarnessRL drives OpenCode through the OpenAI-compatible proxy path (`scaffold_code` `oc`, Harbor `BuiltinOCAgentLoop`). That lets RL training treat OpenCode like other validated scaffolds while preserving its native tool format via scaffold-specific parsing.\n\nBenchmark coverage includes SWE-bench Verified community scores and Artificial Analysis coding-agent components. Rows may appear as \"OpenCode\" or versioned display labels on AA; alias mapping in ingest ties them to this slug.", "card_summary": "SST's open agent shell \u2014 local-first coding agent with OpenAI-compatible tool routing.", "homepage_url": "https://github.com/sst/opencode", "repo_url": "https://github.com/sst/opencode", "docs_url": null, "scaffold_code": "oc", "harbor_agent_name": "BuiltinOCAgentLoop", "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 15, "best_success_rate": 91.01, "github_stars": 6200, "popularity_tier": "popular", "hrl_score": 55.0, "hrl_rank": 9, "catalog_token_volume": 0, "openrouter_icon_url": "https://opencode.ai/favicon.ico", "leaderboard_icon_url": "https://github.com/sst/opencode/raw/dev/packages/web/public/favicon.ico", "aa_coding_index": 0.5962784529614886, "aa_mean_cost_usd": 2.942840985557261, "aa_mean_total_tokens": 7582747}, {"slug": "muse-code", "name": "Muse Code", "vendor": "Muse", "description": "Muse Code is a coding-agent product line with public benchmark submissions pairing Muse's harness with frontier models. Rows appear on Artificial Analysis coding-agent components and occasional SWE-bench Verified community tables.\n\nAs a vendor product harness, scores reflect Muse's agent stack (tools, prompts, retrieval) plus the disclosed model. Compare against other product CLIs rather than minimal bash-only baselines when the question is vendor competitiveness.\n\nCatalog scores use `artificialanalysis` and `swe_bench_site` sources. This page does not publish Muse's internal eval configs \u2014 only linked public leaderboard snapshots.", "card_summary": "Muse coding agent \u2014 SWE-bench Verified and Artificial Analysis leaderboard presence.", "homepage_url": "https://muse.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 81.65, "github_stars": null, "popularity_tier": "popular", "hrl_score": 53.0, "hrl_rank": 10, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.6164044159758908, "aa_mean_cost_usd": 2.072823051942741, "aa_mean_total_tokens": 19963139}, {"slug": "cursor-cli", "name": "Cursor CLI", "vendor": "Cursor", "description": "Cursor CLI exposes Cursor's coding agent as a terminal command: repository indexing, multi-file edits, and tool use aimed at professional IDE-adjacent workflows. Leaderboard entries typically pair Cursor's harness with frontier models on SWE-bench Verified and Artificial Analysis benchmarks.\n\nThis is a product harness (not a minimal research scaffold). Scores reflect Cursor's prompts, tools, and retrieval stack in addition to the backing model. Compare like-with-like: Cursor CLI vs Claude Code vs Codex, not vs bash-only mini-SWE-agent baselines.\n\nScores in this catalog are ingested from `artificialanalysis` and `swe_bench_site`; each row links to its source snapshot.", "card_summary": "Cursor's CLI agent for repo-aware edits, codebase search, and shell workflows.", "homepage_url": "https://cursor.com", "repo_url": "https://github.com/getcursor/cursor", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 14, "best_success_rate": 79.33, "github_stars": null, "popularity_tier": null, "hrl_score": 48.0, "hrl_rank": 11, "catalog_token_volume": 0, "openrouter_icon_url": "https://cursor.com/apple-touch-icon.png", "leaderboard_icon_url": null, "aa_coding_index": 0.47086936005984537, "aa_mean_cost_usd": 0.08505141359918204, "aa_mean_total_tokens": 3705076}, {"slug": "antigravity-sdk-v0-1-8", "name": "Antigravity Sdk V0 1 8", "vendor": "Antigravity", "description": "Antigravity SDK is a coding-agent SDK with versioned public benchmark entries (this slug pins v0.1.8). Submissions pair the SDK's agent loop and tool interfaces with specific foundation models on Artificial Analysis and related leaderboards.\n\nVersion strings in the slug matter: later SDK releases may appear as separate harness rows when authors publish new scores. Scores encode both SDK capabilities and model choice.\n\nPrimarily `artificialanalysis` ingest in this catalog. Check component benchmark slugs (DeepSWE, Terminal-Bench, etc.) in the scores table below \u2014 composite AA index alone does not replace per-benchmark reading.", "card_summary": "Antigravity SDK v0.1.8 \u2014 agent harness for coding benchmark submissions.", "homepage_url": "https://antigravity.dev", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 86.52, "github_stars": null, "popularity_tier": null, "hrl_score": 47.0, "hrl_rank": 12, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.5657152973570364, "aa_mean_cost_usd": 1.396316917192255, "aa_mean_total_tokens": 15113684}, {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "description": "mini-SWE-agent is a deliberately small agent from the SWE-agent team (~100 lines of core loop). The model receives a single bash tool in a ReAct pattern: propose a command, observe output, repeat. There are no rich IDE tools or vendor-specific edit formats \u2014 leaderboard scores emphasize model capability over scaffold engineering.\n\nThe SWE-bench site publishes an official bash-only view where every model runs the same mini-SWE-agent configuration on SWE-bench Verified (500 tasks). Release tags on the mini-SWE-agent repository correspond to leaderboard version numbers.\n\nBecause the harness is minimal, many third-party evaluators (Vals.ai, SWE-rebench model baselines, DeepSWE comparisons) also use it. It is not interchangeable with full SWE-agent, which adds richer tool interfaces and tuned prompts. Catalog scores with `source = swe_bench_site` are community harness submissions using this scaffold.\n\nThis harness is the fairest public baseline when the question is \"how good is the model?\" rather than \"how good is the product?\" If a vendor CLI scores higher than mini-SWE-agent on the same model, that gap is largely scaffold and tooling \u2014 not a contradiction.", "card_summary": "Minimal bash-only ReAct loop \u2014 the reference scaffold for apples-to-apples LM comparisons.", "homepage_url": "https://github.com/SWE-agent/mini-swe-agent", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 212, "best_success_rate": 100.0, "github_stars": 1800, "popularity_tier": null, "hrl_score": 36.0, "hrl_rank": 13, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/mini-icon.svg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aider", "name": "Aider", "vendor": "Aider", "description": "Aider is a mature open-source coding agent focused on pair programming in the terminal: unified diffs, repository map context, and git integration. Public YAML leaderboards on aider.chat publish harness\u00d7model pass rates for Polyglot (225 Exercism tasks), Edit, and Refactor tracks \u2014 ingested as `source = aider_leaderboard`.\n\nOn SWE-bench Verified, Aider also appears as a community harness submission (`swe_bench_site`) with vendor-chosen models. Artificial Analysis publishes component scores where Aider is the agent product.\n\nWhen reading scores, note whether the row is SWE-bench Verified (issue resolution %) or an Aider-native benchmark (edit accuracy, polyglot pass rate) \u2014 filter the scores table by benchmark slug.", "card_summary": "Terminal pair-programming agent \u2014 Polyglot, Edit, Refactor YAML leaderboards plus SWE-bench submissions.", "homepage_url": "https://aider.chat/", "repo_url": "https://github.com/Aider-AI/aider", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 157, "best_success_rate": 92.1, "github_stars": 24800, "popularity_tier": null, "hrl_score": 35.0, "hrl_rank": 14, "catalog_token_volume": 0, "openrouter_icon_url": "https://aider.chat/favicon.ico", "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240523_aider.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "amazon-q-developer", "name": "Amazon Q Developer", "vendor": "Amazon", "description": "Amazon Q Developer is AWS's AI coding assistant (IDE and CLI surfaces). SWE-bench Verified leaderboard entries use Amazon's agent harness with disclosed foundation models, submitted as community results on swebench.com.\n\nScores represent Amazon's product agent (tooling, retrieval, safety filters) plus the backing model \u2014 not a minimal open scaffold. Enterprise features (IAM, repo connectors) may differ from the public eval configuration; rely on each row's `source_url` and submission metadata.\n\nThis catalog ingests those public leaderboard rows via `swe_bench_site`. CLI repository: `aws/amazon-q-developer-cli`.", "card_summary": "Amazon Q Developer agent \u2014 AWS coding assistant with SWE-bench Verified submissions.", "homepage_url": "https://aws.amazon.com/q/developer/", "repo_url": "https://github.com/aws/amazon-q-developer-cli", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": 980, "popularity_tier": null, "hrl_score": 35.0, "hrl_rank": 15, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "openhands", "name": "OpenHands (legacy)", "vendor": "All Hands AI", "description": "This slug tracks the older OpenHands integration in HarnessRL (`scaffold_code` `oh`, Harbor agent `OpenHands`). It targets the same All Hands AI codebase as `openhands-sdk` but through the previous Harbor wiring and configuration surface.\n\nFor new experiments and RL, prefer `openhands-sdk`. This entry remains for historical leaderboard rows and configs that still reference `oh`.\n\nBenchmark appearances mirror the SDK path on SWE-bench Verified and AA ingest aliases (AA product name \"OpenHands\" maps here and to the SDK slug depending on ingest version).", "card_summary": "Legacy OpenHands Harbor integration (`SCAFFOLD=oh`) \u2014 same upstream project as the SDK path.", "homepage_url": "https://github.com/All-Hands-AI/OpenHands", "repo_url": "https://github.com/All-Hands-AI/OpenHands", "docs_url": null, "scaffold_code": "oh", "harbor_agent_name": "OpenHands", "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 26.67, "github_stars": 52300, "popularity_tier": null, "hrl_score": 34.0, "hrl_rank": 16, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240725_opendevin_codeact_v1.8_claude35sonnet.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "composio-swekit", "name": "Composio SWEKit", "vendor": "Composiohq", "description": "Composio SWE-kit is Composio's agent scaffold emphasizing tool integrations (third-party APIs, actions) applied to SWE-bench Verified tasks. Public submissions appear under this harness name on swebench.com.\n\nCommunity `swe_bench_site` scores. Useful when studying tool-rich agents vs minimal bash loops \u2014 compare cost and token columns in the table below.", "card_summary": "Composio SWE-kit \u2014 tool-integration agent on Verified.", "homepage_url": "https://composio.dev", "repo_url": "https://github.com/ComposioHQ/composio/tree/master/python/swe/agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 40.6, "github_stars": 29874, "popularity_tier": null, "hrl_score": 33.0, "hrl_rank": 17, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241016_composio_swekit.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "roo-code", "name": "Roo Code", "vendor": "Roo Code", "description": "VS Code extension and autonomous coding agent (formerly Roo Cline). Listed on OpenRouter's coding CLI directory.", "card_summary": "Roo Code Roo Code coding agent harness.", "homepage_url": "https://roocode.com/", "repo_url": "https://github.com/RooCodeInc/Roo-Code", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": null, "hrl_score": 33.0, "hrl_rank": 18, "catalog_token_volume": 199095095971, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "description": "SWE-agent is the reference open-source agent from Princeton NLP for software engineering benchmarks. It provides structured tools (file editor, bash, search), configurable agent policies, and the ecosystem that spawned mini-SWE-agent for controlled comparisons.\n\nLeaderboard rows labeled SWE-agent usually mean this full scaffold with a disclosed model, not the bash-only mini variant. SWE-bench Verified community submissions and Artificial Analysis ingest map the product name here when aliases match.\n\nHarnessRL lists SWE-agent as a community scaffold (`supports_rl` false) \u2014 useful for tracking public scores but not part of the validated RL scaffold set. For model-only baselines on Verified, see mini-SWE-agent and the SWE-bench bash-only board.", "card_summary": "Princeton NLP's open LM agent loop \u2014 rich tools and prompts for software engineering tasks.", "homepage_url": "https://github.com/SWE-agent/SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 35, "best_success_rate": 72.0, "github_stars": 14200, "popularity_tier": null, "hrl_score": 33.0, "hrl_rank": 19, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240402_sweagent_claude3opus.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "lingma-agent", "name": "Lingma Agent", "vendor": "Alibaba", "description": "Lingma Agent refers to Alibaba's software-engineering agent offerings (Lingma / Tongyi ecosystem) with public ModelScope and SWE-bench leaderboard presence. Submissions pair the Lingma agent harness with disclosed models on Verified tasks.\n\nRows in this catalog are ingested from community SWE-bench submissions. The agent stack includes vendor retrieval, edit tools, and prompts beyond a minimal bash loop.\n\nFor model weights and training details, see ModelScope model cards (for example Lingma-SWE-GPT family); this page tracks harness-level leaderboard scores only.", "card_summary": "Alibaba Lingma SWE agent \u2014 ModelScope-backed submissions on SWE-bench Verified.", "homepage_url": "https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT", "repo_url": "https://github.com/modelscope/modelscope", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 28.8, "github_stars": 7200, "popularity_tier": null, "hrl_score": 32.0, "hrl_rank": 20, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240918_lingma-agent_lingma-swe-gpt-7b.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-agent-javascript", "name": "SWE-agent (JavaScript)", "vendor": "Princeton NLP", "description": "SWE-agent (JavaScript) is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Princeton NLP SWE-agent (JavaScript) coding agent harness.", "homepage_url": "https://github.com/SWE-agent/SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 11.99, "github_stars": 14200, "popularity_tier": null, "hrl_score": 32.0, "hrl_rank": 21, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241006_SWE-agent_JS_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-agent-multimodal", "name": "SWE-agent (multimodal)", "vendor": "Princeton NLP", "description": "SWE-agent (multimodal) is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Princeton NLP SWE-agent (multimodal) coding agent harness.", "homepage_url": "https://github.com/SWE-agent/SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 12.19, "github_stars": 14200, "popularity_tier": null, "hrl_score": 32.0, "hrl_rank": 22, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241006_SWE-agent_M_claude3.5.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "bytedance-marscode-agent", "name": "Bytedance Marscode Agent", "vendor": "ByteDance", "description": "MarsCode Agent is ByteDance's software-engineering agent (MarsCode / TRAE ecosystem) appearing on Artificial Analysis coding-agent benchmarks. Related open scaffold: `trae-agent` for SWE-bench Verified community rows.\n\nThis slug tracks AA ingest rows for the MarsCode agent product naming. `artificialanalysis` source.", "card_summary": "ByteDance MarsCode Agent \u2014 TRAE/MarsCode ecosystem SWE harness.", "homepage_url": "https://www.marscode.com", "repo_url": "https://github.com/bytedance/trae-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 34.0, "github_stars": 2100, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 23, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240723_marscode-agent-dev.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codeshellagent", "name": "Codeshellagent", "vendor": "Wisdomshell", "description": "CodeShell Agent appears on SWE-bench Verified as a community submission harness \u2014 typically pairing the CodeShell ecosystem's agent with a disclosed foundation model.\n\n`swe_bench_site` ingest on `swe-bench-verified`.", "card_summary": "CodeShell agent \u2014 SWE-bench Verified community submission.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/WisdomShell/codeshell", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 44.2, "github_stars": 1621, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 24, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250118_codeshellagent_gemini_2.0_flash_experimental.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codeshelltester", "name": "CodeShellTester", "vendor": "Wisdomshell", "description": "CodeShellTester is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Wisdomshell CodeShellTester coding agent harness.", "homepage_url": "https://github.com/WisdomShell/codeshell", "repo_url": "https://github.com/WisdomShell/codeshell", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 31.33, "github_stars": 1621, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 25, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241111_codeshelltester_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "moatless-tools", "name": "Moatless Tools", "vendor": "Aorwall", "description": "Moatless Tools is an agentic coding toolkit (Moatless / aorwall ecosystem) with public SWE-bench Verified leaderboard entries. The harness provides structured tools and workflows for repository editing and issue resolution benchmarks.\n\nRelated infrastructure repos (for example SWE-bench docker tooling) sometimes appear in submission metadata; the agent loop itself is documented under `aorwall/moatless-tools`.\n\nCommunity `swe_bench_site` scores only. Compare against Moatless rows with matching model and submission date \u2014 Moatless is not interchangeable with SWE-agent or OpenHands without reading each `source_url`.", "card_summary": "Moatless Tools agent scaffold \u2014 SWE-bench Verified community submissions.", "homepage_url": "https://github.com/aorwall/moatless-tools", "repo_url": "https://github.com/aorwall/moatless-tools", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 5, "best_success_rate": 70.8, "github_stars": 1200, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 26, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240617_moatless_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "description": "Terminus-2 is Harbor's reference agent that drives an interactive terminal (tmux) inside the evaluation container. Instead of only issuing one-shot shell strings, it can send keystrokes, scroll, and manage multiplexer sessions \u2014 closer to how engineers use a real terminal.\n\nHarnessRL validates Terminus-2 end-to-end (`scaffold_code` `terminus`, Harbor `Terminus2`, OpenAI-compatible API). It is a strong choice when tasks require interactive CLI tools, pagers, or long-running processes.\n\nPublic catalog scores are primarily HarnessRL-measured or curated seed rows; community SWE-bench submissions more often use SWE-agent family scaffolds. Check `supports_rl` and measured flags on individual score rows.", "card_summary": "Harbor Terminus-2 agent \u2014 in-process tmux control inside the sandbox pod.", "homepage_url": "https://github.com/Elvin-Yiming-Du/harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "docs_url": "https://harness-rl.pages.dev/docs/architecture/agent-loop-workers", "scaffold_code": "terminus", "harbor_agent_name": "Terminus2", "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 5, "best_success_rate": 80.45, "github_stars": 890, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 27, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "trae-agent", "name": "Trae Agent", "vendor": "ByteDance", "description": "TRAE (ByteDance) is an agent scaffold with strong public SWE-bench Verified submissions. The open `trae-agent` repository documents the agent loop, tools, and evaluation setup used for leaderboard runs.\n\nLeaderboard display names may appear as \"TRAE\" or \"Trae\"; ingest aliases map both to this slug. Scores are community submissions (`swe_bench_site`) unless otherwise tagged.\n\nCompare TRAE rows against other product harnesses (Claude Code, Codex, Cursor CLI) when evaluating vendor agents, or against mini-SWE-agent when evaluating raw model capability on a minimal scaffold.", "card_summary": "ByteDance TRAE agent scaffold \u2014 competitive SWE-bench Verified leaderboard entries.", "homepage_url": "https://github.com/bytedance/trae-agent", "repo_url": "https://github.com/bytedance/trae-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 78.8, "github_stars": 2100, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 28, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250612_trae.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "agentless-lite", "name": "Agentless Lite", "vendor": "OpenAutoCoder", "description": "Agentless Lite is a slim variant of the Agentless approach: separate localization and repair stages with minimal agentic looping, aimed at efficient SWE-bench runs. The parent Agentless project demonstrated strong Verified numbers without a traditional tool-using agent.\n\nLeaderboard submissions use this harness name with specific model pairings. It is distinct from full agent scaffolds (SWE-agent, OpenHands) \u2014 scores reflect the agentless pipeline's edit format and search strategy.\n\nCatalog scores are from `swe_bench_site`. Open-source code lives under the Agentless / OpenAutoCoder repositories.", "card_summary": "Lightweight agentless pipeline \u2014 localization and repair without a full agent loop.", "homepage_url": "https://github.com/OpenAutoCoder/Agentless", "repo_url": "https://github.com/OpenAutoCoder/Agentless", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 8, "best_success_rate": 50.8, "github_stars": 680, "popularity_tier": null, "hrl_score": 30.0, "hrl_rank": 29, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241006_Agentless_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "deepseek-harness", "name": "DeepSeek Harness", "vendor": "DeepSeek", "description": "DeepSeek's plugin-style agent harness listed on OpenRouter's coding CLI directory.", "card_summary": "DeepSeek DeepSeek Harness coding agent harness.", "homepage_url": "https://github.com/deepseek-ai/deepseek-harness", "repo_url": "https://github.com/deepseek-ai/deepseek-harness", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": null, "hrl_score": 30.0, "hrl_rank": 30, "catalog_token_volume": 0, "openrouter_icon_url": "https://github.com/deepseek-ai/deepseek-harness", "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "live-swe-agent", "name": "Live SWE Agent", "vendor": "OpenAutoCoder", "description": "live-SWE-agent is a community fork/scaffold from OpenAutoCoder built on the SWE-agent lineage with configurations tuned for high SWE-bench Verified numbers. Public leaderboard rows often pair it with Claude Opus-class models.\n\nIt is a research/community harness (`status` community), not a HarnessRL validated scaffold. Scores reflect the fork's prompts, tool set, and model pairing at submission time.\n\nCatalog data comes from `swe_bench_site` ingest. For apples-to-apples model comparisons, contrast with mini-SWE-agent bash-only rows on the same benchmark rather than assuming identical tooling.", "card_summary": "OpenAutoCoder's live SWE-agent fork \u2014 strong public Verified results with frontier Claude pairings.", "homepage_url": "https://github.com/OpenAutoCoder/live-swe-agent", "repo_url": "https://github.com/OpenAutoCoder/live-swe-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 79.2, "github_stars": 450, "popularity_tier": null, "hrl_score": 30.0, "hrl_rank": 31, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20251120_livesweagent_gemini-3-pro-preview.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-rl-llama3-swe-rl-70b", "name": "SWE Rl Llama3 SWE Rl 70b", "vendor": "Facebookresearch", "description": "This slug tracks a SWE-bench Verified submission for an agent/checkpoint from SWE-RL training on Llama 3 at 70B scale (`SWE-RL-70B`). It connects RL-for-SWE research to the standard Verified leaderboard for comparability with supervised baselines.\n\nCommunity `swe_bench_site` ingest. Compare against same-scale non-RL Llama agents only when eval protocol matches.", "card_summary": "SWE-RL Llama3 SWE-RL-70B \u2014 RL-trained agent on Verified.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/facebookresearch/swe-rl", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 41.2, "github_stars": 719, "popularity_tier": null, "hrl_score": 30.0, "hrl_rank": 32, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250226_swerl_llama3_70b.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aime-coder-v1", "name": "Aime Coder V1", "vendor": "Swe Bench", "description": "AIME-Coder v1 is a research agent submission on SWE-bench Verified \u2014 typically a specialized coding agent or pipeline evaluated on the verified 500-issue split. Narrow research presence in the catalog is expected.\n\n`swe_bench_site` community row(s). Read model and source URL for the exact paper or release tied to the submission.", "card_summary": "AIME-Coder v1 agent \u2014 research harness on SWE-bench Verified.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/SWE-bench/experiments/blob/main/evaluation/verified/20250514_aime_coder/README.md", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 66.4, "github_stars": 280, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 33, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250514_aime_coder.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "coder", "name": "CodeR", "vendor": "Nl2Code", "description": "CodeR is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Nl2Code CodeR coding agent harness.", "homepage_url": "https://github.com/NL2Code/CodeR", "repo_url": "https://github.com/NL2Code/CodeR", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 28.33, "github_stars": null, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 34, "catalog_token_volume": 0, "openrouter_icon_url": "https://coder.techcamp.org.uk/", "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "joycode", "name": "Joycode", "vendor": "Jd Opensource", "description": "JoyCode appears as a harness name on public SWE-bench Verified submissions \u2014 typically a product or research agent scaffold paired with a disclosed model. This catalog tracks the harness-level leaderboard row, not JoyCode's full product surface area.\n\nScores: `swe_bench_site` on `swe-bench-verified`. Use the model column and source link to see which checkpoint and submission date apply.", "card_summary": "JoyCode agent \u2014 community SWE-bench Verified submission harness.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/jd-opensource/joycode-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 74.6, "github_stars": 342, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 35, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250915_JoyCode.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "junie", "name": "Junie", "vendor": "jetbrains", "description": "Junie is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "jetbrains Junie coding agent harness.", "homepage_url": "https://swe-rebench.com/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 84.0, "github_stars": null, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 36, "catalog_token_volume": 0, "openrouter_icon_url": "https://www.jetbrains.com/junie", "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "lingxi", "name": "Lingxi", "vendor": "Nimasteryang", "description": "Lingxi is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Nimasteryang Lingxi coding agent harness.", "homepage_url": "https://github.com/nimasteryang/Lingxi", "repo_url": "https://github.com/nimasteryang/Lingxi", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 42.67, "github_stars": 258, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 37, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250509_Lingxi_claude-3-5-sonnet-20241022.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "lingxi-v1-5", "name": "Lingxi V1.5 Claude 4 Sonnet 20250514", "vendor": "Nimasteryang", "description": "Lingxi V1.5 Claude 4 Sonnet 20250514 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Nimasteryang Lingxi V1.5 Claude 4 Sonnet 20250514 coding agent harness.", "homepage_url": "https://github.com/nimasteryang/Lingxi", "repo_url": "https://github.com/nimasteryang/Lingxi", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 74.6, "github_stars": 258, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 38, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250720_Lingxi-v1.5_claude-4-sonnet-20250514.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "lingxi-v1-5-x", "name": "Lingxi V1.5 X KIMI K2", "vendor": "Lingxi Agent", "description": "Lingxi V1.5 X KIMI K2 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Lingxi Agent Lingxi V1.5 X KIMI K2 coding agent harness.", "homepage_url": "https://github.com/lingxi-agent/Lingxi/tree/master", "repo_url": "https://github.com/lingxi-agent/Lingxi/tree/master", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 71.2, "github_stars": 258, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 39, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20251014_Lingxi_kimi_k2.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "sima", "name": "SIMA", "vendor": "Swe Bench", "description": "SIMA is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Swe Bench SIMA coding agent harness.", "homepage_url": "https://github.com/swe-bench/experiments/tree/main/evaluation/lite/20240706_sima_gpt4o", "repo_url": "https://github.com/swe-bench/experiments/tree/main/evaluation/lite/20240706_sima_gpt4o", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 27.67, "github_stars": 280, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 40, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-fixer", "name": "SWE Fixer", "vendor": "Internlm", "description": "SWE Fixer is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Internlm SWE Fixer coding agent harness.", "homepage_url": "https://github.com/InternLM/SWE-Fixer", "repo_url": "https://github.com/InternLM/SWE-Fixer", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 32.8, "github_stars": 139, "popularity_tier": null, "hrl_score": 29.0, "hrl_rank": 41, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241128_SWE-Fixer_Qwen2.5-7b-retriever_Qwen2.5-72b-editor_20241128.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "entropo", "name": "Entropo", "vendor": "Sherdencooper", "description": "Entropo is listed as a harness name on a public SWE-bench Verified submission. This catalog tracks that leaderboard row for coverage completeness \u2014 typically a research or product agent evaluated on the verified split.\n\nSingle-community submission pattern. `swe_bench_site` source.", "card_summary": "Entropo agent \u2014 SWE-bench Verified community submission.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/sherdencooper/R2E-Gym", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 52.2, "github_stars": 12, "popularity_tier": null, "hrl_score": 28.0, "hrl_rank": 42, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250901_entroPO_R2E_QwenCoder30BA3B.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "patchpilot", "name": "Patchpilot", "vendor": "Ucsb Mlsec", "description": "PatchPilot is an open agent from the InternLM ecosystem focused on generating and applying code patches for issue resolution benchmarks. Versioned rows (for example PatchPilot v1.1) reflect leaderboard submission tags rather than separate products.\n\nCommunity scores on SWE-bench Verified use the PatchPilot harness with disclosed models (`swe_bench_site`). Version slugs in this catalog (for example `patchpilot-v1-1`) map to specific submission labels.\n\nOpen source: `InternLM/PatchPilot`. For RL integration status see `supports_rl` \u2014 PatchPilot is tracked for leaderboard coverage, not HarnessRL validated scaffolds.", "card_summary": "InternLM PatchPilot \u2014 patch-generation agent with SWE-bench community entries.", "homepage_url": "https://github.com/InternLM/PatchPilot", "repo_url": "https://github.com/ucsb-mlsec/Co-PatcheR", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 64.6, "github_stars": 8, "popularity_tier": null, "hrl_score": 28.0, "hrl_rank": 43, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250528_patchpilot_Co-PatcheR.gif", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-exp", "name": "SWE Exp", "vendor": "Yerbapage", "description": "SWE-Exp is a research agent name on public SWE-bench Verified submissions (exploration-focused SWE agent designs). Narrow research footprint in the catalog is expected.\n\n`swe_bench_site` rows. Consult submission `source_url` for paper/repo references.", "card_summary": "SWE-Exp research agent \u2014 Verified submission harness.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/YerbaPage/SWE-Exp", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 42.0, "github_stars": 44, "popularity_tier": null, "hrl_score": 28.0, "hrl_rank": 44, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250806_SWE-Exp_DeepSeek-V3.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "qwen-code", "name": "QWEN Code", "vendor": "Alibaba", "description": "Terminal-native AI coding agent optimized for Qwen Coder models. Open-source CLI with IDE integrations.", "card_summary": "Alibaba QWEN Code coding agent harness.", "homepage_url": "https://qwenlm.github.io/qwen-code-docs/en/users/overview", "repo_url": "https://github.com/QwenLM/qwen-code", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 0, "best_success_rate": null, "github_stars": null, "popularity_tier": null, "hrl_score": 23.0, "hrl_rank": 45, "catalog_token_volume": 122701283891, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "hal-generalist-agent", "name": "HAL Generalist Agent", "vendor": "Princeton PLI", "description": "HAL Generalist Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Princeton PLI HAL Generalist Agent coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": "https://github.com/princeton-pli/hal-harness", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 124, "best_success_rate": 96.25, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kimi-code-cli", "name": "Kimi Code CLI", "vendor": "Moonshot", "description": "Kimi Code CLI is Moonshot AI's terminal-oriented coding agent, paired with Kimi foundation models on public leaderboards. It follows the product CLI pattern: repository context, edits, and shell tools in a managed loop.\n\nScores in this catalog primarily come from Artificial Analysis ingest (multiple model pairings per agent). Community SWE-bench submissions may appear when Moonshot or partners publish Verified runs.\n\nRelated open weights and dev tooling may appear under MoonshotAI GitHub org (for example Kimi-Dev); this page lists harness-level published scores only.", "card_summary": "Moonshot Kimi terminal coding agent with tool-use for SWE-style tasks.", "homepage_url": "https://kimi.moonshot.cn", "repo_url": "https://github.com/MoonshotAI/Kimi-Dev", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 87.64, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.6263880112748019, "aa_mean_cost_usd": 3.081635541610427, "aa_mean_total_tokens": 10383196}, {"slug": "livecodebench", "name": "Livecodebench", "vendor": "LiveCodeBench", "description": "LiveCodeBench is a contamination-aware competitive programming benchmark. The catalog uses harness slug `livecodebench` for model-level generation scores ingested from livecodebench.github.io JSON (`source = livecodebench`, benchmark `livecodebench-generation`).\n\nThis is not a harness\u00d7model matrix like SWE-bench Verified \u2014 one row per model with aggregated pass@1. Compare LiveCodeBench scores only against other model-level code benchmarks, not SWE resolution %.\n\nUpstream project: LiveCodeBench (UC Berkeley / community). Leaderboard JSON is the authoritative public source for catalog ingest.", "card_summary": "Implicit eval harness for LiveCodeBench generation \u2014 model-level pass@1 leaderboard.", "homepage_url": "https://livecodebench.github.io/", "repo_url": "https://github.com/LiveCodeBench/LiveCodeBench", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 26, "best_success_rate": 87.3, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "grok-build", "name": "Grok Build", "vendor": "xAI", "description": "Grok Build refers to xAI's coding agent harness used in public benchmark submissions (Artificial Analysis Coding Agent Index and related leaderboards). Rows pair xAI's agent tooling with Grok foundation models.\n\nAs with other vendor product harnesses, scores encode both model capability and xAI's agent stack (tools, prompts, safety). Use `aa_coding_index` on the harness row for AA's composite index when present \u2014 component breakdowns appear as separate benchmark scores in the table below.\n\nCheck `source` and `source_url` on each score; Grok model identifiers evolve quickly across leaderboard snapshots.", "card_summary": "xAI Grok coding agent scaffold for software engineering benchmarks.", "homepage_url": "https://x.ai", "repo_url": "https://github.com/xai-org/grok", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 84.27, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.6408998279698193, "aa_mean_cost_usd": 2.4389824437627774, "aa_mean_total_tokens": 3598867}, {"slug": "devin-cli", "name": "Devin CLI", "vendor": "Cognition", "description": "Local terminal coding agent with handoff to Cognition's cloud Devin sessions.", "card_summary": "Cognition Devin CLI coding agent harness.", "homepage_url": "https://artificialanalysis.ai/agents/coding-agents", "repo_url": "https://github.com/CognitionAI/devin-cli", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 79.4, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.5221389513465003, "aa_mean_cost_usd": 8.521259638036815, "aa_mean_total_tokens": 15573672}, {"slug": "sonar-foundation-agent", "name": "Sonar Foundation Agent", "vendor": "Sonar", "description": "Sonar Foundation Agent refers to benchmark submissions using the Sonar agent scaffold with disclosed foundation models on SWE-bench Verified. It is a community leaderboard harness, not a HarnessRL validated scaffold.\n\nRows are ingested from `swe_bench_site`. The agent name on swebench.com maps to this catalog slug; verify the model column for which Sonar-backed or third-party model was used in each submission.\n\nFor Sonar model family context, see model detail pages for the paired checkpoint \u2014 this harness page tracks agent-level leaderboard coverage only.", "card_summary": "Sonar foundation agent \u2014 SWE-bench Verified submissions under the Sonar agent harness.", "homepage_url": "https://www.perplexity.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 79.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20251219_sonar-foundation-agent_claude-opus-4-5.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "epam-ai-run-developer-agent", "name": "EPAM AI Run", "vendor": "EPAM Systems, Inc.", "description": "EPAM AI RUN is EPAM's agent scaffold for software-engineering benchmarks. Public SWE-bench Verified submissions appear under several versioned harness names (`epam-ai-run-developer-agent-v20241029`, v20241212, v20250219, v20250719, etc.) when EPAM publishes updated agent builds.\n\nEach version slug reflects a leaderboard submission tag \u2014 prompts, retrieval, and tool wiring can differ between versions even when the backing model is similar. Scores are community `swe_bench_site` rows only; this is not a HarnessRL validated scaffold.\n\nOpen source reference: `epam/AI-RUN` on GitHub. Compare versioned slugs side-by-side rather than merging them into one historical trend line.", "card_summary": "EPAM AI/Run developer agent \u2014 enterprise SWE harness with versioned Verified submissions.", "homepage_url": "https://github.com/epam/AI-RUN", "repo_url": "https://github.com/epam/AI-RUN", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 76.8, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240820_epam-ai-run-gpt-4o.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "gemini-cli", "name": "Gemini CLI", "vendor": "Google", "description": "Gemini CLI is Google's open terminal agent for coding workflows, built around Gemini models with tool use, file operations, and shell access. It competes in the same product harness category as Claude Code, Codex, and Cursor CLI.\n\nPublic scores appear on Artificial Analysis coding-agent benchmarks and occasional SWE-bench Verified submissions when Google or community members publish results. Each row's model column shows the exact Gemini variant used.\n\nRepository: `google-gemini/gemini-cli`. This catalog does not host Google's internal eval configs \u2014 only publicly linked leaderboard snapshots.", "card_summary": "Google's Gemini CLI \u2014 terminal agent with tool use for repo editing and shell tasks.", "homepage_url": "https://github.com/google-gemini/gemini-cli", "repo_url": "https://github.com/google-gemini/gemini-cli", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 5, "best_success_rate": 76.03, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": "https://www.gstatic.com/lamda/images/gemini_favicon_f06995846fc77c897+24.png", "leaderboard_icon_url": null, "aa_coding_index": 0.3293046837328304, "aa_mean_cost_usd": 1.0400415618609393, "aa_mean_total_tokens": 4727804}, {"slug": "refact-ai-agent", "name": "Refact Ai Agent", "vendor": "Refact.ai", "description": "Refact provides an open AI coding agent (Refact.ai) with tooling for repository context and edits. Public SWE-bench Verified submissions use this harness label when authors evaluate Refact's agent loop on the 500-task verified split.\n\nOpen-source agent stack; community leaderboard coverage via `swe_bench_site`. Product features (IDE plugins, enterprise) may differ from the public eval configuration cited on swebench.com.", "card_summary": "Refact AI agent \u2014 open coding agent with Verified leaderboard entry.", "homepage_url": "https://refact.ai", "repo_url": "https://github.com/smallcloudai/refact", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 74.4, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250611_Refact_Agent_claude-4-sonnet.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "tools", "name": "Tools", "vendor": "Anthropic", "description": "On the public SWE-bench Verified leaderboard, \"Tools\" is the harness name attached to several Anthropic Claude submissions that use a tool-using agent setup (distinct from the Claude Code product slug or bash-only mini-SWE-agent baselines). Rows pair this label with specific Claude snapshot dates in the model column.\n\nThis catalog ingests those community rows via `swe_bench_site`. The slug `tools` is a leaderboard artifact \u2014 not a separate open-source repository. For Anthropic's shipping CLI product, see `claude-code`; for minimal model comparisons, see `mini-swe-agent`.\n\nAll published scores here target `swe-bench-verified` (500 verified issues). Check each row's `source_url` for the submission snapshot on swebench.com.", "card_summary": "Anthropic \"Tools\" harness label on SWE-bench Verified \u2014 Claude models with tool-use agent configuration.", "homepage_url": "https://www.anthropic.com", "repo_url": "https://github.com/anthropics", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 5, "best_success_rate": 73.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241022_tools_claude-3-5-haiku.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "prometheus-v1-2", "name": "Prometheus V1 2", "vendor": "EuniAI", "description": "Earlier Prometheus v1.2 submission on SWE-bench Verified. Compare against v1.2.1 only when models and task protocol align \u2014 version bumps may change prompts or tool interfaces.\n\n`GAIR-NLP/Prometheus` on GitHub. `swe_bench_site` ingest.", "card_summary": "Prometheus v1.2 agent \u2014 earlier Verified submission build.", "homepage_url": "https://github.com/GAIR-NLP/Prometheus", "repo_url": "https://github.com/GAIR-NLP/Prometheus", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 71.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250929_Prometheus_v1.2_gpt5.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "usaco-episodic-semantic", "name": "USACO Episodic + Semantic", "vendor": null, "description": "USACO Episodic + Semantic is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "USACO Episodic + Semantic coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 11, "best_success_rate": 69.06, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "glm-4-6", "name": "GLM 4 6", "vendor": "Z.ai", "description": "GLM-4.6 appears as a harness+model pairing on SWE-bench Verified community submissions from the GLM / Zhipu ecosystem. The slug reflects leaderboard naming for agent evaluations using GLM-4.6-class models.\n\nDistinct from `glm-4-5` slug when submissions use different model generations. `swe_bench_site` ingest on `swe-bench-verified`.", "card_summary": "GLM-4.6 agent harness \u2014 Zhipu AI SWE-bench Verified submission.", "homepage_url": "https://www.zhipuai.cn", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 68.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250930_zai_glm4-6.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "nemotron-cortexa", "name": "Nemotron Cortexa", "vendor": "NVIDIA", "description": "Nemotron Cortexa is NVIDIA's agent scaffold used in public SWE-bench Verified submissions pairing Nemotron-family models with a Cortexa agent loop. Rows track NVIDIA's agent tooling plus model capability on issue resolution.\n\nCommunity `swe_bench_site` scores. For Nemotron model weights and training details, follow NVIDIA's model cards \u2014 this page is harness-level leaderboard coverage.", "card_summary": "NVIDIA Nemotron Cortexa agent \u2014 Verified submission harness.", "homepage_url": "https://developer.nvidia.com/nemotron", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 68.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250516_cortexa_o3.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "w&b-programmer-o1-crosscheck5", "name": "W&b Programmer O1 Crosscheck5", "vendor": "Weights & Biases", "description": "This slug captures a SWE-bench Verified submission using Weights & Biases' Programmer agent scaffold with OpenAI O1 in a crosscheck evaluation configuration (submission label \"crosscheck5\" on the leaderboard).\n\nHighly specific eval setup \u2014 single-row or few-row catalog presence. `swe_bench_site` ingest. Not a general W&B product harness; treat as a named reproducibility snapshot.", "card_summary": "Weights & Biases Programmer + O1 crosscheck harness \u2014 Verified submission.", "homepage_url": "https://wandb.ai", "repo_url": "https://github.com/wandb/wandb", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 64.6, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250117_wandb_programmer_o1_crosscheck5.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "famou-agent-2-0", "name": "Famou Agent 2.0", "vendor": "Baidubce", "description": "Famou Agent 2.0 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Baidubce Famou Agent 2.0 coding agent harness.", "homepage_url": "https://github.com/baidubce/FM-Agent", "repo_url": "https://github.com/baidubce/FM-Agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 64.44, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "glm-4-5", "name": "GLM 4 5", "vendor": "Z.ai", "description": "GLM-4.5 leaderboard submission harness on SWE-bench Verified. Use alongside `glm-4-6` rows to see generational changes \u2014 not as duplicate entries for the same eval.\n\nCommunity `swe_bench_site` scores.", "card_summary": "GLM-4.5 agent harness \u2014 Zhipu AI Verified submission label.", "homepage_url": "https://www.zhipuai.cn", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 64.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250728_zai_glm4-5.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aibuildai", "name": "AIBuildAI", "vendor": "Aibuildai", "description": "AIBuildAI is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Aibuildai AIBuildAI coding agent harness.", "homepage_url": "https://github.com/aibuildai/AI-Build-AI", "repo_url": "https://github.com/aibuildai/AI-Build-AI", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 63.11, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "hf-open-deep-research", "name": "HF Open Deep Research", "vendor": null, "description": "HF Open Deep Research is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "HF Open Deep Research coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 14, "best_success_rate": 62.8, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codestory-midwit-agent", "name": "Codestory Midwit Agent", "vendor": "CodeStory", "description": "CodeStory's Midwit agent appears on SWE-bench Verified as a community submission harness. CodeStory builds IDE-adjacent coding agents; public leaderboard rows reflect a specific eval configuration and model pairing at submission time.\n\n`swe_bench_site` ingest. Check `source_url` for swebench.com metadata and the model column for the foundation model used.", "card_summary": "CodeStory Midwit agent \u2014 community Verified submission harness.", "homepage_url": "https://codestory.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 62.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240702_codestory_aide_mixed.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "mlevolve", "name": "MLEvolve", "vendor": "Internscience", "description": "MLEvolve is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Internscience MLEvolve coding agent harness.", "homepage_url": "https://github.com/InternScience/MLEvolve", "repo_url": "https://github.com/InternScience/MLEvolve", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 61.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "experepair-v1-0", "name": "ExpeRepair V1.0", "vendor": "Experepair", "description": "ExpeRepair V1.0 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Experepair ExpeRepair V1.0 coding agent harness.", "homepage_url": "https://github.com/ExpeRepair/ExpeRepair", "repo_url": "https://github.com/ExpeRepair/ExpeRepair", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 60.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "learn-by-interact", "name": "Learn By Interact", "vendor": "Google", "description": "Learn-by-Interact refers to a research agent approach evaluated on SWE-bench Verified (interaction-heavy learning or feedback loops during issue resolution). Appears as a named harness on community leaderboard submissions.\n\n`swe_bench_site` rows. Useful for tracking academic agent ideas on the standard Verified split \u2014 not a shipping vendor CLI.", "card_summary": "Learn-by-Interact agent \u2014 research harness on Verified.", "homepage_url": "https://www.swebench.com/verified", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 60.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "deepswe-preview", "name": "Deepswe Preview", "vendor": "Agentica", "description": "DeepSWE preview marks early or preview DeepSWE agent submissions distinct from the production DeepSWE benchmark component on Artificial Analysis. DeepSWE (DataCurve) is a long-horizon software engineering eval; preview harness slugs capture leaderboard rows before stable release tagging.\n\nMay appear on `swe-bench-verified` or AA-related ingest depending on submission. Contrast with AA `deepswe` benchmark scores \u2014 preview harness slug vs benchmark slug measure different things.\n\nSee DataCurve / DeepSWE official docs for task definitions; this page lists published harness-level scores only.", "card_summary": "DeepSWE preview harness \u2014 early DataCurve / DeepSWE agent eval rows.", "homepage_url": "https://datacurve.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 58.8, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250629_deepswerl_r2eagent_tts.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kgcompass", "name": "KGCompass", "vendor": "Gleam Lab", "description": "KGCompass is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Gleam Lab KGCompass coding agent harness.", "homepage_url": "https://github.com/GLEAM-Lab/KGCompass", "repo_url": "https://github.com/GLEAM-Lab/KGCompass", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 58.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250609_KGCompass_deepseek-v3.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "swe-rizzo", "name": "SWE Rizzo", "vendor": "SWE-Rizzo", "description": "SWE-Rizzo is a research agent name on SWE-bench Verified community submissions. Narrow catalog presence \u2014 typically one paper or repo release evaluated on the verified split.\n\n`swe_bench_site` ingest. Use source links for the authoritative implementation reference.", "card_summary": "SWE-Rizzo research agent \u2014 Verified community submission.", "homepage_url": "https://www.swebench.com/verified", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 56.6, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250405_swe-rizzo_claude37.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "ml-master-2-0", "name": "ML Master 2.0", "vendor": "Sjtu Sai Agents", "description": "ML Master 2.0 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Sjtu Sai Agents ML Master 2.0 coding agent harness.", "homepage_url": "https://github.com/sjtu-sai-agents/ML-Master", "repo_url": "https://github.com/sjtu-sai-agents/ML-Master", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 56.44, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "cair-mars", "name": "Cair Mars", "vendor": "Google", "description": "Cair Mars is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Google CAIR MARS agent harness for software engineering evals.", "homepage_url": "https://research.google/teams/cloud-ai-research/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 56.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "tau-bench-tool-calling", "name": "TAU-bench Tool Calling", "vendor": null, "description": "TAU-bench Tool Calling is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "TAU-bench Tool Calling coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 11, "best_success_rate": 54.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "codesweep-swe-agent", "name": "CodeSweep SWE Agent KIMI K2 Instruct", "vendor": "CodeSweep Inc.", "description": "CodeSweep SWE Agent KIMI K2 Instruct is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "CodeSweep Inc. CodeSweep SWE Agent KIMI K2 Instruct coding agent harness.", "homepage_url": "https://codesweep.ai/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 53.4, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250804_codesweep_sweagent_kimi_k2_instruct.svg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "google-jules", "name": "Google Jules", "vendor": "Google", "description": "Google Jules is Google's agent product for asynchronous coding tasks, with public SWE-bench Verified submissions under the Jules harness name. Scores reflect Google's agent tooling, safety, and retrieval paired with disclosed Gemini or other Google models.\n\nProduct harness \u2014 not minimal bash-only baseline. `swe_bench_site` on `swe-bench-verified`. For Google's terminal CLI scaffold see `gemini-cli`; Jules rows are product-specific.", "card_summary": "Google Jules coding agent \u2014 SWE-bench Verified submission harness.", "homepage_url": "https://jules.google", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 52.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241212_google_jules_gemini_2.0_flash_experimental.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "pievolve", "name": "PiEvolve", "vendor": "Fractalairesearchlabs", "description": "PiEvolve is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Fractalairesearchlabs PiEvolve coding agent harness.", "homepage_url": "https://github.com/FractalAIResearchLabs/PiEvolve", "repo_url": "https://github.com/FractalAIResearchLabs/PiEvolve", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 52.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "semagent-multi-v1-0", "name": "SemAgent Multi V1.0", "vendor": "Columbia", "description": "SemAgent Multi V1.0 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Columbia SemAgent Multi V1.0 coding agent harness.", "homepage_url": "https://arxiv.org/abs/2506.16650", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 51.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250625_SemAgent_Multi-v1_Claude3.7Sonnet_Gemini2.5Pro.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "isea", "name": "Isea", "vendor": "ISEA", "description": "Isea is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "ISEA coding agent submission on SWE-bench Lite leaderboards.", "homepage_url": "https://www.swebench.com/lite", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 51.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "core-agent", "name": "CORE-Agent", "vendor": null, "description": "CORE-Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "CORE-Agent coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 23, "best_success_rate": 51.11, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "leeroo", "name": "Leeroo", "vendor": "Leeroo Ai", "description": "Leeroo is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Leeroo Ai Leeroo coding agent harness.", "homepage_url": "https://github.com/Leeroo-AI/kapso", "repo_url": "https://github.com/Leeroo-AI/kapso", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 50.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "thesis", "name": "Thesis", "vendor": "Thesis Labs", "description": "Thesis is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Thesis Labs coding agent harness on MLE-bench leaderboards.", "homepage_url": "https://thesislabs.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 48.44, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "appmap-navie-v2", "name": "Appmap Navie V2", "vendor": "AppMap", "description": "AppMap Navie is AppMap's AI coding agent with emphasis on application context and security-aware edits. v2 marks a public SWE-bench Verified submission generation.\n\nSee also `appmap-navie` for earlier submission label. `swe_bench_site` community scores.\n\nAppMap integrates with IDE workflows; public eval config may differ from enterprise deployments \u2014 rely on `source_url`.", "card_summary": "AppMap Navie v2 \u2014 security-aware coding agent on Verified.", "homepage_url": "https://appmap.io", "repo_url": "https://github.com/applandeo/appmap", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 47.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241113_navie-2-gpt4o-sonnet.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "dars-agent", "name": "DARS Agent", "vendor": "Darsagent", "description": "DARS Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Darsagent DARS Agent coding agent harness.", "homepage_url": "https://github.com/darsagent/DARS-Agent", "repo_url": "https://github.com/darsagent/DARS-Agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 47.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250205_dars_agent_claude_3.5_sonnet_deepseek_r1.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "skywork-swe-32b", "name": "Skywork SWE 32b", "vendor": "Skywork AI", "description": "Skywork SWE 32B is a leaderboard harness label tied to Skywork's 32B-class software engineering agent/model pairing on SWE-bench Verified. The slug emphasizes model scale in the public submission name.\n\nCommunity `swe_bench_site` row(s). For Skywork model weights see Hugging Face / Skywork model cards \u2014 this page tracks harness-level scores.", "card_summary": "Skywork SWE 32B agent \u2014 model-scoped harness on Verified.", "homepage_url": "https://www.skywork.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 47.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250616_Skywork-SWE-32B+TTS_Bo8.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "autocoderover-v20240620", "name": "Autocoderover V20240620", "vendor": "AutoCodeRover", "description": "AutoCodeRover is an open-source software engineering agent that combines retrieval over the codebase with structured patch proposals. The catalog tracks versioned submission labels (v2.0, v2.1, dated builds) as separate harness slugs when leaderboard metadata distinguishes them.\n\nScores are community SWE-bench Verified submissions. Each version may change prompts, retrieval, or tool interfaces \u2014 do not treat all AutoCodeRover slugs as interchangeable.\n\nUpstream repository: `AutoCodeRover/AutoCodeRover`. EPAM AI RUN and other forks sometimes appear as distinct slugs when submissions use different harness names.", "card_summary": "AutoCodeRover open agent \u2014 iterative retrieval and patch application for issue fixing.", "homepage_url": "https://github.com/AutoCodeRover/AutoCodeRover", "repo_url": "https://github.com/nus-apr/auto-code-rover", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 6, "best_success_rate": 46.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240628_autocoderover-v20240620.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kodu-v1", "name": "Kodu V1", "vendor": "Kodu", "description": "Kodu V1 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Kodu v1 terminal coding agent on SWE-bench Verified runs.", "homepage_url": "https://www.kodu.ai/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 44.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241207_kodu_sonnet_v1.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "famou-agent", "name": "Famou Agent", "vendor": "Baidubce", "description": "Famou Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Baidubce Famou Agent coding agent harness.", "homepage_url": "https://github.com/baidubce/FM-Agent", "repo_url": "https://github.com/baidubce/FM-Agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 43.56, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "seeact", "name": "SeeAct", "vendor": null, "description": "SeeAct is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "SeeAct coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 9, "best_success_rate": 42.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "patchkitty-0-9", "name": "Patchkitty 0 9", "vendor": "PatchKitty", "description": "Patchkitty 0 9 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "PatchKitty 0.9 patch-generation agent on SWE-bench Lite.", "homepage_url": "https://www.swebench.com/lite", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 41.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241220_PatchKitty-0.9_claude-3.5-sonnet-20241022.gif", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "orcaloca", "name": "OrcaLoca", "vendor": "Fishmingyu", "description": "OrcaLoca is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Fishmingyu OrcaLoca coding agent harness.", "homepage_url": "https://github.com/fishmingyu/OrcarLLM", "repo_url": "https://github.com/fishmingyu/OrcarLLM", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 41.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "nebius-ai", "name": "Nebius AI QWEN 2.5 72B Generator", "vendor": "Nebius", "description": "Nebius AI QWEN 2.5 72B Generator is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Nebius Nebius AI QWEN 2.5 72B Generator coding agent harness.", "homepage_url": "https://nebius.com/blog/posts/training-and-search-for-software-engineering-agents", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 40.6, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241113_nebius-search-open-weight-models-11-24.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "browser-use", "name": "Browser-Use", "vendor": null, "description": "Browser-Use is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Browser-Use coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 25, "best_success_rate": 40.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "opencsg-starship-agentic-coder", "name": "Opencsg Starship Agentic Coder", "vendor": "OpenCSG", "description": "Opencsg Starship Agentic Coder is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "OpenCSG StarShip agentic coder harness for repo editing tasks.", "homepage_url": "https://opencsg.com/starship", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 39.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250113_OpenCSG-Starship-Agentic-Coder_gpt4o.svg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "operand", "name": "Operand", "vendor": "Operand", "description": "Operand is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Operand ensemble coding agent on MLE-bench and SWE benchmarks.", "homepage_url": "https://operand.com", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 39.56, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "cair", "name": "Cair", "vendor": "Google", "description": "Cair is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Google Cloud AI Research (CAIR) agent harness on coding benchmarks.", "homepage_url": "https://research.google/teams/cloud-ai-research/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 38.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "abanteai-mentatbot", "name": "Abanteai Mentatbot", "vendor": "AbanteAI", "description": "Abanteai Mentatbot is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "AbanteAI MentatBot coding agent on SWE-bench Verified submissions.", "homepage_url": "https://mentat.ai/blog/mentatbot-sota-coding-agent", "repo_url": "https://github.com/AbanteAI/mentat", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 38.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "patched-codes-patchwork", "name": "Patched.Codes Patchwork", "vendor": "Patched Codes", "description": "Patched.Codes Patchwork is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Patched Codes Patched.Codes Patchwork coding agent harness.", "homepage_url": "https://github.com/patched-codes/patchwork", "repo_url": "https://github.com/patched-codes/patchwork", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 37.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250104_patched_codes_claude-3.5-sonnet-20241022.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "internagent", "name": "InternAgent", "vendor": "Alpha Innovator", "description": "InternAgent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Alpha Innovator InternAgent coding agent harness.", "homepage_url": "https://github.com/Alpha-Innovator/InternAgent/", "repo_url": "https://github.com/Alpha-Innovator/InternAgent/", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 36.44, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "description": "GUIRepair is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "GUIRepair GUIRepair coding agent harness.", "homepage_url": "https://sites.google.com/view/guirepair", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 35.98, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250531_GUIRepair_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "rd-agent", "name": "R&D Agent", "vendor": "Microsoft", "description": "R&D Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Microsoft R&D Agent coding agent harness.", "homepage_url": "https://github.com/microsoft/RD-Agent", "repo_url": "https://github.com/microsoft/RD-Agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 35.11, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "neo", "name": "Neo", "vendor": "Neo", "description": "Neo is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Neo multi-agent coding harness from ProjectDiscovery.", "homepage_url": "https://heyneo.so/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 34.22, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "sab-self-debug", "name": "SAB Self Debug", "vendor": null, "description": "SAB Self Debug is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "SAB Self Debug coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 15, "best_success_rate": 33.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "masai", "name": "Masai", "vendor": "Masai", "description": "Masai appears as a harness label on SWE-bench Verified community submissions. Typically a product or research coding agent evaluated on the verified 500-issue split.\n\n`swe_bench_site` ingest. Read model and source URL per row.", "card_summary": "Masai agent \u2014 SWE-bench Verified community submission.", "homepage_url": "https://www.swebench.com/verified", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 32.6, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aira-dojo", "name": "AIRA Dojo", "vendor": "Facebookresearch", "description": "AIRA Dojo is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Facebookresearch AIRA Dojo coding agent harness.", "homepage_url": "https://github.com/facebookresearch/aira-dojo/", "repo_url": "https://github.com/facebookresearch/aira-dojo/", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 31.6, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aegis-o3-mini-1-0", "name": "Aegis O3 Mini 1.0", "vendor": "Evandiewald", "description": "Aegis O3 Mini 1.0 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Evandiewald Aegis O3 Mini 1.0 coding agent harness.", "homepage_url": "https://github.com/evandiewald/aegis", "repo_url": "https://github.com/evandiewald/aegis", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 30.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aigcode-infant-coder-2024-08-30", "name": "Aigcode Infant Coder 2024 08 30", "vendor": "AIGCode", "description": "Aigcode Infant Coder 2024 08 30 is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "AIGCode Infant-Coder agent harness (August 2024 snapshot).", "homepage_url": "https://aigcode.net/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 30.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "kortix-ai", "name": "Kortix AI", "vendor": null, "description": "Kortix AI is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Kortix AI coding agent harness.", "homepage_url": "https://www.kortix.ai/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 30.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20241203_KortixAI-AgentPress-sonnet-20241022.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "agentless", "name": "Agentless", "vendor": "Ozyyshr", "description": "Agentless is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Ozyyshr Agentless coding agent harness.", "homepage_url": "https://github.com/ozyyshr/RepoGraph", "repo_url": "https://github.com/ozyyshr/RepoGraph", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 29.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240808_RepoGraph_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "ml-master", "name": "ML Master", "vendor": "Zeroxleo", "description": "ML Master is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Zeroxleo ML Master coding agent harness.", "homepage_url": "https://github.com/zeroxleo/ML-Master", "repo_url": "https://github.com/zeroxleo/ML-Master", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 29.33, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "reproducedrg", "name": "Reproducedrg", "vendor": "Research", "description": "Reproducedrg is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Reproduced research-group baseline harness on SWE-bench Lite.", "homepage_url": "https://www.swebench.com/lite", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 28.0, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "appmap-navie", "name": "Appmap Navie", "vendor": "AppMap", "description": "Earlier AppMap Navie harness submission on SWE-bench Verified. Version lineage continues in `appmap-navie-v2` when AppMap published updated agent builds.\n\n`swe_bench_site` ingest.", "card_summary": "AppMap Navie agent \u2014 earlier Verified submission label.", "homepage_url": "https://appmap.io", "repo_url": "https://github.com/applandeo/appmap", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 26.2, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240615_appmap-navie_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "opencsg-starship-codegenagent", "name": "Opencsg Starship Codegenagent", "vendor": "OpenCSG", "description": "Opencsg Starship Codegenagent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "OpenCSG StarShip CodeGenAgent on SWE-bench submissions.", "homepage_url": "https://opencsg.com/product?class=StarShip", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 23.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240524_opencsg_starship_gpt4.svg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "bytedance-autose-based-on-swe-agent", "name": "Bytedance AutoSE (based On SWE Agent)", "vendor": "Bytedance", "description": "Bytedance AutoSE (based On SWE Agent) is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Bytedance Bytedance AutoSE (based On SWE Agent) coding agent harness.", "homepage_url": "https://www.swebench.com/lite", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 21.67, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "aide", "name": "AIDE", "vendor": "Wecoai", "description": "AIDE is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Wecoai AIDE coding agent harness.", "homepage_url": "https://github.com/wecoai/aideml", "repo_url": "https://github.com/wecoai/aideml", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 17.12, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "scicode-tool-calling-agent", "name": "Scicode Tool Calling Agent", "vendor": null, "description": "Scicode Tool Calling Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Scicode Tool Calling Agent coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 14, "best_success_rate": 9.23, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "scicode-zero-shot-agent", "name": "Scicode Zero Shot Agent", "vendor": null, "description": "Scicode Zero Shot Agent is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "Scicode Zero Shot Agent coding agent harness.", "homepage_url": "https://hal.cs.princeton.edu/", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 8, "best_success_rate": 6.15, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, {"slug": "mlab", "name": "Mlab", "vendor": "MLAB", "description": "Mlab is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "MLAB agent harness tracked on MLE-bench and SWE leaderboards.", "homepage_url": "https://github.com/mlab-ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 1.6, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}]