{"harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "description": "SWE-agent is the reference open-source agent from Princeton NLP for software engineering benchmarks. It provides structured tools (file editor, bash, search), configurable agent policies, and the ecosystem that spawned mini-SWE-agent for controlled comparisons.\n\nLeaderboard rows labeled SWE-agent usually mean this full scaffold with a disclosed model, not the bash-only mini variant. SWE-bench Verified community submissions and Artificial Analysis ingest map the product name here when aliases match.\n\nHarnessRL lists SWE-agent as a community scaffold (`supports_rl` false) \u2014 useful for tracking public scores but not part of the validated RL scaffold set. For model-only baselines on Verified, see mini-SWE-agent and the SWE-bench bash-only board.", "card_summary": "Princeton NLP's open LM agent loop \u2014 rich tools and prompts for software engineering tasks.", "homepage_url": "https://github.com/SWE-agent/SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 35, "best_success_rate": 72.0, "github_stars": 14200, "popularity_tier": null, "hrl_score": 33.0, "hrl_rank": 19, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240402_sweagent_claude3opus.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 72.0, "resolved_count": null, "total_count": null, "cost_usd": 9.4673, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet-high", "name": "Claude 4.5 Sonnet (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 68.0, "resolved_count": null, "total_count": null, "cost_usd": 10.3249, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet", "name": "Claude 4.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 66.6, "resolved_count": 333, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2025-05-22T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 62.4, "resolved_count": 312, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2025-02-25T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-sonnet", "name": "Claude 3.7 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 61.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-1-opus", "name": "Claude 4.1 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 56.67, "resolved_count": 170, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2025-05-26T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 54.0, "resolved_count": null, "total_count": null, "cost_usd": 7.9363, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-sonnet-high", "name": "Claude 3.7 Sonnet (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 54.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-1-opus-high", "name": "Claude 4.1 Opus (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 50.0, "resolved_count": null, "total_count": null, "cost_usd": 5.0706, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "o4-mini", "name": "o4 mini", "vendor": "openai"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 50.0, "resolved_count": null, "total_count": null, "cost_usd": 8.2182, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-sonnet", "name": "Claude 3.7 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 50.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-opus", "name": "Claude 4 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 48.0, "resolved_count": 144, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2025-02-26T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-sonnet", "name": "Claude 3.7 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 46.0, "resolved_count": null, "total_count": null, "cost_usd": 3.3251, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-medium", "name": "GPT-5 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 46.0, "resolved_count": null, "total_count": null, "cost_usd": 9.8659, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "o3", "name": "o3", "vendor": "openai"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 44.0, "resolved_count": null, "total_count": null, "cost_usd": 8.0337, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1", "name": "GPT-4.1", "vendor": "openai"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 40.2, "resolved_count": 201, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://swesmith.com/", "observed_at": "2025-05-11T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "swe-agent-lm-32b", "name": "SWE-agent-LM-32B", "vendor": "Princeton NLP"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 38.0, "resolved_count": 190, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://swe-agent.com/", "observed_at": "2025-07-25T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-small-2507", "name": "DevStral Small 2507", "vendor": "Mistral"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 33.83, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Test public submission", "source": "swe_bench_site", "source_url": "https://github.com/swe-agent/swe-agent", "observed_at": "2025-02-27T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-20250219", "name": "Claude 3 7 20250219", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-test", "name": "SWE-bench Test"}}, {"success_rate": 33.6, "resolved_count": 168, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-06-20T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-5-sonnet", "name": "Claude 3.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 24.0, "resolved_count": null, "total_count": null, "cost_usd": 0.0963, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-0-flash", "name": "Gemini 2.0 Flash", "vendor": "Google"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 24.0, "resolved_count": null, "total_count": null, "cost_usd": 0.2402, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-march-2025", "name": "DeepSeek V3 (March 2025)", "vendor": "DeepSeek"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}, {"success_rate": 23.2, "resolved_count": 116, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-07-28T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240513", "name": "GPT-4o (2024-05-13)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 23.0, "resolved_count": 69, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-06-20T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-5-sonnet", "name": "Claude 3.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 22.4, "resolved_count": 112, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-04-02T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4", "name": "GPT-4", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 18.33, "resolved_count": 55, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-07-28T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240513", "name": "GPT-4o (2024-05-13)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 18.13, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Test public submission", "source": "swe_bench_site", "source_url": "https://www.swebench.com/", "observed_at": "2024-06-20T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-5-sonnet", "name": "Claude 3.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-test", "name": "SWE-bench Test"}}, {"success_rate": 18.0, "resolved_count": 54, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-04-02T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4", "name": "GPT-4", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 15.8, "resolved_count": 79, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-04-02T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-opus", "name": "Claude 3 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 12.47, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Test public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-04-02T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4", "name": "GPT-4", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-test", "name": "SWE-bench Test"}}, {"success_rate": 12.19, "resolved_count": 63, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-10-06T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-5-sonnet", "name": "Claude 3.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 11.99, "resolved_count": 62, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-10-06T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 11.99, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Test public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-07-28T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240513", "name": "GPT-4o (2024-05-13)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-test", "name": "SWE-bench Test"}}, {"success_rate": 11.67, "resolved_count": 35, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-04-02T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-opus", "name": "Claude 3 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 10.51, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Test public submission", "source": "swe_bench_site", "source_url": "https://www.swebench.com/", "observed_at": "2024-04-02T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-opus", "name": "Claude 3 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-test", "name": "SWE-bench Test"}}, {"success_rate": 0.0, "resolved_count": null, "total_count": null, "cost_usd": 0.0849, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "HAL SWE-bench Verified Mini (HAL) accuracy", "source": "hal_site", "source_url": "https://hal.cs.princeton.edu/swebench_verified_mini", "observed_at": null, "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-r1-january-2025", "name": "DeepSeek R1 (January 2025)", "vendor": "DeepSeek"}, "benchmark": {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)"}}], "benchmarks": [{"slug": "swe-bench-lite", "name": "SWE-bench Lite", "category": "eval"}, {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)", "category": "eval"}, {"slug": "swe-bench-test", "name": "SWE-bench Test", "category": "eval"}, {"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}, {"slug": "hal-swe-bench-verified-mini", "name": "SWE-bench Verified Mini (HAL)", "category": "eval"}]}