{"harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "description": "mini-SWE-agent is a deliberately small agent from the SWE-agent team (~100 lines of core loop). The model receives a single bash tool in a ReAct pattern: propose a command, observe output, repeat. There are no rich IDE tools or vendor-specific edit formats \u2014 leaderboard scores emphasize model capability over scaffold engineering.\n\nThe SWE-bench site publishes an official bash-only view where every model runs the same mini-SWE-agent configuration on SWE-bench Verified (500 tasks). Release tags on the mini-SWE-agent repository correspond to leaderboard version numbers.\n\nBecause the harness is minimal, many third-party evaluators (Vals.ai, SWE-rebench model baselines, DeepSWE comparisons) also use it. It is not interchangeable with full SWE-agent, which adds richer tool interfaces and tuned prompts. Catalog scores with `source = swe_bench_site` are community harness submissions using this scaffold.\n\nThis harness is the fairest public baseline when the question is \"how good is the model?\" rather than \"how good is the product?\" If a vendor CLI scores higher than mini-SWE-agent on the same model, that gap is largely scaffold and tooling \u2014 not a contradiction.", "card_summary": "Minimal bash-only ReAct loop \u2014 the reference scaffold for apples-to-apples LM comparisons.", "homepage_url": "https://github.com/SWE-agent/mini-swe-agent", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 212, "best_success_rate": 100.0, "github_stars": 1800, "popularity_tier": null, "hrl_score": 36.0, "hrl_rank": 13, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/mini-icon.svg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 100.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-5-opus-high", "name": "Claude 5 Opus (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 84.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "fable-5-[high]", "name": "Fable 5 [high]", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 80.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-5-2-[high]", "name": "GLM 5.2 [high]", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 80.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-5-sonnet-high", "name": "Claude 5 Sonnet (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 76.8, "resolved_count": 384, "total_count": 500, "cost_usd": 0.753907997, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 33, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-opus-high", "name": "Claude 4.5 Opus (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 76.8, "resolved_count": 384, "total_count": 500, "cost_usd": 0.753907997, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/claude", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-opus-high", "name": "Claude 4.5 Opus (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 76.18, "resolved_count": null, "total_count": null, "cost_usd": 0.44505617977528095, "input_tokens": 60996, "output_tokens": 13679953, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/94", "observed_at": "2026-07-09T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "muse-spark-1-1", "name": "Muse Spark 1.1", "vendor": "meta"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}, {"success_rate": 76.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "grok-4-5-[high]", "name": "GROK 4.5 [high]", "vendor": "xAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 75.8, "resolved_count": 379, "total_count": 500, "cost_usd": 0.35596413079999994, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 56, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-flash-high", "name": "Gemini 3 Flash (high)", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 75.8, "resolved_count": 379, "total_count": 500, "cost_usd": 0.073289987844, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 60, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2-5-high", "name": "MiniMax M2.5 (high)", "vendor": "MiniMax"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 75.8, "resolved_count": 379, "total_count": 500, "cost_usd": 0.35596413079999994, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-flash-high", "name": "Gemini 3 Flash (high)", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 75.8, "resolved_count": 379, "total_count": 500, "cost_usd": 0.073289987844, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.minimax.io", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2-5-high", "name": "MiniMax M2.5 (high)", "vendor": "MiniMax"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 75.6, "resolved_count": 378, "total_count": 500, "cost_usd": 0.5515226445, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 29, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-6-opus", "name": "Claude 4.6 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 75.6, "resolved_count": 378, "total_count": 500, "cost_usd": 0.5515226445, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/claude", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-6-opus", "name": "Claude 4.6 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 75.29, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-6-medium", "name": "GPT-5.6 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 74.4, "resolved_count": 372, "total_count": 500, "cost_usd": 0.721244997, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 38, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-11-24T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-opus-medium", "name": "Claude 4.5 Opus (medium)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 74.4, "resolved_count": 372, "total_count": 500, "cost_usd": 0.721244997, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-opus-4-5", "observed_at": "2025-11-24T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-opus-medium", "name": "Claude 4.5 Opus (medium)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 74.2, "resolved_count": 371, "total_count": 500, "cost_usd": 0.45995751895000003, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 40, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-11-18T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 74.2, "resolved_count": 371, "total_count": 500, "cost_usd": 0.45995751895000003, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2025-11-18T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 72.8, "resolved_count": 364, "total_count": 500, "cost_usd": 0.4494238987, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 28, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-19T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-codex", "name": "GPT-5.2 codex", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 72.8, "resolved_count": 364, "total_count": 500, "cost_usd": 0.53438913622, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 76, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-5-high", "name": "GLM 5 (high)", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 72.8, "resolved_count": 364, "total_count": 500, "cost_usd": 0.4494238987, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5", "observed_at": "2026-02-19T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-codex", "name": "GPT-5.2 codex", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 72.8, "resolved_count": 364, "total_count": 500, "cost_usd": 0.53438913622, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://z.ai", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-5-high", "name": "GLM 5 (high)", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 72.7, "resolved_count": 218, "total_count": 300, "cost_usd": 0.3518355868333333, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-flash", "name": "Gemini 3 Flash", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 72.0, "resolved_count": 216, "total_count": 300, "cost_usd": 0.6629735766666667, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-opus-4-5", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-6-opus", "name": "Claude 4.6 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 71.8, "resolved_count": 359, "total_count": 500, "cost_usd": 0.5202590036, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 20, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-11T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-high", "name": "GPT-5.2 (high)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 71.8, "resolved_count": 359, "total_count": 500, "cost_usd": 0.5202590036, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5", "observed_at": "2025-12-11T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-high", "name": "GPT-5.2 (high)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 71.4, "resolved_count": 357, "total_count": 500, "cost_usd": 0.6578964984000001, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 48, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet-high", "name": "Claude 4.5 Sonnet (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 71.4, "resolved_count": 357, "total_count": 500, "cost_usd": 0.6578964984000001, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/claude", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet-high", "name": "Claude 4.5 Sonnet (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 71.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-medium", "name": "GPT-5.2 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 71.11, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-5-xhigh", "name": "GPT-5.5 (xhigh)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 70.8, "resolved_count": 354, "total_count": 500, "cost_usd": 0.14655374074400002, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 51, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-5-high", "name": "KIMI K2.5 (high)", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 70.8, "resolved_count": 354, "total_count": 500, "cost_usd": 0.14655374074400002, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://moonshotai.github.io/Kimi-K2/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-5-high", "name": "KIMI K2.5 (high)", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 70.7, "resolved_count": 212, "total_count": 300, "cost_usd": 0.8335529758333333, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-opus-4-5", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-opus", "name": "Claude 4.5 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 70.6, "resolved_count": 353, "total_count": 500, "cost_usd": 0.5583347409, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 51, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-09-29T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet", "name": "Claude 4.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 70.6, "resolved_count": 353, "total_count": 500, "cost_usd": 0.5583347409, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-sonnet-4-5", "observed_at": "2025-09-29T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet", "name": "Claude 4.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 70.0, "resolved_count": 350, "total_count": 500, "cost_usd": 0.4478464965, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 89, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-2-high", "name": "DeepSeek V3.2 (high)", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 70.0, "resolved_count": 350, "total_count": 500, "cost_usd": 0.4478464965, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://api-docs.deepseek.com", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-2-high", "name": "DeepSeek V3.2 (high)", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 70.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-4-medium", "name": "GPT-5.4 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 69.7, "resolved_count": 209, "total_count": 300, "cost_usd": 0.6424743668666667, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://z.ai", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-5", "name": "GLM 5", "vendor": "z-ai"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 69.6, "resolved_count": 348, "total_count": 500, "cost_usd": 0.9600198564, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 51, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro", "name": "Gemini 3 Pro", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 69.6, "resolved_count": 348, "total_count": 500, "cost_usd": 0.9600198564, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2026-02-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro", "name": "Gemini 3 Pro", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 69.0, "resolved_count": 345, "total_count": 500, "cost_usd": 0.2696674974, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 16, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-11T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2", "name": "GPT-5.2", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 69.0, "resolved_count": 345, "total_count": 500, "cost_usd": 0.2696674974, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/", "observed_at": "2025-12-11T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2", "name": "GPT-5.2", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 68.89, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-opus", "name": "Claude 4.5 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 68.7, "resolved_count": 206, "total_count": 300, "cost_usd": 1.0210855846666667, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro", "name": "Gemini 3 Pro", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 68.3, "resolved_count": 205, "total_count": 300, "cost_usd": 0.10121049956228956, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.minimax.io", "observed_at": "2026-02-16T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2-5", "name": "MiniMax M2.5", "vendor": "minimax"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 67.78, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-5-medium", "name": "GPT-5.5 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 67.6, "resolved_count": 338, "total_count": 500, "cost_usd": 1.131270447, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 31, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-02T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-opus", "name": "Claude 4 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 67.6, "resolved_count": 338, "total_count": 500, "cost_usd": 1.131270447, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-4", "observed_at": "2025-08-02T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-opus", "name": "Claude 4 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 67.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-6-opus-high", "name": "Claude 4.6 Opus (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 67.32, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-xhigh", "name": "GPT-5.2 (xhigh)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 67.3, "resolved_count": 202, "total_count": 300, "cost_usd": 0.6928368501333334, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://moonshotai.github.io/Kimi-K2/", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-5", "name": "KIMI K2.5", "vendor": "moonshotai"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 67.27, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "vendor": "Google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 67.06, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m3", "name": "MiniMax M3", "vendor": "minimax"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 67.0, "resolved_count": 201, "total_count": 300, "cost_usd": 0.669532236, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-opus-4-5", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet", "name": "Claude 4.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 66.7, "resolved_count": 200, "total_count": 300, "cost_usd": 0.5364109905, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-high", "name": "GPT-5.2 (high)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 66.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-7-opus-high", "name": "Claude 4.7 Opus (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 66.6, "resolved_count": 333, "total_count": 500, "cost_usd": 0.33092443610000005, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 66, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-haiku-high", "name": "Claude 4.5 Haiku (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 66.6, "resolved_count": 333, "total_count": 500, "cost_usd": 0.33092443610000005, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/claude", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-haiku-high", "name": "Claude 4.5 Haiku (high)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 66.36, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2025-medium", "name": "GPT-5.2025 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 66.3, "resolved_count": 199, "total_count": 300, "cost_usd": 0.6621715578333333, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5", "observed_at": "2026-02-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-codex", "name": "GPT-5.2 codex", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 66.0, "resolved_count": 330, "total_count": 500, "cost_usd": 0.5888823500000001, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 24, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-11-24T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-1-codex-medium", "name": "GPT-5.1 codex (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 66.0, "resolved_count": 330, "total_count": 500, "cost_usd": 0.3061554725, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 21, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-11-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-1-medium", "name": "GPT-5.1 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 66.0, "resolved_count": 330, "total_count": 500, "cost_usd": 0.5888823500000001, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5", "observed_at": "2025-11-24T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-1-codex-medium", "name": "GPT-5.1 codex (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 66.0, "resolved_count": 330, "total_count": 500, "cost_usd": 0.3061554725, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5", "observed_at": "2025-11-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-1-medium", "name": "GPT-5.1 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 65.56, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-8-opus-xhigh", "name": "Claude 4.8 Opus (xhigh)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 65.45, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-1-codex-max", "name": "GPT-5.1 codex (max)", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 65.0, "resolved_count": 325, "total_count": 500, "cost_usd": 0.2803830175, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 13, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-medium", "name": "GPT-5 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 65.0, "resolved_count": 325, "total_count": 500, "cost_usd": 0.2803830175, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-medium", "name": "GPT-5 (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 64.93, "resolved_count": 325, "total_count": 500, "cost_usd": 0.37145316780000004, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 37, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 64.93, "resolved_count": 325, "total_count": 500, "cost_usd": 0.37145316780000004, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-4", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 64.7, "resolved_count": 194, "total_count": 300, "cost_usd": 0.3794902823333333, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-opus-4-5", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-haiku", "name": "Claude 4.5 Haiku", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 64.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-1-pro-preview", "name": "Gemini 3.1 Pro Preview", "vendor": "google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 64.21, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 64.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-2", "name": "DeepSeek V3.2", "vendor": "deepseek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 64.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-3-codex", "name": "GPT-5.3 codex", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 63.64, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-5-sonnet", "name": "Claude 4.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 63.4, "resolved_count": 317, "total_count": 500, "cost_usd": 0.4383072354, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 47, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-10T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-thinking", "name": "KIMI K2 Thinking", "vendor": "moonshotai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 63.4, "resolved_count": 317, "total_count": 500, "cost_usd": 0.4383072354, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://moonshotai.github.io/Kimi-K2/", "observed_at": "2025-12-10T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-thinking", "name": "KIMI K2 Thinking", "vendor": "moonshotai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 63.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "vendor": "google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.96, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-5", "name": "GLM 5", "vendor": "z-ai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-3-codex-xhigh", "name": "GPT-5.3 codex (xhigh)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-5-397b-a17b", "name": "Qwen3.5 397B A17B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-5-27b", "name": "Qwen3.5 27B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-5-flash", "name": "Gemini 3.5 Flash", "vendor": "google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2-codex", "name": "GPT-5.2 codex", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 62.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "step-3-5-flash", "name": "Step 3.5 Flash", "vendor": "stepfun"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 61.54, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "mimo-v2-5-pro", "name": "MiMo V2.5 Pro", "vendor": "xiaomi"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 61.43, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2025-high", "name": "GPT-5.2025 (high)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 61.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-5", "name": "KIMI K2.5", "vendor": "moonshotai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 61.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-6-sonnet", "name": "Claude 4.6 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 61.0, "resolved_count": 305, "total_count": 500, "cost_usd": 0.42825617729, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 74, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-11-24T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2", "name": "MiniMax M2", "vendor": "minimax"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 61.0, "resolved_count": 305, "total_count": 500, "cost_usd": 0.42825617729, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.minimax.io", "observed_at": "2025-11-24T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2", "name": "MiniMax M2", "vendor": "minimax"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 60.0, "resolved_count": 300, "total_count": 500, "cost_usd": 0.0280763028, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 46, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-01T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-2-reasoner", "name": "DeepSeek V3.2 Reasoner", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 60.0, "resolved_count": 300, "total_count": 500, "cost_usd": 0.0280763028, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.swebench.com/verified", "observed_at": "2025-12-01T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-2-reasoner", "name": "DeepSeek V3.2 Reasoner", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 60.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-codex", "name": "GPT-5 codex", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 60.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-1-codex", "name": "GPT-5.1 codex", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 60.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-7", "name": "GLM 4.7", "vendor": "z-ai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 59.8, "resolved_count": 299, "total_count": 500, "cost_usd": 0.035477067300000005, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 14, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini-medium", "name": "GPT-5 mini (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 59.8, "resolved_count": 299, "total_count": 500, "cost_usd": 0.035477067300000005, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5-mini", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini-medium", "name": "GPT-5 mini (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 59.0, "resolved_count": 177, "total_count": 300, "cost_usd": 0.3838255249, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://api-docs.deepseek.com", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-2", "name": "DeepSeek V3.2", "vendor": "deepseek"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 58.4, "resolved_count": 292, "total_count": 500, "cost_usd": 0.333652748, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 25, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "o3", "name": "o3", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 58.4, "resolved_count": 292, "total_count": 500, "cost_usd": 0.333652748, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/o3", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "o3", "name": "o3", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 58.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-thinking", "name": "KIMI K2 Thinking", "vendor": "moonshotai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 58.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-5-1", "name": "GLM 5 1", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 56.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-5-35b-a3b", "name": "Qwen3.5 35B A3B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 56.4, "resolved_count": 282, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 87, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-09T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-small-2512", "name": "Devstral Small (2512)", "vendor": "Mistral"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 56.4, "resolved_count": 282, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.swebench.com/verified", "observed_at": "2025-12-09T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-small-2512", "name": "Devstral Small (2512)", "vendor": "Mistral"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 56.36, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "o3", "name": "o3", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 56.36, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-4-1-opus", "name": "Claude 4.1 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 56.2, "resolved_count": 281, "total_count": 500, "cost_usd": 0.0472012191, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 20, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini", "name": "GPT-5 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 56.2, "resolved_count": 281, "total_count": 500, "cost_usd": 0.0472012191, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com", "observed_at": "2026-02-17T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini", "name": "GPT-5 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 55.4, "resolved_count": 277, "total_count": 500, "cost_usd": 0.247927108, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 65, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-02T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "vendor": "Qwen"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 55.4, "resolved_count": 277, "total_count": 500, "cost_usd": 0.09661007519999999, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 49, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-01T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-6-t=1", "name": "GLM 4.6 (T=1)", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 55.4, "resolved_count": 277, "total_count": 500, "cost_usd": 0.247927108, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://qwenlm.github.io/blog/qwen3-coder/", "observed_at": "2025-08-02T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "vendor": "Qwen"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 55.4, "resolved_count": 277, "total_count": 500, "cost_usd": 0.09661007519999999, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.swebench.com/verified", "observed_at": "2025-12-01T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-6-t=1", "name": "GLM 4.6 (T=1)", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 55.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-coder-next", "name": "Qwen3 Coder Next", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 54.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2-5", "name": "MiniMax M2.5", "vendor": "minimax"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 54.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-6", "name": "KIMI K2.6", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 54.29, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini-medium", "name": "GPT-5 mini (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 54.29, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-instruct-0905", "name": "KIMI K2 Instruct 0905", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 54.2, "resolved_count": 271, "total_count": 500, "cost_usd": 0.2971903488, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 40, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-22T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-5-2025-08-22", "name": "GLM 4.5 (2025 08 22)", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 54.2, "resolved_count": 271, "total_count": 500, "cost_usd": 0.2971903488, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://z.ai/blog/glm-4.5", "observed_at": "2025-08-22T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-5-2025-08-22", "name": "GLM 4.5 (2025 08 22)", "vendor": "Zhipu"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 54.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2-7", "name": "MiniMax M2.7", "vendor": "minimax"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 53.8, "resolved_count": 269, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 75, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-12-09T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-2512", "name": "Devstral (2512)", "vendor": "mistralai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 53.8, "resolved_count": 269, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.swebench.com/verified", "observed_at": "2025-12-09T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-2512", "name": "Devstral (2512)", "vendor": "mistralai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 53.6, "resolved_count": 268, "total_count": 500, "cost_usd": 0.28837209562500005, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 20, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-pro-20250506", "name": "Gemini 2.5 Pro (2025-05-06)", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 53.6, "resolved_count": 268, "total_count": 500, "cost_usd": 0.28837209562500005, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-pro-20250506", "name": "Gemini 2.5 Pro (2025-05-06)", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 52.8, "resolved_count": 264, "total_count": 500, "cost_usd": 0.3546154482, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 35, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-sonnet-20250219", "name": "Claude 3.7 Sonnet (2025-02-19)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 52.8, "resolved_count": 264, "total_count": 500, "cost_usd": 0.3546154482, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.anthropic.com/news/claude-3-7-sonnet", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-7-sonnet-20250219", "name": "Claude 3.7 Sonnet (2025-02-19)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 52.73, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-2-123b-instruct-2512", "name": "Devstral 2 123B Instruct 2512", "vendor": "Mistral"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 51.58, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-20250414", "name": "GPT-4.1 (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 51.43, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-5", "name": "GLM 4.5", "vendor": "z-ai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 50.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "vendor": "Qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 50.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "grok-4", "name": "GROK 4", "vendor": "xAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 49.6, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-6", "name": "GLM 4.6", "vendor": "z-ai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 48.97, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v4-pro-[high]", "name": "DeepSeek V4 Pro [high]", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 48.57, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "grok-code-fast-1", "name": "GROK Code Fast 1", "vendor": "xAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 48.42, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-0324", "name": "DeepSeek V3 0324", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 48.18, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2-1", "name": "MiniMax M2.1", "vendor": "minimax"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 47.91, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-5-air", "name": "GLM 4.5 Air", "vendor": "z-ai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 46.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-oss-120b-high", "name": "GPT Oss 120b High", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 46.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemma-4-31b", "name": "Gemma 4 31B", "vendor": "Google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 46.3, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini-high", "name": "GPT-5 mini (high)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 45.0, "resolved_count": 225, "total_count": 500, "cost_usd": 0.2099733328, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 23, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "o4-mini", "name": "o4 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 45.0, "resolved_count": 225, "total_count": 500, "cost_usd": 0.2099733328, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/o4-mini", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "o4-mini", "name": "o4 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 44.21, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "claude-3-5-sonnet", "name": "Claude 3.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 43.81, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "minimax-m2", "name": "MiniMax M2", "vendor": "minimax"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 43.8, "resolved_count": 219, "total_count": 500, "cost_usd": 0.531593366, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 38, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-instruct", "name": "KIMI K2 Instruct", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 43.8, "resolved_count": 219, "total_count": 500, "cost_usd": 0.531593366, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://moonshotai.github.io/Kimi-K2/", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2-instruct", "name": "KIMI K2 Instruct", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 43.64, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "o4-mini", "name": "o4 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 43.2, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v4-flash-[high]", "name": "DeepSeek V4 Flash [high]", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 43.16, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3", "name": "DeepSeek V3", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 42.86, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-2025", "name": "GPT-5.2025", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 41.67, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-6-27b", "name": "Qwen3.6 27B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 41.43, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "vendor": "Qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 40.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "kimi-k2", "name": "KIMI K2", "vendor": "moonshotai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 40.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-v3-1", "name": "DeepSeek V3.1", "vendor": "DeepSeek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 40.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-small-2-24b-instruct-2512", "name": "Devstral Small 2 24B Instruct 2512", "vendor": "Mistral"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 39.7, "resolved_count": 119, "total_count": 300, "cost_usd": 0.051632880333333325, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multilingual (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com", "observed_at": "2026-02-13T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-mini", "name": "GPT-5 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual"}}, {"success_rate": 39.58, "resolved_count": 198, "total_count": 500, "cost_usd": 0.146313124, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 20, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-20250414", "name": "GPT-4.1 (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 39.58, "resolved_count": 198, "total_count": 500, "cost_usd": 0.146313124, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-4.1", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-20250414", "name": "GPT-4.1 (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 38.75, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-6-35b-a3b", "name": "Qwen3.6 35B A3B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 37.89, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-235b-a22b-no-thinking", "name": "Qwen3 235B A22B No Thinking", "vendor": "Qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 36.3, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "glm-4-7-flash", "name": "GLM 4.7 Flash", "vendor": "z-ai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 35.71, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-flash", "name": "Gemini 2.5 Flash", "vendor": "google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 35.45, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-oss-120b", "name": "GPT Oss 120b", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 35.24, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22B Thinking 2507", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 34.8, "resolved_count": 174, "total_count": 500, "cost_usd": 0.03807518078, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 40, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-nano-medium", "name": "GPT-5 nano (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 34.8, "resolved_count": 174, "total_count": 500, "cost_usd": 0.03807518078, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-5-nano", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-nano-medium", "name": "GPT-5 nano (medium)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 34.29, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3 Coder 30B A3B Instruct", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 34.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "horizon-beta", "name": "Horizon Beta", "vendor": "openrouter"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 33.68, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-3-3-70b-instruct", "name": "Llama 3.3 70B Instruct", "vendor": "meta-llama"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 33.68, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-flash-preview", "name": "Gemini 2.5 Flash Preview", "vendor": "Google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 33.33, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-pro", "name": "Gemini 2.5 Pro", "vendor": "google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 32.63, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-32b-thinking", "name": "Qwen3 32B Thinking", "vendor": "Qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 32.63, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "devstral-small-2505", "name": "Devstral Small 2505", "vendor": "Mistral"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 31.76, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-next-80b-a3b-instruct", "name": "Qwen3 Next 80B A3B Instruct", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 31.43, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "deepseek-r1-0528", "name": "DeepSeek R1 0528", "vendor": "deepseek"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 30.53, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-32b-no-thinking", "name": "Qwen3 32B No Thinking", "vendor": "Qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 30.0, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "horizon-alpha", "name": "Horizon Alpha", "vendor": "openrouter"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 28.73, "resolved_count": 144, "total_count": 500, "cost_usd": 0.13388571835, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 27, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-flash-20250417", "name": "Gemini 2.5 Flash (2025-04-17)", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 28.73, "resolved_count": 144, "total_count": 500, "cost_usd": 0.13388571835, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-5-flash-20250417", "name": "Gemini 2.5 Flash (2025-04-17)", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 28.42, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-235b-a22b-thinking", "name": "Qwen3 235B A22B Thinking", "vendor": "Qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 26.0, "resolved_count": 130, "total_count": 500, "cost_usd": 0.05711905579999999, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 28, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-oss-120b", "name": "GPT Oss 120b", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 26.0, "resolved_count": 130, "total_count": 500, "cost_usd": 0.05711905579999999, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-oss-120b", "observed_at": "2025-08-07T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-oss-120b", "name": "GPT Oss 120b", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 24.29, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-0-flash", "name": "Gemini 2.0 Flash", "vendor": "Google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 23.94, "resolved_count": 120, "total_count": 500, "cost_usd": 0.43515415279999997, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 89, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-mini-20250414", "name": "GPT-4.1 mini (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 23.94, "resolved_count": 120, "total_count": 500, "cost_usd": 0.43515415279999997, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-4.1-mini", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-mini-20250414", "name": "GPT-4.1 mini (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 23.81, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-mini-20250414", "name": "GPT-4.1 mini (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 22.5, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-4-maverick-17b-128e-instruct", "name": "Llama 4 Maverick 17B 128E Instruct", "vendor": "Meta"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 21.62, "resolved_count": 108, "total_count": 500, "cost_usd": 1.531148125, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 63, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20241120", "name": "GPT-4o (2024-11-20)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 21.62, "resolved_count": 108, "total_count": 500, "cost_usd": 1.531148125, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://platform.openai.com/docs/models/gpt-4o", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20241120", "name": "GPT-4o (2024-11-20)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 21.05, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen2-5-72b-instruct", "name": "Qwen2.5 72B Instruct", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 21.04, "resolved_count": 105, "total_count": 500, "cost_usd": 0.31370609759999996, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 48, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-4-maverick-instruct", "name": "Llama 4 Maverick Instruct", "vendor": "Meta"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 21.04, "resolved_count": 105, "total_count": 500, "cost_usd": 0.31370609759999996, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.together.ai/models/llama-4-maverick", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-4-maverick-instruct", "name": "Llama 4 Maverick Instruct", "vendor": "Meta"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 19.47, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-235b-a22b", "name": "Qwen3 235B A22B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 19.05, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-30b-a3b-thinking-2507", "name": "Qwen3 30B A3B Thinking 2507", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 18.75, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E Instruct", "vendor": "Meta"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 17.14, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 14.29, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-oss-20b", "name": "GPT Oss 20b", "vendor": "openai"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 13.68, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemma-3-27b-it", "name": "Gemma 3 27b It", "vendor": "google"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 13.52, "resolved_count": 68, "total_count": 500, "cost_usd": 0.0, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 0, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-0-flash", "name": "Gemini 2.0 Flash", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 13.52, "resolved_count": 68, "total_count": 500, "cost_usd": 0.0, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://ai.google.dev/gemini-api/docs/models", "observed_at": "2025-07-26T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gemini-2-0-flash", "name": "Gemini 2.0 Flash", "vendor": "Google"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 9.06, "resolved_count": 45, "total_count": 500, "cost_usd": 0.115631523, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 29, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-4-scout-instruct", "name": "Llama 4 Scout Instruct", "vendor": "Meta"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 9.06, "resolved_count": 45, "total_count": 500, "cost_usd": 0.115631523, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://www.together.ai/models/llama-4-scout", "observed_at": "2025-07-20T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "llama-4-scout-instruct", "name": "Llama 4 Scout Instruct", "vendor": "Meta"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 9.0, "resolved_count": 45, "total_count": 500, "cost_usd": 0.06811614065999999, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 48, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://mini-swe-agent.com/latest/", "observed_at": "2025-08-03T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen2-5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "vendor": "qwen"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 9.0, "resolved_count": 45, "total_count": 500, "cost_usd": 0.06811614065999999, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench bash-only (mini-SWE-agent)", "source": "swe_bench_site", "source_url": "https://qwenlm.github.io/blog/qwen2.5-coder/", "observed_at": "2025-08-03T00:00:00+00:00", "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen2-5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "vendor": "qwen"}, "benchmark": {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)"}}, {"success_rate": 8.57, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen3-32b", "name": "Qwen3 32B", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 7.62, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "qwen2-5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "vendor": "qwen"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}, {"success_rate": 1.25, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-rebench resolved rate (best rolling window)", "source": "swe_rebench_site", "source_url": "https://swe-rebench.com/", "observed_at": null, "harness": {"slug": "mini-swe-agent", "name": "mini-SWE-agent", "vendor": "anthropic", "repo_url": "https://github.com/SWE-agent/mini-swe-agent", "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-nano-20250414", "name": "GPT-4.1 nano (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-rebench", "name": "SWE-rebench"}}], "benchmarks": [{"slug": "swe-bench-multilingual", "name": "SWE-bench Multilingual", "category": "eval"}, {"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}, {"slug": "swe-bench-bash-only", "name": "SWE-bench Verified (bash-only)", "category": "eval"}, {"slug": "swe-rebench", "name": "SWE-rebench", "category": "eval"}, {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)", "category": "eval"}]}