{"model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI", "model_family": "gpt-4o", "parameter_count_b": null, "context_length": 400000, "homepage_url": null, "hf_model_id": null, "openrouter_id": null, "card_summary": "OpenAI GPT-4o (2024-08-06) on SWE-bench Verified leaderboards.", "is_rl_checkpoint": false, "score_count": 12, "best_success_rate": 71.4, "input_price_per_million": null, "output_price_per_million": null}, "scores": [{"success_rate": 71.4, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Aider edit leaderboard", "source": "aider_leaderboard", "source_url": "https://aider.chat/docs/leaderboards/", "observed_at": "2024-08-06T00:00:00+00:00", "harness": {"slug": "aider", "name": "Aider", "vendor": "Aider", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "aider-edit", "name": "Aider Edit"}}, {"success_rate": 49.4, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Aider refactor leaderboard", "source": "aider_leaderboard", "source_url": "https://aider.chat/docs/leaderboards/", "observed_at": "2024-08-06T00:00:00+00:00", "harness": {"slug": "aider", "name": "Aider", "vendor": "Aider", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "aider-refactor", "name": "Aider Refactor"}}, {"success_rate": 38.3, "resolved_count": null, "total_count": 1055, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "LiveCodeBench generation pass@1 mean", "source": "livecodebench", "source_url": "https://livecodebench.github.io/leaderboard.html", "observed_at": null, "harness": {"slug": "livecodebench", "name": "Livecodebench", "vendor": "LiveCodeBench", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "livecodebench-generation", "name": "LiveCodeBench (code generation)"}}, {"success_rate": 30.37, "resolved_count": 157, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://sites.google.com/view/guirepair", "observed_at": "2025-05-31T00:00:00+00:00", "harness": {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 23.1, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Aider polyglot leaderboard", "source": "aider_leaderboard", "source_url": "https://aider.chat/docs/leaderboards/", "observed_at": "2024-12-30T00:00:00+00:00", "harness": {"slug": "aider", "name": "Aider", "vendor": "Aider", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "aider-polyglot", "name": "Aider Polyglot"}}, {"success_rate": 12.19, "resolved_count": 63, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-10-06T00:00:00+00:00", "harness": {"slug": "swe-agent-multimodal", "name": "SWE-agent (multimodal)", "vendor": "Princeton NLP", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 11.99, "resolved_count": 62, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-10-06T00:00:00+00:00", "harness": {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 9.28, "resolved_count": 48, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://github.com/SWE-agent/SWE-agent", "observed_at": "2024-10-06T00:00:00+00:00", "harness": {"slug": "swe-agent-javascript", "name": "SWE-agent (JavaScript)", "vendor": "Princeton NLP", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 8.63, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "MLE-bench All split any_medal_percentage", "source": "mle_bench_site", "source_url": "https://github.com/openai/mle-bench", "observed_at": "2024-10-08T00:00:00+00:00", "harness": {"slug": "aide", "name": "AIDE", "vendor": "Wecoai", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "mle-bench", "name": "MLE-bench"}}, {"success_rate": 4.89, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "MLE-bench All split any_medal_percentage", "source": "mle_bench_site", "source_url": "https://github.com/openai/mle-bench", "observed_at": "2024-10-08T00:00:00+00:00", "harness": {"slug": "openhands", "name": "OpenHands (legacy)", "vendor": "All Hands AI", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "mle-bench", "name": "MLE-bench"}}, {"success_rate": 3.09, "resolved_count": 16, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://github.com/OpenAutoCoder/Agentless", "observed_at": "2024-10-06T00:00:00+00:00", "harness": {"slug": "agentless-lite", "name": "Agentless Lite", "vendor": "OpenAutoCoder", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 1.6, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "MLE-bench All split any_medal_percentage", "source": "mle_bench_site", "source_url": "https://github.com/openai/mle-bench", "observed_at": "2024-10-08T00:00:00+00:00", "harness": {"slug": "mlab", "name": "Mlab", "vendor": "MLAB", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "mle-bench", "name": "MLE-bench"}}], "harnesses": [{"slug": "aide", "name": "AIDE", "vendor": "Wecoai", "repo_url": "https://github.com/wecoai/aideml", "openrouter_icon_url": null}, {"slug": "agentless-lite", "name": "Agentless Lite", "vendor": "OpenAutoCoder", "repo_url": "https://github.com/OpenAutoCoder/Agentless", "openrouter_icon_url": null}, {"slug": "aider", "name": "Aider", "vendor": "Aider", "repo_url": "https://github.com/Aider-AI/aider", "openrouter_icon_url": "https://aider.chat/favicon.ico"}, {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "repo_url": null, "openrouter_icon_url": null}, {"slug": "livecodebench", "name": "Livecodebench", "vendor": "LiveCodeBench", "repo_url": "https://github.com/LiveCodeBench/LiveCodeBench", "openrouter_icon_url": null}, {"slug": "mlab", "name": "Mlab", "vendor": "MLAB", "repo_url": null, "openrouter_icon_url": null}, {"slug": "openhands", "name": "OpenHands (legacy)", "vendor": "All Hands AI", "repo_url": "https://github.com/All-Hands-AI/OpenHands", "openrouter_icon_url": null}, {"slug": "swe-agent", "name": "SWE-agent", "vendor": "SWE-agent", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, {"slug": "swe-agent-javascript", "name": "SWE-agent (JavaScript)", "vendor": "Princeton NLP", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}, {"slug": "swe-agent-multimodal", "name": "SWE-agent (multimodal)", "vendor": "Princeton NLP", "repo_url": "https://github.com/SWE-agent/SWE-agent", "openrouter_icon_url": null}]}