{"model": {"slug": "gpt-5-6", "name": "GPT-5.6", "vendor": "OpenAI", "model_family": "gpt-5-6", "parameter_count_b": null, "context_length": 400000, "homepage_url": null, "hf_model_id": null, "openrouter_id": null, "card_summary": "OpenAI GPT-5.6 Luna base variant (no extra reasoning tier).", "is_rl_checkpoint": false, "score_count": 5, "best_success_rate": 78.43, "input_price_per_million": null, "output_price_per_million": null}, "scores": [{"success_rate": 78.43, "resolved_count": null, "total_count": null, "cost_usd": 0.9464044943820225, "input_tokens": 29803660, "output_tokens": 8707506, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/115", "observed_at": "2026-07-11T00:00:00+00:00", "harness": {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-5-6", "name": "GPT-5.6", "vendor": "OpenAI"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}, {"success_rate": 67.2, "resolved_count": null, "total_count": null, "cost_usd": 0.9463, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 60, "is_harnessrl_measured": false, "eval_protocol": "WildClawBench overall score (60 tasks)", "source": "wildclawbench_site", "source_url": "https://github.com/internlm/WildClawBench", "observed_at": null, "harness": {"slug": "openclaw", "name": "OpenClaw", "vendor": "OpenClaw", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-5-6", "name": "GPT-5.6", "vendor": "OpenAI"}, "benchmark": {"slug": "wildclawbench", "name": "WildClawBench"}}, {"success_rate": 60.3, "resolved_count": null, "total_count": null, "cost_usd": 1.0868546433537856, "input_tokens": 1731981, "output_tokens": 7689, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-5-6", "name": "GPT-5.6", "vendor": "OpenAI"}, "benchmark": {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)"}}, {"success_rate": 35.4, "resolved_count": null, "total_count": null, "cost_usd": 1.0868546433537856, "input_tokens": 1731981, "output_tokens": 7689, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-5-6", "name": "GPT-5.6", "vendor": "OpenAI"}, "benchmark": {"slug": "deepswe", "name": "DeepSWE (AA component)"}}, {"success_rate": 34.41, "resolved_count": null, "total_count": null, "cost_usd": 1.0868546433537856, "input_tokens": 1731981, "output_tokens": 7689, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "codex", "name": "Codex", "vendor": "OpenAI", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-5-6", "name": "GPT-5.6", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)"}}], "harnesses": [{"slug": "codex", "name": "Codex", "vendor": "OpenAI", "repo_url": "https://github.com/openai/codex", "openrouter_icon_url": "https://openai.com/favicon.ico"}, {"slug": "openclaw", "name": "OpenClaw", "vendor": "OpenClaw", "repo_url": "https://github.com/openclaw/openclaw", "openrouter_icon_url": null}]}