{"harness": {"slug": "grok-build", "name": "Grok Build", "vendor": "xAI", "description": "Grok Build refers to xAI's coding agent harness used in public benchmark submissions (Artificial Analysis Coding Agent Index and related leaderboards). Rows pair xAI's agent tooling with Grok foundation models.\n\nAs with other vendor product harnesses, scores encode both model capability and xAI's agent stack (tools, prompts, safety). Use `aa_coding_index` on the harness row for AA's composite index when present \u2014 component breakdowns appear as separate benchmark scores in the table below.\n\nCheck `source` and `source_url` on each score; Grok model identifiers evolve quickly across leaderboard snapshots.", "card_summary": "xAI Grok coding agent scaffold for software engineering benchmarks.", "homepage_url": "https://x.ai", "repo_url": "https://github.com/xai-org/grok", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 84.27, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.6408998279698193, "aa_mean_cost_usd": 2.4389824437627774, "aa_mean_total_tokens": 3598867}, "scores": [{"success_rate": 84.27, "resolved_count": null, "total_count": null, "cost_usd": 2.4389824437627774, "input_tokens": 1837535, "output_tokens": 25699, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "grok-build", "name": "Grok Build", "vendor": "xAI", "repo_url": "https://github.com/xai-org/grok", "openrouter_icon_url": null}, "model": {"slug": "grok-4-5-high", "name": "GROK 4.5 (high)", "vendor": "xAI"}, "benchmark": {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)"}}, {"success_rate": 59.88, "resolved_count": null, "total_count": null, "cost_usd": 2.4389824437627774, "input_tokens": 1837535, "output_tokens": 25699, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "grok-build", "name": "Grok Build", "vendor": "xAI", "repo_url": "https://github.com/xai-org/grok", "openrouter_icon_url": null}, "model": {"slug": "grok-4-5-high", "name": "GROK 4.5 (high)", "vendor": "xAI"}, "benchmark": {"slug": "deepswe", "name": "DeepSWE (AA component)"}}, {"success_rate": 48.12, "resolved_count": null, "total_count": null, "cost_usd": 2.4389824437627774, "input_tokens": 1837535, "output_tokens": 25699, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "grok-build", "name": "Grok Build", "vendor": "xAI", "repo_url": "https://github.com/xai-org/grok", "openrouter_icon_url": null}, "model": {"slug": "grok-4-5-high", "name": "GROK 4.5 (high)", "vendor": "xAI"}, "benchmark": {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)"}}], "benchmarks": [{"slug": "deepswe", "name": "DeepSWE (AA component)", "category": "eval"}, {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)", "category": "eval"}, {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)", "category": "eval"}]}