{"harness": {"slug": "kimi-code-cli", "name": "Kimi Code CLI", "vendor": "Moonshot", "description": "Kimi Code CLI is Moonshot AI's terminal-oriented coding agent, paired with Kimi foundation models on public leaderboards. It follows the product CLI pattern: repository context, edits, and shell tools in a managed loop.\n\nScores in this catalog primarily come from Artificial Analysis ingest (multiple model pairings per agent). Community SWE-bench submissions may appear when Moonshot or partners publish Verified runs.\n\nRelated open weights and dev tooling may appear under MoonshotAI GitHub org (for example Kimi-Dev); this page lists harness-level published scores only.", "card_summary": "Moonshot Kimi terminal coding agent with tool-use for SWE-style tasks.", "homepage_url": "https://kimi.moonshot.cn", "repo_url": "https://github.com/MoonshotAI/Kimi-Dev", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 87.64, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": 0.6263880112748019, "aa_mean_cost_usd": 3.081635541610427, "aa_mean_total_tokens": 10383196}, "scores": [{"success_rate": 87.64, "resolved_count": null, "total_count": null, "cost_usd": 3.081635541610427, "input_tokens": 5296768, "output_tokens": 54498, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "kimi-code-cli", "name": "Kimi Code CLI", "vendor": "Moonshot", "repo_url": "https://github.com/MoonshotAI/Kimi-Dev", "openrouter_icon_url": null}, "model": {"slug": "kimi-k3", "name": "KIMI K3", "vendor": "Moonshot"}, "benchmark": {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)"}}, {"success_rate": 63.72, "resolved_count": null, "total_count": null, "cost_usd": 3.081635541610427, "input_tokens": 5296768, "output_tokens": 54498, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "kimi-code-cli", "name": "Kimi Code CLI", "vendor": "Moonshot", "repo_url": "https://github.com/MoonshotAI/Kimi-Dev", "openrouter_icon_url": null}, "model": {"slug": "kimi-k3", "name": "KIMI K3", "vendor": "Moonshot"}, "benchmark": {"slug": "deepswe", "name": "DeepSWE (AA component)"}}, {"success_rate": 36.56, "resolved_count": null, "total_count": null, "cost_usd": 3.081635541610427, "input_tokens": 5296768, "output_tokens": 54498, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "Artificial Analysis Coding Agent Index component (pass@1)", "source": "artificialanalysis", "source_url": "https://artificialanalysis.ai/agents/coding-agents", "observed_at": "2026-08-27T21:54:11.189561+00:00", "harness": {"slug": "kimi-code-cli", "name": "Kimi Code CLI", "vendor": "Moonshot", "repo_url": "https://github.com/MoonshotAI/Kimi-Dev", "openrouter_icon_url": null}, "model": {"slug": "kimi-k3", "name": "KIMI K3", "vendor": "Moonshot"}, "benchmark": {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)"}}], "benchmarks": [{"slug": "deepswe", "name": "DeepSWE (AA component)", "category": "eval"}, {"slug": "swe-atlas-qna", "name": "SWE-Atlas-QnA (AA component)", "category": "eval"}, {"slug": "terminal-bench-v2-1", "name": "Terminal-Bench v2.1 (AA component)", "category": "eval"}]}