{"harness": {"slug": "refact-ai-agent", "name": "Refact Ai Agent", "vendor": "Refact.ai", "description": "Refact provides an open AI coding agent (Refact.ai) with tooling for repository context and edits. Public SWE-bench Verified submissions use this harness label when authors evaluate Refact's agent loop on the 500-task verified split.\n\nOpen-source agent stack; community leaderboard coverage via `swe_bench_site`. Product features (IDE plugins, enterprise) may differ from the public eval configuration cited on swebench.com.", "card_summary": "Refact AI agent \u2014 open coding agent with Verified leaderboard entry.", "homepage_url": "https://refact.ai", "repo_url": "https://github.com/smallcloudai/refact", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 74.4, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250611_Refact_Agent_claude-4-sonnet.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 74.4, "resolved_count": 372, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://refact.ai", "observed_at": "2025-06-03T00:00:00+00:00", "harness": {"slug": "refact-ai-agent", "name": "Refact Ai Agent", "vendor": "Refact.ai", "repo_url": "https://github.com/smallcloudai/refact", "openrouter_icon_url": null}, "model": {"slug": "ensemble-claude-4-sonnet-o4-mini", "name": "Ensemble (Claude 4 Sonnet, o4 mini)", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 35.59, "resolved_count": 184, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://refact.ai", "observed_at": "2025-06-11T00:00:00+00:00", "harness": {"slug": "refact-ai-agent", "name": "Refact Ai Agent", "vendor": "Refact.ai", "repo_url": "https://github.com/smallcloudai/refact", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}], "benchmarks": [{"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)", "category": "eval"}, {"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}]}