{"harness": {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "description": "GUIRepair is tracked in this catalog as a software-engineering agent harness. Public benchmark rows use this slug on leaderboards such as SWE-bench Verified or Artificial Analysis coding-agent evals.\n\nCheck each score row's `source` and `source_url` for the authoritative snapshot. Harness-level metadata here does not invent scores \u2014 only documents what the harness is and where published results come from.", "card_summary": "GUIRepair GUIRepair coding agent harness.", "homepage_url": "https://sites.google.com/view/guirepair", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 35.98, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250531_GUIRepair_gpt4o.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 35.98, "resolved_count": 186, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://sites.google.com/view/guirepair", "observed_at": "2025-07-01T00:00:00+00:00", "harness": {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "o3", "name": "o3", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 33.85, "resolved_count": 175, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://sites.google.com/view/guirepair", "observed_at": "2025-05-31T00:00:00+00:00", "harness": {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "o4-mini", "name": "o4 mini", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 31.14, "resolved_count": 161, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://sites.google.com/view/guirepair", "observed_at": "2025-05-31T00:00:00+00:00", "harness": {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4-1-20250414", "name": "GPT-4.1 (2025-04-14)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}, {"success_rate": 30.37, "resolved_count": 157, "total_count": 517, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Multimodal public submission", "source": "swe_bench_site", "source_url": "https://sites.google.com/view/guirepair", "observed_at": "2025-05-31T00:00:00+00:00", "harness": {"slug": "guirepair", "name": "GUIRepair", "vendor": "GUIRepair", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)"}}], "benchmarks": [{"slug": "swe-bench-multimodal-site", "name": "SWE-bench Multimodal (site)", "category": "eval"}]}