{"harness": {"slug": "openhands", "name": "OpenHands (legacy)", "vendor": "All Hands AI", "description": "This slug tracks the older OpenHands integration in HarnessRL (`scaffold_code` `oh`, Harbor agent `OpenHands`). It targets the same All Hands AI codebase as `openhands-sdk` but through the previous Harbor wiring and configuration surface.\n\nFor new experiments and RL, prefer `openhands-sdk`. This entry remains for historical leaderboard rows and configs that still reference `oh`.\n\nBenchmark appearances mirror the SDK path on SWE-bench Verified and AA ingest aliases (AA product name \"OpenHands\" maps here and to the SDK slug depending on ingest version).", "card_summary": "Legacy OpenHands Harbor integration (`SCAFFOLD=oh`) \u2014 same upstream project as the SDK path.", "homepage_url": "https://github.com/All-Hands-AI/OpenHands", "repo_url": "https://github.com/All-Hands-AI/OpenHands", "docs_url": null, "scaffold_code": "oh", "harbor_agent_name": "OpenHands", "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 2, "best_success_rate": 26.67, "github_stars": 52300, "popularity_tier": null, "hrl_score": 34.0, "hrl_rank": 16, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240725_opendevin_codeact_v1.8_claude35sonnet.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 26.67, "resolved_count": 80, "total_count": 300, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Lite public submission", "source": "swe_bench_site", "source_url": "https://docs.all-hands.dev/", "observed_at": "2024-07-25T00:00:00+00:00", "harness": {"slug": "openhands", "name": "OpenHands (legacy)", "vendor": "All Hands AI", "repo_url": "https://github.com/All-Hands-AI/OpenHands", "openrouter_icon_url": null}, "model": {"slug": "codeact-v1-8", "name": "CodeAct v1.8", "vendor": null}, "benchmark": {"slug": "swe-bench-lite", "name": "SWE-bench Lite"}}, {"success_rate": 4.89, "resolved_count": null, "total_count": null, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "MLE-bench All split any_medal_percentage", "source": "mle_bench_site", "source_url": "https://github.com/openai/mle-bench", "observed_at": "2024-10-08T00:00:00+00:00", "harness": {"slug": "openhands", "name": "OpenHands (legacy)", "vendor": "All Hands AI", "repo_url": "https://github.com/All-Hands-AI/OpenHands", "openrouter_icon_url": null}, "model": {"slug": "gpt-4o-20240806", "name": "GPT-4o (2024-08-06)", "vendor": "OpenAI"}, "benchmark": {"slug": "mle-bench", "name": "MLE-bench"}}], "benchmarks": [{"slug": "mle-bench", "name": "MLE-bench", "category": "eval"}, {"slug": "swe-bench-lite", "name": "SWE-bench Lite", "category": "eval"}]}