{"harness": {"slug": "swe-rl-llama3-swe-rl-70b", "name": "SWE Rl Llama3 SWE Rl 70b", "vendor": "Facebookresearch", "description": "This slug tracks a SWE-bench Verified submission for an agent/checkpoint from SWE-RL training on Llama 3 at 70B scale (`SWE-RL-70B`). It connects RL-for-SWE research to the standard Verified leaderboard for comparability with supervised baselines.\n\nCommunity `swe_bench_site` ingest. Compare against same-scale non-RL Llama agents only when eval protocol matches.", "card_summary": "SWE-RL Llama3 SWE-RL-70B \u2014 RL-trained agent on Verified.", "homepage_url": "https://www.swebench.com/verified", "repo_url": "https://github.com/facebookresearch/swe-rl", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 41.2, "github_stars": 719, "popularity_tier": null, "hrl_score": 30.0, "hrl_rank": 32, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250226_swerl_llama3_70b.png", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 41.2, "resolved_count": 206, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://github.com/facebookresearch/swe-rl", "observed_at": "2025-02-26T00:00:00+00:00", "harness": {"slug": "swe-rl-llama3-swe-rl-70b", "name": "SWE Rl Llama3 SWE Rl 70b", "vendor": "Facebookresearch", "repo_url": "https://github.com/facebookresearch/swe-rl", "openrouter_icon_url": null}, "model": {"slug": "agentless-mini-20250226", "name": "Agentless Mini) (20250226", "vendor": "Agentless"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}], "benchmarks": [{"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}]}