{"harness": {"slug": "epam-ai-run-developer-agent", "name": "EPAM AI Run", "vendor": "EPAM Systems, Inc.", "description": "EPAM AI RUN is EPAM's agent scaffold for software-engineering benchmarks. Public SWE-bench Verified submissions appear under several versioned harness names (`epam-ai-run-developer-agent-v20241029`, v20241212, v20250219, v20250719, etc.) when EPAM publishes updated agent builds.\n\nEach version slug reflects a leaderboard submission tag \u2014 prompts, retrieval, and tool wiring can differ between versions even when the backing model is similar. Scores are community `swe_bench_site` rows only; this is not a HarnessRL validated scaffold.\n\nOpen source reference: `epam/AI-RUN` on GitHub. Compare versioned slugs side-by-side rather than merging them into one historical trend line.", "card_summary": "EPAM AI/Run developer agent \u2014 enterprise SWE harness with versioned Verified submissions.", "homepage_url": "https://github.com/epam/AI-RUN", "repo_url": "https://github.com/epam/AI-RUN", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 3, "best_success_rate": 76.8, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240820_epam-ai-run-gpt-4o.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 76.8, "resolved_count": 384, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.epam.com/services/artificial-intelligence", "observed_at": "2025-08-04T00:00:00+00:00", "harness": {"slug": "epam-ai-run-developer-agent", "name": "EPAM AI Run", "vendor": "EPAM Systems, Inc.", "repo_url": "https://github.com/epam/AI-RUN", "openrouter_icon_url": null}, "model": {"slug": "claude-4-sonnet", "name": "Claude 4 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 39.6, "resolved_count": 198, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.epam.com/services/artificial-intelligence", "observed_at": "2024-10-29T00:00:00+00:00", "harness": {"slug": "epam-ai-run-developer-agent", "name": "EPAM AI Run", "vendor": "EPAM Systems, Inc.", "repo_url": "https://github.com/epam/AI-RUN", "openrouter_icon_url": null}, "model": {"slug": "claude-3-5-sonnet", "name": "Claude 3.5 Sonnet", "vendor": "Anthropic"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 24.0, "resolved_count": 120, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.epam.com/services/artificial-intelligence", "observed_at": "2024-08-20T00:00:00+00:00", "harness": {"slug": "epam-ai-run-developer-agent", "name": "EPAM AI Run", "vendor": "EPAM Systems, Inc.", "repo_url": "https://github.com/epam/AI-RUN", "openrouter_icon_url": null}, "model": {"slug": "gpt-4", "name": "GPT-4", "vendor": "openai"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}], "benchmarks": [{"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}]}