{"harness": {"slug": "lingma-agent", "name": "Lingma Agent", "vendor": "Alibaba", "description": "Lingma Agent refers to Alibaba's software-engineering agent offerings (Lingma / Tongyi ecosystem) with public ModelScope and SWE-bench leaderboard presence. Submissions pair the Lingma agent harness with disclosed models on Verified tasks.\n\nRows in this catalog are ingested from community SWE-bench submissions. The agent stack includes vendor retrieval, edit tools, and prompts beyond a minimal bash loop.\n\nFor model weights and training details, see ModelScope model cards (for example Lingma-SWE-GPT family); this page tracks harness-level leaderboard scores only.", "card_summary": "Alibaba Lingma SWE agent \u2014 ModelScope-backed submissions on SWE-bench Verified.", "homepage_url": "https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT", "repo_url": "https://github.com/modelscope/modelscope", "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 4, "best_success_rate": 28.8, "github_stars": 7200, "popularity_tier": null, "hrl_score": 32.0, "hrl_rank": 20, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20240918_lingma-agent_lingma-swe-gpt-7b.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 28.8, "resolved_count": 144, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT (https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT-v20240925)", "observed_at": "2024-10-02T00:00:00+00:00", "harness": {"slug": "lingma-agent", "name": "Lingma Agent", "vendor": "Alibaba", "repo_url": "https://github.com/modelscope/modelscope", "openrouter_icon_url": null}, "model": {"slug": "lingma-swe-gpt-72b-v0925", "name": "Lingma SWE GPT 72b (v0925)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 25.0, "resolved_count": 125, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT", "observed_at": "2024-09-18T00:00:00+00:00", "harness": {"slug": "lingma-agent", "name": "Lingma Agent", "vendor": "Alibaba", "repo_url": "https://github.com/modelscope/modelscope", "openrouter_icon_url": null}, "model": {"slug": "lingma-swe-gpt-72b-v0918", "name": "Lingma SWE GPT 72b (v0918)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 18.2, "resolved_count": 91, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT (https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT-v20240925)", "observed_at": "2024-10-02T00:00:00+00:00", "harness": {"slug": "lingma-agent", "name": "Lingma Agent", "vendor": "Alibaba", "repo_url": "https://github.com/modelscope/modelscope", "openrouter_icon_url": null}, "model": {"slug": "lingma-swe-gpt-7b-v0925", "name": "Lingma SWE GPT 7b (v0925)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}, {"success_rate": 10.2, "resolved_count": 51, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "https://www.modelscope.cn/models/yingwei/Lingma-SWE-GPT", "observed_at": "2024-09-18T00:00:00+00:00", "harness": {"slug": "lingma-agent", "name": "Lingma Agent", "vendor": "Alibaba", "repo_url": "https://github.com/modelscope/modelscope", "openrouter_icon_url": null}, "model": {"slug": "lingma-swe-gpt-7b-v0918", "name": "Lingma SWE GPT 7b (v0918)", "vendor": "OpenAI"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}], "benchmarks": [{"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}]}