{"harness": {"slug": "deepswe-preview", "name": "Deepswe Preview", "vendor": "Agentica", "description": "DeepSWE preview marks early or preview DeepSWE agent submissions distinct from the production DeepSWE benchmark component on Artificial Analysis. DeepSWE (DataCurve) is a long-horizon software engineering eval; preview harness slugs capture leaderboard rows before stable release tagging.\n\nMay appear on `swe-bench-verified` or AA-related ingest depending on submission. Contrast with AA `deepswe` benchmark scores \u2014 preview harness slug vs benchmark slug measure different things.\n\nSee DataCurve / DeepSWE official docs for task definitions; this page lists published harness-level scores only.", "card_summary": "DeepSWE preview harness \u2014 early DataCurve / DeepSWE agent eval rows.", "homepage_url": "https://datacurve.ai", "repo_url": null, "docs_url": null, "scaffold_code": null, "harbor_agent_name": null, "api_protocol": null, "status": "validated", "supports_rl": true, "score_count": 1, "best_success_rate": 58.8, "github_stars": null, "popularity_tier": null, "hrl_score": null, "hrl_rank": null, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": "https://www.swebench.com/img/logos/20250629_deepswerl_r2eagent_tts.jpg", "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 58.8, "resolved_count": 294, "total_count": 500, "cost_usd": null, "input_tokens": null, "output_tokens": null, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": null, "is_harnessrl_measured": false, "eval_protocol": "SWE-bench Verified public submission", "source": "swe_bench_site", "source_url": "agentica-project.com/", "observed_at": "2025-06-29T00:00:00+00:00", "harness": {"slug": "deepswe-preview", "name": "Deepswe Preview", "vendor": "Agentica", "repo_url": null, "openrouter_icon_url": null}, "model": {"slug": "tts-bo16", "name": "tts-bo16", "vendor": "TTS"}, "benchmark": {"slug": "swe-bench-verified", "name": "SWE-bench Verified"}}], "benchmarks": [{"slug": "swe-bench-verified", "name": "SWE-bench Verified", "category": "eval"}]}