{"harness": {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "description": "Terminus-2 is Harbor's reference agent that drives an interactive terminal (tmux) inside the evaluation container. Instead of only issuing one-shot shell strings, it can send keystrokes, scroll, and manage multiplexer sessions \u2014 closer to how engineers use a real terminal.\n\nHarnessRL validates Terminus-2 end-to-end (`scaffold_code` `terminus`, Harbor `Terminus2`, OpenAI-compatible API). It is a strong choice when tasks require interactive CLI tools, pagers, or long-running processes.\n\nPublic catalog scores are primarily HarnessRL-measured or curated seed rows; community SWE-bench submissions more often use SWE-agent family scaffolds. Check `supports_rl` and measured flags on individual score rows.", "card_summary": "Harbor Terminus-2 agent \u2014 in-process tmux control inside the sandbox pod.", "homepage_url": "https://github.com/Elvin-Yiming-Du/harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "docs_url": "https://harness-rl.pages.dev/docs/architecture/agent-loop-workers", "scaffold_code": "terminus", "harbor_agent_name": "Terminus2", "api_protocol": "openai", "status": "validated", "supports_rl": true, "score_count": 5, "best_success_rate": 80.45, "github_stars": 890, "popularity_tier": null, "hrl_score": 31.0, "hrl_rank": 27, "catalog_token_volume": null, "openrouter_icon_url": null, "leaderboard_icon_url": null, "aa_coding_index": null, "aa_mean_cost_usd": null, "aa_mean_total_tokens": null}, "scores": [{"success_rate": 80.45, "resolved_count": null, "total_count": null, "cost_usd": 0.9857078651685393, "input_tokens": 7874199, "output_tokens": 7412519, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/78", "observed_at": "2026-06-05T00:00:00+00:00", "harness": {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "openrouter_icon_url": null}, "model": {"slug": "fable-5", "name": "Fable 5", "vendor": "Anthropic"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}, {"success_rate": 77.98, "resolved_count": null, "total_count": null, "cost_usd": 1.1097752808988766, "input_tokens": 81469069, "output_tokens": 10768129, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/47", "observed_at": "2026-05-01T00:00:00+00:00", "harness": {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "openrouter_icon_url": null}, "model": {"slug": "gpt-5-5", "name": "GPT-5.5", "vendor": "openai"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}, {"success_rate": 73.93, "resolved_count": null, "total_count": null, "cost_usd": 0.5043595505617977, "input_tokens": 26094316, "output_tokens": 12862730, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/48", "observed_at": "2026-05-01T00:00:00+00:00", "harness": {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-pro", "name": "Gemini 3 Pro", "vendor": "Google"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}, {"success_rate": 66.07, "resolved_count": null, "total_count": null, "cost_usd": 1.3084494382022471, "input_tokens": 16646631, "output_tokens": 12451178, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/46", "observed_at": "2026-05-01T00:00:00+00:00", "harness": {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "openrouter_icon_url": null}, "model": {"slug": "claude-4-7-opus", "name": "Claude 4.7 Opus", "vendor": "Anthropic"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}, {"success_rate": 65.62, "resolved_count": null, "total_count": null, "cost_usd": 0.5168314606741573, "input_tokens": 27525560, "output_tokens": 12935831, "latency_ms": null, "temperature": null, "max_turns": null, "context_budget_tokens": null, "harness_version": null, "trial_count": 445, "is_harnessrl_measured": false, "eval_protocol": null, "source": "tbench_official", "source_url": "https://github.com/harbor-framework/terminal-bench-2-1/pull/69", "observed_at": "2026-05-05T00:00:00+00:00", "harness": {"slug": "terminus-2", "name": "Terminus 2", "vendor": "Harbor", "repo_url": "https://github.com/Elvin-Yiming-Du/harbor", "openrouter_icon_url": null}, "model": {"slug": "gemini-3-1-pro", "name": "Gemini 3.1 Pro", "vendor": "Google"}, "benchmark": {"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)"}}], "benchmarks": [{"slug": "terminal-bench-2-1-official", "name": "Terminal-Bench 2.1 (official)", "category": "eval"}]}