{"harness": {"recipes": 1, "summary": "Crush (github.com/charmbracelet/crush) is a terminal-based AI coding assistant built in Go. Its CLI entrypoint is `crush run [prompt]` for non-interactive use, which sends the prompt to an LLM and streams the response while the agent uses tools (view, edit, bash, glob, grep, etc.) to complete coding tasks. The harness is built from source using `golang:latest` with `CGO_ENABLED=0 GOEXPERIMENT=greenteagc`. A crushrc config file at `$CRUSH_GLOBAL_CONFIG/crushrc` configures a custom OpenAI-compatible provider pointing to `$PROXY_URL/v1` with model `claude-3-5-sonnet-20241022`, disables the default provider catalog and auto-update to avoid network calls, and pre-approves all tool permissions so the agent can run without user interaction. The run command is `crush run --quiet \"$TASK\"` from the task working directory.", "recipe": "charmbracelet-crush@c9b45348a4f1b7695e70830a4408aa4c31688851", "compatibility": null, "harness": "charmbracelet-crush", "first_run": "2026-09-26T07:48:48", "domains": ["swe"], "runs": 1, "base_image": "buildpack-deps:jammy", "finished": 1, "last_run": "2026-09-26T07:48:48", "repo": "https://github.com/charmbracelet/crush", "last_run_id": "20260926T073808-charmbracele-308ef1539f9a-polyglot_cpp_allergies", "use_case": "Writes, edits, tests, and debugs code in a terminal by executing shell commands, reading and modifying files, and using LSP for code intelligence.", "tasksets": ["aider_polyglot"], "scored": 0, "results": {"aider_polyglot/polyglot_cpp_allergies": {"best_reward": null, "last": "2026-09-26T07:48:48", "scored": 0, "last_run": "20260926T073808-charmbracele-308ef1539f9a-polyglot_cpp_allergies", "last_tests": null, "last_outcome": "error", "last_reward": null, "runs": 1}}, "models": ["bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"], "tasks_tried": 1, "api_style": "openai", "commit": "c9b45348a4f1b7695e70830a4408aa4c31688851"}, "profile": {"evidence": "README.md:9 (\"Your new coding bestie, now available in your favourite terminal\"), internal/agent/templates/coder.md.tpl:1-435 (system prompt emphasizing autonomous coding: read files, edit via find-replace, run tests, commit with git), internal/agent/coordinator.go:875-920 (assembles tools: bash, edit, multiedit, write, view, grep, glob, ls, fetch, download, plus LSP tools for diagnostics/symbols/rename), internal/agent/tools/*.md (tool descriptions for file operations and shell execution), AGENTS.md:4-10 (describes Crush as \"terminal-based AI coding assistant built in Go...connects to LLMs and gives them tools to read, write, and execute code\").", "harness": "charmbracelet-crush", "domains": ["swe"], "source": "claude -p", "cost_usd": 0.3166032, "capabilities": ["edits-files", "runs-shell", "runs-tests", "uses-git", "reads-docs", "long-horizon", "calls-apis"], "seconds": 79, "languages": ["bash", "cpp", "go", "java", "javascript", "nix", "python", "rust"], "use_case": "Writes, edits, tests, and debugs code in a terminal by executing shell commands, reading and modifying files, and using LSP for code intelligence.", "at": "2026-09-26T07:51:28", "not_for": ["Interactive web browsing (can fetch URLs but not navigate pages)", "GUI-based workflows (terminal/CLI only)", "Direct SQL database querying (no native SQL tool)"], "commit": "c9b45348a4f1b7695e70830a4408aa4c31688851"}, "recommendations": {"model": "haiku", "at": "2026-09-26T07:52:21", "profile_source": "claude -p", "harness": "charmbracelet-crush", "recs": [{"language": "python", "score": null, "why": "Start with a straightforward Python implementation task to establish a working baseline after the initial C++ error.", "task": "14", "domain": "swe", "taskset": "evoeval"}, {"taskset": "aider_polyglot", "task": "polyglot_javascript_triangle", "why": "Test JavaScript, a core language, with a geometry classification problem that's fundamentally simpler than the allergies domain.", "language": "javascript", "score": null, "domain": "swe"}, {"domain": "swe", "why": "Assess Rust capability on a complex stateful scoring problem that requires careful algorithmic thinking and comprehensive test validation.", "taskset": "aider_polyglot", "score": null, "language": "rust", "task": "polyglot_rust_bowling"}, {"language": "java", "taskset": "quixbugs", "why": "Evaluate Java support on a focused bug-fixing task with a single-line constraint, testing code analysis and precise modification ability.", "task": "quixbugs-java-flatten", "score": null, "domain": "swe"}, {"language": null, "task": "sphinx-doc__sphinx-8595", "domain": "swe", "score": null, "why": "Test real-world debugging on an established open-source project with git history, test suites, and documentation\u2014exercising the harness's full toolset.", "taskset": "swebench-verified"}], "source": "llm", "cost_usd": 0.046505, "based_on_run": "20260926T073808-charmbracele-308ef1539f9a-polyglot_cpp_allergies"}, "runs": [{"run": "20260926T073808-charmbracele-308ef1539f9a-polyglot_cpp_allergies", "started": "2026-09-26T07:48:48", "finished": "2026-09-26T07:49:59", "status": "done", "kind": "harbor", "harness": "charmbracelet-crush", "task": {"taskset": "aider_polyglot", "name": "polyglot_cpp_allergies"}, "model": "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", "reward": null, "verifier_rc": 1, "tests": null, "calls": 32, "seconds": 68, "input_tokens": 529212, "output_tokens": 7228, "errors": 0, "last_action": "view: /app/allergies.cpp", "outcome": "error", "verifier_says": "no reward written"}]}