{"harness": {"recipes": 1, "summary": "gptme is a terminal-first AI agent CLI with tool use. The entrypoint is `gptme` (Python console script from `gptme.cli.main:main`). It is invoked with `gptme --non-interactive \"$TASK\"`, which starts a non-interactive session: no confirmation prompts, no tty required. The harness uses the Anthropic provider: `ANTHROPIC_API_KEY=proxy` provides the key, `LLM_PROXY_URL=$PROXY_URL` redirects all Anthropic SDK calls to the proxy (the Anthropic SDK appends `/v1/messages` to this base URL, matching the proxy's Anthropic endpoint), and `GPTME_MODEL=anthropic/claude-sonnet-4-6` selects the model. A summary/title model (claude-haiku-4-5) is also called automatically. The harness is installed via `uv` into `/opt/harness/venv` from the repo source.", "recipe": "gptme-gptme@9e3ab7cf5418ae623d4b01ad7710fc9cfcf43a9c", "compatibility": null, "harness": "gptme-gptme", "first_run": "2026-09-26T07:31:42", "domains": ["swe", "devops-sre"], "runs": 1, "base_image": "buildpack-deps:jammy", "finished": 1, "last_run": "2026-09-26T07:31:42", "repo": "https://github.com/gptme/gptme", "last_run_id": "20260926T072636-gptme-gptme-c876fe156da2-polyglot_cpp_allergies", "use_case": "Runs a coding agent in any terminal to write code, fix tests, edit files, run shell commands, browse the web, and operate autonomously on a schedule", "tasksets": ["aider_polyglot"], "scored": 0, "results": {"aider_polyglot/polyglot_cpp_allergies": {"best_reward": null, "last": "2026-09-26T07:31:42", "scored": 0, "last_run": "20260926T072636-gptme-gptme-c876fe156da2-polyglot_cpp_allergies", "last_tests": null, "last_outcome": "error", "last_reward": null, "runs": 1}}, "models": ["bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"], "tasks_tried": 1, "api_style": "anthropic", "commit": "9e3ab7cf5418ae623d4b01ad7710fc9cfcf43a9c"}, "profile": {"evidence": "README.md lines 58-64 describe it as \"a personal AI agent that runs anywhere a terminal runs... A great coding agent\"; lines 242-266 list all tools (shell, python, save, patch, browser, vision, screenshot, computer, tmux, subagent, gh, etc.); gptme/prompts/templates.py lines 91-338 show the system prompt instructing it as a programming assistant with code execution and file editing; docs/tools.rst lines 1-150 document the full tool interface; docs/features.rst lines 19-23 highlight autonomous agents like Bob with 5000+ merged PRs; AGENTS.md documents git workflows and testing requirements for development work", "harness": "gptme-gptme", "domains": ["swe", "devops-sre"], "source": "claude -p", "cost_usd": 0.29963055, "capabilities": ["edits-files", "runs-shell", "runs-tests", "browses-web", "uses-git", "multi-agent", "long-horizon", "reads-docs", "calls-apis", "gui"], "seconds": 66, "languages": ["bash", "javascript", "python", "rust"], "use_case": "Runs a coding agent in any terminal to write code, fix tests, edit files, run shell commands, browse the web, and operate autonomously on a schedule", "at": "2026-09-26T07:33:14", "not_for": ["Windows native (requires WSL or Docker)", "direct SQL database querying without Python/shell wrapper", "specialized data science workflows (general-purpose with some data analysis capability)"], "commit": "9e3ab7cf5418ae623d4b01ad7710fc9cfcf43a9c"}, "recommendations": {"model": "haiku", "at": "2026-09-26T07:33:45", "profile_source": "claude -p", "harness": "gptme-gptme", "recs": [{"task": "polyglot_javascript_triangle", "why": "JavaScript is a core language; validating it after the C++ error confirms basic multi-language support.", "score": null, "domain": "swe", "taskset": "aider_polyglot", "language": "javascript"}, {"domain": "swe", "task": "polyglot_python_bowling", "why": "Python dominates your use case; bowling scoring is a medium-complexity stateful algorithm that exercises file editing and test-running.", "taskset": "aider_polyglot", "score": null, "language": "python"}, {"score": null, "taskset": "aider_polyglot", "domain": "swe", "why": "Completes coverage of core compiled languages; same problem structure in Rust reveals language-specific handling.", "language": "rust", "task": "polyglot_rust_bowling"}, {"task": "14", "domain": "swe", "why": "New taskset (untried), simple Python baseline to build confidence before harder benchmarks.", "taskset": "evoeval", "score": null, "language": "python"}, {"task": "sphinx-doc__sphinx-8595", "why": "Real-world Python tool bug with long-horizon traits (git, multi-file edits); tests whether your agent can handle production-scale repos.", "score": null, "language": null, "domain": "swe", "taskset": "swebench-verified"}], "source": "llm", "cost_usd": 0.035005, "based_on_run": "20260926T072636-gptme-gptme-c876fe156da2-polyglot_cpp_allergies"}, "runs": [{"run": "20260926T072636-gptme-gptme-c876fe156da2-polyglot_cpp_allergies", "started": "2026-09-26T07:31:42", "finished": "2026-09-26T07:32:03", "status": "done", "kind": "harbor", "harness": "gptme-gptme", "task": {"taskset": "aider_polyglot", "name": "polyglot_cpp_allergies"}, "model": "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", "reward": null, "verifier_rc": 1, "tests": null, "calls": 3, "seconds": 17, "input_tokens": 287, "output_tokens": 2324, "errors": 0, "last_action": null, "outcome": "error", "verifier_says": "no reward written"}]}