{"harness": {"recipes": 1, "summary": "Company Brain is a Cloudflare Workers-based Slack bot that uses Durable Objects (CompanyBrainAgent) to maintain conversational state. It handles Slack events via HTTP webhooks and makes LLM calls through the Vercel AI SDK with support for Anthropic (@ai-sdk/anthropic), OpenAI (@ai-sdk/openai), Google (@ai-sdk/google), and xAI (@ai-sdk/xai) providers, routing through Cloudflare AI Gateway when configured. The application has no CLI entrypoint - src/worker.ts exports a Hono HTTP handler, not a task runner. Model selection happens in src/brain/turn/brain-model.ts via getBrainModel(), which resolves providers based on available API keys and routes through wrapBrainGateway().", "recipe": "supermemoryai-company-brain@0071d6164991ce5dccddbd645bcac631ee477572", "harness": "supermemoryai-company-brain", "first_run": "2026-09-26T04:19:17", "domains": ["other"], "runs": 1, "base_image": "buildpack-deps:jammy", "finished": 1, "last_run": "2026-09-26T04:19:17", "repo": "https://github.com/supermemoryai/company-brain", "last_run_id": "20260926T041003-supermemorya-6304726e1230-polyglot_cpp_allergies", "use_case": "Deploys a Slack bot that remembers company conversations, answers from memory, and acts in connected tools like GitHub and Linear", "tasksets": ["aider_polyglot"], "scored": 0, "results": {"aider_polyglot/polyglot_cpp_allergies": {"best_reward": null, "last": "2026-09-26T04:19:17", "scored": 0, "last_run": "20260926T041003-supermemorya-6304726e1230-polyglot_cpp_allergies", "last_tests": null, "last_outcome": "error", "last_reward": null, "runs": 1}}, "models": ["bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"], "tasks_tried": 1, "api_style": "anthropic", "commit": "0071d6164991ce5dccddbd645bcac631ee477572"}, "profile": {"evidence": "README.md lines 1-56 describe it as \"A teammate in your Slack that truly knows and understands your company\" - a deployed product. src/brain/prompt/system.ts lines 64-239 contain the agent's system prompt showing it's a Slack bot persona. package.json lines 43-44 show it uses Anthropic's \"agents\" SDK (0.17.4) to build this specific bot, not as a harness. docs/README.md lines 1-27 confirm it's an \"org-scoped agent\" application running on Cloudflare Workers.", "harness": "supermemoryai-company-brain", "domains": ["other"], "source": "claude -p", "cost_usd": 0.18385755, "capabilities": [], "seconds": 51, "languages": ["javascript"], "at": "2026-09-26T04:20:14", "use_case": "Deploys a Slack bot that remembers company conversations, answers from memory, and acts in connected tools like GitHub and Linear", "not_for": ["This is not an AI agent harness for testing or benchmarking agents", "This is a complete Slack bot application that IS an agent, not a framework for building/testing agents", "Not designed for general-purpose agent development or evaluation"], "commit": "0071d6164991ce5dccddbd645bcac631ee477572"}, "recommendations": {"model": "haiku", "at": "2026-09-26T04:20:52", "profile_source": "claude -p", "harness": "supermemoryai-company-brain", "recs": [{"score": null, "domain": "swe", "language": "javascript", "task": "polyglot_javascript_triangle", "why": "Your agent is JavaScript-based\u2014test native language capability to recover from the prior C++ error and establish a reliable baseline.", "taskset": "aider_polyglot"}, {"taskset": "humanevalfix", "score": null, "domain": "swe", "language": null, "why": "Bug-fixing directly parallels your GitHub integration purpose; this simple XOR fix tests whether your agent can reason about logic errors.", "task": "python-11"}, {"score": null, "task": "card_games__372", "domain": "data-sql", "why": "SQL query generation from natural language is essential for querying Linear/GitHub APIs; validate basic intent-to-database reasoning.", "taskset": "bird-bench", "language": null}, {"domain": "swe", "why": "Simple Python coding validates core agent capability before harder problems; the prefix-suffix task tests string reasoning similar to parsing intent.", "score": null, "language": "python", "taskset": "evoeval", "task": "14"}, {"taskset": "swebench-verified", "score": null, "task": "sphinx-doc__sphinx-8595", "domain": "swe", "why": "Real production bugs from major projects validate whether your agent handles authentic complexity, where it matters most for real GitHub issue triage.", "language": null}], "source": "llm", "cost_usd": 0.036762500000000004, "based_on_run": "20260926T041003-supermemorya-6304726e1230-polyglot_cpp_allergies"}, "runs": [{"run": "20260926T041003-supermemorya-6304726e1230-polyglot_cpp_allergies", "started": "2026-09-26T04:19:17", "finished": "2026-09-26T04:19:21", "status": "done", "kind": "harbor", "harness": "supermemoryai-company-brain", "task": {"taskset": "aider_polyglot", "name": "polyglot_cpp_allergies"}, "model": "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", "reward": null, "verifier_rc": 1, "tests": null, "calls": 0, "seconds": 0, "input_tokens": 0, "output_tokens": 0, "errors": 0, "last_action": null, "outcome": "error", "verifier_says": "no reward written"}]}