From 1dae5a8ea624741b91b858a4d141b9e492668617 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Sun, 12 Jul 2026 22:33:01 +0100 Subject: [PATCH 01/31] feat(chat): add read_file tool so the orchestrator can inspect running work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Orcabot chat could write to terminals (terminal_send_input) and manipulate the canvas, but had no way to READ anything back — so it was a blind launcher and couldn't answer "how's it going?" with real data. Add a read_file tool (workspace-scoped, read-only) that reuses the existing internal-token file API (sandboxFetch → GET /sessions/:id/file), with a max_bytes tail for large/live logs and a no-traversal path guard. This is the linchpin that turns chat from blind launcher into a real orchestrator — it reads progress logs / results (.scb-run.log, runs.jsonl, result.json) instead of guessing. (Chose read_file over terminal_read: the scrollback endpoint requires per-PTY X-MCP-Secret the control plane can't supply, and benchmark progress lives in files anyway.) Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- controlplane/src/chat/handler.ts | 68 ++++++++++++++++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/controlplane/src/chat/handler.ts b/controlplane/src/chat/handler.ts index 332d60f7..0f4a3753 100644 --- a/controlplane/src/chat/handler.ts +++ b/controlplane/src/chat/handler.ts @@ -25,6 +25,7 @@ import * as dashboards from '../dashboards/handler'; import * as secrets from '../secrets/handler'; import * as integrationPolicies from '../integration-policies/handler'; import { SandboxClient } from '../sandbox/client'; +import { sandboxFetch } from '../sandbox/fetch'; import { HELP_DOCS_GROUNDING } from './help-docs'; // System prompt for Orcabot @@ -518,6 +519,22 @@ function friendlyProviderError(raw: string | undefined, providerId: string): str return 'Something went wrong — please try again.'; } +const WORKSPACE_TOOLS: GeminiTool[] = [ + { + name: 'read_file', + description: 'Read a text file from the dashboard sandbox workspace (read-only). Use this to check on running work — progress logs, results, or anything an agent/benchmark wrote (e.g. ".scb-run.log", ".scb_tmux/runs.jsonl", "outputs//result.json"). This is how you answer "how is it going?" with real data instead of guessing.', + inputSchema: { + type: 'object', + properties: { + dashboard_id: { type: 'string', description: 'The dashboard whose sandbox workspace to read from' }, + path: { type: 'string', description: 'Workspace-relative path (e.g. ".scb-run.log"). Absolute "/workspace/..." is also accepted.' }, + max_bytes: { type: 'number', description: 'Return only the LAST N bytes (tail) — useful for large/live logs. Default 8000.' }, + }, + required: ['dashboard_id', 'path'], + }, + }, +]; + function getOrcabotTools(): GeminiTool[] { return [ ...DASHBOARD_TOOLS, @@ -525,6 +542,7 @@ function getOrcabotTools(): GeminiTool[] { ...TERMINAL_TOOLS, ...SECRETS_TOOLS, ...GUIDANCE_TOOLS, + ...WORKSPACE_TOOLS, ...convertMcpToGeminiTools(UI_TOOLS), ]; } @@ -792,6 +810,56 @@ async function executeTool( // Phase 4: Terminal Control Tools // ========================================== + if (toolName === 'read_file') { + const dashboardId = args.dashboard_id as string; + const rawPath = String(args.path || '').trim(); + const maxBytes = typeof args.max_bytes === 'number' && args.max_bytes > 0 + ? Math.min(args.max_bytes, 200_000) + : 8000; + + const access = await env.DB.prepare(` + SELECT role FROM dashboard_members WHERE dashboard_id = ? AND user_id = ? + `).bind(dashboardId, userId).first<{ role: string }>(); + if (!access) { + return { result: { error: 'Access denied. User does not have access to this dashboard.' }, isError: true }; + } + + const sb = await env.DB.prepare(` + SELECT sandbox_session_id, sandbox_machine_id FROM dashboard_sandboxes WHERE dashboard_id = ? + `).bind(dashboardId).first<{ sandbox_session_id: string; sandbox_machine_id: string }>(); + if (!sb) { + return { result: { error: 'No sandbox is running for this dashboard yet. Start a terminal or benchmark first.' }, isError: true }; + } + + // Workspace-scoped, no traversal. + if (!rawPath || rawPath.includes('..')) { + return { result: { error: 'Invalid path.' }, isError: true }; + } + const absPath = rawPath.startsWith('/workspace/') + ? rawPath + : `/workspace/${rawPath.replace(/^\/+/, '')}`; + + try { + const res = await sandboxFetch( + env, + `/sessions/${sb.sandbox_session_id}/file?path=${encodeURIComponent(absPath)}`, + { machineId: sb.sandbox_machine_id || undefined } + ); + if (!res.ok) { + return { result: { error: `Could not read ${rawPath} (status ${res.status}). It may not exist yet.` }, isError: true }; + } + let content = await res.text(); + let truncated = false; + if (content.length > maxBytes) { + content = content.slice(-maxBytes); + truncated = true; + } + return { result: { path: absPath, truncated, bytes: content.length, content }, isError: false }; + } catch (error) { + return { result: { error: `Read failed: ${error instanceof Error ? error.message : String(error)}` }, isError: true }; + } + } + if (toolName === 'terminal_send_input') { const dashboardId = args.dashboard_id as string; const terminalItemId = args.terminal_item_id as string; From cb860ebede87f8061b25c0cf36f78cdee02d56a8 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Sun, 12 Jul 2026 22:38:18 +0100 Subject: [PATCH 02/31] feat(benchmarks): live per-agent visualization + chat-as-orchestrator template MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - scb-visualize: watches the host-tmux executor's .scb_tmux/runs.jsonl and, for each agent-under-test that starts, creates a read-only Orcabot terminal tailing its per-run logfile with a note above naming the problem — via the local sandbox MCP server (create_note/create_terminal), authed with ORCABOT_MCP_SECRET from its own PTY (dashboard auto-resolved server-side). Read-only by construction: a viewer only tails a file, so it can't steer/corrupt the run. `run` mode wraps the benchmark; `watch` mode observes an existing run. - SlopCodeBench template: drop the auto-opened Claude Code terminal — Orcabot chat is now the orchestrator. Keep the note + results blocks; rewrite the setup guide to drive the run via chat's tools (create_terminal / read_file / secrets_create) and launch through scb-visualize for the live per-agent view. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- benchmarks/slopcodebench/bin/scb-visualize | 170 +++++++++++++++++++++ benchmarks/slopcodebench/template.json | 36 ++--- 2 files changed, 179 insertions(+), 27 deletions(-) create mode 100755 benchmarks/slopcodebench/bin/scb-visualize diff --git a/benchmarks/slopcodebench/bin/scb-visualize b/benchmarks/slopcodebench/bin/scb-visualize new file mode 100755 index 00000000..eadcf1cc --- /dev/null +++ b/benchmarks/slopcodebench/bin/scb-visualize @@ -0,0 +1,170 @@ +#!/usr/bin/env python3 +"""scb-visualize — turn a slop-code-bench run into a live Orcabot canvas. + +Watches the host-tmux executor's run manifest and, for each agent-under-test that +starts, surfaces a **read-only** Orcabot terminal tailing that run's logfile, with +a note above it naming the problem. Orcabot becomes a live, per-problem view of +the benchmark as it runs. + +Why this is safe (read-only by construction): each viewer only runs +`tail -F ` on the run's output. It has no path to the agent CLI, so +nothing typed into a viewer can steer or corrupt the run — the exact property the +viewer-smoke test asserts. (This is why we tail the per-run logfile rather than +attach a tmux control socket, which would grant cross-session inject.) + +Detection surface (from the `feat/host-tmux-executor` fork): + /.scb_tmux/runs.jsonl — one JSON line per run: + {"target","session","window":"","logfile":"...","created":...} + +Runs inside an Orcabot PTY: it creates canvas components via the local sandbox +MCP server using ORCABOT_MCP_SECRET; the dashboard is resolved server-side from +the session, so no dashboard id is needed. + +Usage: + scb-visualize watch [--workdir DIR] # watch an existing run's manifest + scb-visualize run [--workdir DIR] -- CMD… # run the benchmark AND visualize +""" +import json +import os +import shlex +import subprocess +import sys +import threading +import time +import urllib.error +import urllib.request + +MCP_BASE = os.environ.get("ORCABOT_MCP_BASE", "http://127.0.0.1:8081") +SID = os.environ.get("ORCABOT_SESSION_ID", "").strip() +PTY = os.environ.get("ORCABOT_PTY_ID", "").strip() +SECRET = os.environ.get("ORCABOT_MCP_SECRET", "").strip() + +# Canvas layout: a grid of columns; each run = a note stacked over its terminal. +COL_W, NOTE_H, TERM_H, GAP, COLS = 380, 96, 300, 24, 3 +BASE_X, BASE_Y, COL_GAP, ROW_GAP = 40, 40, 40, 48 + + +def _log(msg): + sys.stderr.write(f"[scb-visualize] {msg}\n") + sys.stderr.flush() + + +def mcp_ready(): + return bool(SID and PTY and SECRET) + + +def call_tool(name, arguments): + """Invoke an Orcabot MCP UI tool via the local sandbox MCP server.""" + url = f"{MCP_BASE}/sessions/{SID}/mcp/tools/call?pty_id={PTY}" + body = json.dumps({"name": name, "arguments": arguments}).encode() + req = urllib.request.Request( + url, data=body, method="POST", + headers={"Content-Type": "application/json", "X-MCP-Secret": SECRET}, + ) + try: + with urllib.request.urlopen(req, timeout=15) as resp: + return json.load(resp) + except urllib.error.HTTPError as e: + _log(f"MCP {name} failed: {e.code} {e.read()[:200]!r}") + except Exception as e: # noqa: BLE001 — best-effort visualization + _log(f"MCP {name} error: {e}") + return None + + +def slot(idx): + col, row = idx % COLS, idx // COLS + x = BASE_X + col * (COL_W + COL_GAP) + y = BASE_Y + row * (NOTE_H + TERM_H + GAP + ROW_GAP) + return x, y + + +def surface_run(idx, problem, logfile): + x, y = slot(idx) + call_tool("create_note", { + "content": f"### Solving: {problem}\n\nAgent-under-test — live, read-only.", + "position": {"x": x, "y": y}, + "size": {"width": COL_W, "height": NOTE_H}, + "color": "blue", + }) + call_tool("create_terminal", { + "name": f"agent: {problem}", + "boot_command": f"tail -n +1 -F {shlex.quote(logfile)}", + "agentic": False, + "position": {"x": x, "y": y + NOTE_H + GAP}, + "size": {"width": COL_W, "height": TERM_H}, + }) + _log(f"surfaced run '{problem}' -> {logfile}") + + +def watch(workdir, stop=None): + manifest = os.path.join(workdir, ".scb_tmux", "runs.jsonl") + _log(f"watching {manifest}") + if not mcp_ready(): + _log("not in an Orcabot PTY (no ORCABOT_MCP_SECRET) — visualization disabled; still tailing manifest for logs") + seen, idx = set(), 0 + while stop is None or not stop.is_set(): + try: + with open(manifest) as f: + lines = f.readlines() + except FileNotFoundError: + time.sleep(1.0) + continue + for line in lines: + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + logfile = rec.get("logfile") + problem = rec.get("window") or rec.get("target") or "run" + key = logfile or line + if key in seen: + continue + seen.add(key) + if not logfile: + continue + if not os.path.isabs(logfile): + logfile = os.path.join(workdir, logfile) + if mcp_ready(): + surface_run(idx, problem, logfile) + else: + _log(f"(disabled) would surface: {problem} -> {logfile}") + idx += 1 + time.sleep(1.0) + + +def main(): + argv = sys.argv[1:] + if not argv or argv[0] in ("-h", "--help"): + print(__doc__) + sys.exit(0 if argv else 2) + + mode = argv[0] + rest = argv[1:] + workdir = os.getcwd() + if "--workdir" in rest: + i = rest.index("--workdir") + workdir = rest[i + 1] + rest = rest[:i] + rest[i + 2:] + + if mode == "watch": + watch(workdir) + elif mode == "run": + cmd = rest[1:] if rest[:1] == ["--"] else rest + if not cmd: + _log("usage: scb-visualize run [--workdir DIR] -- ") + sys.exit(2) + t = threading.Thread(target=watch, args=(workdir,), daemon=True) + t.start() + rc = subprocess.call(cmd, cwd=workdir) + time.sleep(2) # let the watcher surface the final run + sys.exit(rc) + else: + print(__doc__) + sys.exit(2) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/slopcodebench/template.json b/benchmarks/slopcodebench/template.json index c581a0db..94a10d1d 100644 --- a/benchmarks/slopcodebench/template.json +++ b/benchmarks/slopcodebench/template.json @@ -1,33 +1,20 @@ { - "_comment": "Importable Orcabot dashboard template (DashboardTemplateWithData). Inserted into D1 via benchmarks/slopcodebench/bin/insert-template or the templates API. The orchestrator's bootCommand appends a briefing to CLAUDE.md so Claude Code boots knowing the validated slop-code commands. Manual one-time VM setup (fork + venv) still required.", + "_comment": "Importable Orcabot dashboard template. Orcabot chat is the orchestrator (no auto-opened agent terminal); it drives runs via its tools (create_terminal/read_file/secrets_create) and surfaces each agent-under-test as a live read-only component via bin/scb-visualize. One-time VM setup (fork + venv) still required. Seed with bin/seed-benchmark-templates.", "name": "SlopCodeBench Runner", - "description": "Run slop-code-bench arms with a Claude Code orchestrator you drive in English. One-time VM setup required (fork + venv).", + "description": "Run slop-code-bench arms driven by Orcabot chat in plain English; each agent-under-test opens as a live read-only terminal on the canvas. One-time VM setup required (fork + venv).", "category": "coding", "items": [ - { - "placeholderId": "orchestrator", - "type": "terminal", - "content": "{\"name\": \"Claude Code\", \"agentic\": true, \"bootCommand\": \"B='CjwhLS0gU0NCLU9SQ0ggLS0+CiMgU2xvcENvZGVCZW5jaCBPcmNoZXN0cmF0b3IKCllvdSBvcGVyYXRlIHNsb3AtY29kZS1iZW5jaCBpbiB0aGlzIE9yY2Fib3QgVk0uIFRoZSB1c2VyIGRpcmVjdHMgeW91IGluIEVuZ2xpc2g7CnlvdSB0cmFuc2xhdGUgdGhhdCBpbnRvIGBzbG9wLWNvZGVgIGNvbW1hbmRzIGFuZCByZXBvcnQgd2hhdCB5b3UgcmFuICsgb2JzZXJ2ZWQuCllvdSBkbyBOT1Qgc29sdmUgdGhlIGJlbmNobWFyayBwcm9ibGVtcyB5b3Vyc2VsZiAtLSB0aGUgYWdlbnQtdW5kZXItdGVzdCBkb2VzLgoKIyMgRW52aXJvbm1lbnQgKG1hbnVhbCBzZXR1cCAtLSBhc3N1bWUgZG9uZSBpZiAvd29ya3NwYWNlL3NjYi12ZW52IGV4aXN0cykKLSBGb3JrIGNoZWNrb3V0ID0gdGhpcyBkaXJlY3RvcnkgKGAvd29ya3NwYWNlL3Nsb3AtY29kZS1iZW5jaGApLgotIFJ1biB0aGUgQ0xJIHZpYSB0aGUgdmVudiBkaXJlY3RseTogYC93b3Jrc3BhY2Uvc2NiLXZlbnYvYmluL3Nsb3AtY29kZWAgKG5vIGB1dmApLgotIFRoZSBwcm9ibGVtIGNhdGFsb2cgYXV0by1kb3dubG9hZHMgdG8gYH4vLmNhY2hlL3NjYmVuY2gvcHJvYmxlbXNgIG9uIGZpcnN0IHJ1bi4KCiMjIFNlY3JldHMgLS0gTkVWRVIgZWNobywgY2F0LCBvciBwcmludCBhIGtleQpQcm92aWRlciBrZXlzIGFycml2ZSBmcm9tIHRoZSBPcmNhYm90IHNlY3JldHMgYnJva2VyIGFzIGVudiB2YXJzIChlLmcuCmBPUEVOUk9VVEVSX0FQSV9LRVlgIGhvbGRzIGEgYFtCUk9LRVJFRF1gIHBsYWNlaG9sZGVyOyB0aGUgYnJva2VyIGluamVjdHMgdGhlCnJlYWwga2V5IHNlcnZlci1zaWRlKS4gSWYgYSBuZWVkZWQga2V5IGlzIG1pc3NpbmcsIGFzayB0aGUgdXNlciB0byBhZGQgaXQgaW4gdGhlCnNlY3JldHMgcGFuZWwgLS0gbmV2ZXIgcmVxdWVzdCB0aGUgdmFsdWUgaW4gY2hhdC4KCiMjIEJyb2tlciBiYXNlIFVSTCAtLSB1c2UgMTI3LjAuMC4xLCBOT1QgbG9jYWxob3N0CkJlZm9yZSBlYWNoIHJ1biwgcG9pbnQgcHJvdmlkZXJzLnlhbWwncyBicm9rZXIgYGFwaV9iYXNlYCBhdCAxMjcuMC4wLjEuIG9wZW5jb2RlCihOb2RlL3VuZGljaSkgcmVzb2x2ZXMgYGxvY2FsaG9zdGAgdG8gSVB2NiBgOjoxYCwgd2hpY2ggdGhlIElQdjQtb25seSBicm9rZXIKbmV2ZXIgYW5zd2VycyAtPiBhIG11bHRpLW1pbnV0ZSAiaGFuZyIuIFBhdGNoIGl0OgpgYGAKY3AgY29uZmlncy9wcm92aWRlcnMueWFtbC5vcmlnIGNvbmZpZ3MvcHJvdmlkZXJzLnlhbWwgMj4vZGV2L251bGwKcHl0aG9uMyAtIDw8J1BZJwppbXBvcnQgb3M7IHA9ImNvbmZpZ3MvcHJvdmlkZXJzLnlhbWwiOyBzPW9wZW4ocCkucmVhZCgpCnVybD1vcy5lbnZpcm9uWyJPUEVOUk9VVEVSX0JBU0VfVVJMIl0ucmVwbGFjZSgibG9jYWxob3N0IiwiMTI3LjAuMC4xIikKb3BlbihwLCJ3Iikud3JpdGUocy5yZXBsYWNlKCJhcGlfYmFzZTogaHR0cHM6Ly9vcGVucm91dGVyLmFpL2FwaS92MSIsImFwaV9iYXNlOiAiK3VybCkpClBZCmBgYAoKIyMgUnVuIG9uZSBhcm0gKHZhbGlkYXRlZDogb3BlbmNvZGUgKyBLaW1pIEsyLjYgdmlhIE9wZW5Sb3V0ZXIpCmBgYAovd29ya3NwYWNlL3NjYi12ZW52L2Jpbi9zbG9wLWNvZGUgcnVuIC0tYWdlbnQgb3BlbmNvZGUgLS1tb2RlbCBvcGVucm91dGVyL2tpbWktazIuNiBcCiAgLS1lbnZpcm9ubWVudCBjb25maWdzL2Vudmlyb25tZW50cy9sb2NhbC10bXV4LXB5LnlhbWwgXAogIC0tcHJvbXB0IGNvbmZpZ3MvcHJvbXB0cy9qdXN0LXNvbHZlLmppbmphIFwKICAtLXByb2JsZW0gZmlsZV9iYWNrdXAgLS1uby1ldmFsdWF0ZSB0aGlua2luZz1sb3cKYGBgCkxhdW5jaCBkZXRhY2hlZCB0byBhIGxvZ2ZpbGUgc28gdGhlIHVzZXIgY2FuIHdhdGNoLCB0aGVuIHRhaWwgaXQ6CmBgYApzZXRzaWQgYmFzaCAtYyAnPHRoZSBydW4gY29tbWFuZCBhYm92ZT4nID4gL3dvcmtzcGFjZS8uc2NiLXJ1bi5sb2cgMj4mMSA8IC9kZXYvbnVsbCAmCnRhaWwgLWYgL3dvcmtzcGFjZS8uc2NiLXJ1bi5sb2cKYGBgClN1Y2Nlc3MgPSB0aGUgcnVuIGFkdmFuY2VzIHBhc3QgYGNoZWNrcG9pbnRfMWAgd2l0aCBgY29zdD4wYCBhbmQgYHN0ZXBzPjBgLgoKIyMgU2NvcmUKRHJvcCBgLS1uby1ldmFsdWF0ZWAgdG8gc2NvcmUgaW5saW5lLCBvciBydW4KYC93b3Jrc3BhY2Uvc2NiLXZlbnYvYmluL3Nsb3AtY29kZSBldmFsIG91dHB1dHMvPHJ1bi1kaXI+L2AgYWZ0ZXIuCgojIyBXYXRjaCAocmVhZC1vbmx5IG9ubHkpClBlci1ydW4gbG9nZmlsZXMgbGl2ZSB1bmRlciB0aGUgcnVuJ3Mgb3V0cHV0IGRpciAvIGAuc2NiX3RtdXgvcnVucy5qc29ubGAuIFRhaWwKdGhlbS4gRG8gTk9UIGJyaWRnZSBhIGNyb3NzLXNlc3Npb24gdG11eCBjb250cm9sIHNvY2tldCAtLSB0aGF0IGdyYW50cwpyZWFkK2luamVjdCBhY3Jvc3Mgc2Vzc2lvbnMgYW5kIGJ5cGFzc2VzIG91dHB1dCByZWRhY3Rpb24uCg=='; grep -q SCB-ORCH CLAUDE.md 2>/dev/null || echo \\\"$B\\\" | base64 -d >> CLAUDE.md; exec claude\", \"workingDir\": \"slop-code-bench\", \"skipApprovals\": true}", - "position": { - "x": 0, - "y": 0 - }, - "size": { - "width": 640, - "height": 520 - } - }, { "placeholderId": "runbook", "type": "note", - "content": "# SlopCodeBench Runner\n\nRun slop-code-bench arms with a **Claude Code orchestrator** you drive in plain English.\n\n## One-time setup (per VM)\n1. Fork at `/workspace/slop-code-bench` with venv `/workspace/scb-venv` (uv sync `--python 3.12`). On desktop this is the host-shared workspace.\n2. Add your provider key in the **secrets panel**, broker-protected (e.g. `OPENROUTER_API_KEY`).\n\n## Drive the Claude Code terminal (it already knows these from CLAUDE.md)\n- \"Run `file_backup` with opencode on kimi-k2.6\"\n- \"Show me the run log\" / \"is it past checkpoint_1 with cost>0?\"\n- \"Score the finished run\"\n\nValidated arm: opencode + Kimi K2.6 via OpenRouter. Broker URL must be `127.0.0.1`, not `localhost`. Full docs: `benchmarks/slopcodebench/RESUME-kimi-opencode.md`.", + "content": "# SlopCodeBench Runner\n\nRun slop-code-bench arms driven by **Orcabot chat** — just tell it what to run in\nplain English. Each agent-under-test opens as a **live, read-only terminal** on\nthe canvas, labelled with the problem it's solving.\n\n## One-time setup (per VM)\n1. Fork at `/workspace/slop-code-bench` with venv `/workspace/scb-venv`.\n2. Add your provider key in the **secrets panel**, broker-protected (e.g.\n `OPENROUTER_API_KEY`).\n\n## Drive it from chat\n- \"Run `file_backup` with opencode on kimi-k2.6\"\n- \"How's the run going?\" (chat reads the progress log for you)\n- \"Score the finished run\"\n\nValidated arm: opencode + Kimi K2.6 via OpenRouter. Broker URL must be\n`127.0.0.1`, not `localhost`.", "position": { - "x": 680, + "x": 0, "y": 0 }, "size": { - "width": 400, - "height": 320 + "width": 420, + "height": 340 } }, { @@ -35,7 +22,7 @@ "type": "browser", "content": "{\"name\": \"SlopCodeBench Results\", \"url\": \"http://localhost:8050\"}", "position": { - "x": 680, + "x": 0, "y": 360 }, "size": { @@ -44,16 +31,11 @@ } } ], - "edges": [ - { - "sourcePlaceholderId": "orchestrator", - "targetPlaceholderId": "results" - } - ], + "edges": [], "viewport": { "x": 0, "y": 0, "zoom": 1 }, - "setupGuide": "# SlopCodeBench setup walkthrough\n\nGoal: get this dashboard ready to run a slop-code-bench arm, then run one.\nBe brief; do each step with your tools and confirm before the next.\n\n1. **Choose the harness source.** Ask the user which to use:\n - (a) **Our fork** `https://github.com/robdmac/slop-code-bench` (branch\n `feat/host-tmux-executor`) \u2014 recommended; it has the no-Docker watchable\n executor + the broker-compatible routing this needs. Default to this.\n - (b) a **custom fork/branch URL** they paste.\n - (c) **already present** at `/workspace/slop-code-bench` \u2014 skip the clone.\n\n2. **Clone it** (unless (c)) via terminal_input in a shell terminal:\n `git clone --branch /workspace/slop-code-bench`\n (needs network; if a domain is held by the egress proxy, tell the user to approve it.)\n\n3. **Create the venv** (one-time, slow): in `/workspace/slop-code-bench` run\n `UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12`\n (Python 3.12 \u2014 3.14 has no wheels and the VM has no compiler.)\n\n4. **Provider key.** Ensure an `OPENROUTER_API_KEY` broker secret exists for this\n dashboard. If not, ask the user to add it (secrets_create, broker-protected).\n Never display or echo the value.\n\n5. **Point the broker URL at 127.0.0.1** (not localhost \u2014 Node/opencode can't\n reach the IPv4-only broker via `localhost`). In the checkout:\n `cp configs/providers.yaml.orig configs/providers.yaml` then replace the\n openrouter `api_base` with `$OPENROUTER_BASE_URL` after swapping\n `localhost`->`127.0.0.1`.\n\n6. **Run one arm** and watch it: launch detached to a logfile, then tail it:\n `/workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low`\n Success = advances past `checkpoint_1` with `cost>0` and `steps>0`. Report when confirmed.\n" + "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Ensure the harness is set up (skip if /workspace/scb-venv already exists)\nIf not set up, create a shell terminal and run:\n git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench\n cd /workspace/slop-code-bench && UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12\n(`uv sync` is slow and one-time — tell the user to expect a wait. If a domain is\nheld by the egress proxy, ask them to Allow it.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low\")\nIf `scb-visualize` isn't on PATH yet, run the slop-code command directly and tail\n/workspace/.scb-run.log instead (no per-agent components).\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." } \ No newline at end of file From 18ab8acab7794e8adb547d6d68a763a4f89b6f4b Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Sun, 12 Jul 2026 22:51:04 +0100 Subject: [PATCH 03/31] feat(benchmarks): --sql mode for seed-benchmark-templates (local wrangler-dev) The HTTP seeder targets the desktop d1-shim (:9001), which doesn't reach a local wrangler-dev miniflare D1. Add `--sql` to emit the DELETE+INSERT (status 'approved', with setup_guide) for `wrangler d1 execute --local --file`. Author is AUTHOR_ID (local D1 enforces the author_id FK, unlike D1 remote, so it must be a real user id). Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- benchmarks/bin/seed-benchmark-templates | 40 ++++++++++++++++++++++++- 1 file changed, 39 insertions(+), 1 deletion(-) diff --git a/benchmarks/bin/seed-benchmark-templates b/benchmarks/bin/seed-benchmark-templates index 0228d3e1..43bd16d2 100755 --- a/benchmarks/bin/seed-benchmark-templates +++ b/benchmarks/bin/seed-benchmark-templates @@ -76,8 +76,46 @@ def upsert(tpl): return tid +def _sql_lit(v): + if v is None: + return "NULL" + if isinstance(v, int): + return str(v) + return "'" + str(v).replace("'", "''") + "'" + + +def emit_sql(tpl): + """Emit DELETE+INSERT SQL for a template (for `wrangler d1 execute --file`).""" + tid = str(uuid.uuid4()) + now = datetime.datetime.utcnow().isoformat() + cols = ("id,name,description,category,author_id,author_name,items_json,edges_json," + "viewport_json,setup_guide,item_count,is_featured,use_count,status,created_at,updated_at") + vals = [ + tid, tpl["name"], tpl["description"], tpl.get("category", "coding"), + AUTHOR_ID, AUTHOR_NAME, + json.dumps(tpl["items"]), json.dumps(tpl["edges"]), + json.dumps(tpl.get("viewport")) if tpl.get("viewport") is not None else None, + tpl.get("setupGuide"), len(tpl["items"]), 1, 0, "approved", now, now, + ] + literals = ", ".join(_sql_lit(v) for v in vals) + return (f"DELETE FROM dashboard_templates WHERE name = {_sql_lit(tpl['name'])};\n" + f"INSERT INTO dashboard_templates ({cols}) VALUES ({literals});") + + def main(): - paths = discover(sys.argv[1:]) + args = sys.argv[1:] + sql_mode = "--sql" in args + args = [a for a in args if a != "--sql"] + paths = discover(args) + + # --sql: print SQL to stdout for a direct D1 apply (e.g. local wrangler-dev): + # benchmarks/bin/seed-benchmark-templates --sql slopcodebench > /tmp/seed.sql + # (cd controlplane && npx wrangler d1 execute orcabot-db --local --file /tmp/seed.sql) + if sql_mode: + for path in paths: + print(emit_sql(json.loads(path.read_text()))) + return + # Ensure the setup_guide column exists (idempotent) so older DBs accept the insert. try: q("ALTER TABLE dashboard_templates ADD COLUMN setup_guide TEXT", []) From 30564454c1b71ddc51ea5e04b317d20d22e00b92 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Sun, 12 Jul 2026 23:20:48 +0100 Subject: [PATCH 04/31] fix(benchmarks): idempotent setup + robust runner in SlopCodeBench guide MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two failures hit during a live localhost run: - git clone wedges on a leftover /workspace/slop-code-bench from a prior attempt ("destination path already exists"). Make the clone idempotent (reuse a valid checkout, else rm -rf + clone) and guard the venv build too. - the runner PTY died instantly on a bad boot command (dead PTY → "failed to connect"), and bare `scb-visualize` isn't on PATH. Use bin/scb-visualize and append `; echo "[runner exited $?]"; exec bash` so failures stay visible. Also have the orchestrator verify setup via read_file before launching instead of declaring "done" blind. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- benchmarks/slopcodebench/template.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmarks/slopcodebench/template.json b/benchmarks/slopcodebench/template.json index 94a10d1d..055d6974 100644 --- a/benchmarks/slopcodebench/template.json +++ b/benchmarks/slopcodebench/template.json @@ -37,5 +37,5 @@ "y": 0, "zoom": 1 }, - "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Ensure the harness is set up (skip if /workspace/scb-venv already exists)\nIf not set up, create a shell terminal and run:\n git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench\n cd /workspace/slop-code-bench && UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12\n(`uv sync` is slow and one-time — tell the user to expect a wait. If a domain is\nheld by the egress proxy, ask them to Allow it.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low\")\nIf `scb-visualize` isn't on PATH yet, run the slop-code command directly and tail\n/workspace/.scb-run.log instead (no per-agent components).\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." + "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Ensure the harness is set up — and VERIFY it (don't declare done blind)\nBefore launching, confirm setup actually finished with read_file:\n- read_file(\"slop-code-bench/bin/scb-visualize\") — clone succeeded.\n- read_file(\"scb-venv/pyvenv.cfg\") — the venv exists (uv sync finished).\nIf either is missing/errors, setup is not done — wait or re-run it; do NOT start the\nbenchmark yet.\n\n### Set up (idempotent — safe to re-run; won't wedge on leftovers)\nCreate a shell terminal and run this exactly:\n if [ -e /workspace/slop-code-bench/bin/scb-visualize ]; then echo \"checkout present\"; else rm -rf /workspace/slop-code-bench && git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench; fi\n cd /workspace/slop-code-bench && { [ -e /workspace/scb-venv/bin/slop-code ] || UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12; }\n(`uv sync` is slow and one-time — tell the user to expect a wait. If github.com is\nheld by the egress proxy, ask them to Allow it.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && bin/scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low ; echo \\\"[runner exited $?]\\\" ; exec bash\")\nscb-visualize lives at bin/scb-visualize in the checkout (not on PATH). Always\nappend `; echo \"[runner exited $?]\" ; exec bash` so a boot failure stays visible\ninstead of killing the PTY (a dead PTY shows as \"failed to connect\").\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." } \ No newline at end of file From 997cba4a09b95cbccb39fd9151d50cada29b32b1 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Sun, 12 Jul 2026 23:27:48 +0100 Subject: [PATCH 05/31] fix(benchmarks): run setup in a persistent shell (terminal_send_input), not a self-terminating boot_command A boot_command terminal dies the moment its command finishes; when setup is already done the idempotent setup is a <1s no-op, so the setup terminal vanished as 'pty not found'/'failed to connect'. Have the orchestrator create a plain shell and send setup via terminal_send_input so it stays alive. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- benchmarks/slopcodebench/template.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmarks/slopcodebench/template.json b/benchmarks/slopcodebench/template.json index 055d6974..0f1cd14e 100644 --- a/benchmarks/slopcodebench/template.json +++ b/benchmarks/slopcodebench/template.json @@ -37,5 +37,5 @@ "y": 0, "zoom": 1 }, - "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Ensure the harness is set up — and VERIFY it (don't declare done blind)\nBefore launching, confirm setup actually finished with read_file:\n- read_file(\"slop-code-bench/bin/scb-visualize\") — clone succeeded.\n- read_file(\"scb-venv/pyvenv.cfg\") — the venv exists (uv sync finished).\nIf either is missing/errors, setup is not done — wait or re-run it; do NOT start the\nbenchmark yet.\n\n### Set up (idempotent — safe to re-run; won't wedge on leftovers)\nCreate a shell terminal and run this exactly:\n if [ -e /workspace/slop-code-bench/bin/scb-visualize ]; then echo \"checkout present\"; else rm -rf /workspace/slop-code-bench && git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench; fi\n cd /workspace/slop-code-bench && { [ -e /workspace/scb-venv/bin/slop-code ] || UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12; }\n(`uv sync` is slow and one-time — tell the user to expect a wait. If github.com is\nheld by the egress proxy, ask them to Allow it.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && bin/scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low ; echo \\\"[runner exited $?]\\\" ; exec bash\")\nscb-visualize lives at bin/scb-visualize in the checkout (not on PATH). Always\nappend `; echo \"[runner exited $?]\" ; exec bash` so a boot failure stays visible\ninstead of killing the PTY (a dead PTY shows as \"failed to connect\").\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." + "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Ensure the harness is set up — and VERIFY it (don't declare done blind)\nBefore launching, confirm setup actually finished with read_file:\n- read_file(\"slop-code-bench/bin/scb-visualize\") — clone succeeded.\n- read_file(\"scb-venv/pyvenv.cfg\") — the venv exists (uv sync finished).\nIf either is missing/errors, setup is not done — wait or re-run it; do NOT start the\nbenchmark yet.\n\n### Set up (idempotent — safe to re-run; won't wedge on leftovers)\nCreate a PLAIN shell terminal (create_terminal with NO boot_command). A terminal\nWITH a boot_command dies the instant the command finishes — and when setup is\nalready done these are near-instant no-ops, so it would vanish as \"failed to\nconnect\". Then send the setup as input (terminal_send_input) so the shell stays:\n if [ -e /workspace/slop-code-bench/bin/scb-visualize ]; then echo \"checkout present\"; else rm -rf /workspace/slop-code-bench && git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench; fi\n cd /workspace/slop-code-bench && { [ -e /workspace/scb-venv/bin/slop-code ] || UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12; }\n(`uv sync` is slow and one-time — tell the user to expect a wait. If github.com is\nheld by the egress proxy, ask them to Allow it.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && bin/scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low ; echo \\\"[runner exited $?]\\\" ; exec bash\")\nscb-visualize lives at bin/scb-visualize in the checkout (not on PATH). Always\nappend `; echo \"[runner exited $?]\" ; exec bash` so a boot failure stays visible\ninstead of killing the PTY (a dead PTY shows as \"failed to connect\").\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." } \ No newline at end of file From cbf922d6725ea137af618d7e804f6789d3b7d317 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Sun, 12 Jul 2026 23:32:36 +0100 Subject: [PATCH 06/31] feat(sandbox): bake uv into the image for benchmark venvs The sandbox image had python3/pip/venv but not uv, so slop-code-bench's `uv sync` failed with "uv: command not found" on a freshly-built docker sandbox (desktop worked only via a host-mounted/manually-set-up workspace). Copy the official uv binary into /usr/local/bin so `uv sync` works out of the box after a rebuild (localhost docker + desktop VM, whose rootfs derives from this image). Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- sandbox/docker/Dockerfile | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/sandbox/docker/Dockerfile b/sandbox/docker/Dockerfile index 9059e5bc..80836203 100644 --- a/sandbox/docker/Dockerfile +++ b/sandbox/docker/Dockerfile @@ -230,6 +230,13 @@ COPY --from=builder /build/orcabot-server /usr/local/bin/orcabot-server COPY --from=builder /build/browser-ctl /usr/local/bin/browser-ctl COPY --from=builder /build/mcp-bridge /usr/local/bin/mcp-bridge +# uv (fast Python package manager) — baked in so benchmark harnesses +# (e.g. slop-code-bench `uv sync`) can build a venv without a network install. +# NOTE: :latest is unpinned — pin to a specific uv version once confirmed, per the +# repo's unpinned-dependency guidance (an image rebuild can otherwise inherit an +# upstream uv change). +COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /usr/local/bin/ + # Force the CURRENT chromium (placed late so this cache-bust only re-runs this cheap # layer, not the heavy npm/agent installs above). The base apt layer is Docker-cached # and pins whatever chromium was current when it was first built — and 150.0.7871.46 From ea937301b2223238ec9cee027a711be5a25fb157 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Mon, 13 Jul 2026 00:01:53 +0100 Subject: [PATCH 07/31] =?UTF-8?q?feat(chat):=20add=20run=5Fcommand=20?= =?UTF-8?q?=E2=80=94=20blocking=20exec=20so=20the=20orchestrator=20can=20s?= =?UTF-8?q?equence=20setup?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit terminal_send_input is fire-and-forget, so chat had no way to know when a command finished — setup (clone, uv sync) would run while chat sat idle, with no callback to continue. run_command runs a shell command in the sandbox workspace and BLOCKS until it exits, returning {exit_code, output}. Chat's agentic loop then sequences automatically: run_command(setup) -> see exit 0 -> read_file verify -> launch. Implemented control-plane-only (no VM rebuild): wrap the command to write stdout/stderr to a .out file and the exit code to a .exit marker (base64'd to sidestep quoting), launch it as a headless PTY (createPty), then poll the existing internal-token file API for the .exit marker until it appears or timeout (default 120s, max 300s). Cleans up marker files + the PTY. Long-running processes should still use create_terminal + read_file, not this. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- controlplane/src/chat/handler.ts | 106 +++++++++++++++++++++++++++++++ 1 file changed, 106 insertions(+) diff --git a/controlplane/src/chat/handler.ts b/controlplane/src/chat/handler.ts index 0f4a3753..2329c49a 100644 --- a/controlplane/src/chat/handler.ts +++ b/controlplane/src/chat/handler.ts @@ -533,6 +533,19 @@ const WORKSPACE_TOOLS: GeminiTool[] = [ required: ['dashboard_id', 'path'], }, }, + { + name: 'run_command', + description: 'Run a shell command in the dashboard sandbox workspace and WAIT for it to finish, returning its exit code and output. Use this for finite setup/operational steps where you need the result before continuing (git clone, uv sync, installs, checks) — unlike terminal_send_input (fire-and-forget), this blocks until the command exits, so you can sequence dependent steps automatically. Do NOT use it to start long-running processes (a benchmark run, a server) — those will time out; launch those with create_terminal and check progress with read_file.', + inputSchema: { + type: 'object', + properties: { + dashboard_id: { type: 'string', description: 'The dashboard whose sandbox to run in' }, + command: { type: 'string', description: 'The shell command to run (bash). Starts in /workspace; cd as needed.' }, + timeout_s: { type: 'number', description: 'Max seconds to wait (default 120, max 300). Exceeding this returns timed_out=true.' }, + }, + required: ['dashboard_id', 'command'], + }, + }, ]; function getOrcabotTools(): GeminiTool[] { @@ -860,6 +873,99 @@ async function executeTool( } } + if (toolName === 'run_command') { + const dashboardId = args.dashboard_id as string; + const command = String(args.command || ''); + const timeoutS = Math.min( + typeof args.timeout_s === 'number' && args.timeout_s > 0 ? args.timeout_s : 120, + 300 + ); + if (!command.trim()) { + return { result: { error: 'command is required' }, isError: true }; + } + + const access = await env.DB.prepare(` + SELECT role FROM dashboard_members WHERE dashboard_id = ? AND user_id = ? + `).bind(dashboardId, userId).first<{ role: string }>(); + if (!access) { + return { result: { error: 'Access denied. User does not have access to this dashboard.' }, isError: true }; + } + + const sb = await env.DB.prepare(` + SELECT sandbox_session_id, sandbox_machine_id FROM dashboard_sandboxes WHERE dashboard_id = ? + `).bind(dashboardId).first<{ sandbox_session_id: string; sandbox_machine_id: string }>(); + if (!sb) { + return { result: { error: 'No sandbox is running for this dashboard yet. Start a terminal first.' }, isError: true }; + } + + const sessionId = sb.sandbox_session_id; + const machineId = sb.sandbox_machine_id || undefined; + const runId = generateId(); + const outPath = `/workspace/.orc-run-${runId}.out`; + const exitPath = `/workspace/.orc-run-${runId}.exit`; + + // base64 the command so arbitrary quoting/metachars can't break the wrapper. + const bytes = new TextEncoder().encode(command); + let bin = ''; + for (const b of bytes) bin += String.fromCharCode(b); + const b64 = btoa(bin); + // Run it, capture stdout+stderr to .out, then write the exit code to .exit + // last (so the marker never appears before the output is flushed). + const wrapped = `echo ${b64} | base64 -d | bash > ${outPath} 2>&1; echo $? > ${exitPath}`; + + const client = new SandboxClient(env.SANDBOX_URL, env.SANDBOX_INTERNAL_TOKEN); + const readMarker = async (p: string) => { + const res = await sandboxFetch(env, `/sessions/${sessionId}/file?path=${encodeURIComponent(p)}`, { machineId }); + return res.ok ? await res.text() : null; + }; + + let pty: { id: string } | null = null; + try { + pty = await client.createPty(sessionId, '', wrapped, machineId); + + // Poll the (internal-token) file API for the exit marker — no new sandbox + // endpoint needed. Interval kept at 2s to bound subrequest count. + const started = Date.now(); + let exitRaw: string | null = null; + while (Date.now() - started < timeoutS * 1000) { + await new Promise((r) => setTimeout(r, 2000)); + exitRaw = await readMarker(exitPath); + if (exitRaw !== null) break; + } + + let output = (await readMarker(outPath)) || ''; + const truncated = output.length > 8000; + if (truncated) output = output.slice(-8000); + + // Best-effort cleanup: marker files + the headless PTY (it self-reaps when + // the wrapper exits, but delete to be tidy). + for (const p of [outPath, exitPath]) { + try { await sandboxFetch(env, `/sessions/${sessionId}/file?path=${encodeURIComponent(p)}`, { method: 'DELETE', machineId }); } catch { /* ignore */ } + } + if (pty?.id) { try { await client.deletePty(sessionId, pty.id); } catch { /* ignore */ } } + + if (exitRaw === null) { + return { + result: { + timed_out: true, + message: `Command still running after ${timeoutS}s. For long-running processes use create_terminal instead and check progress with read_file.`, + output, + truncated, + }, + isError: false, + }; + } + const exitCode = Number.parseInt(exitRaw.trim(), 10); + return { + result: { exit_code: Number.isNaN(exitCode) ? null : exitCode, output, truncated }, + isError: !Number.isNaN(exitCode) && exitCode !== 0, + }; + } catch (error) { + if (pty?.id) { try { await client.deletePty(sessionId, pty.id); } catch { /* ignore */ } } + return { result: { error: `run_command failed: ${error instanceof Error ? error.message : String(error)}` }, isError: true }; + } + } + if (toolName === 'terminal_send_input') { const dashboardId = args.dashboard_id as string; const terminalItemId = args.terminal_item_id as string; From 9f4855ba8be6613af13deabdbd42cb7f5c99031d Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Mon, 13 Jul 2026 00:02:58 +0100 Subject: [PATCH 08/31] fix(benchmarks): drive setup via run_command (blocking) + install uv if missing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Now that run_command exists, the setup guide uses it (timeout 300s) instead of fire-and-forget terminal_send_input — so chat blocks until clone+uv sync finish and continues automatically instead of stalling. Also installs uv if the image lacks it (until the Dockerfile-baked uv ships), and ends with SETUP_OK as the success signal. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- benchmarks/slopcodebench/template.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmarks/slopcodebench/template.json b/benchmarks/slopcodebench/template.json index 0f1cd14e..7a66e2fc 100644 --- a/benchmarks/slopcodebench/template.json +++ b/benchmarks/slopcodebench/template.json @@ -37,5 +37,5 @@ "y": 0, "zoom": 1 }, - "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Ensure the harness is set up — and VERIFY it (don't declare done blind)\nBefore launching, confirm setup actually finished with read_file:\n- read_file(\"slop-code-bench/bin/scb-visualize\") — clone succeeded.\n- read_file(\"scb-venv/pyvenv.cfg\") — the venv exists (uv sync finished).\nIf either is missing/errors, setup is not done — wait or re-run it; do NOT start the\nbenchmark yet.\n\n### Set up (idempotent — safe to re-run; won't wedge on leftovers)\nCreate a PLAIN shell terminal (create_terminal with NO boot_command). A terminal\nWITH a boot_command dies the instant the command finishes — and when setup is\nalready done these are near-instant no-ops, so it would vanish as \"failed to\nconnect\". Then send the setup as input (terminal_send_input) so the shell stays:\n if [ -e /workspace/slop-code-bench/bin/scb-visualize ]; then echo \"checkout present\"; else rm -rf /workspace/slop-code-bench && git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench; fi\n cd /workspace/slop-code-bench && { [ -e /workspace/scb-venv/bin/slop-code ] || UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12; }\n(`uv sync` is slow and one-time — tell the user to expect a wait. If github.com is\nheld by the egress proxy, ask them to Allow it.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && bin/scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low ; echo \\\"[runner exited $?]\\\" ; exec bash\")\nscb-visualize lives at bin/scb-visualize in the checkout (not on PATH). Always\nappend `; echo \"[runner exited $?]\" ; exec bash` so a boot failure stays visible\ninstead of killing the PTY (a dead PTY shows as \"failed to connect\").\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." + "setupGuide": "# SlopCodeBench orchestrator (you are Orcabot chat)\n\nYou run slop-code-bench for the user in this dashboard's VM using your tools — you\nare the orchestrator; do NOT open a separate agent terminal to drive it. Translate\nplain English into slop-code commands, launch them, and report progress by reading\nfiles. You do NOT solve the benchmark problems — the agent-under-test does.\n\nBe brief. Do each step with a tool and confirm before the next.\n\n## Your tools\n- create_terminal(boot_command=…) — run setup/benchmark commands in the VM.\n- terminal_send_input — type follow-up commands into a terminal you created.\n- read_file — check progress/results; this is how you answer \"how's it going?\".\n- secrets_create — help the user add a broker-protected provider key.\n\n## 1. Set up + VERIFY the harness — use run_command so it BLOCKS until done\nRun this with run_command (timeout_s: 300). run_command waits for the command to\nfinish and returns its exit_code + output, so you continue only when setup\nactually succeeds — no guessing, no stalling. It is idempotent (safe to re-run,\nwon't wedge on leftovers) and installs uv if the image lacks it:\n\n command -v uv >/dev/null || { curl -LsSf https://astral.sh/uv/install.sh | sh; export PATH=\"$HOME/.local/bin:$PATH\"; }\n if [ ! -e /workspace/slop-code-bench/bin/scb-visualize ]; then rm -rf /workspace/slop-code-bench && git clone --branch feat/host-tmux-executor https://github.com/robdmac/slop-code-bench /workspace/slop-code-bench; fi\n cd /workspace/slop-code-bench && { [ -e /workspace/scb-venv/bin/slop-code ] || UV_PYTHON_INSTALL_DIR=/workspace/.uv-python UV_PROJECT_ENVIRONMENT=/workspace/scb-venv uv sync --python 3.12; }\n ls /workspace/scb-venv/bin/slop-code && echo SETUP_OK\n\nProceed only if exit_code is 0 and the output ends with SETUP_OK. Otherwise show the\nuser the output; if github.com or astral.sh is held by the egress proxy, ask them to\nAllow it, then re-run. (uv sync is slow the first time — hence the generous timeout.)\n\n## 2. Provider key\nEnsure an OPENROUTER_API_KEY broker secret exists. If not, ask the user to add it\n(secrets_create, broker-protected). Never display or echo the value.\n\n## 3. Broker URL -> 127.0.0.1 (not localhost)\nPatch configs/providers.yaml so the openrouter api_base uses 127.0.0.1 (opencode /\nNode resolves localhost to IPv6 ::1, which the IPv4-only broker never answers).\n\n## 4. Run + visualize\nLaunch the run THROUGH the visualizer so each agent opens as a live read-only\ncomponent with a note naming its problem (scb-visualize watches\n.scb_tmux/runs.jsonl and surfaces each run):\n create_terminal(name=\"benchmark runner\", boot_command=\"cd /workspace/slop-code-bench && bin/scb-visualize run -- /workspace/scb-venv/bin/slop-code run --agent opencode --model openrouter/kimi-k2.6 --environment configs/environments/local-tmux-py.yaml --prompt configs/prompts/just-solve.jinja --problem file_backup --no-evaluate thinking=low ; echo \\\"[runner exited $?]\\\" ; exec bash\")\nscb-visualize lives at bin/scb-visualize in the checkout (not on PATH). Always\nappend `; echo \"[runner exited $?]\" ; exec bash` so a boot failure stays visible\ninstead of killing the PTY (a dead PTY shows as \"failed to connect\").\n\n## 5. Report progress\nAnswer \"how's it going?\" with read_file(\".scb-run.log\") or\nread_file(\".scb_tmux/runs.jsonl\"). Success = advances past checkpoint_1 with\ncost>0 and steps>0. To score: drop --no-evaluate or run\n`slop-code eval outputs//`." } \ No newline at end of file From af3166af2fb44b20169db472c5ab754d3d57e445 Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Mon, 13 Jul 2026 00:03:48 +0100 Subject: [PATCH 09/31] fix(frontend): shared SecretInput + no native form-submit so password managers don't pop on secrets Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- .../components/blocks/ASRSettingsDialog.tsx | 4 +- .../src/components/blocks/GoogleChatBlock.tsx | 4 +- .../src/components/blocks/MatrixBlock.tsx | 4 +- frontend/src/components/blocks/TeamsBlock.tsx | 4 +- .../src/components/blocks/TelegramBlock.tsx | 4 +- .../src/components/blocks/TerminalBlock.tsx | 75 +++++++++---------- .../src/components/blocks/WhatsAppBlock.tsx | 4 +- frontend/src/components/ui/SecretInput.tsx | 46 ++++++++++++ frontend/src/components/ui/index.ts | 1 + 9 files changed, 95 insertions(+), 51 deletions(-) create mode 100644 frontend/src/components/ui/SecretInput.tsx diff --git a/frontend/src/components/blocks/ASRSettingsDialog.tsx b/frontend/src/components/blocks/ASRSettingsDialog.tsx index 245df926..415c86d6 100644 --- a/frontend/src/components/blocks/ASRSettingsDialog.tsx +++ b/frontend/src/components/blocks/ASRSettingsDialog.tsx @@ -15,6 +15,7 @@ import { } from "@/components/ui/dialog"; import { Button } from "@/components/ui/button"; import { Input } from "@/components/ui/input"; +import { SecretInput } from "@/components/ui/SecretInput"; import { cn } from "@/lib/utils"; import { useASRSettingsStore, @@ -219,8 +220,7 @@ export function ASRSettingsDialog({ trigger, open: controlledOpen, onOpenChange {keyConfig.label}
- setKeyInputs((prev) => ({ ...prev, [keyConfig.key]: e.target.value })) diff --git a/frontend/src/components/blocks/GoogleChatBlock.tsx b/frontend/src/components/blocks/GoogleChatBlock.tsx index 69d8813d..6cc0584b 100644 --- a/frontend/src/components/blocks/GoogleChatBlock.tsx +++ b/frontend/src/components/blocks/GoogleChatBlock.tsx @@ -25,6 +25,7 @@ import { BlockWrapper } from "./BlockWrapper"; import { ConnectionHandles } from "./ConnectionHandles"; import { MinimizedBlockView, MINIMIZED_SIZE } from "./MinimizedBlockView"; import { Button } from "@/components/ui/button"; +import { SecretInput } from "@/components/ui/SecretInput"; import { DropdownMenu, DropdownMenuContent, @@ -394,8 +395,7 @@ export function GoogleChatBlock({ id, data, selected }: NodeProps
- setTokenInput(e.target.value)} placeholder="Paste your OAuth2 access token" diff --git a/frontend/src/components/blocks/MatrixBlock.tsx b/frontend/src/components/blocks/MatrixBlock.tsx index 58b56168..aa11e130 100644 --- a/frontend/src/components/blocks/MatrixBlock.tsx +++ b/frontend/src/components/blocks/MatrixBlock.tsx @@ -24,6 +24,7 @@ import { BlockWrapper } from "./BlockWrapper"; import { ConnectionHandles } from "./ConnectionHandles"; import { MinimizedBlockView, MINIMIZED_SIZE } from "./MinimizedBlockView"; import { Button } from "@/components/ui/button"; +import { SecretInput } from "@/components/ui/SecretInput"; import { DropdownMenu, DropdownMenuContent, @@ -480,8 +481,7 @@ export function MatrixBlock({ id, data, selected }: NodeProps) { placeholder="Homeserver URL" className="w-full px-2 py-1.5 text-xs rounded border border-[var(--border)] bg-[var(--background)] text-[var(--text-primary)] focus:outline-none focus:ring-1 focus:ring-[#0DBD8B]" /> - setTokenInput(e.target.value)} placeholder="Paste your Matrix access token" diff --git a/frontend/src/components/blocks/TeamsBlock.tsx b/frontend/src/components/blocks/TeamsBlock.tsx index cee30b6a..4d1006d0 100644 --- a/frontend/src/components/blocks/TeamsBlock.tsx +++ b/frontend/src/components/blocks/TeamsBlock.tsx @@ -25,6 +25,7 @@ import { BlockWrapper } from "./BlockWrapper"; import { ConnectionHandles } from "./ConnectionHandles"; import { MinimizedBlockView, MINIMIZED_SIZE } from "./MinimizedBlockView"; import { Button } from "@/components/ui/button"; +import { SecretInput } from "@/components/ui/SecretInput"; import { DropdownMenu, DropdownMenuContent, @@ -621,8 +622,7 @@ export function TeamsBlock({ id, data, selected }: NodeProps) { placeholder="Bot App ID (from Azure Bot Service)" className="w-full px-2 py-1.5 text-xs rounded border border-[var(--border)] bg-[var(--background)] text-[var(--text-primary)] focus:outline-none focus:ring-1 focus:ring-[#6264A7]" /> - setTokenInput(e.target.value)} placeholder="Bot App Secret" diff --git a/frontend/src/components/blocks/TelegramBlock.tsx b/frontend/src/components/blocks/TelegramBlock.tsx index cade7357..89f08306 100644 --- a/frontend/src/components/blocks/TelegramBlock.tsx +++ b/frontend/src/components/blocks/TelegramBlock.tsx @@ -24,6 +24,7 @@ import { BlockWrapper } from "./BlockWrapper"; import { ConnectionHandles } from "./ConnectionHandles"; import { MinimizedBlockView, MINIMIZED_SIZE } from "./MinimizedBlockView"; import { Button } from "@/components/ui/button"; +import { SecretInput } from "@/components/ui/SecretInput"; import { DropdownMenu, DropdownMenuContent, @@ -472,8 +473,7 @@ export function TelegramBlock({ id, data, selected }: NodeProps) { Connect Telegram to send and receive messages

- setTokenInput(e.target.value)} placeholder="Paste your Telegram bot token" diff --git a/frontend/src/components/blocks/TerminalBlock.tsx b/frontend/src/components/blocks/TerminalBlock.tsx index ee09dd09..f548d248 100644 --- a/frontend/src/components/blocks/TerminalBlock.tsx +++ b/frontend/src/components/blocks/TerminalBlock.tsx @@ -73,6 +73,7 @@ import { DropdownMenuSeparator, DropdownMenuTrigger, Input, + SecretInput, } from "@/components/ui"; import { ConnectionHandles } from "./ConnectionHandles"; import { ConnectionMarkers } from "./ConnectionMarkers"; @@ -2903,17 +2904,7 @@ export function TerminalBlock({
Secrets are brokered - the LLM cannot read them directly.
-
{ - e.preventDefault(); - if (newSecretName.trim() && newSecretValue.trim()) { - handleAddSecret(); - } - }} - className="flex gap-1" - > +
- setNewSecretValue(e.target.value)} className="h-6 text-xs flex-1 nodrag" - autoComplete="off" - data-form-type="other" - data-lpignore="true" - style={{ WebkitTextSecurity: "disc" } as React.CSSProperties} + onKeyDown={(e) => { + if (e.key === "Enter") { + e.preventDefault(); + if (newSecretName.trim() && newSecretValue.trim()) { + handleAddSecret(); + } + } + }} /> - +
{/* Pending domain approvals */} {pendingApprovalsQuery.data && pendingApprovalsQuery.data.length > 0 && (
@@ -3058,17 +3057,7 @@ export function TerminalBlock({
)} -
{ - e.preventDefault(); - if (newEnvVarName.trim() && newEnvVarValue.trim()) { - handleAddEnvVar(); - } - }} - className="flex gap-1" - > +
- setNewEnvVarValue(e.target.value)} className="h-6 text-xs flex-1 nodrag" - autoComplete="off" - data-form-type="other" - data-lpignore="true" + onKeyDown={(e) => { + if (e.key === "Enter") { + e.preventDefault(); + if (newEnvVarName.trim() && newEnvVarValue.trim()) { + handleAddEnvVar(); + } + } + }} /> - +
{/* Env vars list */}
{secretsQuery.isLoading && ( @@ -3612,10 +3610,9 @@ export function TerminalBlock({ - setCpApiKey(e.target.value)} - type="password" placeholder="API key (optional)" className="flex-1 min-w-0 px-2 py-1 text-[11px] rounded border border-[var(--border)] bg-[var(--background-elevated)] text-[var(--foreground)]" /> diff --git a/frontend/src/components/blocks/WhatsAppBlock.tsx b/frontend/src/components/blocks/WhatsAppBlock.tsx index 4c14b223..c787e2eb 100644 --- a/frontend/src/components/blocks/WhatsAppBlock.tsx +++ b/frontend/src/components/blocks/WhatsAppBlock.tsx @@ -24,6 +24,7 @@ import { BlockWrapper } from "./BlockWrapper"; import { ConnectionHandles } from "./ConnectionHandles"; import { MinimizedBlockView, MINIMIZED_SIZE } from "./MinimizedBlockView"; import { Button } from "@/components/ui/button"; +import { SecretInput } from "@/components/ui/SecretInput"; import { DropdownMenu, DropdownMenuContent, @@ -704,8 +705,7 @@ export function WhatsAppBlock({ id, data, selected }: NodeProps) {

Connect WhatsApp Business API

- setTokenInput(e.target.value)} placeholder="Access token" diff --git a/frontend/src/components/ui/SecretInput.tsx b/frontend/src/components/ui/SecretInput.tsx new file mode 100644 index 00000000..cde4c7f9 --- /dev/null +++ b/frontend/src/components/ui/SecretInput.tsx @@ -0,0 +1,46 @@ +// Copyright 2026 Rob Macrae. All rights reserved. +// SPDX-License-Identifier: LicenseRef-Proprietary + +// REVISION: secret-input-v1-masked-no-password-field +"use client"; + +import * as React from "react"; +import { Input, type InputProps } from "./input"; + +const SECRET_INPUT_REVISION = "secret-input-v1-masked-no-password-field"; +if (typeof window !== "undefined" && !(window as unknown as { __secretInputLogged?: boolean }).__secretInputLogged) { + (window as unknown as { __secretInputLogged?: boolean }).__secretInputLogged = true; + console.log(`[SecretInput] REVISION: ${SECRET_INPUT_REVISION} loaded at ${new Date().toISOString()}`); +} + +/** + * SecretInput — a masked text input for secrets/tokens/API keys that NEVER uses + * `type="password"`. Using a real password field triggers browser password + * managers (notably Safari's built-in Passwords, which pops a "save password" + * prompt whenever a masked field is submitted through a native form). Instead we + * render a plain text input visually masked via `-webkit-text-security` and set + * every "please ignore me" hint the common managers respect. + */ +const SecretInput = React.forwardRef( + ({ style, ...props }, ref) => { + return ( + + ); + } +); +SecretInput.displayName = "SecretInput"; + +export { SecretInput }; diff --git a/frontend/src/components/ui/index.ts b/frontend/src/components/ui/index.ts index 2884538c..bbbcbf3c 100644 --- a/frontend/src/components/ui/index.ts +++ b/frontend/src/components/ui/index.ts @@ -6,6 +6,7 @@ export { Button, buttonVariants, type ButtonProps } from "./button"; export { Badge, badgeVariants, type BadgeProps } from "./badge"; export { Avatar, AvatarGroup, avatarVariants, type AvatarProps, type AvatarGroupProps } from "./avatar"; export { Input, Textarea, type InputProps, type TextareaProps } from "./input"; +export { SecretInput } from "./SecretInput"; export { Tooltip, TooltipProvider, From 1496e31d4f36f0e182413d69f5a9bb484f7c76fd Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Mon, 13 Jul 2026 00:04:36 +0100 Subject: [PATCH 10/31] feat(chat): dedicated in-window input when expanded; title bar focuses it Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- frontend/src/components/chat/ChatPanel.tsx | 157 ++++++++++++++++----- 1 file changed, 119 insertions(+), 38 deletions(-) diff --git a/frontend/src/components/chat/ChatPanel.tsx b/frontend/src/components/chat/ChatPanel.tsx index 44ef4e79..de286be1 100644 --- a/frontend/src/components/chat/ChatPanel.tsx +++ b/frontend/src/components/chat/ChatPanel.tsx @@ -1,6 +1,6 @@ // Copyright 2026 Rob Macrae. All rights reserved. // SPDX-License-Identifier: LicenseRef-Proprietary -// REVISION: chat-v34-no-key-provider-card +// REVISION: chat-v35-in-window-input "use client"; @@ -12,7 +12,7 @@ * Supports smooth handoff from splash page transition overlay. */ -const CHAT_PANEL_REVISION = "chat-v34-no-key-provider-card"; +const CHAT_PANEL_REVISION = "chat-v35-in-window-input"; const AI_ONBOARD_KEYWORD = "force_ai_onboard"; console.log(`[ChatPanel] REVISION: ${CHAT_PANEL_REVISION} loaded at ${new Date().toISOString()}`); @@ -66,6 +66,10 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o const [inputValue, setInputValue] = React.useState(""); const inputBarRef = React.useRef(null); const inputRef = React.useRef(null); + // Separate ref for the dedicated compose input rendered inside the expanded + // window. Shares the single `inputValue` source of truth with the collapsed + // top-bar input; only the focus target differs. + const inWindowInputRef = React.useRef(null); const panelRef = React.useRef(null); const initialPromptConsumedRef = React.useRef(false); // Auto-minimize the chat the first time the user interacts with the page @@ -373,28 +377,47 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o borderBottom: isExpanded ? "1px solid rgba(0, 229, 255, 0.15)" : undefined, }} > - setInputValue(e.target.value)} - onKeyDown={handleKeyDown} - onFocus={() => setIsExpanded(true)} - placeholder="Ask Orcabot..." - disabled={isStreaming} - autoComplete="one-time-code" - data-form-type="other" - data-lpignore="true" - data-1p-ignore - className={cn( - "flex-1 border-0 outline-none chat-input-splash focus-visible:outline-none", - "text-sm", - "disabled:opacity-50", - "placeholder:text-[#5a7a9e]" - )} - style={{ color: "#e8edf5", caretColor: "#00e5ff", backgroundColor: "transparent", outline: "none" }} - /> + {isExpanded ? ( + /* Expanded: the top bar is a title/header, not an input. Clicking it + focuses the dedicated in-window compose input below. */ + + ) : ( + setInputValue(e.target.value)} + onKeyDown={handleKeyDown} + onFocus={() => { + setIsExpanded(true); + // The collapsed input unmounts on expand; hand focus to the + // in-window compose input once it has mounted. + requestAnimationFrame(() => inWindowInputRef.current?.focus()); + }} + placeholder="Ask Orcabot..." + disabled={isStreaming} + autoComplete="one-time-code" + data-form-type="other" + data-lpignore="true" + data-1p-ignore + className={cn( + "flex-1 border-0 outline-none chat-input-splash focus-visible:outline-none", + "text-sm", + "disabled:opacity-50", + "placeholder:text-[#5a7a9e]" + )} + style={{ color: "#e8edf5", caretColor: "#00e5ff", backgroundColor: "transparent", outline: "none" }} + /> + )} {/* Trash button (when has messages) */} {hasMessages && ( @@ -429,22 +452,80 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o )} - {/* Send Button */} - + {/* Send Button — collapsed only; expanded uses the in-window compose box */} + {!isExpanded && ( + + )}
+ {/* In-window compose box (expanded only). Pinned between the title bar and + the messages scroll area — since messages render newest-at-top, a new + message appears directly beneath this box. Right-aligned and styled to + echo the user's own message bubbles (rounded-2xl, primary accent) so it + reads as "your next message". White in light mode, elevated surface in + dark. */} + {isExpanded && ( +
+
+ setInputValue(e.target.value)} + onKeyDown={handleKeyDown} + placeholder="Ask Orcabot..." + disabled={isStreaming} + autoComplete="one-time-code" + data-form-type="other" + data-lpignore="true" + data-1p-ignore + className={cn( + "flex-1 bg-transparent border-0 outline-none focus-visible:outline-none", + "text-sm placeholder:text-muted-foreground disabled:opacity-50" + )} + style={{ color: "var(--foreground)", caretColor: "var(--accent-primary)", outline: "none" }} + /> + +
+
+ )} + {/* Messages area (when expanded, below input) — newest at top */} {isExpanded && (
Date: Mon, 13 Jul 2026 00:21:52 +0100 Subject: [PATCH 11/31] fix(frontend): unmask SecretInput by default so Safari stops flagging it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The `-webkit-text-security: disc` CSS mask makes Safari classify a text input as a password just like `type="password"` does — popping the "save password" prompt and offering to autofill saved site credentials (the Touch-ID thumbprint) right on the field. Since a key is pasted once and saved secrets are already masked in the list, render the entry field unmasked by default; add an opt-in `masked` prop for callers that accept the manager popups. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- frontend/src/components/ui/SecretInput.tsx | 28 +++++++++++++--------- 1 file changed, 17 insertions(+), 11 deletions(-) diff --git a/frontend/src/components/ui/SecretInput.tsx b/frontend/src/components/ui/SecretInput.tsx index cde4c7f9..44a9cee4 100644 --- a/frontend/src/components/ui/SecretInput.tsx +++ b/frontend/src/components/ui/SecretInput.tsx @@ -1,28 +1,34 @@ // Copyright 2026 Rob Macrae. All rights reserved. // SPDX-License-Identifier: LicenseRef-Proprietary -// REVISION: secret-input-v1-masked-no-password-field +// REVISION: secret-input-v2-unmasked-default "use client"; import * as React from "react"; import { Input, type InputProps } from "./input"; -const SECRET_INPUT_REVISION = "secret-input-v1-masked-no-password-field"; +const SECRET_INPUT_REVISION = "secret-input-v2-unmasked-default"; if (typeof window !== "undefined" && !(window as unknown as { __secretInputLogged?: boolean }).__secretInputLogged) { (window as unknown as { __secretInputLogged?: boolean }).__secretInputLogged = true; console.log(`[SecretInput] REVISION: ${SECRET_INPUT_REVISION} loaded at ${new Date().toISOString()}`); } /** - * SecretInput — a masked text input for secrets/tokens/API keys that NEVER uses - * `type="password"`. Using a real password field triggers browser password - * managers (notably Safari's built-in Passwords, which pops a "save password" - * prompt whenever a masked field is submitted through a native form). Instead we - * render a plain text input visually masked via `-webkit-text-security` and set - * every "please ignore me" hint the common managers respect. + * SecretInput — a text input for secrets/tokens/API keys that NEVER trips browser + * password managers. BOTH `type="password"` AND the `-webkit-text-security` CSS + * mask make browsers (notably Safari) classify the field as a password — popping + * a "save password" prompt and offering to autofill saved site passwords (the + * Touch-ID thumbprint) right on top of the field. So we render a plain text input + * with every "ignore me" hint the managers respect, and leave it UNMASKED by + * default: a key is pasted once, and saved secrets are shown masked in the list. + * Pass `masked` to re-enable the CSS mask if you accept the manager popups. */ -const SecretInput = React.forwardRef( - ({ style, ...props }, ref) => { +interface SecretInputProps extends InputProps { + masked?: boolean; +} + +const SecretInput = React.forwardRef( + ({ style, masked = false, ...props }, ref) => { return ( ( data-1p-ignore="true" data-bwignore="true" data-form-type="other" - style={{ WebkitTextSecurity: "disc", ...style } as React.CSSProperties} + style={masked ? ({ WebkitTextSecurity: "disc", ...style } as React.CSSProperties) : style} {...props} /> ); From 2e8921a33ea2b3011af2af558bb7e83aed4080bd Mon Sep 17 00:00:00 2001 From: Rob Macrae Date: Mon, 13 Jul 2026 00:37:27 +0100 Subject: [PATCH 12/31] style(frontend): retheme chat input bar to match app theme Replace the hardcoded splash dark-blue glass (--chat-input-bg, #e8edf5 text, #8ba3c4 icons, cyan caret, cyan hairline borders) on the chat input/title bar and in-window compose row with theme tokens (--background-elevated bar, --background-surface compose pill, --foreground / --muted-foreground text, --accent-primary caret, --border). The bar now blends with the dashboard and follows light/dark/midnight themes. Drops the chat-input-splash autofill class from the collapsed input (autofill already suppressed via nofill name + one-time-code + ignore attrs). Send button stays primary blue. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- frontend/src/components/chat/ChatPanel.tsx | 30 ++++++++++------------ 1 file changed, 13 insertions(+), 17 deletions(-) diff --git a/frontend/src/components/chat/ChatPanel.tsx b/frontend/src/components/chat/ChatPanel.tsx index de286be1..ed51014c 100644 --- a/frontend/src/components/chat/ChatPanel.tsx +++ b/frontend/src/components/chat/ChatPanel.tsx @@ -1,6 +1,6 @@ // Copyright 2026 Rob Macrae. All rights reserved. // SPDX-License-Identifier: LicenseRef-Proprietary -// REVISION: chat-v35-in-window-input +// REVISION: chat-v36-theme-input-bar "use client"; @@ -12,7 +12,7 @@ * Supports smooth handoff from splash page transition overlay. */ -const CHAT_PANEL_REVISION = "chat-v35-in-window-input"; +const CHAT_PANEL_REVISION = "chat-v36-theme-input-bar"; const AI_ONBOARD_KEYWORD = "force_ai_onboard"; console.log(`[ChatPanel] REVISION: ${CHAT_PANEL_REVISION} loaded at ${new Date().toISOString()}`); @@ -371,10 +371,8 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o ref={inputBarRef} className="flex items-center gap-2 px-4 py-2.5 rounded-t-2xl" style={{ - background: "var(--chat-input-bg)", - backdropFilter: "blur(20px)", - WebkitBackdropFilter: "blur(20px)", - borderBottom: isExpanded ? "1px solid rgba(0, 229, 255, 0.15)" : undefined, + background: "var(--background-elevated)", + borderBottom: isExpanded ? "1px solid var(--border)" : undefined, }} > {isExpanded ? ( @@ -384,7 +382,7 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o type="button" onClick={() => inWindowInputRef.current?.focus()} className="flex-1 text-left text-sm font-medium truncate focus-visible:outline-none" - style={{ color: "#e8edf5" }} + style={{ color: "var(--foreground)" }} title="Focus the message box" > Ask Orcabot @@ -410,12 +408,12 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o data-lpignore="true" data-1p-ignore className={cn( - "flex-1 border-0 outline-none chat-input-splash focus-visible:outline-none", + "flex-1 border-0 outline-none focus-visible:outline-none", "text-sm", "disabled:opacity-50", - "placeholder:text-[#5a7a9e]" + "placeholder:text-muted-foreground" )} - style={{ color: "#e8edf5", caretColor: "#00e5ff", backgroundColor: "transparent", outline: "none" }} + style={{ color: "var(--foreground)", caretColor: "var(--accent-primary)", backgroundColor: "transparent", outline: "none" }} /> )} @@ -442,8 +440,8 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o if (isExpanded && isAtSplashPosition) setIsAtSplashPosition(false); setIsExpanded(!isExpanded); }} - className="h-7 w-7 p-0 rounded-full hover:bg-white/10" - style={{ color: "#8ba3c4" }} + className="h-7 w-7 p-0 rounded-full hover:bg-background-hover" + style={{ color: "var(--muted-foreground)" }} > {isExpanded ? ( @@ -480,15 +478,13 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o
Date: Mon, 13 Jul 2026 01:19:13 +0100 Subject: [PATCH 13/31] style(frontend): chat input is a plain white box, no panel behind it Expanded chat: make the compose input a white pill with black text in every theme (dark/midnight would otherwise render dark-on-dark), and drop the surrounding bar/border backgrounds so the input reads as part of the chat window rather than sitting on its own panel. Collapsed floating pill keeps its themed surface. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XBxVYcEidYfRm5Gf4j7nPL --- frontend/src/components/chat/ChatPanel.tsx | 21 ++++++++++----------- 1 file changed, 10 insertions(+), 11 deletions(-) diff --git a/frontend/src/components/chat/ChatPanel.tsx b/frontend/src/components/chat/ChatPanel.tsx index ed51014c..b4cf4a4e 100644 --- a/frontend/src/components/chat/ChatPanel.tsx +++ b/frontend/src/components/chat/ChatPanel.tsx @@ -1,6 +1,6 @@ // Copyright 2026 Rob Macrae. All rights reserved. // SPDX-License-Identifier: LicenseRef-Proprietary -// REVISION: chat-v36-theme-input-bar +// REVISION: chat-v37-white-input-no-panel "use client"; @@ -12,7 +12,7 @@ * Supports smooth handoff from splash page transition overlay. */ -const CHAT_PANEL_REVISION = "chat-v36-theme-input-bar"; +const CHAT_PANEL_REVISION = "chat-v37-white-input-no-panel"; const AI_ONBOARD_KEYWORD = "force_ai_onboard"; console.log(`[ChatPanel] REVISION: ${CHAT_PANEL_REVISION} loaded at ${new Date().toISOString()}`); @@ -371,8 +371,9 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o ref={inputBarRef} className="flex items-center gap-2 px-4 py-2.5 rounded-t-2xl" style={{ - background: "var(--background-elevated)", - borderBottom: isExpanded ? "1px solid var(--border)" : undefined, + // Expanded: no panel — the bar is transparent so it reads as part of + // the chat window. Collapsed: a themed floating pill on the dashboard. + background: isExpanded ? "transparent" : "var(--background-elevated)", }} > {isExpanded ? ( @@ -477,14 +478,10 @@ export function ChatPanel({ dashboardId, className, onUICommand, needsAiSetup, o {isExpanded && (
- setNewSecretValue(e.target.value)} className="w-full" - autoComplete="off" - data-1p-ignore - data-lpignore="true" - data-form-type="other" - style={{ WebkitTextSecurity: "disc" } as React.CSSProperties} />