Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
31 commits
Select commit Hold shift + click to select a range
1dae5a8
feat(chat): add read_file tool so the orchestrator can inspect runnin…
robdmac Jul 12, 2026
cb860eb
feat(benchmarks): live per-agent visualization + chat-as-orchestrator…
robdmac Jul 12, 2026
18ab8ac
feat(benchmarks): --sql mode for seed-benchmark-templates (local wran…
robdmac Jul 12, 2026
3056445
fix(benchmarks): idempotent setup + robust runner in SlopCodeBench guide
robdmac Jul 12, 2026
997cba4
fix(benchmarks): run setup in a persistent shell (terminal_send_input…
robdmac Jul 12, 2026
cbf922d
feat(sandbox): bake uv into the image for benchmark venvs
robdmac Jul 12, 2026
ea93730
feat(chat): add run_command — blocking exec so the orchestrator can s…
robdmac Jul 12, 2026
9f4855b
fix(benchmarks): drive setup via run_command (blocking) + install uv …
robdmac Jul 12, 2026
af3166a
fix(frontend): shared SecretInput + no native form-submit so password…
robdmac Jul 12, 2026
1496e31
feat(chat): dedicated in-window input when expanded; title bar focuse…
robdmac Jul 12, 2026
a180686
fix(frontend): unmask SecretInput by default so Safari stops flagging it
robdmac Jul 12, 2026
2e8921a
style(frontend): retheme chat input bar to match app theme
robdmac Jul 12, 2026
e9cd41a
style(frontend): chat input is a plain white box, no panel behind it
robdmac Jul 13, 2026
7fde717
fix(frontend): finish SecretInput sweep — dashboard secrets + AI key …
robdmac Jul 13, 2026
4c609c9
chore(controlplane): instrument run_command / executeTool for visibility
robdmac Jul 13, 2026
8e50449
feat(frontend): mask secrets with a disc font (no password-manager po…
robdmac Jul 13, 2026
0a06e20
chore(controlplane): heartbeat log in run_command poll loop
robdmac Jul 13, 2026
21b3fdc
fix(benchmarks): run slopcodebench setup in a visible terminal, not r…
robdmac Jul 13, 2026
70c77e4
fix(frontend): don't auto-minimize chat when it's the primary UI
robdmac Jul 13, 2026
1e4a7df
chore(controlplane): log emitted create_terminal boot_command
robdmac Jul 13, 2026
ea0818a
fix(benchmarks): robust one-line setup boot_command + honest monitoring
robdmac Jul 13, 2026
1ac757f
fix(frontend): don't clear terminal on first connect (wiped boot output)
robdmac Jul 13, 2026
13d0ae5
fix(benchmarks): sleep 2 in setup boot_command so output isn't lost p…
robdmac Jul 13, 2026
027e202
fix(controlplane): read_file/run_command must use workspace-RELATIVE …
robdmac Jul 13, 2026
ec96d03
fix(controlplane): gate terminal-executing chat tools on owner/editor…
robdmac Jul 13, 2026
d5bdb0c
fix(benchmarks): scb-visualize only surfaces NEW runs, not the whole …
robdmac Jul 13, 2026
7bf5d59
fix(frontend): SecretInput font-display block (P2) + bump stale revis…
robdmac Jul 13, 2026
6b52925
fix(frontend): SecretInput masking fails closed if font unavailable (…
robdmac Jul 13, 2026
fdffcc6
fix(frontend): SecretInput fails closed BEFORE font verification (P2)
robdmac Jul 13, 2026
0913229
fix(frontend): mask SecretInput 'unknown' state against selection rev…
robdmac Jul 13, 2026
10ef2d7
fix(frontend): keep caret visible in SecretInput pending state (P3)
robdmac Jul 13, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 39 additions & 1 deletion benchmarks/bin/seed-benchmark-templates
Original file line number Diff line number Diff line change
Expand Up @@ -76,8 +76,46 @@ def upsert(tpl):
return tid


def _sql_lit(v):
if v is None:
return "NULL"
if isinstance(v, int):
return str(v)
return "'" + str(v).replace("'", "''") + "'"


def emit_sql(tpl):
"""Emit DELETE+INSERT SQL for a template (for `wrangler d1 execute --file`)."""
tid = str(uuid.uuid4())
now = datetime.datetime.utcnow().isoformat()
cols = ("id,name,description,category,author_id,author_name,items_json,edges_json,"
"viewport_json,setup_guide,item_count,is_featured,use_count,status,created_at,updated_at")
vals = [
tid, tpl["name"], tpl["description"], tpl.get("category", "coding"),
AUTHOR_ID, AUTHOR_NAME,
json.dumps(tpl["items"]), json.dumps(tpl["edges"]),
json.dumps(tpl.get("viewport")) if tpl.get("viewport") is not None else None,
tpl.get("setupGuide"), len(tpl["items"]), 1, 0, "approved", now, now,
]
literals = ", ".join(_sql_lit(v) for v in vals)
return (f"DELETE FROM dashboard_templates WHERE name = {_sql_lit(tpl['name'])};\n"
f"INSERT INTO dashboard_templates ({cols}) VALUES ({literals});")


def main():
paths = discover(sys.argv[1:])
args = sys.argv[1:]
sql_mode = "--sql" in args
args = [a for a in args if a != "--sql"]
paths = discover(args)

# --sql: print SQL to stdout for a direct D1 apply (e.g. local wrangler-dev):
# benchmarks/bin/seed-benchmark-templates --sql slopcodebench > /tmp/seed.sql
# (cd controlplane && npx wrangler d1 execute orcabot-db --local --file /tmp/seed.sql)
if sql_mode:
for path in paths:
print(emit_sql(json.loads(path.read_text())))
return

# Ensure the setup_guide column exists (idempotent) so older DBs accept the insert.
try:
q("ALTER TABLE dashboard_templates ADD COLUMN setup_guide TEXT", [])
Expand Down
191 changes: 191 additions & 0 deletions benchmarks/slopcodebench/bin/scb-visualize
Original file line number Diff line number Diff line change
@@ -0,0 +1,191 @@
#!/usr/bin/env python3
"""scb-visualize — turn a slop-code-bench run into a live Orcabot canvas.

Watches the host-tmux executor's run manifest and, for each agent-under-test that
starts, surfaces a **read-only** Orcabot terminal tailing that run's logfile, with
a note above it naming the problem. Orcabot becomes a live, per-problem view of
the benchmark as it runs.

Why this is safe (read-only by construction): each viewer only runs
`tail -F <logfile>` on the run's output. It has no path to the agent CLI, so
nothing typed into a viewer can steer or corrupt the run — the exact property the
viewer-smoke test asserts. (This is why we tail the per-run logfile rather than
attach a tmux control socket, which would grant cross-session inject.)

Detection surface (from the `feat/host-tmux-executor` fork):
<workdir>/.scb_tmux/runs.jsonl — one JSON line per run:
{"target","session","window":"<problem>","logfile":"...","created":...}

Runs inside an Orcabot PTY: it creates canvas components via the local sandbox
MCP server using ORCABOT_MCP_SECRET; the dashboard is resolved server-side from
the session, so no dashboard id is needed.

Usage:
scb-visualize watch [--workdir DIR] # watch an existing run's manifest
scb-visualize run [--workdir DIR] -- CMD… # run the benchmark AND visualize
"""
import json
import os
import shlex
import subprocess
import sys
import threading
import time
import urllib.error
import urllib.request

MCP_BASE = os.environ.get("ORCABOT_MCP_BASE", "http://127.0.0.1:8081")
SID = os.environ.get("ORCABOT_SESSION_ID", "").strip()
PTY = os.environ.get("ORCABOT_PTY_ID", "").strip()
SECRET = os.environ.get("ORCABOT_MCP_SECRET", "").strip()

# Canvas layout: a grid of columns; each run = a note stacked over its terminal.
COL_W, NOTE_H, TERM_H, GAP, COLS = 380, 96, 300, 24, 3
BASE_X, BASE_Y, COL_GAP, ROW_GAP = 40, 40, 40, 48


def _log(msg):
sys.stderr.write(f"[scb-visualize] {msg}\n")
sys.stderr.flush()


def mcp_ready():
return bool(SID and PTY and SECRET)


def call_tool(name, arguments):
"""Invoke an Orcabot MCP UI tool via the local sandbox MCP server."""
url = f"{MCP_BASE}/sessions/{SID}/mcp/tools/call?pty_id={PTY}"
body = json.dumps({"name": name, "arguments": arguments}).encode()
req = urllib.request.Request(
url, data=body, method="POST",
headers={"Content-Type": "application/json", "X-MCP-Secret": SECRET},
)
try:
with urllib.request.urlopen(req, timeout=15) as resp:
return json.load(resp)
except urllib.error.HTTPError as e:
_log(f"MCP {name} failed: {e.code} {e.read()[:200]!r}")
except Exception as e: # noqa: BLE001 — best-effort visualization
_log(f"MCP {name} error: {e}")
return None


def slot(idx):
col, row = idx % COLS, idx // COLS
x = BASE_X + col * (COL_W + COL_GAP)
y = BASE_Y + row * (NOTE_H + TERM_H + GAP + ROW_GAP)
return x, y


def surface_run(idx, problem, logfile):
x, y = slot(idx)
call_tool("create_note", {
"content": f"### Solving: {problem}\n\nAgent-under-test — live, read-only.",
"position": {"x": x, "y": y},
"size": {"width": COL_W, "height": NOTE_H},
"color": "blue",
})
call_tool("create_terminal", {
"name": f"agent: {problem}",
"boot_command": f"tail -n +1 -F {shlex.quote(logfile)}",
"agentic": False,
"position": {"x": x, "y": y + NOTE_H + GAP},
"size": {"width": COL_W, "height": TERM_H},
})
_log(f"surfaced run '{problem}' -> {logfile}")


def watch(workdir, stop=None, skip_existing=False):
manifest = os.path.join(workdir, ".scb_tmux", "runs.jsonl")
_log(f"watching {manifest}")
if not mcp_ready():
_log("not in an Orcabot PTY (no ORCABOT_MCP_SECRET) — visualization disabled; still tailing manifest for logs")
seen, idx = set(), 0
# runs.jsonl is append-only across ALL runs. In `run` mode we launch a fresh
# watcher per benchmark run, so without this it would re-surface a note + a
# permanent `tail -F` terminal for every historical run on each launch.
# Snapshot the pre-existing records as already-seen so only runs THIS
# invocation appends get surfaced.
if skip_existing:
try:
with open(manifest) as f:
for line in f:
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
seen.add(rec.get("logfile") or line)
except FileNotFoundError:
pass
idx = len(seen) # place new terminals past any from prior invocations
_log(f"skip_existing: snapshotted {len(seen)} prior run(s)")
while stop is None or not stop.is_set():
try:
with open(manifest) as f:
lines = f.readlines()
except FileNotFoundError:
time.sleep(1.0)
continue
for line in lines:
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
logfile = rec.get("logfile")
problem = rec.get("window") or rec.get("target") or "run"
key = logfile or line
if key in seen:
continue
seen.add(key)
if not logfile:
continue
if not os.path.isabs(logfile):
logfile = os.path.join(workdir, logfile)
if mcp_ready():
surface_run(idx, problem, logfile)
else:
_log(f"(disabled) would surface: {problem} -> {logfile}")
idx += 1
time.sleep(1.0)


def main():
argv = sys.argv[1:]
if not argv or argv[0] in ("-h", "--help"):
print(__doc__)
sys.exit(0 if argv else 2)

mode = argv[0]
rest = argv[1:]
workdir = os.getcwd()
if "--workdir" in rest:
i = rest.index("--workdir")
workdir = rest[i + 1]
rest = rest[:i] + rest[i + 2:]

if mode == "watch":
watch(workdir)
elif mode == "run":
cmd = rest[1:] if rest[:1] == ["--"] else rest
if not cmd:
_log("usage: scb-visualize run [--workdir DIR] -- <command...>")
sys.exit(2)
t = threading.Thread(target=watch, args=(workdir,), kwargs={"skip_existing": True}, daemon=True)
t.start()
rc = subprocess.call(cmd, cwd=workdir)
time.sleep(2) # let the watcher surface the final run
sys.exit(rc)
else:
print(__doc__)
sys.exit(2)


if __name__ == "__main__":
main()
Loading
Loading