From 7414918c016a2397668acd20e162a0e2ab89c3e8 Mon Sep 17 00:00:00 2001 From: Matt Rubens <2600+mrubens@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:02:04 -0400 Subject: [PATCH 1/5] [Improve] Check the work before the agent reports to a person or ships, not only at turn end --- apps/docs/tasks.mdx | 18 +- ...code-completion-gate-plugin-script.test.ts | 148 ++++++++++++++ apps/worker/src/run-task/agent-home.ts | 14 +- .../opencode-completion-gate-plugin-script.ts | 90 +++++++++ apps/worker/src/sandbox-server/lib/harness.ts | 8 + .../opencode-completion-gate-diff.test.ts | 63 ++++++ .../opencode-server-bootstrap.test.ts | 2 +- .../opencode-server-completion-gate.test.ts | 102 ++++++++++ .../opencode-server/completion-gate.ts | 76 +++++++ .../lib/harnesses/opencode-server/harness.ts | 185 +++++++++++++++--- .../checkCompletionBeforeTool.test.ts | 53 +++++ .../procedures/checkCompletionBeforeTool.ts | 23 +++ .../src/sandbox-server/procedures/index.ts | 1 + .../src/sandbox-server/routers/commands.ts | 2 + 14 files changed, 747 insertions(+), 38 deletions(-) create mode 100644 apps/worker/src/run-task/__tests__/opencode-completion-gate-plugin-script.test.ts create mode 100644 apps/worker/src/run-task/opencode-completion-gate-plugin-script.ts create mode 100644 apps/worker/src/sandbox-server/procedures/__tests__/checkCompletionBeforeTool.test.ts create mode 100644 apps/worker/src/sandbox-server/procedures/checkCompletionBeforeTool.ts diff --git a/apps/docs/tasks.mdx b/apps/docs/tasks.mdx index 2adb831b89..df4e93c6a3 100644 --- a/apps/docs/tasks.mdx +++ b/apps/docs/tasks.mdx @@ -171,17 +171,21 @@ video. Roomote skips visual proof when it would not add useful evidence. Judge the evidence against the stated result rather than requiring a recording for every change. -When a [judgment model](/models#judgment-model) is configured and a task turn -ends with code changes, Roomote automatically holds the agent's closing report -against the evidence: what you asked for, the agent's own checklist, the diff, -and the shell commands it actually ran. It looks for a requested item or +When a [judgment model](/models#judgment-model) is configured, Roomote +automatically holds the agent's work against the evidence at the moments that +matter: before the agent reports to you, before it pushes or opens a pull +request, and when its turn ends with code changes. The evidence is what you +asked for, the agent's own checklist, the diff, and the shell commands it +actually ran. It looks for a requested item or checklist item left undone without explanation, a claimed change the diff does not contain, a claimed test or build result the recorded commands do not support, code shipped with no validation and no reason, an interface change with visual proof waved off, a plain defect in the changed lines, and leftovers -such as debug logging or a disabled test. If anything is found, the agent gets -one chance to fix it or explain before the task reports back. The check takes -about a second and does not replace pull request review. Without a judgment +such as debug logging or a disabled test. If anything is found, the report or +push is held once with the reasons, so the agent fixes or explains before a +person sees the claim or the code ships; a second attempt for the same work +always goes through. The check takes about a second and does not replace pull +request review. Without a judgment model, the agent runs a slower review pass of its own before delivery instead. Visual proof should use genuine application, authentication, database, and diff --git a/apps/worker/src/run-task/__tests__/opencode-completion-gate-plugin-script.test.ts b/apps/worker/src/run-task/__tests__/opencode-completion-gate-plugin-script.test.ts new file mode 100644 index 0000000000..c9daaa0ab2 --- /dev/null +++ b/apps/worker/src/run-task/__tests__/opencode-completion-gate-plugin-script.test.ts @@ -0,0 +1,148 @@ +import fs from 'node:fs'; +import http from 'node:http'; +import os from 'node:os'; +import path from 'node:path'; +import { pathToFileURL } from 'node:url'; + +import { OPENCODE_COMPLETION_GATE_PLUGIN_SCRIPT } from '../opencode-completion-gate-plugin-script'; + +type ToolHooks = { + 'tool.execute.before': ( + input: { tool: string; args?: unknown }, + context?: { args?: unknown }, + ) => Promise; +}; + +describe('OPENCODE_COMPLETION_GATE_PLUGIN_SCRIPT', () => { + let tempDir: string; + let server: http.Server | undefined; + const requests: Array<{ + path: string; + auth: string | undefined; + body: unknown; + }> = []; + let respondWith: { allowed: boolean; reason?: string } = { allowed: true }; + const originalEnv = { ...process.env }; + + beforeEach(() => { + tempDir = fs.mkdtempSync( + path.join(os.tmpdir(), 'roomote-opencode-completion-gate-plugin-'), + ); + requests.length = 0; + respondWith = { allowed: true }; + process.env.ROOMOTE_COMPLETION_GATE = 'true'; + process.env.ROOMOTE_CLOUD_TOKEN = 'run-token'; + }); + + afterEach(async () => { + process.env = { ...originalEnv }; + fs.rmSync(tempDir, { recursive: true, force: true }); + await new Promise((resolve) => { + if (!server) return resolve(); + server.close(() => resolve()); + server = undefined; + }); + }); + + async function startServer(): Promise { + server = http.createServer((req, res) => { + let body = ''; + req.on('data', (chunk) => (body += chunk)); + req.on('end', () => { + requests.push({ + path: req.url ?? '', + auth: req.headers.authorization, + body: JSON.parse(body), + }); + res.setHeader('Content-Type', 'application/json'); + res.end(JSON.stringify({ result: { data: { json: respondWith } } })); + }); + }); + await new Promise((resolve) => + server!.listen(0, '127.0.0.1', resolve), + ); + const { port } = server!.address() as { port: number }; + process.env.ROOMOTE_SANDBOX_SERVER_URL = `http://127.0.0.1:${port}`; + } + + async function loadHooks(): Promise { + const pluginPath = path.join(tempDir, 'roomote-completion-gate.mjs'); + fs.writeFileSync( + pluginPath, + OPENCODE_COMPLETION_GATE_PLUGIN_SCRIPT, + 'utf8', + ); + const module = (await import( + /* @vite-ignore */ pathToFileURL(pluginPath).href + )) as { RoomoteOpenCodeCompletionGate: () => Promise }; + + return module.RoomoteOpenCodeCompletionGate(); + } + + it('fails a held tool call with the reasons from the sandbox server', async () => { + await startServer(); + respondWith = { allowed: false, reason: 'Roomote held this report.' }; + const hooks = await loadHooks(); + + await expect( + hooks['tool.execute.before']( + { tool: 'roomote_report_to_parent_session' }, + { args: { text: 'Done.' } }, + ), + ).rejects.toThrow('Roomote held this report.'); + expect(requests).toEqual([ + { + path: '/trpc/commands.checkCompletionBeforeTool', + auth: 'Bearer run-token', + body: { + json: { + tool: 'roomote_report_to_parent_session', + args: { text: 'Done.' }, + }, + }, + }, + ]); + }); + + it('lets an allowed call through and skips tools that cannot be a trigger', async () => { + await startServer(); + const hooks = await loadHooks(); + + await expect( + hooks['tool.execute.before']( + { tool: 'bash' }, + { args: { command: 'git push origin HEAD' } }, + ), + ).resolves.toBeUndefined(); + await hooks['tool.execute.before']( + { tool: 'bash' }, + { args: { command: 'pnpm vitest run' } }, + ); + await hooks['tool.execute.before']( + { tool: 'read' }, + { args: { filePath: 'a' } }, + ); + expect(requests).toHaveLength(1); + }); + + it('allows the call when the check is off or the server is unreachable', async () => { + process.env.ROOMOTE_SANDBOX_SERVER_URL = 'http://127.0.0.1:1'; + const unreachable = await loadHooks(); + + await expect( + unreachable['tool.execute.before']( + { tool: 'bash' }, + { args: { command: 'git push' } }, + ), + ).resolves.toBeUndefined(); + + await startServer(); + process.env.ROOMOTE_COMPLETION_GATE = 'false'; + const off = await loadHooks(); + await off['tool.execute.before']( + { tool: 'bash' }, + { args: { command: 'git push' } }, + ); + expect(requests).toHaveLength(0); + }); +}); diff --git a/apps/worker/src/run-task/agent-home.ts b/apps/worker/src/run-task/agent-home.ts index 31d2e3b93b..172b4376e2 100644 --- a/apps/worker/src/run-task/agent-home.ts +++ b/apps/worker/src/run-task/agent-home.ts @@ -83,6 +83,7 @@ import { SLACK_STOP_HOOK_SCRIPT } from './slack-stop-hook-script'; import { OPENCODE_SLACK_HOOKS_PLUGIN_SCRIPT } from './opencode-slack-hooks-plugin-script'; import { OPENCODE_CHATGPT_GATEWAY_PLUGIN_SCRIPT } from './opencode-chatgpt-gateway-plugin-script'; import { OPENCODE_TOOL_SAFETY_PLUGIN_SCRIPT } from './opencode-tool-safety-plugin-script'; +import { OPENCODE_COMPLETION_GATE_PLUGIN_SCRIPT } from './opencode-completion-gate-plugin-script'; import { resolveOpenCodeModelSelection } from './opencode-model'; import { getRepoLocalSkillInvocations, @@ -187,6 +188,8 @@ const ROOMOTE_OPENCODE_CHATGPT_GATEWAY_PLUGIN_FILE_NAME = 'roomote-chatgpt-gateway.js'; const ROOMOTE_OPENCODE_TOOL_SAFETY_PLUGIN_FILE_NAME = 'roomote-tool-safety.js'; +const ROOMOTE_OPENCODE_COMPLETION_GATE_PLUGIN_FILE_NAME = + 'roomote-completion-gate.js'; const ROOMOTE_OPENCODE_IDENTITY_PLUGIN_FILE_NAME = 'roomote-identity.js'; @@ -967,6 +970,10 @@ function writeOpenCodeManagedFiles(openCodeConfigDir: string): void { pluginsDir, ROOMOTE_OPENCODE_IDENTITY_PLUGIN_FILE_NAME, ); + const completionGatePluginPath = path.join( + pluginsDir, + ROOMOTE_OPENCODE_COMPLETION_GATE_PLUGIN_FILE_NAME, + ); const silenceHookPath = path.join( openCodeConfigDir, ROOMOTE_OPENCODE_SLACK_SILENCE_HOOK_FILE_NAME, @@ -989,6 +996,11 @@ function writeOpenCodeManagedFiles(openCodeConfigDir: string): void { 'utf8', ); fs.writeFileSync(identityPluginPath, OPENCODE_IDENTITY_PLUGIN_SCRIPT, 'utf8'); + fs.writeFileSync( + completionGatePluginPath, + OPENCODE_COMPLETION_GATE_PLUGIN_SCRIPT, + 'utf8', + ); fs.writeFileSync(silenceHookPath, SLACK_SILENCE_HOOK_SCRIPT, 'utf8'); fs.writeFileSync(stopHookPath, SLACK_STOP_HOOK_SCRIPT, 'utf8'); fs.chmodSync(silenceHookPath, 0o755); @@ -1416,7 +1428,7 @@ function createProofOnlyJudgeModelInstructions(): string { '', 'When `R_VISION_MODEL` is configured, the judge runs on that vision model so it can open proof screenshots directly. Otherwise it falls back to the active coding model for the task.', '', - `Delegate one focused pass to the \`${ROOMOTE_OPENCODE_JUDGE_AGENT_NAME}\` subagent with the Task tool only when a pre-delivery \`capture-visual-proof\` step for this shipped change kept screenshots or keyframes. When that step kept no images (a no-op, not-applicable, unnecessary, or blocked result), or the workflow required no proof step, do not spawn the judge. Whether the work matches the request is checked by the platform automatically when your turn ends: it compares the request, your closing report, and the diff, and sends you a follow-up only if something needs another look.`, + `Delegate one focused pass to the \`${ROOMOTE_OPENCODE_JUDGE_AGENT_NAME}\` subagent with the Task tool only when a pre-delivery \`capture-visual-proof\` step for this shipped change kept screenshots or keyframes. When that step kept no images (a no-op, not-applicable, unnecessary, or blocked result), or the workflow required no proof step, do not spawn the judge. Whether the work matches the request is checked by the platform automatically: before you report to a person, before you push or open a pull request, and when your turn ends. It compares the request, your report, the commands you ran, and the diff. If something needs another look, the tool call fails once with the reasons (fix or explain, then call it again) or you get a follow-up after the turn.`, '', 'Include in the judge brief: the plan or requested outcome, the validation results, the proof report verbatim, the path `/tmp/capture-visual-proof/diff-at-start.patch`, and the local paths of every kept screenshot and keyframe so the judge can open them.', '', diff --git a/apps/worker/src/run-task/opencode-completion-gate-plugin-script.ts b/apps/worker/src/run-task/opencode-completion-gate-plugin-script.ts new file mode 100644 index 0000000000..ccd9f998eb --- /dev/null +++ b/apps/worker/src/run-task/opencode-completion-gate-plugin-script.ts @@ -0,0 +1,90 @@ +/** + * OpenCode plugin: before a tool call that reports to a person or ships the + * work, ask the sandbox server to run the completion check. A denial fails + * the tool call with the reasons, which the agent reads as the tool result. + * Everything else, including any failure to reach the server, lets the call + * through. Written into the OpenCode plugins directory by agent-home. + */ +export const OPENCODE_COMPLETION_GATE_PLUGIN_SCRIPT = `const SANDBOX_SERVER_URL = + process.env.ROOMOTE_SANDBOX_SERVER_URL || 'http://127.0.0.1:4200'; +const CHECK_TIMEOUT_MS = 20_000; +// A cheap pre-filter; the sandbox server makes the real classification. +const CANDIDATE_TOOL = /report_to_parent_session|send_chat_reply|send_chat_message|manage_source_control|^(bash|shell)$/; +const CANDIDATE_COMMAND = /\\bgit\\b[^|;&\\n]*\\bpush\\b|\\bgh\\s+pr\\b|\\bglab\\s+mr\\b/; + +function isCandidate(tool, args) { + if (!CANDIDATE_TOOL.test(tool)) { + return false; + } + + if (tool === 'bash' || tool === 'shell') { + return CANDIDATE_COMMAND.test(String(args?.command ?? '')); + } + + return true; +} + +async function checkCompletionBeforeTool(tool, args) { + const controller = new AbortController(); + const timer = setTimeout(() => controller.abort(), CHECK_TIMEOUT_MS); + + try { + const response = await fetch( + SANDBOX_SERVER_URL + '/trpc/commands.checkCompletionBeforeTool', + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + Authorization: 'Bearer ' + process.env.ROOMOTE_CLOUD_TOKEN, + }, + // superjson envelope, as the sandbox server's tRPC expects. + body: JSON.stringify({ json: { tool, args } }), + signal: controller.signal, + }, + ); + + if (!response.ok) { + return { allowed: true }; + } + + const body = await response.json(); + return body?.result?.data?.json ?? { allowed: true }; + } finally { + clearTimeout(timer); + } +} + +export const RoomoteOpenCodeCompletionGate = async () => ({ + 'tool.execute.before': async (input, context) => { + if (process.env.ROOMOTE_COMPLETION_GATE !== 'true') { + return; + } + + const tool = typeof input?.tool === 'string' ? input.tool : ''; + const args = context?.args ?? input?.args; + + if (!isCandidate(tool, args)) { + return; + } + + let decision; + + try { + decision = await checkCompletionBeforeTool(tool, args); + } catch (error) { + process.stderr.write( + 'WARN [CompletionGate] check before ' + + tool + + ' failed; allowing the call: ' + + (error instanceof Error ? error.message : String(error)) + + '\\n', + ); + return; + } + + if (decision && decision.allowed === false && decision.reason) { + throw new Error(decision.reason); + } + }, +}); +`; diff --git a/apps/worker/src/sandbox-server/lib/harness.ts b/apps/worker/src/sandbox-server/lib/harness.ts index 93b4b9ad00..9807996630 100644 --- a/apps/worker/src/sandbox-server/lib/harness.ts +++ b/apps/worker/src/sandbox-server/lib/harness.ts @@ -314,6 +314,14 @@ export interface Harness extends EventEmitter { * Returns an unsubscribe function. */ subscribeRuntimeOutput(listener: (event: AcpMessage) => void): () => void; + /** + * Run the completion check before a tool call that reports to a person or + * ships the work. Harnesses without the check allow every call. + */ + checkCompletionBeforeTool?(input: { + tool: string; + args?: unknown; + }): Promise<{ allowed: boolean; reason?: string }>; /** * Subscribe to persisted Roomote runtime envelope events. diff --git a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts index d13a56bac1..84ee62603b 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts @@ -4,7 +4,9 @@ import os from 'node:os'; import path from 'node:path'; import { + buildCompletionGateDenial, buildCompletionGateReminder, + classifyCompletionCheckTool, clipDiffByFile, collectShippedDiff, isCompletionGateEligible, @@ -285,6 +287,67 @@ describe('isCompletionGateEligible', () => { }); }); +describe('classifyCompletionCheckTool', () => { + it('treats reporting to a person as a report trigger carrying the report text', () => { + expect( + classifyCompletionCheckTool('roomote_report_to_parent_session', { + text: 'Removed the guard. Tests pass.', + }), + ).toEqual({ trigger: 'report', report: 'Removed the guard. Tests pass.' }); + expect( + classifyCompletionCheckTool('roomote_send_chat_reply', { + message: 'Done, PR is up.', + }), + ).toEqual({ trigger: 'report', report: 'Done, PR is up.' }); + }); + + it.each([ + 'git push origin HEAD', + 'git -C packages/api push --force-with-lease', + 'gh pr create --draft --title x --body-file /tmp/b.md', + 'gh pr ready 12', + 'git add -A && git commit -m wip && git push', + ])('treats %s as shipping', (command) => { + expect(classifyCompletionCheckTool('bash', { command })).toEqual({ + trigger: 'ship', + }); + }); + + it('treats opening a pull request through the platform as shipping', () => { + expect( + classifyCompletionCheckTool('roomote_manage_source_control', { + action: 'create_pull_request', + }), + ).toEqual({ trigger: 'ship' }); + }); + + it.each([ + ['bash', { command: 'git status --short' }], + ['bash', { command: 'git fetch origin && git log --oneline -3' }], + ['bash', { command: 'pnpm vitest run' }], + [ + 'roomote_manage_source_control', + { action: 'create_pull_request_comment' }, + ], + ['read', { filePath: '/tmp/a.ts' }], + ['roomote_save_task_memory', { outcome: 'x' }], + ])('leaves %s alone', (tool, args) => { + expect(classifyCompletionCheckTool(tool, args)).toBeNull(); + }); +}); + +describe('buildCompletionGateDenial', () => { + it('tells the agent the next call for the same work goes through', () => { + const denial = buildCompletionGateDenial('ship', [ + { id: 'validationMissing', probability: 0.9 }, + ]); + + expect(denial).toContain('before shipping it'); + expect(denial).toContain('no test, type check, lint, or build was run'); + expect(denial).toContain('will not be held back a second time'); + }); +}); + describe('buildCompletionGateReminder', () => { it('names each flagged point and tells the agent the check can be wrong', () => { const reminder = buildCompletionGateReminder([ diff --git a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-bootstrap.test.ts b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-bootstrap.test.ts index a78ea75735..886cbed47d 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-bootstrap.test.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-bootstrap.test.ts @@ -944,7 +944,7 @@ describe('opencode-server bootstrap', () => { 'When that step kept no images (a no-op, not-applicable, unnecessary, or blocked result), or the workflow required no proof step, do not spawn the judge.', ); expect(instructions).toContain( - 'Whether the work matches the request is checked by the platform automatically when your turn ends', + 'Whether the work matches the request is checked by the platform automatically: before you report to a person, before you push or open a pull request, and when your turn ends', ); expect(instructions).not.toContain('delegate one focused compare pass'); }); diff --git a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts index 56a83afcec..d4c8a33916 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts @@ -715,6 +715,108 @@ describe('OpenCode harness completion check', () => { } }); + it('holds a report to a person once when flagged, then lets the retry through', async () => { + mockRequestTaskCompletionCheck.mockResolvedValue({ + status: 'flagged', + flags: [{ id: 'reportOverclaims', probability: 0.9 }], + }); + const { harness } = await startTask(); + + try { + const first = await harness.checkCompletionBeforeTool({ + tool: 'roomote_report_to_parent_session', + args: { text: 'Removed the guard and added a retry helper.' }, + }); + + expect(first.allowed).toBe(false); + expect(first.reason).toContain('before sending it'); + expect(first.reason).toContain( + 'Your report describes a code change that the diff does not contain.', + ); + // The report under check is the tool's own text. + expect(mockRequestTaskCompletionCheck.mock.calls[0]![1].report).toBe( + 'Removed the guard and added a retry helper.', + ); + + // Same work: one hold only, and no second request to the platform. + const second = await harness.checkCompletionBeforeTool({ + tool: 'roomote_report_to_parent_session', + args: { text: 'Removed the guard and added a retry helper.' }, + }); + + expect(second).toEqual({ allowed: true }); + expect(mockRequestTaskCompletionCheck).toHaveBeenCalledTimes(1); + } finally { + harness.dispose(); + } + }); + + it('checks before a push against the latest message, and does not re-check the same work at turn end', async () => { + const { client, harness, prompts, completed } = await startTask(); + + try { + // A finalized parent message is the closest thing to a report so far. + client.message.mockResolvedValueOnce( + finalMessage('msg_0', 'Tests pass; pushing now.'), + ); + await client.emit({ + type: 'message.updated', + properties: { + info: { + id: 'msg_0', + sessionID: 'ses_1', + role: 'assistant', + time: { completed: 1 }, + }, + }, + }); + + await expect( + harness.checkCompletionBeforeTool({ + tool: 'bash', + args: { command: 'git push origin HEAD' }, + }), + ).resolves.toEqual({ allowed: true }); + expect(mockRequestTaskCompletionCheck).toHaveBeenCalledTimes(1); + expect(mockRequestTaskCompletionCheck.mock.calls[0]![1].report).toBe( + 'Tests pass; pushing now.', + ); + + await completeTurn(client, 'msg_1', 'Pushed. Done.'); + await vi.waitFor(() => expect(completed()).toHaveLength(1)); + // The diff has not changed since the pre-push check. + expect(mockRequestTaskCompletionCheck).toHaveBeenCalledTimes(1); + expect(prompts).toHaveLength(1); + } finally { + harness.dispose(); + } + }); + + it('allows tool calls that are neither a report nor shipping, and everything when ineligible', async () => { + const { harness } = await startTask(); + const { ROOMOTE_COMPLETION_GATE: _gate, ...withoutGate } = GATE_ENV; + const off = await startTask(withoutGate); + + try { + await expect( + harness.checkCompletionBeforeTool({ + tool: 'bash', + args: { command: 'pnpm vitest run' }, + }), + ).resolves.toEqual({ allowed: true }); + await expect( + off.harness.checkCompletionBeforeTool({ + tool: 'bash', + args: { command: 'git push' }, + }), + ).resolves.toEqual({ allowed: true }); + expect(mockRequestTaskCompletionCheck).not.toHaveBeenCalled(); + } finally { + harness.dispose(); + off.harness.dispose(); + } + }); + it('skips the check when nothing changed or the task is ineligible', async () => { mockCollectShippedDiff.mockResolvedValue(null); const unchanged = await startTask(); diff --git a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts index 8924059de5..966ba580af 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts @@ -533,6 +533,62 @@ const FLAG_GUIDANCE: Record = { 'The diff appears to add something that should not ship: debug logging, commented-out code, a placeholder standing in for requested behavior, or a disabled test.', }; +type CompletionCheckTrigger = + /** The agent is about to tell a person the work is done. */ + | 'report' + /** The agent is about to push or open a pull request. */ + | 'ship'; + +const REPORT_TOOL_NAMES = new Set([ + 'report_to_parent_session', + 'send_chat_reply', + 'send_chat_message', +]); +const SHIP_MCP_ACTIONS = /^(create|update)_pull_request$/; +const SHIP_SHELL_COMMAND = + /\bgit\b[^|;&\n]*\bpush\b|\bgh\s+pr\s+(create|ready|edit)\b|\bglab\s+mr\s+create\b/; + +/** + * Whether a tool call is a moment to check the work: the agent reporting to + * a person, or shipping. For a report the tool's own text is the report to + * hold against the diff. MCP tools arrive flattened (`roomote_send_chat_reply`), + * so names are matched by suffix. + */ +export function classifyCompletionCheckTool( + tool: string, + args: unknown, +): { trigger: CompletionCheckTrigger; report?: string } | null { + const name = tool.trim().toLowerCase(); + const record = + args && typeof args === 'object' ? (args as Record) : {}; + + for (const reportTool of REPORT_TOOL_NAMES) { + if (name === reportTool || name.endsWith(`_${reportTool}`)) { + const report = [record.text, record.message, record.summary, record.body] + .find((value) => typeof value === 'string' && value.trim()) + ?.toString(); + + return { trigger: 'report', ...(report ? { report } : {}) }; + } + } + + if (name === 'bash' || name === 'shell') { + const command = typeof record.command === 'string' ? record.command : ''; + + return SHIP_SHELL_COMMAND.test(command) ? { trigger: 'ship' } : null; + } + + if ( + name.endsWith('manage_source_control') && + typeof record.action === 'string' && + SHIP_MCP_ACTIONS.test(record.action) + ) { + return { trigger: 'ship' }; + } + + return null; +} + /** The hidden prompt that reopens a turn the completion check flagged. */ export function buildCompletionGateReminder( flags: TaskCompletionCheckResponse['flags'], @@ -545,3 +601,23 @@ export function buildCompletionGateReminder( 'This check is a quick automated read and can be wrong. Re-read the request and your diff against each point. If a point is right, make the smallest fix, re-run the validation it affects, deliver the update the same way you delivered the change, and send a short corrected report. If a point is wrong, change nothing and say in one sentence why the work is complete as it stands. Do not restart the task, do not repeat work that is already done, and do not mention this check to the user.', ].join('\n'); } + +/** + * The error a flagged tool call fails with. The agent reads it as the tool + * result, fixes or explains, and calls the tool again; the second call is + * never denied for the same work. + */ +export function buildCompletionGateDenial( + trigger: CompletionCheckTrigger, + flags: TaskCompletionCheckResponse['flags'], +): string { + return [ + trigger === 'report' + ? 'Roomote held this report against what was asked, the commands you ran, and everything this task changed before sending it, and flagged the following:' + : 'Roomote compared what was asked, the commands you ran, and everything this task changed before shipping it, and flagged the following:', + '', + ...flags.map((flag) => `- ${FLAG_GUIDANCE[flag.id]}`), + '', + 'This check is a quick automated read and can be wrong. Re-read the request and your diff against each point. If a point is right, make the smallest fix and re-run the validation it affects. If a point is wrong, say in one sentence why the work is complete as it stands. Then call this tool again; it will not be held back a second time for the same work.', + ].join('\n'); +} diff --git a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts index a7e472a388..15e26f457c 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts @@ -20,6 +20,7 @@ import { TASK_COMPLETION_GATE_LIMITS, TERMINAL_PROVIDER_ERROR_PAYLOAD_KEY, type TaskCompletionCommand, + type TaskCompletionCheckResponse, TaskEventName, } from '@roomote/types'; import { redactSecrets } from '@roomote/communication/redact-secrets'; @@ -115,7 +116,9 @@ import { resolveOpenCodeModelSelection, } from '../../../../run-task/opencode-model'; import { + buildCompletionGateDenial, buildCompletionGateReminder, + classifyCompletionCheckTool, collectShippedDiff, isCompletionGateEligible, requestTaskCompletionCheck, @@ -1788,6 +1791,17 @@ export class OpenCodeServerHarness // The report the agent gave before the check reopened its turn. The // follow-up turn only adds a short correction, so the two are joined. private completionGateHeldReport: string | null = null; + // Work (generation plus diff) a tool call was already denied for once. The + // agent gets one hold per piece of work, then its next call goes through. + private completionGateDeniedKeys = new Set(); + // Tool calls arriving while a check runs share that one evaluation. + private completionGateInFlight: Promise<{ + verdict: TaskCompletionCheckResponse; + checkedKey: string; + } | null> | null = null; + // The parent agent's latest message text, the closest thing to a report a + // ship-time check has when the agent has not written one yet. + private latestParentAssistantText = ''; private stopHookReminderStallTimer: ReturnType | null = null; // OpenCode 1.17 emits session.status(idle) followed by session.idle for the @@ -2332,6 +2346,8 @@ export class OpenCodeServerHarness this.completionGateRequestGeneration += 1; this.completionGateHeldReport = null; this.completionGateCommands = []; + this.completionGateDeniedKeys.clear(); + this.latestParentAssistantText = ''; this.ignoreNextStopHookSessionIdle = false; this.ignoreNextQueuedDrainSessionIdle = false; this.currentWorkflowPhase = command.data.workflowPhase ?? null; @@ -5606,18 +5622,86 @@ export class OpenCodeServerHarness try { // Read before the first await: anything that moves it mid-check (a new // task, a cancel) makes this verdict stale. + const generation = this.completionGateRequestGeneration; + const evaluated = await this.evaluateCompletionGate({ + sessionId, + report, + // A stale idle can arrive while tool work is still settling; the + // genuine idle that follows runs the check against the finished diff. + skipWhenToolWorkUnsettled: true, + }); + + if (!evaluated) { + return false; + } + + const { verdict } = evaluated; + + if ( + verdict.status !== 'flagged' || + this.disposed || + this.sessionId !== sessionId || + this.completionGateRequestGeneration !== generation + ) { + return false; + } + + this.completionGateReminderCount += 1; + this.completionGateHeldReport = finalized?.text ?? null; + this.finalizedAssistantTurn = null; + this.clearVisualProofTimeout(); + await this.submitPrompt({ + text: buildCompletionGateReminder(verdict.flags), + visibleInTranscript: false, + source: 'opencode-completion-gate', + }); + // Same pairing as the closeout reminder: the session.idle that follows a + // status-sourced idle belongs to the turn that just ended. + this.ignoreNextStopHookSessionIdle = source === 'session_status'; + this.armStopHookReminderStall(sessionId); + return true; + } catch (error) { + this.logger.warn( + `OpenCode completion check failed; completing the turn without it. ${ + error instanceof Error ? error.message : String(error) + }`, + ); + return false; + } + } + + /** + * One evaluation of the completion check against the current diff, shared + * by the turn-end check and the tool-time checks. Returns null when there + * is nothing new to check: no diff, or the same work already checked in + * this request generation. Concurrent callers share one evaluation. + */ + private async evaluateCompletionGate(input: { + sessionId: string; + report: string; + skipWhenToolWorkUnsettled?: boolean; + }): Promise<{ + verdict: TaskCompletionCheckResponse; + checkedKey: string; + } | null> { + if (this.completionGateInFlight) { + return this.completionGateInFlight; + } + + const run = async () => { const generation = this.completionGateRequestGeneration; const shipped = await collectShippedDiff(this.workspacePath); const checkedKey = `${generation}:${shipped?.key}`; if (!shipped || checkedKey === this.completionGateLastCheckedKey) { - return false; + return null; } - // A stale idle can arrive while tool work is still settling; the - // genuine idle that follows runs the check against the finished diff. - if (await this.hasUnsettledToolWork(sessionId)) { - return false; + if ( + input.skipWhenToolWorkUnsettled && + (await this.hasUnsettledToolWork(input.sessionId)) + ) { + return null; } this.completionGateLastCheckedKey = checkedKey; @@ -5633,8 +5717,8 @@ export class OpenCodeServerHarness ), ); const startedAt = Date.now(); - const verdict = await requestTaskCompletionCheck(this.commandEnv, { - report: report.slice(-TASK_COMPLETION_GATE_LIMITS.reportMaxChars), + const verdict = await requestTaskCompletionCheck(this.commandEnv!, { + report: input.report.slice(-TASK_COMPLETION_GATE_LIMITS.reportMaxChars), diff: shipped.diff, diffStat: shipped.diffStat, diffTruncated: shipped.diffTruncated, @@ -5646,39 +5730,78 @@ export class OpenCodeServerHarness verdict.flags.map((flag) => flag.id).join(',') || 'none' } diffChars=${shipped.diff.length} truncated=${shipped.diffTruncated} elapsedMs=${ Date.now() - startedAt - } sessionId=${sessionId}`, + } sessionId=${input.sessionId}`, ); + return { verdict, checkedKey }; + }; + + this.completionGateInFlight = run().finally(() => { + this.completionGateInFlight = null; + }); + + return this.completionGateInFlight; + } + + /** + * The completion check at the moments that matter more than turn end: when + * the agent is about to tell a person the work is done, or about to push + * or open a pull request. Called by the sandbox's OpenCode plugin before + * the tool runs. A flagged verdict denies the call once, with the reasons + * as the tool error; the next call for the same work goes through, so the + * agent is never stuck. Anything that fails allows the call. + */ + async checkCompletionBeforeTool(input: { + tool: string; + args?: unknown; + }): Promise<{ allowed: boolean; reason?: string }> { + const sessionId = this.sessionId; + const classified = classifyCompletionCheckTool(input.tool, input.args); + + if ( + !classified || + !sessionId || + !this.commandEnv || + !isCompletionGateEligible(this.commandEnv) + ) { + return { allowed: true }; + } + + try { + const evaluated = await this.evaluateCompletionGate({ + sessionId, + report: classified.report ?? this.latestParentAssistantText, + }); + if ( - verdict.status !== 'flagged' || - this.disposed || - this.sessionId !== sessionId || - this.completionGateRequestGeneration !== generation + !evaluated || + evaluated.verdict.status !== 'flagged' || + this.completionGateDeniedKeys.has(evaluated.checkedKey) ) { - return false; + return { allowed: true }; } - this.completionGateReminderCount += 1; - this.completionGateHeldReport = finalized?.text ?? null; - this.finalizedAssistantTurn = null; - this.clearVisualProofTimeout(); - await this.submitPrompt({ - text: buildCompletionGateReminder(verdict.flags), - visibleInTranscript: false, - source: 'opencode-completion-gate', - }); - // Same pairing as the closeout reminder: the session.idle that follows a - // status-sourced idle belongs to the turn that just ended. - this.ignoreNextStopHookSessionIdle = source === 'session_status'; - this.armStopHookReminderStall(sessionId); - return true; + this.completionGateDeniedKeys.add(evaluated.checkedKey); + this.logger.info( + `OpenCode completion check held a ${classified.trigger} tool call tool=${input.tool} flags=${evaluated.verdict.flags + .map((flag) => flag.id) + .join(',')} sessionId=${sessionId}`, + ); + + return { + allowed: false, + reason: buildCompletionGateDenial( + classified.trigger, + evaluated.verdict.flags, + ), + }; } catch (error) { this.logger.warn( - `OpenCode completion check failed; completing the turn without it. ${ + `OpenCode completion check before a tool call failed; allowing the call. ${ error instanceof Error ? error.message : String(error) }`, ); - return false; + return { allowed: true }; } } @@ -5935,6 +6058,10 @@ export class OpenCodeServerHarness tokenUsage, }; + if (options?.finalizeParentTurn !== false && text.trim()) { + this.latestParentAssistantText = text; + } + this.persistedMessageIds.add(message.info.id); // Persist the turn's reasoning as one consolidated thought (before the // answer) so the transcript renders a single reasoning block, matching the diff --git a/apps/worker/src/sandbox-server/procedures/__tests__/checkCompletionBeforeTool.test.ts b/apps/worker/src/sandbox-server/procedures/__tests__/checkCompletionBeforeTool.test.ts new file mode 100644 index 0000000000..3c7b9e86dc --- /dev/null +++ b/apps/worker/src/sandbox-server/procedures/__tests__/checkCompletionBeforeTool.test.ts @@ -0,0 +1,53 @@ +import type { RunTokenContext } from '@roomote/types'; + +import { appRouter } from '../../routers'; +import type { Context } from '../../trpc'; + +function createCaller(harness: Record) { + const ctx = { + workingDirectory: '/tmp/workspace', + harness, + auth: { + runId: 1, + userId: 'user-1', + principal: 'user', + tokenType: 'run', + version: 1, + } satisfies RunTokenContext, + runId: 1, + } as unknown as Context; + + return appRouter.createCaller(ctx); +} + +describe('checkCompletionBeforeTool', () => { + it('hands the tool call to the harness and returns its decision', async () => { + const checkCompletionBeforeTool = vi.fn(async () => ({ + allowed: false, + reason: 'Roomote held this report.', + })); + const caller = createCaller({ + isConnected: true, + checkCompletionBeforeTool, + }); + + await expect( + caller.commands.checkCompletionBeforeTool({ + tool: 'roomote_send_chat_reply', + args: { message: 'Done.' }, + }), + ).resolves.toEqual({ allowed: false, reason: 'Roomote held this report.' }); + expect(checkCompletionBeforeTool).toHaveBeenCalledWith({ + tool: 'roomote_send_chat_reply', + args: { message: 'Done.' }, + }); + }); + + it('allows every call for a harness without the check', async () => { + const caller = createCaller({ isConnected: true }); + + await expect( + caller.commands.checkCompletionBeforeTool({ tool: 'bash', args: {} }), + ).resolves.toEqual({ allowed: true }); + }); +}); diff --git a/apps/worker/src/sandbox-server/procedures/checkCompletionBeforeTool.ts b/apps/worker/src/sandbox-server/procedures/checkCompletionBeforeTool.ts new file mode 100644 index 0000000000..132b9179e3 --- /dev/null +++ b/apps/worker/src/sandbox-server/procedures/checkCompletionBeforeTool.ts @@ -0,0 +1,23 @@ +import { z } from 'zod'; + +import { publicProcedure } from '../trpc'; + +const inputSchema = z.object({ + tool: z.string().min(1).max(200), + args: z.unknown().optional(), +}); + +/** + * Called by the sandbox's OpenCode plugin before a tool call that reports to + * a person or ships the work. The harness decides whether the call goes + * through; a denial carries the reasons the agent should read. + */ +export const checkCompletionBeforeTool = publicProcedure + .input(inputSchema) + .mutation(async ({ ctx, input }) => { + if (!ctx.harness.checkCompletionBeforeTool) { + return { allowed: true as const }; + } + + return ctx.harness.checkCompletionBeforeTool(input); + }); diff --git a/apps/worker/src/sandbox-server/procedures/index.ts b/apps/worker/src/sandbox-server/procedures/index.ts index a013351f5b..f2b4160615 100644 --- a/apps/worker/src/sandbox-server/procedures/index.ts +++ b/apps/worker/src/sandbox-server/procedures/index.ts @@ -18,6 +18,7 @@ export { scrubSnapshotSecrets } from './scrubSnapshotSecrets'; export { restoreScrubbedCredentials } from './restoreScrubbedCredentials'; export { prepareRepository } from './prepareRepository'; export { listRepositories } from './listRepositories'; +export { checkCompletionBeforeTool } from './checkCompletionBeforeTool'; // Queries export { getRuntimeState } from './getRuntimeState'; diff --git a/apps/worker/src/sandbox-server/routers/commands.ts b/apps/worker/src/sandbox-server/routers/commands.ts index b7d6892723..34f02cff61 100644 --- a/apps/worker/src/sandbox-server/routers/commands.ts +++ b/apps/worker/src/sandbox-server/routers/commands.ts @@ -22,6 +22,7 @@ import { restoreScrubbedCredentials, prepareRepository, listRepositories, + checkCompletionBeforeTool, } from '../procedures'; export const commandsRouter = router({ @@ -46,4 +47,5 @@ export const commandsRouter = router({ restoreScrubbedCredentials, prepareRepository, listRepositories, + checkCompletionBeforeTool, }); From 35a4a863f7041a3903562f776a8ede94479cd9a0 Mon Sep 17 00:00:00 2001 From: Matt Rubens <2600+mrubens@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:05:00 -0400 Subject: [PATCH 2/5] Check a turn that ends with an empty message against the agent's last message --- .../opencode-server-completion-gate.test.ts | 31 +++++++++++++++++++ .../lib/harnesses/opencode-server/harness.ts | 5 ++- 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts index d4c8a33916..34317dc4e8 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-server-completion-gate.test.ts @@ -817,6 +817,37 @@ describe('OpenCode harness completion check', () => { } }); + it('checks a turn that ends with an empty message against the last thing the agent said', async () => { + const { client, harness, completed } = await startTask(); + + try { + client.message.mockResolvedValueOnce( + finalMessage('msg_0', 'Removed the guard; reporting to the session.'), + ); + await client.emit({ + type: 'message.updated', + properties: { + info: { + id: 'msg_0', + sessionID: 'ses_1', + role: 'assistant', + time: { completed: 1 }, + }, + }, + }); + // The report tool call is the last action; the closing text is empty. + await completeTurn(client, 'msg_1', ''); + + await vi.waitFor(() => expect(completed()).toHaveLength(1)); + expect(mockRequestTaskCompletionCheck).toHaveBeenCalledTimes(1); + expect(mockRequestTaskCompletionCheck.mock.calls[0]![1].report).toBe( + 'Removed the guard; reporting to the session.', + ); + } finally { + harness.dispose(); + } + }); + it('skips the check when nothing changed or the task is ineligible', async () => { mockCollectShippedDiff.mockResolvedValue(null); const unchanged = await startTask(); diff --git a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts index 15e26f457c..3327ce4b4d 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/harness.ts @@ -5608,7 +5608,10 @@ export class OpenCodeServerHarness finalized: FinalizedAssistantTurn | null, source: 'session_status' | 'session_idle', ): Promise { - const report = finalized?.text.trim(); + // A delegated task reports through a tool and often ends its turn with + // no text at all, so fall back to the last thing the agent said. + const report = + finalized?.text.trim() || this.latestParentAssistantText.trim(); if ( !report || From 66a022ebf5ef6fe157cdc8836cd29560aac98cc2 Mon Sep 17 00:00:00 2001 From: Matt Rubens <2600+mrubens@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:07:22 -0400 Subject: [PATCH 3/5] Recognize the source-control tool's real pull request actions as shipping --- .../lib/harnesses/opencode-server/completion-gate.ts | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts index 966ba580af..012d978f20 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts @@ -544,7 +544,13 @@ const REPORT_TOOL_NAMES = new Set([ 'send_chat_reply', 'send_chat_message', ]); -const SHIP_MCP_ACTIONS = /^(create|update)_pull_request$/; +// The source-control tool's shipping actions; getting, commenting on, closing, +// or reopening a pull request are not. +const SHIP_MCP_ACTIONS = new Set([ + 'create_or_update_pull_request', + 'update_pull_request', + 'reopen_pull_request', +]); const SHIP_SHELL_COMMAND = /\bgit\b[^|;&\n]*\bpush\b|\bgh\s+pr\s+(create|ready|edit)\b|\bglab\s+mr\s+create\b/; @@ -581,7 +587,7 @@ export function classifyCompletionCheckTool( if ( name.endsWith('manage_source_control') && typeof record.action === 'string' && - SHIP_MCP_ACTIONS.test(record.action) + SHIP_MCP_ACTIONS.has(record.action) ) { return { trigger: 'ship' }; } From 360bf2b7529d5640ffdbd991a495f146881e27f3 Mon Sep 17 00:00:00 2001 From: Matt Rubens <2600+mrubens@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:07:52 -0400 Subject: [PATCH 4/5] Cover the source-control pull request actions in the classifier tests --- .../__tests__/opencode-completion-gate-diff.test.ts | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts index 84ee62603b..1d4087bcd8 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts @@ -313,11 +313,13 @@ describe('classifyCompletionCheckTool', () => { }); }); - it('treats opening a pull request through the platform as shipping', () => { + it.each([ + 'create_or_update_pull_request', + 'update_pull_request', + 'reopen_pull_request', + ])('treats the platform action %s as shipping', (action) => { expect( - classifyCompletionCheckTool('roomote_manage_source_control', { - action: 'create_pull_request', - }), + classifyCompletionCheckTool('roomote_manage_source_control', { action }), ).toEqual({ trigger: 'ship' }); }); @@ -329,6 +331,8 @@ describe('classifyCompletionCheckTool', () => { 'roomote_manage_source_control', { action: 'create_pull_request_comment' }, ], + ['roomote_manage_source_control', { action: 'get_pull_request' }], + ['roomote_manage_source_control', { action: 'close_pull_request' }], ['read', { filePath: '/tmp/a.ts' }], ['roomote_save_task_memory', { outcome: 'x' }], ])('leaves %s alone', (tool, args) => { From 6b283dee7c74613afb66d25c8f855d16f87fb8ad Mon Sep 17 00:00:00 2001 From: Matt Rubens <2600+mrubens@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:10:04 -0400 Subject: [PATCH 5/5] Reopening a pull request is not shipping --- .../opencode-completion-gate-diff.test.ts | 20 ++++++++++--------- .../opencode-server/completion-gate.ts | 1 - 2 files changed, 11 insertions(+), 10 deletions(-) diff --git a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts index 1d4087bcd8..0b2cc9dd7c 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/__tests__/opencode-completion-gate-diff.test.ts @@ -313,15 +313,16 @@ describe('classifyCompletionCheckTool', () => { }); }); - it.each([ - 'create_or_update_pull_request', - 'update_pull_request', - 'reopen_pull_request', - ])('treats the platform action %s as shipping', (action) => { - expect( - classifyCompletionCheckTool('roomote_manage_source_control', { action }), - ).toEqual({ trigger: 'ship' }); - }); + it.each(['create_or_update_pull_request', 'update_pull_request'])( + 'treats the platform action %s as shipping', + (action) => { + expect( + classifyCompletionCheckTool('roomote_manage_source_control', { + action, + }), + ).toEqual({ trigger: 'ship' }); + }, + ); it.each([ ['bash', { command: 'git status --short' }], @@ -333,6 +334,7 @@ describe('classifyCompletionCheckTool', () => { ], ['roomote_manage_source_control', { action: 'get_pull_request' }], ['roomote_manage_source_control', { action: 'close_pull_request' }], + ['roomote_manage_source_control', { action: 'reopen_pull_request' }], ['read', { filePath: '/tmp/a.ts' }], ['roomote_save_task_memory', { outcome: 'x' }], ])('leaves %s alone', (tool, args) => { diff --git a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts index 012d978f20..6fd9b249ba 100644 --- a/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts +++ b/apps/worker/src/sandbox-server/lib/harnesses/opencode-server/completion-gate.ts @@ -549,7 +549,6 @@ const REPORT_TOOL_NAMES = new Set([ const SHIP_MCP_ACTIONS = new Set([ 'create_or_update_pull_request', 'update_pull_request', - 'reopen_pull_request', ]); const SHIP_SHELL_COMMAND = /\bgit\b[^|;&\n]*\bpush\b|\bgh\s+pr\s+(create|ready|edit)\b|\bglab\s+mr\s+create\b/;