diff --git a/.github/workflows/desktop-e2e.yml b/.github/workflows/desktop-e2e.yml index a86ea559a71..fb673b362c8 100644 --- a/.github/workflows/desktop-e2e.yml +++ b/.github/workflows/desktop-e2e.yml @@ -84,6 +84,8 @@ jobs: - name: Run Playwright _electron smoke suite working-directory: apps/desktop run: bunx playwright test + env: + BACKGROUND_EXECUTOR_REPORT_PATH: test-results/background-executor-report.json - name: Upload test results if: failure() diff --git a/apps/desktop/e2e/background-executor.spec.ts b/apps/desktop/e2e/background-executor.spec.ts new file mode 100644 index 00000000000..4dd67c50458 --- /dev/null +++ b/apps/desktop/e2e/background-executor.spec.ts @@ -0,0 +1,741 @@ +import { execFileSync } from 'node:child_process' +import { mkdtempSync, readFileSync, writeFileSync } from 'node:fs' +import { createServer, type IncomingMessage, type Server, type ServerResponse } from 'node:http' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { + type ElectronApplication, + _electron as electron, + expect, + type Page, + test, +} from '@playwright/test' +import type { SimDesktopApi } from '@sim/desktop-bridge' +import { getErrorMessage } from '@sim/utils/errors' +import { sleep } from '@sim/utils/helpers' + +/** + * The Sim desktop app's background executor against a fixture Sim that speaks the executor's + * device protocol (register, inbox, doorbell, claim, lease, complete) the way Sim's routes do. + * The window navigates, reloads and leaves the chats while their calls run: nothing in this + * suite depends on a chat view, which is the point. Each scenario's checks land in a JSON + * report at BACKGROUND_EXECUTOR_REPORT_PATH. + */ + +const DESKTOP_DIR = fileURLToPath(new URL('..', import.meta.url)) +const WORKSPACE = 'ws-e2e' +const CHAT_A = 'chat-browser-a' +const CHAT_B = 'chat-terminal-b' +const CHAT_C = 'chat-idle-c' +const LEASE_RENEW_MS = 1_000 +const RECONCILE_MS = 2_000 + +type CallStatus = 'pending' | 'awaiting_approval' | 'running' | 'completed' | 'failed' | 'cancelled' + +interface Completion { + status: string + message: string + data?: Record + outcome: 'recorded' | 'duplicate' | 'superseded' + at: number +} + +interface FixtureCall { + toolCallId: string + toolName: string + args: Record + chatId: string + deviceId: string + status: CallStatus + issuedAt: number + claimedAt?: number + token?: string + claims: number + renewals: number + completions: Completion[] + acknowledged: boolean +} + +interface ReportCheck { + name: string + status: 'passed' | 'failed' + durationMs: number + error?: string +} + +const report: ReportCheck[] = [] + +class FixtureSim { + readonly calls = new Map() + readonly devices = new Map() + readonly requests: string[] = [] + readonly hits = new Map() + readonly streams = new Map>() + enabled = true + offline = false + droppedWhileOffline = 0 + private server: Server | null = null + origin = '' + private nextToken = 0 + + async start(): Promise { + this.server = createServer((request, response) => { + void this.handle(request, response) + }) + await new Promise((resolve) => this.server?.listen(0, '127.0.0.1', resolve)) + const address = this.server.address() + if (!address || typeof address === 'string') throw new Error('Missing fixture address') + this.origin = `http://127.0.0.1:${address.port}` + } + + async stop(): Promise { + for (const streams of this.streams.values()) for (const stream of streams) stream.destroy() + this.server?.closeAllConnections() + await new Promise((resolve) => this.server?.close(() => resolve())) + } + + reset(): void { + this.calls.clear() + this.devices.clear() + this.requests.length = 0 + this.hits.clear() + this.enabled = true + this.offline = false + this.droppedWhileOffline = 0 + } + + /** Persists a call for a device and rings its doorbell, as Sim's pre-persist and offer do. */ + issue( + deviceId: string, + chatId: string, + toolName: string, + args: Record, + status: CallStatus = 'pending' + ): string { + const toolCallId = `call-${this.calls.size + 1}-${toolName}` + this.calls.set(toolCallId, { + toolCallId, + toolName, + args, + chatId, + deviceId, + status, + issuedAt: Date.now(), + claims: 0, + renewals: 0, + completions: [], + acknowledged: false, + }) + this.ring(deviceId, status === 'awaiting_approval' ? 'approval' : 'call') + return toolCallId + } + + /** Stop from any surface: Sim settles the call and tells the device to cancel. */ + stopCall(toolCallId: string): void { + const call = this.requireCall(toolCallId) + call.status = 'cancelled' + this.ring(call.deviceId, 'cancel') + } + + approve(toolCallId: string): void { + const call = this.requireCall(toolCallId) + call.status = 'pending' + this.ring(call.deviceId, 'approval') + } + + requireCall(toolCallId: string): FixtureCall { + const call = this.calls.get(toolCallId) + if (!call) throw new Error(`No fixture call ${toolCallId}`) + return call + } + + ring(deviceId: string, reason: string): void { + for (const stream of this.streams.get(deviceId) ?? []) { + stream.write(`event: inbox_changed\ndata: ${JSON.stringify({ reason })}\n\n`) + } + } + + private async body(request: IncomingMessage): Promise> { + let text = '' + for await (const chunk of request) text += chunk.toString() + return text ? (JSON.parse(text) as Record) : {} + } + + private json(response: ServerResponse, status: number, body: unknown): void { + response.writeHead(status, { 'Content-Type': 'application/json' }) + response.end(JSON.stringify(body)) + } + + private signedIn(request: IncomingMessage): boolean { + return request.headers.cookie?.includes('better-auth.session_token=fixture') ?? false + } + + private async handle(request: IncomingMessage, response: ServerResponse): Promise { + const url = new URL(request.url ?? '/', this.origin) + const path = url.pathname + this.requests.push(`${request.method} ${path}`) + if (path.startsWith('/api/desktop/') && this.offline) { + this.droppedWhileOffline += 1 + request.socket.destroy() + return + } + if (path === '/api/auth/get-session') { + this.json( + response, + 200, + this.signedIn(request) ? { user: { id: 'user-e2e' }, session: { id: 'session-e2e' } } : null + ) + return + } + if (path.startsWith('/api/desktop/') && !this.signedIn(request)) { + this.json(response, 401, { error: 'Unauthorized' }) + return + } + if (path === '/api/desktop/devices' && request.method === 'POST') { + const body = await this.body(request) + if (this.enabled) { + this.devices.set(String(body.deviceId), { + name: String(body.name), + platform: String(body.platform), + }) + } + this.json(response, 200, { + enabled: this.enabled, + protocolVersion: 1, + leaseMs: 60_000, + leaseRenewMs: LEASE_RENEW_MS, + reconcileMs: RECONCILE_MS, + }) + return + } + const deviceId = url.searchParams.get('deviceId') + if (path === '/api/desktop/inbox/stream') { + if (!deviceId || !this.devices.has(deviceId)) { + this.json(response, 401, { error: 'Unregistered' }) + return + } + response.writeHead(200, { 'Content-Type': 'text/event-stream', 'Cache-Control': 'no-cache' }) + response.write(': connected\n\n') + const streams = this.streams.get(deviceId) ?? new Set() + streams.add(response) + this.streams.set(deviceId, streams) + request.on('close', () => streams.delete(response)) + return + } + if (path === '/api/desktop/inbox') { + if (!deviceId || !this.devices.has(deviceId)) { + this.json(response, 401, { error: 'Unregistered' }) + return + } + const items = [...this.calls.values()] + .filter((call) => call.deviceId === deviceId) + .flatMap((call) => { + if (call.status === 'pending' && !call.token) + return [ + { + kind: 'call', + toolCallId: call.toolCallId, + toolName: call.toolName, + chatId: call.chatId, + workspaceId: WORKSPACE, + createdAt: new Date(call.issuedAt).toISOString(), + }, + ] + if (call.status === 'awaiting_approval') + return [ + { + kind: 'approval_needed', + toolCallId: call.toolCallId, + toolName: call.toolName, + chatId: call.chatId, + chatTitle: 'Fix CI', + workspaceId: WORKSPACE, + summary: String((call.args.args as { command?: string } | undefined)?.command), + }, + ] + if (call.token && call.status === 'cancelled' && !call.acknowledged) + return [{ kind: 'cancel', toolCallId: call.toolCallId }] + return [] + }) + this.json(response, 200, { items }) + return + } + if (path.startsWith('/api/desktop/tool/')) { + const body = await this.body(request) + const call = this.calls.get(String(body.toolCallId)) + if (!call || call.deviceId !== body.deviceId) { + this.json(response, 404, { error: 'Desktop tool call not found' }) + return + } + if (path === '/api/desktop/tool/claim') { + if (call.status !== 'pending' || call.token) { + this.json(response, 404, { error: 'This call is no longer waiting for this device' }) + return + } + call.claims += 1 + call.status = 'running' + call.claimedAt = Date.now() + call.token = `token-${++this.nextToken}` + this.json(response, 200, { + toolName: call.toolName, + args: call.args, + chatId: call.chatId, + workspaceId: WORKSPACE, + executionToken: call.token, + }) + return + } + if (body.executionToken !== call.token) { + this.json(response, 404, { error: 'Desktop tool call not found' }) + return + } + if (path === '/api/desktop/tool/lease') { + call.renewals += 1 + if (call.status !== 'running') { + this.json(response, 410, { error: 'This call was stopped or settled. Stop running it.' }) + return + } + this.json(response, 200, { renewed: true }) + return + } + if (path === '/api/desktop/tool/complete') { + const status = String(body.status) + const settled = + status === 'success' ? 'completed' : status === 'cancelled' ? 'cancelled' : 'failed' + const outcome = + call.status === 'running' + ? 'recorded' + : call.completions.length > 0 + ? 'duplicate' + : 'superseded' + if (outcome === 'recorded') call.status = settled + if (call.status === 'cancelled') call.acknowledged = true + call.completions.push({ + status, + message: String(body.message ?? ''), + ...(body.data && typeof body.data === 'object' + ? { data: body.data as Record } + : {}), + outcome, + at: Date.now(), + }) + this.json(response, 200, { + outcome, + status: call.status === 'running' ? settled : call.status, + }) + return + } + } + if (path === '/hit') { + const chat = url.searchParams.get('chat') ?? '' + this.hits.set(chat, (this.hits.get(chat) ?? 0) + 1) + this.json(response, 200, { count: this.hits.get(chat) }) + return + } + response.writeHead(200, { + 'Content-Type': 'text/html', + 'Set-Cookie': 'better-auth.session_token=fixture; HttpOnly; SameSite=Lax; Path=/', + }) + response.end( + path === '/counter' + ? `Counter ${url.searchParams.get('chat')} + +

0

` + : `Sim fixture

${path}

` + ) + } +} + +const sim = new FixtureSim() + +async function launch(userData: string): Promise<{ app: ElectronApplication; window: Page }> { + const app = await electron.launch({ + args: [process.env.SIM_DESKTOP_E2E_MAIN ?? '.'], + cwd: DESKTOP_DIR, + env: { + ...process.env, + SIM_DESKTOP_ORIGIN: sim.origin, + SIM_DESKTOP_USER_DATA: userData, + }, + }) + const window = await app.firstWindow() + return { app, window } +} + +/** The device this app registered, once it has opened its doorbell. */ +async function registeredDevice(knownDevices: ReadonlySet = new Set()): Promise { + let deviceId: string | undefined + await expect + .poll( + () => { + deviceId = [...sim.devices.keys()].find( + (id) => !knownDevices.has(id) && (sim.streams.get(id)?.size ?? 0) > 0 + ) + return deviceId + }, + { timeout: 30_000 } + ) + .toBeTruthy() + return deviceId as string +} + +async function settled(toolCallId: string, timeout = 60_000): Promise { + await expect + .poll(() => sim.requireCall(toolCallId).completions.length, { timeout }) + .toBeGreaterThan(0) + return sim.requireCall(toolCallId).completions[0] as Completion +} + +async function check(name: string, body: () => Promise): Promise { + const startedAt = Date.now() + try { + await body() + report.push({ name, status: 'passed', durationMs: Date.now() - startedAt }) + } catch (error) { + report.push({ + name, + status: 'failed', + durationMs: Date.now() - startedAt, + error: getErrorMessage(error), + }) + throw error + } +} + +function refFor(outline: string, name: string): number { + const match = outline + .split('\n') + .find((line) => line.includes(`"${name}"`) && /\[ref=\d+\]/.test(line)) + ?.match(/\[ref=(\d+)\]/) + if (!match) throw new Error(`No reference for ${name}: ${outline}`) + return Number(match[1]) +} + +function processRunning(pattern: string): boolean { + try { + execFileSync('pgrep', ['-f', pattern]) + return true + } catch { + return false + } +} + +test.describe('background executor', () => { + let app: ElectronApplication | null = null + + test.beforeAll(async () => { + await sim.start() + }) + + test.afterEach(async () => { + await app?.close().catch(() => {}) + app = null + sim.reset() + }) + + test.afterAll(async () => { + await sim.stop() + const reportPath = process.env.BACKGROUND_EXECUTOR_REPORT_PATH + if (reportPath) { + writeFileSync( + reportPath, + JSON.stringify({ suite: 'background-executor', checks: report }, null, 2) + ) + } + }) + + test('A: two chats run browser and terminal work while the user is elsewhere and reloads', async () => { + const userData = mkdtempSync(join(tmpdir(), 'sim-executor-a-')) + const launched = await launch(userData) + app = launched.app + const window = launched.window + const deviceId = await registeredDevice() + await window.goto(`${sim.origin}/workspace/${WORKSPACE}/chat/${CHAT_C}`) + + const marker = join(userData, 'terminal-marker.txt') + const readable = join(userData, 'notes.txt') + writeFileSync(readable, 'background read fixture') + + await check('A: chat A opens its page in the background', async () => { + const opened = sim.issue(deviceId, CHAT_A, 'browser_open_url', { + url: `${sim.origin}/counter?chat=A`, + }) + const completion = await settled(opened) + expect(completion.status, completion.message).toBe('success') + }) + const opened = [...sim.calls.values()][0] as FixtureCall + const outline = (opened.completions[0]?.data?.snapshot as { outline: string }).outline + const button = refFor(outline, 'Count visit') + + const clicks = Array.from({ length: 10 }, () => + sim.issue(deviceId, CHAT_A, 'browser_click', { elementId: button }) + ) + const terminalRuns = [1, 2, 3].map((n) => + sim.issue(deviceId, CHAT_B, 'terminal', { + operation: 'run', + args: { command: `sleep 1; echo B-${n} >> '${marker}'`, waitSeconds: 30 }, + }) + ) + const localRead = sim.issue(deviceId, CHAT_B, 'read_local_file', { path: readable }) + + await window.goto(`${sim.origin}/workspace/ws-other/home`) + await window.reload() + await window.goto(`${sim.origin}/workspace/${WORKSPACE}/chat/${CHAT_C}`) + + await check('A: every call completes exactly once with its own token', async () => { + for (const id of [...clicks, ...terminalRuns, localRead]) { + const completion = await settled(id, 90_000) + const call = sim.requireCall(id) + expect(completion.status, `${call.toolName}: ${completion.message}`).toBe('success') + expect(completion.outcome).toBe('recorded') + expect(call.claims).toBe(1) + expect(call.completions).toHaveLength(1) + } + }) + + await check('A: the page saw each click exactly once', async () => { + await expect.poll(() => sim.hits.get('A')).toBe(10) + }) + + await check('A: chat B ran each command once, in order, in its own terminal', async () => { + expect(readFileSync(marker, 'utf8').trim().split('\n')).toEqual(['B-1', 'B-2', 'B-3']) + const read = sim.requireCall(localRead).completions[0] + expect(JSON.stringify(read?.data)).toContain('background read fixture') + }) + + await check("A: chat B cannot see chat A's tabs", async () => { + const listed = sim.issue(deviceId, CHAT_B, 'browser_list_tabs', {}) + const completion = await settled(listed) + expect(JSON.stringify(completion.data)).not.toContain('/counter') + }) + + await check('A: no chat view executed or reported a call', async () => { + expect(sim.requests.filter((request) => request.includes('/api/copilot/confirm'))).toEqual([]) + expect( + sim.requests.filter((request) => request.includes('/api/desktop/tool/authorize')) + ).toEqual([]) + }) + + await check('A: calls are picked up within 1.5 s at p95', async () => { + const latencies = [...sim.calls.values()] + .map((call) => (call.claimedAt ?? Number.POSITIVE_INFINITY) - call.issuedAt) + .sort((a, b) => a - b) + const p95 = latencies[Math.ceil(latencies.length * 0.95) - 1] ?? Number.POSITIVE_INFINITY + expect(p95).toBeLessThan(1_500) + }) + + await check('A: the lease is renewed while calls wait and run', async () => { + expect(Math.max(...terminalRuns.map((id) => sim.requireCall(id).renewals))).toBeGreaterThan(0) + }) + }) + + test('B: a result produced while offline is delivered once after reconnecting', async () => { + const userData = mkdtempSync(join(tmpdir(), 'sim-executor-b-')) + app = (await launch(userData)).app + const deviceId = await registeredDevice() + + const run = sim.issue(deviceId, CHAT_B, 'terminal', { + operation: 'run', + args: { command: 'sleep 2; echo offline-done', waitSeconds: 30 }, + }) + await expect.poll(() => sim.requireCall(run).claims).toBe(1) + sim.offline = true + await sleep(6_000) + + await check('B: nothing reached Sim while offline', async () => { + expect(sim.requireCall(run).completions).toHaveLength(0) + expect(sim.droppedWhileOffline).toBeGreaterThan(0) + }) + sim.offline = false + + await check('B: the result arrives once after reconnecting', async () => { + const completion = await settled(run, 60_000) + expect(completion.status).toBe('success') + expect(JSON.stringify(completion.data)).toContain('offline-done') + await sleep(3_000) + expect(sim.requireCall(run).completions).toHaveLength(1) + }) + }) + + test('C: a crash mid-command reports the outcome as unknown and never reruns it', async () => { + const userData = mkdtempSync(join(tmpdir(), 'sim-executor-c-')) + const marker = join(userData, 'crash-marker.txt') + const first = await launch(userData) + const deviceId = await registeredDevice() + + const run = sim.issue(deviceId, CHAT_B, 'terminal', { + operation: 'run', + args: { command: `echo started >> '${marker}'; sleep 40`, waitSeconds: 60 }, + }) + await expect.poll(() => readFileSafe(marker), { timeout: 30_000 }).toContain('started') + first.app.process().kill('SIGKILL') + sim.streams.get(deviceId)?.clear() + + app = (await launch(userData)).app + + await check('C: the restarted app reports the lost result as outcome unknown', async () => { + const completion = await settled(run, 60_000) + expect(completion.data).toMatchObject({ outcomeUnknown: true, doNotRetry: true }) + }) + + await check('C: the command ran exactly once', async () => { + expect(sim.requireCall(run).claims).toBe(1) + expect(readFileSafe(marker).trim().split('\n')).toEqual(['started']) + }) + }) + + test('E: Stop from another chat stops a running browser wait and terminal command', async () => { + const userData = mkdtempSync(join(tmpdir(), 'sim-executor-e-')) + app = (await launch(userData)).app + const deviceId = await registeredDevice() + + const opened = sim.issue(deviceId, CHAT_A, 'browser_open_url', { + url: `${sim.origin}/counter?chat=E`, + }) + await settled(opened) + const wait = sim.issue(deviceId, CHAT_A, 'browser_wait_for', { + text: 'never appears', + timeoutMs: 30_000, + }) + const command = 'sleep 47' + const run = sim.issue(deviceId, CHAT_B, 'terminal', { + operation: 'run', + args: { command, waitSeconds: 60 }, + }) + await expect.poll(() => sim.requireCall(wait).claims).toBe(1) + await expect.poll(() => processRunning(command), { timeout: 20_000 }).toBe(true) + + const stoppedAt = Date.now() + sim.stopCall(wait) + sim.stopCall(run) + + await check('E: both stopped calls are acknowledged within seconds', async () => { + await settled(wait, 10_000) + await settled(run, 15_000) + expect(sim.requireCall(wait).completions[0]?.outcome).toBe('superseded') + expect(sim.requireCall(run).completions[0]?.outcome).toBe('superseded') + expect((sim.requireCall(wait).completions[0]?.at ?? 0) - stoppedAt).toBeLessThan(5_000) + }) + + await check('E: the stopped command no longer runs on the machine', async () => { + await expect.poll(() => processRunning(command), { timeout: 10_000 }).toBe(false) + }) + }) + + test('F: a call waiting for approval in a background chat notifies, then runs once approved', async () => { + const userData = mkdtempSync(join(tmpdir(), 'sim-executor-f-')) + app = (await launch(userData)).app + await app.evaluate(({ Notification }) => { + const shown: Array<{ title: string; body: string }> = [] + const target = globalThis as typeof globalThis & { __shownNotifications?: typeof shown } + target.__shownNotifications = shown + Notification.prototype.show = function show(this: { title: string; body: string }) { + shown.push({ title: this.title, body: this.body }) + } + }) + const deviceId = await registeredDevice() + + const gated = sim.issue( + deviceId, + CHAT_B, + 'terminal', + { operation: 'run', args: { command: 'echo approved-run', waitSeconds: 30 } }, + 'awaiting_approval' + ) + + await check('F: the user is notified, without the command in the notification', async () => { + await expect + .poll( + () => + app?.evaluate( + () => + (globalThis as { __shownNotifications?: Array<{ body: string }> }) + .__shownNotifications ?? [] + ), + { timeout: 15_000 } + ) + .toHaveLength(1) + const shown = await app?.evaluate( + () => + (globalThis as { __shownNotifications?: Array<{ body: string }> }).__shownNotifications + ) + expect(JSON.stringify(shown)).not.toContain('approved-run') + }) + + await check('F: nothing runs before approval', async () => { + await sleep(RECONCILE_MS * 2) + expect(sim.requireCall(gated).claims).toBe(0) + }) + + sim.approve(gated) + const approvedAt = Date.now() + await check('F: the approved call is claimed promptly and runs once', async () => { + const completion = await settled(gated) + expect((sim.requireCall(gated).claimedAt ?? 0) - approvedAt).toBeLessThan(1_500) + expect(JSON.stringify(completion.data)).toContain('approved-run') + }) + }) + + test('G: only the device a turn is bound to claims its calls', async () => { + const first = await launch(mkdtempSync(join(tmpdir(), 'sim-executor-g1-'))) + const firstDevice = await registeredDevice() + const second = await launch(mkdtempSync(join(tmpdir(), 'sim-executor-g2-'))) + const secondDevice = await registeredDevice(new Set([firstDevice])) + app = first.app + + const call = sim.issue(firstDevice, CHAT_A, 'browser_list_tabs', {}) + await check('G: the bound device runs it and the other never claims it', async () => { + await settled(call) + expect(sim.requireCall(call).claims).toBe(1) + expect( + sim.requests.filter((request) => request === 'POST /api/desktop/tool/claim').length + ).toBe(1) + expect(secondDevice).not.toBe(firstDevice) + }) + await second.app.close() + }) + + test('H: a device Sim has not enabled offers no binding', async () => { + sim.enabled = false + const launched = await launch(mkdtempSync(join(tmpdir(), 'sim-executor-h-'))) + app = launched.app + await launched.window.goto(`${sim.origin}/workspace/${WORKSPACE}/chat/${CHAT_C}`) + await expect.poll(() => sim.requests.includes('POST /api/desktop/devices')).toBe(true) + + await check( + 'H: the composer gets no device, so its turns stay with the chat view', + async () => { + const device = await launched.window.evaluate(() => + ( + globalThis as typeof globalThis & { simDesktop: SimDesktopApi } + ).simDesktop.desktopExecutor?.getDevice() + ) + expect(device).toBeNull() + } + ) + + sim.enabled = true + await launched.app.close() + const enabled = await launch(mkdtempSync(join(tmpdir(), 'sim-executor-h2-'))) + app = enabled.app + const deviceId = await registeredDevice() + await enabled.window.goto(`${sim.origin}/workspace/${WORKSPACE}/chat/${CHAT_C}`) + await check('H: an enabled device offers itself for binding', async () => { + const device = await enabled.window.evaluate(() => + ( + globalThis as typeof globalThis & { simDesktop: SimDesktopApi } + ).simDesktop.desktopExecutor?.getDevice() + ) + expect(device).toEqual({ deviceId, protocolVersion: 1 }) + }) + }) +}) + +function readFileSafe(path: string): string { + try { + return readFileSync(path, 'utf8') + } catch { + return '' + } +} diff --git a/apps/desktop/src/main/browser-agent/driver.ts b/apps/desktop/src/main/browser-agent/driver.ts index c624a00beca..711fa28c7a8 100644 --- a/apps/desktop/src/main/browser-agent/driver.ts +++ b/apps/desktop/src/main/browser-agent/driver.ts @@ -938,6 +938,12 @@ export function restoreBrowserScope(scopeId: string): BrowserTabsState { }) } +/** Whether a chat's browser has a live page for a tool to act on. */ +export function hasBrowserScopeSession(scopeId: string): boolean { + const resolved = resolveDriverScopeId(scopeId) + return session.withBrowserScope(resolved, () => session.hasSession()) +} + /** Moves pending-new-chat driver and tab state to the server-issued chat id. */ export function migrateBrowserScope(fromScopeId: string, toScopeId: string): boolean { const from = resolveDriverScopeId(fromScopeId) diff --git a/apps/desktop/src/main/desktop-executor/approval-notifier.test.ts b/apps/desktop/src/main/desktop-executor/approval-notifier.test.ts new file mode 100644 index 00000000000..23118c45a02 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/approval-notifier.test.ts @@ -0,0 +1,110 @@ +import { describe, expect, it } from 'vitest' +import { createApprovalNotifier } from '@/main/desktop-executor/approval-notifier' +import type { DesktopApprovalItem } from '@/main/desktop-executor/executor' + +function approval(toolCallId: string, chatId = 'chat-b'): DesktopApprovalItem { + return { + kind: 'approval_needed', + toolCallId, + toolName: 'terminal', + chatId, + chatTitle: 'Fix CI', + workspaceId: 'ws-1', + summary: 'rm -rf build && npm test', + } +} + +function harness(options: { enabled?: boolean; focusedChatId?: string | null } = {}) { + /** Each OS notification the notifier made, and what the user would see of it now. */ + const notifications: Array<{ + options: { title: string; body: string; silent: boolean } + state: 'created' | 'shown' | 'closed' + click: () => void + }> = [] + /** Routes the app was asked to open. */ + const opened: Array = [] + let focusedChatId = options.focusedChatId ?? null + const notifier = createApprovalNotifier({ + preferences: () => ({ + notificationsEnabled: options.enabled ?? true, + notificationSounds: true, + }), + focusedChatId: () => focusedChatId, + openRoute: (route) => { + opened.push(route) + }, + createNotification: (notificationOptions) => { + let click = () => {} + const notification = { + options: notificationOptions, + state: 'created' as 'created' | 'shown' | 'closed', + get click() { + return click + }, + } + notifications.push(notification) + return { + show: () => { + notification.state = 'shown' + }, + close: () => { + notification.state = 'closed' + }, + on: (_event, listener) => { + click = listener + }, + } + }, + }) + return { + notifier, + notifications, + opened, + focus: (chatId: string | null) => { + focusedChatId = chatId + }, + } +} + +describe('approval notifications', () => { + it('notifies once per waiting call and opens its chat, naming neither chat nor command', () => { + const { notifier, notifications, opened } = harness() + + notifier.update([approval('call-1')]) + notifier.update([approval('call-1')]) + + expect(notifications).toHaveLength(1) + expect(notifications[0]?.state).toBe('shown') + expect(notifications[0]?.options.body).not.toContain('rm -rf') + expect(notifications[0]?.options.body).not.toContain('Fix CI') + notifications[0]?.click() + expect(opened).toEqual(['/workspace/ws-1/chat/chat-b']) + }) + + it('closes the notification once the call is decided', () => { + const { notifier, notifications } = harness() + notifier.update([approval('call-1')]) + + notifier.update([]) + + expect(notifications[0]?.state).toBe('closed') + }) + + it('stays quiet for the chat the user is looking at, even after they leave it', () => { + const { notifier, notifications, focus } = harness({ focusedChatId: 'chat-b' }) + notifier.update([approval('call-1')]) + focus('chat-c') + + notifier.update([approval('call-1')]) + + expect(notifications).toHaveLength(0) + }) + + it('stays quiet when notifications are switched off', () => { + const { notifier, notifications } = harness({ enabled: false }) + + notifier.update([approval('call-1')]) + + expect(notifications).toHaveLength(0) + }) +}) diff --git a/apps/desktop/src/main/desktop-executor/approval-notifier.ts b/apps/desktop/src/main/desktop-executor/approval-notifier.ts new file mode 100644 index 00000000000..ac823de1971 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/approval-notifier.ts @@ -0,0 +1,73 @@ +/** + * Tells the user when a chat they are not looking at waits for their approval. The notification + * only opens the chat at its approval card; approving happens there, where the user sees what + * they are approving. It closes itself once the call is decided, and names neither the chat nor + * the command, since a notification can show on a locked screen. + */ +import type { DesktopApprovalItem } from '@/main/desktop-executor/executor' + +interface ApprovalNotification { + show(): void + close(): void + on(event: 'click', listener: () => void): void +} + +export interface ApprovalNotifierDeps { + /** Whether the user allows desktop notifications, and with sound. */ + preferences: () => { notificationsEnabled: boolean; notificationSounds: boolean } + /** The chat the focused Sim window shows, if any; its own approvals need no notification. */ + focusedChatId: () => string | null + openRoute: (route: string | undefined) => void + createNotification: (options: { + title: string + body: string + silent: boolean + }) => ApprovalNotification | null +} + +/** + * Creates the notifier the executor feeds each inbox read's approval items to. `update` notifies + * once per newly waiting call and closes notifications for calls decided since; `clear` closes + * them all, for sign-out. + */ +export function createApprovalNotifier(deps: ApprovalNotifierDeps) { + /** Calls already brought to the user's attention, by notification or by being on screen. */ + const shown = new Map() + + return { + /** Shows a notification for each newly waiting call and closes those no longer waiting. */ + update(items: DesktopApprovalItem[]): void { + const waiting = new Set(items.map((item) => item.toolCallId)) + for (const [toolCallId, notification] of shown) { + if (waiting.has(toolCallId)) continue + notification?.close() + shown.delete(toolCallId) + } + const preferences = deps.preferences() + if (!preferences.notificationsEnabled) return + for (const item of items) { + if (shown.has(item.toolCallId)) continue + if (deps.focusedChatId() === item.chatId) { + shown.set(item.toolCallId, null) + continue + } + const notification = deps.createNotification({ + title: 'Approval needed', + body: 'A chat is waiting for your approval.', + silent: !preferences.notificationSounds, + }) + if (!notification) return + const route = item.workspaceId + ? `/workspace/${encodeURIComponent(item.workspaceId)}/chat/${encodeURIComponent(item.chatId)}` + : undefined + notification.on('click', () => deps.openRoute(route)) + notification.show() + shown.set(item.toolCallId, notification) + } + }, + clear(): void { + for (const notification of shown.values()) notification?.close() + shown.clear() + }, + } +} diff --git a/apps/desktop/src/main/desktop-executor/client.test.ts b/apps/desktop/src/main/desktop-executor/client.test.ts new file mode 100644 index 00000000000..3dcf4e08f23 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/client.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it } from 'vitest' +import { createDesktopExecutorClient, UnsendableRequestError } from '@/main/desktop-executor/client' + +describe('desktop executor client', () => { + it('refuses a request it cannot encode without sending anything', async () => { + let sent = 0 + const client = createDesktopExecutorClient({ + origin: () => 'https://sim.test', + fetch: async () => { + sent += 1 + return Response.json({ outcome: 'recorded', status: 'completed' }) + }, + deviceId: '00000000-0000-4000-8000-000000000000', + }) + const cyclic: Record = {} + cyclic.self = cyclic + + await expect( + client.complete({ + toolCallId: 'call-1', + executionToken: 'token-1', + completion: { status: 'success', message: 'done', data: cyclic }, + }) + ).rejects.toBeInstanceOf(UnsendableRequestError) + expect(sent).toBe(0) + }) + + it('never splits a character when it shortens a long completion message', async () => { + const sent: string[] = [] + const client = createDesktopExecutorClient({ + origin: () => 'https://sim.test', + fetch: async (_url, init) => { + sent.push(String(init.body)) + return Response.json({ outcome: 'recorded', status: 'completed' }) + }, + deviceId: '00000000-0000-4000-8000-000000000000', + }) + // An emoji straddles the cut point, so a cut by code units would leave half of it behind. + const message = `${'a'.repeat(9_996)}😀${'b'.repeat(10)}` + + await client.complete({ + toolCallId: 'call-1', + executionToken: 'token-1', + completion: { status: 'success', message }, + }) + + const sentMessage: string = JSON.parse(sent[0] ?? '{}').message + const loneSurrogate = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?= 500 + } +} + +/** + * A request this device could not even encode (a result holding a cycle or a BigInt). Nothing was + * sent, and sending it again cannot help. + */ +export class UnsendableRequestError extends Error { + constructor(message: string) { + super(message) + this.name = 'UnsendableRequestError' + } +} + +function encode(body: Record): string { + try { + return JSON.stringify(body) + } catch (error) { + throw new UnsendableRequestError(getErrorMessage(error)) + } +} + +type DeviceFetch = (url: string, init: RequestInit) => Promise + +export interface DesktopExecutorClient { + register(registration: DesktopDeviceRegistration): Promise + listInbox(): Promise + /** Opens the doorbell; the caller reads the event stream and aborts it to close. */ + openInboxStream(signal: AbortSignal): Promise> + claim(toolCallId: string): Promise + renewLease(toolCallId: string, executionToken: string): Promise + complete(request: DesktopCompletionRequest): Promise +} + +interface DesktopExecutorClientOptions { + origin: () => string + fetch: DeviceFetch + deviceId: string +} + +async function errorMessage(response: Response): Promise { + const body = (await response.json().catch(() => null)) as { error?: unknown } | null + return typeof body?.error === 'string' ? body.error : `HTTP ${response.status}` +} + +/** + * Creates the client for one install id. Each request times out on its own and fails with a + * {@link DeviceRequestError}, so the executor decides what to retry; responses are parsed before + * they are returned, and a malformed one is a 502. + */ +export function createDesktopExecutorClient( + options: DesktopExecutorClientOptions +): DesktopExecutorClient { + const { deviceId } = options + + async function send( + method: 'GET' | 'POST', + path: string, + body?: Record, + signal?: AbortSignal + ): Promise { + const encoded = body ? encode(body) : undefined + const timeout = AbortSignal.timeout(REQUEST_TIMEOUT_MS) + let response: Response + try { + response = await options.fetch(`${options.origin()}${path}`, { + method, + credentials: 'include', + headers: { + Accept: 'application/json', + ...(body ? { 'Content-Type': 'application/json' } : {}), + }, + ...(encoded !== undefined ? { body: encoded } : {}), + signal: signal ? AbortSignal.any([signal, timeout]) : timeout, + }) + } catch (error) { + throw new DeviceRequestError(0, getErrorMessage(error)) + } + if (!response.ok) { + throw new DeviceRequestError( + response.status, + await errorMessage(response), + parseRetryAfter(response.headers.get('retry-after')) + ) + } + return response + } + + async function json( + method: 'GET' | 'POST', + path: string, + body?: Record + ): Promise { + const response = await send(method, path, body) + return response.json().catch(() => null) + } + + function malformed(route: string): DeviceRequestError { + return new DeviceRequestError(502, `Sim returned a malformed ${route} response`) + } + + return { + async register(registration) { + const timing = parseRegistration( + await json('POST', '/api/desktop/devices', { ...registration }) + ) + if (!timing) throw malformed('registration') + return timing + }, + async listInbox() { + const items = parseInbox( + await json('GET', `/api/desktop/inbox?deviceId=${encodeURIComponent(deviceId)}`) + ) + if (!items) throw malformed('inbox') + return items + }, + async openInboxStream(signal) { + let response: Response + try { + response = await options.fetch( + `${options.origin()}/api/desktop/inbox/stream?deviceId=${encodeURIComponent(deviceId)}`, + { + method: 'GET', + credentials: 'include', + headers: { Accept: 'text/event-stream' }, + cache: 'no-store', + signal, + } + ) + } catch (error) { + throw new DeviceRequestError(0, getErrorMessage(error)) + } + if (!response.ok || !response.body) { + throw new DeviceRequestError(response.status, `Doorbell refused: HTTP ${response.status}`) + } + return response.body + }, + async claim(toolCallId) { + const claimed = parseClaim( + toolCallId, + await json('POST', '/api/desktop/tool/claim', { deviceId, toolCallId }) + ) + if (!claimed) throw malformed('claim') + return claimed + }, + async renewLease(toolCallId, executionToken) { + await send('POST', '/api/desktop/tool/lease', { deviceId, toolCallId, executionToken }) + }, + async complete({ toolCallId, executionToken, completion }) { + const outcome = parseCompletionOutcome( + await json('POST', '/api/desktop/tool/complete', { + deviceId, + toolCallId, + executionToken, + status: completion.status, + // Leaves room for the ellipsis, so a cut message still fits Sim's limit. + message: truncateAtCodePoint(completion.message, COMPLETION_MESSAGE_MAX_CHARS - 3), + ...(completion.data !== undefined ? { data: completion.data } : {}), + }) + ) + if (!outcome) throw malformed('completion') + return outcome + }, + } +} diff --git a/apps/desktop/src/main/desktop-executor/doorbell.test.ts b/apps/desktop/src/main/desktop-executor/doorbell.test.ts new file mode 100644 index 00000000000..6f6609402ea --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/doorbell.test.ts @@ -0,0 +1,166 @@ +import { describe, expect, it, vi } from 'vitest' +import { DeviceRequestError } from '@/main/desktop-executor/client' +import { InboxDoorbell, parseServerSentEvents } from '@/main/desktop-executor/doorbell' + +const encoder = new TextEncoder() + +/** A stream the test writes SSE text into, closing when told to or when its reader aborts. */ +function controllableStream(signal: AbortSignal) { + let controller: ReadableStreamDefaultController | undefined + const stream = new ReadableStream({ + start(c) { + controller = c + }, + }) + signal.addEventListener('abort', () => { + try { + controller?.error(new Error('aborted')) + } catch {} + }) + return { + stream, + write: (text: string) => controller?.enqueue(encoder.encode(text)), + end: () => controller?.close(), + } +} + +function harness(options: { staleAfterMs?: number; retryBaseMs?: number } = {}) { + const connections: ReturnType[] = [] + const failures: DeviceRequestError[] = [] + /** How many times the doorbell rang, and asked to register again. */ + const counts = { rings: 0, unregistered: 0 } + const onRing = () => { + counts.rings += 1 + } + const onUnregistered = () => { + counts.unregistered += 1 + } + const doorbell = new InboxDoorbell({ + client: { + openInboxStream: async (signal) => { + const failure = failures.shift() + if (failure) throw failure + const connection = controllableStream(signal) + connections.push(connection) + return connection.stream + }, + }, + onRing, + onUnregistered, + retryBaseMs: 5, + ...options, + }) + return { doorbell, connections, failures, counts } +} + +describe('parseServerSentEvents', () => { + it('keeps a partial event for the next chunk and skips comments', () => { + const { events, rest } = parseServerSentEvents( + ': heartbeat\n\nevent: inbox_changed\ndata: {"reason":"call"}\n\nevent: rot' + ) + expect(events).toEqual([{ event: 'inbox_changed', data: '{"reason":"call"}' }]) + expect(rest).toBe('event: rot') + }) +}) + +describe('InboxDoorbell', () => { + it('rings on connect and on every inbox change', async () => { + const { doorbell, connections, counts } = harness() + doorbell.start() + await vi.waitFor(() => expect(counts.rings).toBe(1)) + + connections[0]?.write('event: inbox_changed\ndata: {"reason":"call"}\n\n') + await vi.waitFor(() => expect(counts.rings).toBe(2)) + doorbell.stop() + }) + + it('opens a new connection when Sim rotates the stream', async () => { + const { doorbell, connections } = harness() + doorbell.start() + await vi.waitFor(() => expect(connections).toHaveLength(1)) + + connections[0]?.write('event: rotate\ndata: {}\n\n') + await vi.waitFor(() => expect(connections).toHaveLength(2)) + doorbell.stop() + }) + + it('reconnects after the server drops the stream or refuses a connection', async () => { + const { doorbell, connections, failures, counts } = harness() + failures.push(new DeviceRequestError(0, 'offline'), new DeviceRequestError(502, 'deploy')) + doorbell.start() + await vi.waitFor(() => expect(connections).toHaveLength(1)) + + connections[0]?.end() + await vi.waitFor(() => expect(connections).toHaveLength(2)) + expect(counts.rings).toBe(2) + doorbell.stop() + }) + + it('backs off when every stream ends as soon as it opens', async () => { + vi.useFakeTimers() + const opened: number[] = [] + const doorbell = new InboxDoorbell({ + client: { + openInboxStream: async () => { + opened.push(Date.now()) + return new ReadableStream({ start: (controller) => controller.close() }) + }, + }, + onRing: () => {}, + onUnregistered: () => {}, + retryBaseMs: 5, + }) + try { + doorbell.start() + await vi.advanceTimersByTimeAsync(400) + doorbell.stop() + } finally { + vi.useRealTimers() + } + + // A fixed 5 ms retry would open 80; doubling from 5 ms opens a handful. + expect(opened.length).toBeGreaterThan(2) + expect(opened.length).toBeLessThan(15) + }) + + it('replaces a connection that goes silent past the heartbeat', async () => { + const { doorbell, connections } = harness({ staleAfterMs: 30 }) + doorbell.start() + + await vi.waitFor(() => expect(connections.length).toBeGreaterThanOrEqual(2)) + doorbell.stop() + }) + + it('reports a device Sim no longer recognizes', async () => { + const { doorbell, failures, counts } = harness() + failures.push(new DeviceRequestError(401, 'unregistered')) + doorbell.start() + + await vi.waitFor(() => expect(counts.unregistered).toBeGreaterThan(0)) + doorbell.stop() + }) + + it('reconnects at once on wake, without waiting out a backoff', async () => { + const { doorbell, connections, counts } = harness({ retryBaseMs: 60_000 }) + doorbell.start() + await vi.waitFor(() => expect(connections).toHaveLength(1)) + + doorbell.wake() + + await vi.waitFor(() => expect(connections).toHaveLength(2)) + await vi.waitFor(() => expect(counts.rings).toBe(2)) + doorbell.stop() + }) + + it('starts on wake after sleep stopped it', async () => { + const { doorbell, connections } = harness({ retryBaseMs: 60_000 }) + doorbell.start() + await vi.waitFor(() => expect(connections).toHaveLength(1)) + doorbell.stop() + + doorbell.wake() + + await vi.waitFor(() => expect(connections).toHaveLength(2)) + doorbell.stop() + }) +}) diff --git a/apps/desktop/src/main/desktop-executor/doorbell.ts b/apps/desktop/src/main/desktop-executor/doorbell.ts new file mode 100644 index 00000000000..3c81c30a62f --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/doorbell.ts @@ -0,0 +1,186 @@ +/** + * The inbox doorbell: a long-lived SSE stream on which Sim hints that this device's inbox + * changed. Hints only; the inbox read they trigger is the record, and the executor also reads + * it on a timer, so a lost event costs latency, never correctness. The stream reconnects with + * backoff, opens its replacement when Sim announces a rotation, and is replaced when it goes + * silent past the server's heartbeat. + */ +import { createLogger } from '@sim/logger' +import { getErrorMessage } from '@sim/utils/errors' +import { interruptibleSleep } from '@sim/utils/helpers' +import { backoffWithJitter } from '@sim/utils/retry' +import { type DesktopExecutorClient, DeviceRequestError } from '@/main/desktop-executor/client' + +const logger = createLogger('DesktopExecutorDoorbell') + +/** Sim heartbeats every 30 s; two missed ones mean the connection is gone. */ +const STALE_STREAM_MS = 75_000 +const RECONNECT_MAX_MS = 30_000 +/** A stream that stayed open this long was healthy; its clean end starts backoff afresh. */ +const HEALTHY_STREAM_MS = 60_000 +const HANDSHAKE_TIMEOUT_MS = 15_000 + +interface DoorbellOptions { + client: Pick + /** The inbox may have changed: on every event, and on every (re)connect. */ + onRing: () => void + onUnregistered: () => void + staleAfterMs?: number + retryBaseMs?: number +} + +interface ServerSentEvent { + event: string + data: string +} + +/** Parses complete SSE blocks out of `buffer`, returning them and the unparsed remainder. */ +export function parseServerSentEvents(buffer: string): { + events: ServerSentEvent[] + rest: string +} { + const blocks = buffer.split(/\r?\n\r?\n/) + const rest = blocks.pop() ?? '' + const events: ServerSentEvent[] = [] + for (const block of blocks) { + let event = 'message' + const data: string[] = [] + for (const line of block.split(/\r?\n/)) { + if (line.startsWith(':')) continue + if (line.startsWith('event:')) event = line.slice(6).trim() + else if (line.startsWith('data:')) data.push(line.slice(5).trimStart()) + } + if (data.length > 0 || event !== 'message') events.push({ event, data: data.join('\n') }) + } + return { events, rest } +} + +export class InboxDoorbell { + private running = false + /** Which loop is current; a loop left over from before a stop ends at its next turn. */ + private loopGeneration = 0 + /** The connection was dropped on purpose; open the next one without backing off. */ + private reconnectNow = false + private connection: AbortController | null = null + private sleeper: AbortController | null = null + + constructor(private readonly options: DoorbellOptions) {} + + start(): void { + if (this.running) return + this.running = true + this.loopGeneration += 1 + void this.loop(this.loopGeneration) + } + + stop(): void { + this.running = false + this.loopGeneration += 1 + this.connection?.abort() + this.sleeper?.abort() + } + + /** + * After waking or coming back online: a stopped doorbell starts, and a running one drops its + * connection, which may be dead without knowing it, and reconnects at once. + */ + wake(): void { + if (!this.running) { + this.start() + return + } + this.reconnectNow = true + this.connection?.abort() + this.sleeper?.abort() + } + + private async loop(generation: number): Promise { + let attempt = 0 + const current = () => this.loopGeneration === generation + while (current()) { + const openedAt = Date.now() + const rotated = await this.connectOnce().then( + (result) => { + if (result === 'rotated') { + attempt = 0 + return true + } + // A stream something keeps closing right away is a failing connection, not a healthy one. + attempt = Date.now() - openedAt >= HEALTHY_STREAM_MS ? 0 : attempt + 1 + return false + }, + (error: unknown) => { + if (this.reconnectNow) { + this.reconnectNow = false + attempt = 0 + return true + } + attempt += 1 + if (error instanceof DeviceRequestError && error.unregistered) { + this.options.onUnregistered() + } + logger.info('Doorbell disconnected', { + attempt, + error: getErrorMessage(error), + }) + return false + } + ) + if (!current() || rotated) continue + this.sleeper = new AbortController() + await interruptibleSleep( + attempt === 0 + ? (this.options.retryBaseMs ?? 1_000) + : backoffWithJitter(attempt, null, { + baseMs: this.options.retryBaseMs ?? 1_000, + maxMs: RECONNECT_MAX_MS, + }), + this.sleeper.signal + ) + this.sleeper = null + } + } + + /** Holds one connection open; resolves when it ends cleanly or Sim asks for a rotation. */ + private async connectOnce(): Promise<'ended' | 'rotated'> { + const connection = new AbortController() + this.connection = connection + const staleAfterMs = this.options.staleAfterMs ?? STALE_STREAM_MS + let staleTimer: ReturnType | null = null + const touch = () => { + if (staleTimer) clearTimeout(staleTimer) + staleTimer = setTimeout(() => connection.abort(), staleAfterMs) + } + try { + // A stream whose response never starts is as dead as one that went quiet. + const handshake = setTimeout(() => connection.abort(), HANDSHAKE_TIMEOUT_MS) + const stream = await this.options.client + .openInboxStream(connection.signal) + .finally(() => clearTimeout(handshake)) + touch() + this.options.onRing() + const reader = stream.getReader() + const decoder = new TextDecoder() + let buffer = '' + for (;;) { + const { value, done } = await reader.read() + if (done) return 'ended' + touch() + buffer += decoder.decode(value, { stream: true }) + const parsed = parseServerSentEvents(buffer) + buffer = parsed.rest + for (const event of parsed.events) { + if (event.event === 'rotate') { + void reader.cancel().catch(() => {}) + return 'rotated' + } + if (event.event === 'inbox_changed') this.options.onRing() + } + } + } finally { + if (staleTimer) clearTimeout(staleTimer) + connection.abort() + if (this.connection === connection) this.connection = null + } + } +} diff --git a/apps/desktop/src/main/desktop-executor/executor.test.ts b/apps/desktop/src/main/desktop-executor/executor.test.ts new file mode 100644 index 00000000000..e512119a266 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/executor.test.ts @@ -0,0 +1,770 @@ +import type { DesktopToolCompletion } from '@sim/desktop-bridge/tool-results' +import { sleep } from '@sim/utils/helpers' +import { describe, expect, it, vi } from 'vitest' +import { + type DesktopExecutorClient, + DeviceRequestError, + UnsendableRequestError, +} from '@/main/desktop-executor/client' +import { + type DesktopApprovalItem, + DesktopExecutor, + type DesktopToolRunner, +} from '@/main/desktop-executor/executor' +import type { ExecutorJournal, JournalEntry } from '@/main/desktop-executor/journal' +import type { + ClaimedDesktopCall, + DesktopCompletionRequest, + DesktopInboxItem, +} from '@/main/desktop-executor/protocol' + +/** + * Failure modes of the executor's state machine, each against fakes of its three boundaries: + * Sim's device routes, the encrypted journal, and the tools on this machine. + */ + +interface Deferred { + promise: Promise + resolve(value: T): void + reject(error: unknown): void +} + +function deferred(): Deferred { + let resolve: (value: T) => void = () => {} + let reject: (error: unknown) => void = () => {} + const promise = new Promise((res, rej) => { + resolve = res + reject = rej + }) + return { promise, resolve, reject } +} + +function callItem( + toolCallId: string, + chatId: string, + toolName = 'browser_click' +): DesktopInboxItem { + return { kind: 'call', toolCallId, toolName, chatId, workspaceId: 'ws-1' } +} + +const DONE: DesktopToolCompletion = { status: 'success', message: 'done', data: { ok: true } } + +class FakeSim { + inbox: DesktopInboxItem[] = [] + readonly claims: string[] = [] + readonly renewals: string[] = [] + readonly completions: DesktopCompletionRequest[] = [] + claimError: ((toolCallId: string) => DeviceRequestError | null) | null = null + renewError: ((toolCallId: string) => DeviceRequestError | null) | null = null + completeErrors: DeviceRequestError[] = [] + completionOutcome: 'recorded' | 'duplicate' | 'superseded' = 'recorded' + + readonly client: DesktopExecutorClient = { + register: async () => { + throw new Error('not used') + }, + listInbox: async () => [...this.inbox], + openInboxStream: async () => { + throw new Error('not used') + }, + claim: async (toolCallId) => { + this.claims.push(toolCallId) + const error = this.claimError?.(toolCallId) + if (error) throw error + const item = this.inbox.find( + (entry): entry is Extract => + entry.kind === 'call' && entry.toolCallId === toolCallId + ) + if (!item) throw new DeviceRequestError(404, 'gone') + this.inbox = this.inbox.filter((entry) => entry !== item) + return { + toolCallId, + toolName: item.toolName, + args: { step: toolCallId }, + chatId: item.chatId, + workspaceId: item.workspaceId, + executionToken: `token-${toolCallId}`, + } satisfies ClaimedDesktopCall + }, + renewLease: async (toolCallId) => { + this.renewals.push(toolCallId) + const error = this.renewError?.(toolCallId) + if (error) throw error + }, + complete: async (request) => { + const error = this.completeErrors.shift() + if (error) throw error + this.completions.push(request) + return this.completionOutcome + }, + } +} + +class MemoryJournal implements ExecutorJournal { + readonly entries = new Map() + readonly history: JournalEntry[] = [] + /** A transition the disk refuses, as a full disk or a failed encryption would. */ + failOn: JournalEntry['state'] | null = null + async load() { + return [...this.entries.values()] + } + async put(entry: JournalEntry) { + if (entry.state === this.failOn) throw new Error('disk full') + this.entries.set(entry.toolCallId, entry) + this.history.push(entry) + } + async remove(toolCallId: string) { + this.entries.delete(toolCallId) + } + async clear() { + this.entries.clear() + } +} + +class FakeRunner implements DesktopToolRunner { + readonly started: string[] = [] + readonly cancelled: string[] = [] + readonly pending = new Map>() + /** When set, every call finishes at once with this completion. */ + immediate: DesktopToolCompletion | null = null + onStart: ((call: ClaimedDesktopCall) => void) | null = null + /** An action that finishes with its full result even after being told to stop. */ + ignoresAbort = false + + async run(call: ClaimedDesktopCall, signal: AbortSignal) { + this.started.push(call.toolCallId) + this.onStart?.(call) + if (this.immediate) return this.immediate + const result = deferred() + this.pending.set(call.toolCallId, result) + signal.addEventListener('abort', () => { + if (!this.ignoresAbort) + result.resolve({ status: 'error', message: 'This browser action was cancelled.' }) + }) + return result.promise + } + + async cancel(call: ClaimedDesktopCall) { + this.cancelled.push(call.toolCallId) + } + + finish(toolCallId: string, completion: DesktopToolCompletion = DONE) { + const result = this.pending.get(toolCallId) + if (!result) throw new Error(`${toolCallId} is not running`) + result.resolve(completion) + } +} + +function setup( + options: { leaseRenewMs?: number; maxHeldCalls?: number; deliveryAwakeLimitMs?: number } = {} +) { + /** Approval items the executor handed to the notifier, one array per inbox read. */ + const approvals: DesktopApprovalItem[][] = [] + const sim = new FakeSim() + const journal = new MemoryJournal() + const runner = new FakeRunner() + const onUnregistered = vi.fn() + const busy: boolean[] = [] + const executor = new DesktopExecutor({ + client: sim.client, + journal, + runner, + leaseRenewMs: options.leaseRenewMs ?? 60_000, + retryBaseMs: 5, + onUnregistered, + onBusyChange: (value) => busy.push(value), + onApprovals: (items) => approvals.push(items), + ...(options.maxHeldCalls ? { maxHeldCalls: options.maxHeldCalls } : {}), + ...(options.deliveryAwakeLimitMs !== undefined + ? { deliveryAwakeLimitMs: options.deliveryAwakeLimitMs } + : {}), + }) + return { sim, journal, runner, executor, onUnregistered, busy, approvals } +} + +describe('claiming', () => { + it('claims an offered call, runs it from the server record, and reports with its token', async () => { + const { sim, journal, runner, executor } = setup() + runner.immediate = DONE + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + expect(sim.completions[0]).toEqual({ + toolCallId: 'call-1', + executionToken: 'token-call-1', + completion: DONE, + }) + await vi.waitFor(() => expect(journal.entries.size).toBe(0)) + expect(executor.heldCallCount()).toBe(0) + }) + + it('claims a whole backlog at once, before any of it runs', async () => { + const { sim, runner, executor } = setup() + sim.inbox = [callItem('a-1', 'chat-a'), callItem('a-2', 'chat-a'), callItem('a-3', 'chat-a')] + + await executor.reconcile() + + expect(sim.claims).toEqual(['a-1', 'a-2', 'a-3']) + await vi.waitFor(() => expect(runner.started).toEqual(['a-1'])) + }) + + it('runs one chat in inbox order and different chats side by side', async () => { + const { sim, runner, executor } = setup() + sim.inbox = [ + callItem('a-1', 'chat-a'), + callItem('b-1', 'chat-b', 'terminal'), + callItem('a-2', 'chat-a'), + ] + + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['a-1', 'b-1'])) + + runner.finish('b-1') + runner.finish('a-1') + await vi.waitFor(() => expect(runner.started).toEqual(['a-1', 'b-1', 'a-2'])) + }) + + it('does not claim a call it already holds when the inbox lists it again', async () => { + const { sim, executor } = setup() + const item = callItem('call-1', 'chat-a') + sim.inbox = [item] + await executor.reconcile() + sim.inbox = [item] + + await executor.reconcile() + + expect(sim.claims).toEqual(['call-1']) + }) + + it.each([ + [404, 'claimed elsewhere or settled'], + [403, 'still awaiting approval'], + [409, 'turn ended'], + ])('drops a claim Sim refuses with %i (%s) without running anything', async (status) => { + const { sim, journal, runner, executor } = setup() + sim.claimError = () => new DeviceRequestError(status, 'refused') + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + expect(runner.started).toEqual([]) + expect(journal.entries.size).toBe(0) + expect(executor.heldCallCount()).toBe(0) + }) + + it('stops claiming at its in-flight cap and claims the rest once a slot frees', async () => { + const { sim, runner, executor } = setup({ maxHeldCalls: 2 }) + sim.inbox = [callItem('a-1', 'chat-a'), callItem('b-1', 'chat-b'), callItem('c-1', 'chat-c')] + + await executor.reconcile() + expect(sim.claims).toEqual(['a-1', 'b-1']) + + await vi.waitFor(() => expect(runner.started).toContain('a-1')) + runner.finish('a-1') + await vi.waitFor(() => expect(executor.heldCallCount()).toBe(1)) + await executor.reconcile() + expect(sim.claims).toEqual(['a-1', 'b-1', 'c-1']) + }) + + it('does not claim while paused for sleep', async () => { + const { sim, executor } = setup() + sim.inbox = [callItem('call-1', 'chat-a')] + executor.setPaused(true) + + await executor.reconcile() + + expect(sim.claims).toEqual([]) + }) + + it('records that a call started before the action begins', async () => { + const { sim, journal, runner, executor } = setup() + let stateAtStart: JournalEntry['state'] | undefined + runner.onStart = (call) => { + stateAtStart = journal.entries.get(call.toolCallId)?.state + } + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + await vi.waitFor(() => expect(stateAtStart).toBe('started')) + }) +}) + +describe('leases', () => { + it('renews the lease while a call waits in its chat queue and while it runs', async () => { + const { sim, runner, executor } = setup({ leaseRenewMs: 20 }) + sim.inbox = [callItem('a-1', 'chat-a'), callItem('a-2', 'chat-a')] + + await executor.reconcile() + + await vi.waitFor(() => { + expect(sim.renewals).toContain('a-1') + expect(sim.renewals).toContain('a-2') + }) + expect(runner.started).toEqual(['a-1']) + runner.finish('a-1') + await vi.waitFor(() => expect(runner.started).toEqual(['a-1', 'a-2'])) + runner.finish('a-2') + await vi.waitFor(() => expect(executor.heldCallCount()).toBe(0)) + const renewalsAfterAck = sim.renewals.length + await sleep(80) + expect(sim.renewals.length).toBe(renewalsAfterAck) + }) + + it('stops the action when Sim revokes its lease', async () => { + const { sim, runner, executor } = setup({ leaseRenewMs: 20 }) + sim.inbox = [callItem('call-1', 'chat-a')] + sim.renewError = () => new DeviceRequestError(410, 'revoked') + + await executor.reconcile() + + await vi.waitFor(() => expect(runner.cancelled).toEqual(['call-1'])) + await vi.waitFor(() => expect(sim.completions.map((c) => c.toolCallId)).toEqual(['call-1'])) + expect(executor.heldCallCount()).toBe(0) + }) +}) + +describe('reporting', () => { + it('retries a result Sim did not acknowledge, without running the action again', async () => { + const { sim, journal, runner, executor } = setup() + runner.immediate = DONE + sim.completeErrors = [ + new DeviceRequestError(0, 'offline'), + new DeviceRequestError(503, 'deploying'), + ] + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + await vi.waitFor(() => expect(journal.history.map((entry) => entry.state)).toContain('result')) + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + expect(runner.started).toEqual(['call-1']) + await vi.waitFor(() => expect(journal.entries.size).toBe(0)) + }) + + it.each(['duplicate', 'superseded'] as const)( + 'treats a %s completion as acknowledged', + async (outcome) => { + const { sim, journal, runner, executor } = setup() + runner.immediate = DONE + sim.completionOutcome = outcome + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + await vi.waitFor(() => expect(journal.entries.size).toBe(0)) + expect(sim.completions).toHaveLength(1) + } + ) +}) + +describe('stopping', () => { + it('stops a running call the inbox lists as cancelled, and acknowledges it', async () => { + const { sim, runner, executor } = setup() + sim.inbox = [callItem('call-1', 'chat-a')] + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['call-1'])) + + sim.inbox = [{ kind: 'cancel', toolCallId: 'call-1' }] + await executor.reconcile() + + await vi.waitFor(() => expect(runner.cancelled).toEqual(['call-1'])) + await vi.waitFor(() => expect(sim.completions.map((c) => c.toolCallId)).toEqual(['call-1'])) + await vi.waitFor(() => expect(executor.heldCallCount()).toBe(0)) + }) + + it('never starts a queued call that was stopped, and acknowledges the stop', async () => { + const { sim, runner, executor } = setup() + sim.inbox = [callItem('a-1', 'chat-a'), callItem('a-2', 'chat-a')] + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['a-1'])) + + sim.inbox = [{ kind: 'cancel', toolCallId: 'a-2' }] + await executor.reconcile() + await vi.waitFor(() => expect(sim.completions.map((c) => c.toolCallId)).toEqual(['a-2'])) + expect(sim.completions[0]?.completion.status).toBe('cancelled') + + runner.finish('a-1') + await vi.waitFor(() => expect(executor.heldCallCount()).toBe(0)) + expect(runner.started).toEqual(['a-1']) + }) + + it('sends nothing the stopped action produced, only the acknowledgement', async () => { + const { sim, runner, executor } = setup() + runner.ignoresAbort = true + sim.inbox = [callItem('call-1', 'chat-a', 'read_local_file')] + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['call-1'])) + + sim.inbox = [{ kind: 'cancel', toolCallId: 'call-1' }] + await executor.reconcile() + await vi.waitFor(() => expect(runner.cancelled).toEqual(['call-1'])) + runner.finish('call-1', { status: 'success', message: 'read', data: { text: 'secret' } }) + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + expect(sim.completions[0]?.completion.status).toBe('cancelled') + expect(JSON.stringify(sim.completions[0])).not.toContain('secret') + }) + + it('ignores a cancel for a call it never held', async () => { + const { sim, runner, executor } = setup() + sim.inbox = [{ kind: 'cancel', toolCallId: 'someone-else' }] + + await executor.reconcile() + + expect(runner.cancelled).toEqual([]) + expect(sim.completions).toEqual([]) + }) +}) + +describe('restarting', () => { + it('reports what each journaled call reached, and never runs any of them again', async () => { + const { sim, journal, runner, executor } = setup() + await journal.put({ toolCallId: 'claiming-1', state: 'claiming' }) + await journal.put({ toolCallId: 'claimed-1', state: 'claimed', executionToken: 't-claimed' }) + await journal.put({ toolCallId: 'started-1', state: 'started', executionToken: 't-started' }) + await journal.put({ + toolCallId: 'result-1', + state: 'result', + executionToken: 't-result', + completion: DONE, + }) + + await executor.recover() + + await vi.waitFor(() => expect(sim.completions).toHaveLength(3)) + const reported = new Map(sim.completions.map((request) => [request.toolCallId, request])) + expect(reported.get('claimed-1')?.completion.data).toMatchObject({ notStarted: true }) + expect(reported.get('claimed-1')?.executionToken).toBe('t-claimed') + expect(reported.get('started-1')?.completion.data).toMatchObject({ + outcomeUnknown: true, + doNotRetry: true, + }) + expect(reported.get('result-1')?.completion).toEqual(DONE) + expect(reported.has('claiming-1')).toBe(false) + expect(runner.started).toEqual([]) + await vi.waitFor(() => expect(journal.entries.size).toBe(0)) + }) +}) + +describe('delivery', () => { + it('retries a result whose request timed out', async () => { + const { sim, journal, executor } = setup() + sim.completeErrors = [new DeviceRequestError(408, 'request timeout')] + await journal.put({ + toolCallId: 'r-1', + state: 'result', + executionToken: 't-1', + completion: DONE, + }) + + await executor.recover() + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + }) + + it('reports a result it cannot encode as outcome unknown, once, instead of retrying it', async () => { + const { sim, journal, executor } = setup() + const complete = sim.client.complete + let attempts = 0 + sim.client.complete = async (request) => { + attempts += 1 + if (request.completion.data && 'cyclic' in request.completion.data) { + throw new UnsendableRequestError('Converting circular structure to JSON') + } + return complete(request) + } + await journal.put({ + toolCallId: 'r-1', + state: 'result', + executionToken: 't-1', + completion: { status: 'success', message: 'done', data: { cyclic: true } }, + }) + + await executor.recover() + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + expect(attempts).toBe(2) + expect(sim.completions[0]?.completion).toMatchObject({ + status: 'error', + data: { outcomeUnknown: true, doNotRetry: true }, + }) + }) + + it('holds a result Sim refuses as unregistered until the device registers again', async () => { + const { sim, journal, executor, onUnregistered } = setup() + sim.completeErrors = [new DeviceRequestError(401, 'unregistered')] + await journal.put({ + toolCallId: 'r-1', + state: 'result', + executionToken: 't-1', + completion: DONE, + }) + + await executor.recover() + await vi.waitFor(() => expect(onUnregistered).toHaveBeenCalledTimes(1)) + await sleep(40) + + expect(onUnregistered).toHaveBeenCalledTimes(1) + expect(sim.completions).toHaveLength(0) + expect(journal.entries.get('r-1')?.state).toBe('result') + + executor.resumeParked() + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + await vi.waitFor(() => expect(journal.entries.size).toBe(0)) + }) +}) + +describe('keeping the machine awake', () => { + it('is busy from the claim until Sim has the result', async () => { + const { sim, runner, executor, busy } = setup() + sim.inbox = [callItem('call-1', 'chat-a')] + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['call-1'])) + expect(busy).toEqual([true]) + + runner.finish('call-1') + + await vi.waitFor(() => expect(busy).toEqual([true, false])) + expect(sim.completions).toHaveLength(1) + }) + + it('leaves nothing of the previous session behind when sign-out lands mid-recovery', async () => { + const { sim, journal, executor } = setup() + await journal.put({ toolCallId: 'c-1', state: 'claiming' }) + await journal.put({ + toolCallId: 'r-1', + state: 'result', + executionToken: 't-1', + completion: DONE, + }) + const loaded = deferred() + const load = journal.load.bind(journal) + journal.load = async () => { + await loaded.promise + return load() + } + + const recovering = executor.recover() + const signingOut = executor.dispose() + loaded.resolve() + await Promise.all([recovering, signingOut]) + await sleep(40) + + expect(sim.completions).toEqual([]) + expect(journal.entries.size).toBe(0) + }) + + it('goes idle at sign-out and stays silent when a recovered delivery settles afterwards', async () => { + const { sim, journal, executor, busy } = setup() + const answer = deferred() + const complete = sim.client.complete + let sending = false + sim.client.complete = async (request) => { + sending = true + await answer.promise + return complete(request) + } + await journal.put({ + toolCallId: 'r-1', + state: 'result', + executionToken: 't-1', + completion: DONE, + }) + await executor.recover() + await vi.waitFor(() => expect(sending).toBe(true)) + expect(busy).toEqual([true]) + + await executor.dispose() + expect(busy).toEqual([true, false]) + answer.resolve() + await sleep(40) + + expect(busy).toEqual([true, false]) + }) + + it('lets the machine sleep once a result has kept failing to reach Sim, and keeps retrying', async () => { + const { sim, journal, executor, busy } = setup({ deliveryAwakeLimitMs: 20 }) + sim.completeErrors = Array.from({ length: 6 }, () => new DeviceRequestError(503, 'deploying')) + await journal.put({ + toolCallId: 'r-1', + state: 'result', + executionToken: 't-1', + completion: DONE, + }) + + await executor.recover() + + await vi.waitFor(() => expect(busy).toEqual([true, false])) + expect(sim.completions).toHaveLength(0) + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + expect(busy).toEqual([true, false]) + }) + + it('stays busy while a result a previous run left is still on its way to Sim', async () => { + const { sim, journal, executor, busy } = setup() + sim.completeErrors = [new DeviceRequestError(503, 'deploying')] + await journal.put({ + toolCallId: 'result-1', + state: 'result', + executionToken: 't-result', + completion: DONE, + }) + + await executor.recover() + expect(busy).toEqual([true]) + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + await vi.waitFor(() => expect(busy).toEqual([true, false])) + }) +}) + +describe('registration', () => { + it('asks to register again when Sim no longer recognizes the device', async () => { + const { sim, executor, onUnregistered } = setup() + sim.claimError = () => new DeviceRequestError(401, 'unregistered') + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + expect(onUnregistered).toHaveBeenCalled() + }) + + it('drops everything it holds on sign-out', async () => { + const { sim, journal, runner, executor } = setup() + sim.inbox = [callItem('call-1', 'chat-a')] + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['call-1'])) + + await executor.dispose() + + expect(runner.cancelled).toEqual(['call-1']) + expect(journal.entries.size).toBe(0) + expect(executor.heldCallCount()).toBe(0) + }) + + it('stops running actions at sign-out without waiting on a claim still in flight', async () => { + const { sim, journal, runner, executor } = setup() + sim.inbox = [callItem('call-1', 'chat-a')] + await executor.reconcile() + await vi.waitFor(() => expect(runner.started).toEqual(['call-1'])) + const answer = deferred() + const claim = sim.client.claim + sim.client.claim = async (toolCallId) => { + await answer.promise + return claim(toolCallId) + } + sim.inbox = [callItem('call-2', 'chat-b')] + const reading = executor.reconcile() + await vi.waitFor(() => expect(journal.entries.get('call-2')?.state).toBe('claiming')) + + const signingOut = executor.dispose() + await vi.waitFor(() => expect(runner.cancelled).toEqual(['call-1'])) + answer.resolve() + await Promise.all([reading, signingOut]) + + expect(runner.started).toEqual(['call-1']) + expect(executor.heldCallCount()).toBe(0) + expect(journal.entries.size).toBe(0) + }) + + it('still claims the calls of an inbox read whose notifier throws', async () => { + const sim = new FakeSim() + const runner = new FakeRunner() + runner.immediate = DONE + const executor = new DesktopExecutor({ + client: sim.client, + journal: new MemoryJournal(), + runner, + leaseRenewMs: 60_000, + retryBaseMs: 5, + onUnregistered: () => {}, + onApprovals: () => { + throw new Error('notifications are unavailable') + }, + }) + + sim.inbox = [callItem('call-1', 'chat-a')] + + await expect(executor.reconcile()).resolves.toBeUndefined() + await vi.waitFor(() => expect(runner.started).toEqual(['call-1'])) + }) + + it('raises no approval from an inbox read that Sim answers after sign-out', async () => { + const { sim, executor, approvals } = setup() + const answer = deferred() + const listInbox = sim.client.listInbox + sim.client.listInbox = async () => { + await answer.promise + return listInbox() + } + sim.inbox = [ + { + kind: 'approval_needed', + toolCallId: 'gated-1', + toolName: 'terminal', + chatId: 'chat-a', + chatTitle: 'Fix CI', + workspaceId: 'ws-1', + summary: 'npm publish', + }, + ] + const reading = executor.reconcile() + + const signingOut = executor.dispose() + answer.resolve() + await Promise.all([reading, signingOut]) + + expect(approvals).toEqual([]) + }) + + it('does not keep a claim that Sim answers after sign-out', async () => { + const { sim, journal, runner, executor } = setup({ leaseRenewMs: 10 }) + const answer = deferred() + const claim = sim.client.claim + sim.client.claim = async (toolCallId) => { + await answer.promise + return claim(toolCallId) + } + sim.inbox = [callItem('call-1', 'chat-a')] + const reading = executor.reconcile() + await vi.waitFor(() => expect(journal.entries.get('call-1')?.state).toBe('claiming')) + + const signingOut = executor.dispose() + answer.resolve() + await Promise.all([reading, signingOut]) + await sleep(40) + + expect(executor.heldCallCount()).toBe(0) + expect(runner.started).toEqual([]) + expect(sim.renewals).toEqual([]) + expect(journal.entries.size).toBe(0) + }) +}) + +describe('recording', () => { + it('leaves a call on offer when it cannot record the claim', async () => { + const { sim, journal, executor } = setup() + journal.failOn = 'claiming' + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + expect(sim.claims).toEqual([]) + }) + + it('never starts an action it could not record, and reports it as not run', async () => { + const { sim, journal, runner, executor } = setup() + journal.failOn = 'started' + sim.inbox = [callItem('call-1', 'chat-a')] + + await executor.reconcile() + + await vi.waitFor(() => expect(sim.completions).toHaveLength(1)) + expect(sim.completions[0]?.completion.data).toMatchObject({ notStarted: true }) + expect(runner.started).toEqual([]) + }) +}) diff --git a/apps/desktop/src/main/desktop-executor/executor.ts b/apps/desktop/src/main/desktop-executor/executor.ts new file mode 100644 index 00000000000..0d56584c027 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/executor.ts @@ -0,0 +1,545 @@ +/** + * The background executor's state machine: takes the calls Sim offers this device, runs each in + * its chat's own browser, terminal or file scope, and delivers every result until Sim + * acknowledges it, whether or not any window shows that chat. + * + * A call is claimed as soon as it is offered, then waits in a local queue per chat and surface, + * so a busy chat never lets a backlog miss Sim's pickup window. Its lease is renewed from claim + * until Sim acknowledges the result. Each step is journaled before the step it guards, so a + * restart reports a call that never started as not started, one that did as outcome unknown, and + * a finished one with its real result; none is ever run twice. + */ +import type { DesktopToolCompletion } from '@sim/desktop-bridge/tool-results' +import { createLogger } from '@sim/logger' +import { getErrorMessage } from '@sim/utils/errors' +import { sleep } from '@sim/utils/helpers' +import { backoffWithJitter } from '@sim/utils/retry' +import { + type DesktopExecutorClient, + DeviceRequestError, + UnsendableRequestError, +} from '@/main/desktop-executor/client' +import type { ExecutorJournal, JournalEntry } from '@/main/desktop-executor/journal' +import type { ClaimedDesktopCall, DesktopInboxItem } from '@/main/desktop-executor/protocol' + +const logger = createLogger('DesktopExecutor') + +/** Claimed calls this device holds at once, across every chat. */ +const DEFAULT_MAX_HELD_CALLS = 32 +/** Ceiling on the wait between attempts to deliver one result. */ +const DELIVERY_RETRY_MAX_MS = 30_000 +/** + * How long a result that keeps failing to reach Sim holds the machine awake. Delivery keeps + * retrying after this; it just stops counting as work that needs the machine awake. + */ +const DELIVERY_AWAKE_LIMIT_MS = 10 * 60_000 + +const NOT_STARTED_AFTER_RESTART = + 'Not run: this action never started, because the Sim desktop app restarted before it began. Nothing happened on the user’s computer. Do not retry it in this turn; tell the user, who can ask again.' +const OUTCOME_UNKNOWN_AFTER_RESTART = + 'The Sim desktop app restarted while this action was running, so its result was lost. It may already have taken effect: inspect the current state before repeating it, and do not retry it automatically.' +const STOPPED_BEFORE_START = 'Stopped before the Sim desktop app started this action.' +const STOPPED_WHILE_RUNNING = 'Stopped while the Sim desktop app was running this action.' +const NOT_RECORDED = + 'Not run: this action never started, because the Sim desktop app could not record it on the user’s computer first. Nothing happened on the user’s computer. Do not retry it in this turn; tell the user, who can ask again later.' +const RESULT_UNSENDABLE = + 'The action finished, but its result could not be encoded to send back. It may have taken effect: inspect the current state before repeating it, and do not retry it automatically.' +const RESULT_TOO_LARGE = + 'The action finished, but its result was too large to send back. Do not repeat a side-effecting action; inspect the current state instead.' + +/** Runs one claimed call on this machine. Implementations never throw; failures are completions. */ +export interface DesktopToolRunner { + run(call: ClaimedDesktopCall, signal: AbortSignal): Promise + /** Stops the action a running call started, through its surface's own cancel. */ + cancel(call: ClaimedDesktopCall): Promise +} + +export type DesktopApprovalItem = Extract + +export interface DesktopExecutorOptions { + client: DesktopExecutorClient + journal: ExecutorJournal + runner: DesktopToolRunner + leaseRenewMs: number + /** Called when Sim no longer recognizes this device for the current session. */ + onUnregistered: () => void + /** Every inbox read's calls that wait for the user's approval. */ + onApprovals?: (items: DesktopApprovalItem[]) => void + /** Called whenever the number of held calls changes between zero and more. */ + onBusyChange?: (busy: boolean) => void + maxHeldCalls?: number + /** First delivery retry delay; tests shorten it. */ + retryBaseMs?: number + /** How long a failing delivery keeps the machine awake; tests shorten it. */ + deliveryAwakeLimitMs?: number +} + +type HeldPhase = 'queued' | 'running' | 'reporting' + +interface HeldCall { + call: ClaimedDesktopCall + phase: HeldPhase + controller: AbortController + renewTimer: ReturnType | null + stopped: boolean +} + +function surfaceOf(toolName: string): 'browser' | 'terminal' | 'files' { + if (toolName.startsWith('browser_')) return 'browser' + if (toolName === 'terminal') return 'terminal' + return 'files' +} + +export class DesktopExecutor { + private readonly held = new Map() + /** Each chat surface's tail, so calls in one chat run in the order they were offered. */ + private readonly queues = new Map>() + private reconciling: Promise | null = null + private reconcileAgain = false + private paused = false + private disposed = false + private busy = false + /** The journal walk after a restart; sign-out waits for it before clearing the journal. */ + private recovering: Promise | null = null + /** Results a previous app run left, still on their way to Sim. */ + private readonly recoveringIds = new Set() + /** Results that have failed to reach Sim for longer than {@link DELIVERY_AWAKE_LIMIT_MS}. */ + private readonly stalledDeliveries = new Set() + /** Results Sim refused because it no longer recognized this device; sent once it registers again. */ + private readonly parked = new Map< + string, + { executionToken: string; completion: DesktopToolCompletion } + >() + + constructor(private readonly options: DesktopExecutorOptions) {} + + heldCallCount(): number { + return this.held.size + } + + /** Suspend stops new claims; held calls keep running and reporting. */ + setPaused(paused: boolean): void { + this.paused = paused + } + + /** Reports every call the journal says a previous run of the app left unfinished. */ + async recover(): Promise { + this.recovering = this.recoverEntries() + await this.recovering + } + + private async recoverEntries(): Promise { + const entries = await this.options.journal.load() + for (const entry of entries) { + // Signed out mid-recovery: the previous session's results stay unsent and are cleared. + if (this.disposed) return + if (entry.state === 'claiming') { + // Unknown whether the claim landed. If it did not, the inbox offers the call again. If it + // did, its token never reached this device, so nothing here can report it; Sim settles it + // as outcome unknown once its lease lapses, which is conservative because nothing ran. + await this.forget(entry.toolCallId) + continue + } + const completion: DesktopToolCompletion = + entry.state === 'result' + ? entry.completion + : entry.state === 'claimed' + ? { + status: 'error', + message: NOT_STARTED_AFTER_RESTART, + data: { error: NOT_STARTED_AFTER_RESTART, notStarted: true }, + } + : { + status: 'error', + message: OUTCOME_UNKNOWN_AFTER_RESTART, + data: { + error: OUTCOME_UNKNOWN_AFTER_RESTART, + outcomeUnknown: true, + doNotRetry: true, + }, + } + logger.info('Reporting a call left unfinished by the previous app run', { + toolCallId: entry.toolCallId, + state: entry.state, + }) + // Held awake like a running call: the result exists only on this machine until Sim has it. + this.recoveringIds.add(entry.toolCallId) + this.updateBusy() + void this.deliver(entry.toolCallId, entry.executionToken, completion).finally(() => { + this.recoveringIds.delete(entry.toolCallId) + this.updateBusy() + }) + } + } + + /** After registering again: sends the results parked while Sim did not recognize the device. */ + resumeParked(): void { + const parked = [...this.parked] + this.parked.clear() + for (const [toolCallId, { executionToken, completion }] of parked) { + void this.deliver(toolCallId, executionToken, completion) + } + } + + /** + * Reads the inbox and acts on it. Concurrent requests coalesce: one more read runs after the + * current one, so a doorbell that rings mid-read is never lost. + */ + reconcile(): Promise { + if (this.reconciling) { + this.reconcileAgain = true + return this.reconciling + } + const run = async () => { + try { + do { + this.reconcileAgain = false + await this.reconcileOnce() + } while (this.reconcileAgain && !this.disposed) + } catch (error) { + // Callers fire and forget; one bad read must never become an unhandled rejection. + logger.error('Desktop inbox read failed unexpectedly', { error: getErrorMessage(error) }) + } finally { + this.reconciling = null + } + } + this.reconciling = run() + return this.reconciling + } + + /** Sign-out: stops every action and forgets every call; the session that owned them is gone. */ + async dispose(): Promise { + this.disposed = true + const stopping = this.dropHeld() + // A claim still in flight sees `disposed` once Sim answers and is never held; the journal is + // cleared only after it, so its `claiming` record does not outlive the session. + await this.reconciling?.catch(() => {}) + await Promise.all([stopping, this.dropHeld()]) + await this.recovering?.catch(() => {}) + this.updateBusy() + await this.options.journal.clear() + } + + /** Releases every held call and stops the actions already running. */ + private async dropHeld(): Promise { + const held = [...this.held.values()] + for (const entry of held) this.release(entry) + await Promise.allSettled( + held + .filter((entry) => entry.phase === 'running') + .map((entry) => { + entry.controller.abort() + return this.options.runner.cancel(entry.call) + }) + ) + } + + private async reconcileOnce(): Promise { + if (this.disposed) return + let items: DesktopInboxItem[] + try { + items = await this.options.client.listInbox() + } catch (error) { + this.noteRequestFailure('Could not read the desktop inbox', error) + return + } + // Signed out while the read was in flight: its approvals and calls belong to the old account. + if (this.disposed) return + for (const item of items) { + if (item.kind === 'cancel') void this.stop(item.toolCallId, 'Stopped by the user.') + } + try { + this.options.onApprovals?.( + items.filter((item): item is DesktopApprovalItem => item.kind === 'approval_needed') + ) + } catch (error) { + // Notifications are a courtesy; the calls in this read still get claimed. + logger.warn('Could not notify about desktop approvals', { error: getErrorMessage(error) }) + } + for (const item of items) { + if (item.kind !== 'call' || this.held.has(item.toolCallId)) continue + if (this.paused || this.disposed) return + if (this.held.size >= (this.options.maxHeldCalls ?? DEFAULT_MAX_HELD_CALLS)) return + await this.claim(item.toolCallId) + } + } + + /** Writes a journal transition; false when it could not be made durable. */ + private async record(entry: JournalEntry): Promise { + try { + await this.options.journal.put(entry) + return true + } catch (error) { + logger.warn('Could not record a desktop call locally', { + toolCallId: entry.toolCallId, + state: entry.state, + error: getErrorMessage(error), + }) + return false + } + } + + private async forget(toolCallId: string): Promise { + await this.options.journal.remove(toolCallId).catch((error) => + logger.warn('Could not forget a desktop call locally', { + toolCallId, + error: getErrorMessage(error), + }) + ) + } + + private async claim(toolCallId: string): Promise { + // Unrecorded, a claim that then crashed could never be accounted for; leave it on offer. + if (!(await this.record({ toolCallId, state: 'claiming' }))) return + let call: ClaimedDesktopCall + try { + call = await this.options.client.claim(toolCallId) + } catch (error) { + await this.forget(toolCallId) + this.noteRequestFailure('Desktop call was not claimed', error, { toolCallId }) + return + } + if (this.disposed) return + // Best effort: an unrecorded token leaves `claiming`, which recovery treats conservatively. + await this.record({ toolCallId, state: 'claimed', executionToken: call.executionToken }) + const entry: HeldCall = { + call, + phase: 'queued', + controller: new AbortController(), + renewTimer: null, + stopped: false, + } + this.held.set(toolCallId, entry) + this.updateBusy() + entry.renewTimer = setInterval(() => void this.renew(entry), this.options.leaseRenewMs) + logger.info('Claimed a desktop call', { + toolCallId, + toolName: call.toolName, + chatId: call.chatId, + }) + this.enqueue(entry) + } + + private enqueue(entry: HeldCall): void { + const key = `${entry.call.chatId}:${surfaceOf(entry.call.toolName)}` + const tail = this.queues.get(key) ?? Promise.resolve() + const next = tail.then(() => this.execute(entry)) + this.queues.set(key, next) + void next.finally(() => { + if (this.queues.get(key) === next) this.queues.delete(key) + }) + } + + private async execute(entry: HeldCall): Promise { + if (entry.stopped || this.disposed) return + const { call } = entry + const recorded = await this.record({ + toolCallId: call.toolCallId, + state: 'started', + executionToken: call.executionToken, + }) + if (entry.stopped || this.disposed) return + entry.phase = 'reporting' + if (!recorded) { + // Running it unrecorded could let a crash report an action that ran as never started. + await this.deliver(call.toolCallId, call.executionToken, { + status: 'error', + message: NOT_RECORDED, + data: { error: NOT_RECORDED, notStarted: true }, + }) + this.release(entry) + return + } + entry.phase = 'running' + const completion = await this.options.runner.run(call, entry.controller.signal) + if (this.disposed || this.held.get(call.toolCallId) !== entry) return + entry.phase = 'reporting' + // A stopped call's result is only an acknowledgement: Sim already settled it, so nothing the + // action produced (page text, file contents) leaves the machine. + await this.deliver( + call.toolCallId, + call.executionToken, + entry.stopped ? { status: 'cancelled', message: STOPPED_WHILE_RUNNING } : completion + ) + this.release(entry) + } + + /** + * Delivers a result until Sim acknowledges it. Any answer Sim gives for the token acknowledges + * it: recorded, a duplicate of one already recorded, or superseded by Sim settling it first. + */ + private async deliver( + toolCallId: string, + executionToken: string, + completion: DesktopToolCompletion + ): Promise { + if (this.disposed) return + // Best effort: unrecorded, a crash reports the call from its `started` entry as outcome unknown. + await this.record({ toolCallId, state: 'result', executionToken, completion }) + const sendingSince = Date.now() + try { + await this.sendResult(toolCallId, executionToken, completion, sendingSince) + } finally { + // Whoever holds the delivery (a held call, a recovered result) updates the busy state as it + // lets go; doing it here first would flash the machine awake again in between. + this.stalledDeliveries.delete(toolCallId) + } + } + + private async sendResult( + toolCallId: string, + executionToken: string, + completion: DesktopToolCompletion, + sendingSince: number + ): Promise { + let pending = completion + for (let attempt = 1; !this.disposed; attempt++) { + try { + const outcome = await this.options.client.complete({ + toolCallId, + executionToken, + completion: pending, + }) + logger.info('Desktop call result acknowledged', { toolCallId, outcome }) + break + } catch (error) { + // Encoding failed on this machine, so nothing was sent; the same data would fail again. + if (error instanceof UnsendableRequestError && pending.data !== undefined) { + pending = { + status: 'error', + message: RESULT_UNSENDABLE, + data: { error: RESULT_UNSENDABLE, outcomeUnknown: true, doNotRetry: true }, + } + continue + } + if (!(error instanceof DeviceRequestError)) { + logger.error('Could not send a desktop call result; dropping it', { + toolCallId, + error: getErrorMessage(error), + }) + break + } + if (error.unregistered) { + // Retrying cannot help until the device registers again; the journal keeps the result. + this.parked.set(toolCallId, { executionToken, completion: pending }) + this.options.onUnregistered() + return + } + if (error.status === 413 && pending.data !== undefined) { + pending = { + status: pending.status, + message: RESULT_TOO_LARGE, + data: { error: RESULT_TOO_LARGE, resultOmitted: true }, + } + continue + } + if (!error.transient) { + logger.warn('Sim refused a desktop call result; dropping it', { + toolCallId, + status: error.status, + }) + break + } + if ( + Date.now() - sendingSince > + (this.options.deliveryAwakeLimitMs ?? DELIVERY_AWAKE_LIMIT_MS) && + !this.stalledDeliveries.has(toolCallId) + ) { + // Still retried, but no longer a reason to keep the machine awake. + this.stalledDeliveries.add(toolCallId) + this.updateBusy() + } + await sleep( + backoffWithJitter(attempt, error.retryAfterMs, { + baseMs: this.options.retryBaseMs, + maxMs: DELIVERY_RETRY_MAX_MS, + }) + ) + } + } + if (!this.disposed) await this.forget(toolCallId) + } + + private async renew(entry: HeldCall): Promise { + if (this.held.get(entry.call.toolCallId) !== entry) return + try { + await this.options.client.renewLease(entry.call.toolCallId, entry.call.executionToken) + } catch (error) { + if (error instanceof DeviceRequestError && error.status === 410) { + this.clearRenewal(entry) + await this.stop(entry.call.toolCallId, 'Sim no longer holds this call for this device.') + return + } + this.noteRequestFailure('Could not renew a desktop call lease', error, { + toolCallId: entry.call.toolCallId, + }) + } + } + + /** + * Stops a held call Sim settled without it. A queued call never starts; a running one is + * interrupted. Either way its result is still delivered, which acknowledges the stop. + */ + private async stop(toolCallId: string, reason: string): Promise { + const entry = this.held.get(toolCallId) + if (!entry || entry.stopped) return + entry.stopped = true + logger.info('Stopping a desktop call', { toolCallId, phase: entry.phase, reason }) + if (entry.phase === 'queued') { + entry.phase = 'reporting' + await this.deliver(toolCallId, entry.call.executionToken, { + status: 'cancelled', + message: STOPPED_BEFORE_START, + data: { error: STOPPED_BEFORE_START, notStarted: true }, + }) + this.release(entry) + return + } + if (entry.phase === 'running') { + entry.controller.abort() + await this.options.runner.cancel(entry.call).catch((error) => + logger.warn('Could not stop a desktop action', { + toolCallId, + error: getErrorMessage(error), + }) + ) + } + } + + private release(entry: HeldCall): void { + this.clearRenewal(entry) + if (this.held.get(entry.call.toolCallId) === entry) this.held.delete(entry.call.toolCallId) + this.updateBusy() + } + + private clearRenewal(entry: HeldCall): void { + if (entry.renewTimer) clearInterval(entry.renewTimer) + entry.renewTimer = null + } + + private updateBusy(): void { + // A disposed executor reports idle once, at dispose; a delivery that settles later must not + // speak for the executor that replaced it. + const awake = (toolCallId: string) => !this.stalledDeliveries.has(toolCallId) + const busy = + !this.disposed && ([...this.held.keys()].some(awake) || [...this.recoveringIds].some(awake)) + if (busy === this.busy) return + this.busy = busy + this.options.onBusyChange?.(busy) + } + + private noteRequestFailure( + message: string, + error: unknown, + context: Record = {} + ): void { + if (error instanceof DeviceRequestError && error.unregistered) { + this.options.onUnregistered() + } + logger.warn(message, { + ...context, + ...(error instanceof DeviceRequestError ? { status: error.status } : {}), + error: getErrorMessage(error), + }) + } +} diff --git a/apps/desktop/src/main/desktop-executor/journal.test.ts b/apps/desktop/src/main/desktop-executor/journal.test.ts new file mode 100644 index 00000000000..7087b2b468f --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/journal.test.ts @@ -0,0 +1,134 @@ +import { mkdtemp, readFile, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => import('@/test/electron-mock')) + +import { createExecutorJournal } from '@/main/desktop-executor/journal' + +function testEncryption(available = true) { + return { + isEncryptionAvailable: vi.fn(() => available), + encryptString: vi.fn((value: string) => Buffer.from(`protected:${value}`, 'utf8')), + decryptString: vi.fn((value: Buffer) => value.toString('utf8').replace(/^protected:/, '')), + } +} + +async function journalPath(): Promise { + return join(await mkdtemp(join(tmpdir(), 'sim-executor-journal-')), 'journal.json') +} + +const RESULT = { + toolCallId: 'call-1', + state: 'result' as const, + executionToken: 'token-1', + completion: { status: 'success' as const, message: 'done', data: { secret: 'page text' } }, +} + +describe('executor journal', () => { + it('survives a restart, encrypted at rest', async () => { + const filePath = await journalPath() + const encryption = testEncryption() + await createExecutorJournal(filePath, encryption).put(RESULT) + + const raw = await readFile(filePath, 'utf8') + expect(raw).not.toContain('page text') + expect(raw).not.toContain('token-1') + await expect(createExecutorJournal(filePath, encryption).load()).resolves.toEqual([RESULT]) + }) + + it('removes the file once every call is acknowledged', async () => { + const filePath = await journalPath() + const journal = createExecutorJournal(filePath, testEncryption()) + await journal.put(RESULT) + await journal.remove('call-1') + + await expect(readFile(filePath, 'utf8')).rejects.toMatchObject({ code: 'ENOENT' }) + }) + + it('applies transitions in order, so an earlier write cannot resurrect a removed call', async () => { + const filePath = await journalPath() + const encryption = testEncryption() + const journal = createExecutorJournal(filePath, encryption) + await Promise.all([ + journal.put({ toolCallId: 'call-1', state: 'claimed', executionToken: 'token-1' }), + journal.put(RESULT), + journal.remove('call-1'), + ]) + + await expect(createExecutorJournal(filePath, encryption).load()).resolves.toEqual([]) + }) + + it('writes nothing in plaintext when OS encryption is unavailable', async () => { + const filePath = await journalPath() + const journal = createExecutorJournal(filePath, testEncryption(false)) + await journal.put(RESULT) + + await expect(readFile(filePath, 'utf8')).rejects.toMatchObject({ code: 'ENOENT' }) + }) + + it('reports a transition it could not make durable', async () => { + const encryption = testEncryption() + encryption.encryptString.mockImplementation(() => { + throw new Error('keychain locked') + }) + + await expect( + createExecutorJournal(await journalPath(), encryption).put(RESULT) + ).rejects.toThrow('keychain locked') + }) + + it('keeps results within what a restart can read back, dropping the data of the excess', async () => { + const filePath = await journalPath() + const encryption = testEncryption() + const journal = createExecutorJournal(filePath, encryption) + const screenshot = 'x'.repeat(20 * 1024 * 1024) + for (const toolCallId of ['call-1', 'call-2']) { + await journal.put({ + toolCallId, + state: 'result', + executionToken: `token-${toolCallId}`, + completion: { status: 'success', message: 'done', data: { screenshot } }, + }) + } + + const restored = await createExecutorJournal(filePath, encryption).load() + expect(restored).toHaveLength(2) + const kept = restored.filter( + (entry) => entry.state === 'result' && entry.completion.data?.screenshot === screenshot + ) + expect(kept).toHaveLength(1) + expect(JSON.stringify(restored)).toContain('resultOmitted') + }) + + it('budgets results by their encoded size, not their length in characters', async () => { + const filePath = await journalPath() + const encryption = testEncryption() + const journal = createExecutorJournal(filePath, encryption) + // Three bytes per character: within the budget by length, three times over it in bytes. + const pageText = '€'.repeat(6 * 1024 * 1024) + for (const toolCallId of ['call-1', 'call-2', 'call-3']) { + await journal.put({ + toolCallId, + state: 'result', + executionToken: `token-${toolCallId}`, + completion: { status: 'success', message: 'done', data: { pageText } }, + }) + } + + const restored = await createExecutorJournal(filePath, encryption).load() + expect(restored).toHaveLength(3) + const kept = restored.filter( + (entry) => entry.state === 'result' && entry.completion.data?.pageText === pageText + ) + expect(kept).toHaveLength(1) + }) + + it('starts empty from a corrupt or foreign file instead of failing', async () => { + const filePath = await journalPath() + await writeFile(filePath, '{"version":1,"ciphertext":"not-base64-json"}') + + await expect(createExecutorJournal(filePath, testEncryption()).load()).resolves.toEqual([]) + }) +}) diff --git a/apps/desktop/src/main/desktop-executor/journal.ts b/apps/desktop/src/main/desktop-executor/journal.ts new file mode 100644 index 00000000000..81cbfc630e8 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/journal.ts @@ -0,0 +1,205 @@ +/** + * The background executor's durable outbox: one entry per call this device took, advanced before + * each step it guards, so a restart knows exactly how far each call got. + * + * - `claiming`: a claim request is in flight; nothing ran. + * - `claimed`: this device owns the call (its token is recorded) but has not started it. + * - `started`: the action may have begun on this machine. + * - `result`: the action finished; its completion is waiting for Sim to acknowledge it. + * + * An acknowledged entry is removed. The file is encrypted with Electron safeStorage and written + * atomically, like every other account-bearing store in userData; without OS encryption nothing + * is written and the journal lives in memory only. + */ +import type { DesktopToolCompletion } from '@sim/desktop-bridge/tool-results' +import { createLogger } from '@sim/logger' +import { getErrorMessage } from '@sim/utils/errors' +import { isRecordLike } from '@sim/utils/object' +import { safeStorage } from 'electron' +import { + FileResourceLimitError, + readFileWithinLimit, + removeFileIfPresent, + writeJsonFileAtomically, +} from '@/main/atomic-json-file' + +const logger = createLogger('DesktopExecutorJournal') + +const JOURNAL_VERSION = 1 +/** Results can carry a screenshot, so the bound is the size of a few of them. */ +const MAX_JOURNAL_BYTES = 64 * 1024 * 1024 +/** + * Result data kept on disk across all entries, in UTF-8 bytes. Encryption and base64 grow it by + * about a third, so a journal within this budget always reads back under {@link MAX_JOURNAL_BYTES}. + */ +const MAX_PERSISTED_RESULT_BYTES = 32 * 1024 * 1024 +const RESULT_NOT_KEPT = + 'The action finished, but its result was too large to keep on this computer. Do not repeat a side-effecting action; inspect the current state instead.' + +/** + * The entries as written to disk: a result whose data would push the journal past its budget is + * kept as finished without its data, so the file never grows past what a restart can read. + */ +function boundedEntries(entries: Map): JournalEntry[] { + let budget = MAX_PERSISTED_RESULT_BYTES + return [...entries.values()].map((entry) => { + if (entry.state !== 'result' || entry.completion.data === undefined) return entry + const size = Buffer.byteLength(JSON.stringify(entry.completion.data), 'utf8') + if (size <= budget) { + budget -= size + return entry + } + return { + ...entry, + completion: { + status: entry.completion.status, + message: RESULT_NOT_KEPT, + data: { error: RESULT_NOT_KEPT, resultOmitted: true }, + }, + } + }) +} + +export type JournalEntry = + | { toolCallId: string; state: 'claiming' } + | { toolCallId: string; state: 'claimed' | 'started'; executionToken: string } + | { + toolCallId: string + state: 'result' + executionToken: string + completion: DesktopToolCompletion + } + +export interface ExecutorJournal { + load(): Promise + put(entry: JournalEntry): Promise + remove(toolCallId: string): Promise + clear(): Promise +} + +interface EncryptionProvider { + isEncryptionAvailable(): boolean + encryptString(value: string): Buffer + decryptString(value: Buffer): string +} + +function encryptionAvailable(encryption: EncryptionProvider): boolean { + try { + return encryption.isEncryptionAvailable() + } catch { + return false + } +} + +function isCompletion(value: unknown): value is DesktopToolCompletion { + return ( + isRecordLike(value) && + (value.status === 'success' || value.status === 'error' || value.status === 'cancelled') && + typeof value.message === 'string' && + (value.data === undefined || isRecordLike(value.data)) + ) +} + +function parseEntry(value: unknown): JournalEntry | null { + if (!isRecordLike(value) || typeof value.toolCallId !== 'string' || !value.toolCallId) return null + const { toolCallId } = value + if (value.state === 'claiming') return { toolCallId, state: 'claiming' } + if (typeof value.executionToken !== 'string' || !value.executionToken) return null + const { executionToken } = value + if (value.state === 'claimed' || value.state === 'started') { + return { toolCallId, state: value.state, executionToken } + } + if (value.state === 'result' && isCompletion(value.completion)) { + return { toolCallId, state: 'result', executionToken, completion: value.completion } + } + return null +} + +export function createExecutorJournal( + filePath: string, + encryption: EncryptionProvider = safeStorage +): ExecutorJournal { + const entries = new Map() + let mutationTail = Promise.resolve() + + const enqueue = (operation: () => Promise): Promise => { + const result = mutationTail.then(operation) + mutationTail = result.catch(() => undefined) + return result + } + + /** + * Rewrites the whole journal. A failure rejects, so the caller knows the transition is not + * durable; the in-memory journal stays current and the next transition rewrites it all. + */ + const persist = async (): Promise => { + if (!encryptionAvailable(encryption)) return + if (entries.size === 0) { + await removeFileIfPresent(filePath) + return + } + const payload = JSON.stringify({ version: JOURNAL_VERSION, entries: boundedEntries(entries) }) + const ciphertext = encryption.encryptString(payload).toString('base64') + await writeJsonFileAtomically(filePath, { version: JOURNAL_VERSION, ciphertext }) + } + + return { + async load() { + await enqueue(async () => { + entries.clear() + if (!encryptionAvailable(encryption)) return + let raw: Buffer + try { + raw = await readFileWithinLimit(filePath, MAX_JOURNAL_BYTES) + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return + logger.warn('Could not read the executor journal', { + error: error instanceof FileResourceLimitError ? 'too large' : getErrorMessage(error), + }) + return + } + try { + const envelope = JSON.parse(raw.toString('utf8')) as unknown + if ( + !isRecordLike(envelope) || + envelope.version !== JOURNAL_VERSION || + typeof envelope.ciphertext !== 'string' + ) { + return + } + const payload = JSON.parse( + encryption.decryptString(Buffer.from(envelope.ciphertext, 'base64')) + ) as unknown + if (!isRecordLike(payload) || !Array.isArray(payload.entries)) return + for (const candidate of payload.entries) { + const entry = parseEntry(candidate) + if (entry) entries.set(entry.toolCallId, entry) + } + } catch (error) { + logger.warn('Discarding an unreadable executor journal', { + error: getErrorMessage(error), + }) + } + }) + return [...entries.values()] + }, + put(entry) { + return enqueue(async () => { + entries.set(entry.toolCallId, entry) + await persist() + }) + }, + remove(toolCallId) { + return enqueue(async () => { + if (!entries.delete(toolCallId)) return + await persist() + }) + }, + clear() { + return enqueue(async () => { + entries.clear() + await removeFileIfPresent(filePath) + }) + }, + } +} diff --git a/apps/desktop/src/main/desktop-executor/protocol.ts b/apps/desktop/src/main/desktop-executor/protocol.ts new file mode 100644 index 00000000000..34b162dcaf0 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/protocol.ts @@ -0,0 +1,177 @@ +/** + * The device side of Sim's background executor protocol (`/api/desktop/devices`, `/inbox`, + * `/inbox/stream`, `/tool/claim`, `/tool/lease`, `/tool/complete`). Sim's contracts are the + * source of truth; responses are parsed defensively here because a malformed one must never + * reach a tool. + */ +import { isDesktopScopeId } from '@sim/desktop-bridge' +import type { DesktopToolCompletion } from '@sim/desktop-bridge/tool-results' +import { isRecordLike } from '@sim/utils/object' + +/** The executor protocol version this build speaks. */ +export const DESKTOP_EXECUTOR_PROTOCOL_VERSION = 1 + +/** Sim refuses a longer completion message, so longer ones are cut before they are sent. */ +export const COMPLETION_MESSAGE_MAX_CHARS = 10_000 + +export interface DesktopDeviceRegistration { + deviceId: string + name: string + appVersion: string + platform: string + capabilities: { executor: number; browser: boolean; terminal: boolean; localFiles: boolean } +} + +/** Sim's answer to a registration. `enabled: false` means no new turn will bind to this device. */ +export interface DesktopExecutorTiming { + enabled: boolean + protocolVersion: number + leaseMs: number + leaseRenewMs: number + reconcileMs: number +} + +export type DesktopInboxItem = + | { + kind: 'call' + toolCallId: string + toolName: string + chatId: string + workspaceId: string | null + } + | { + kind: 'approval_needed' + toolCallId: string + toolName: string + chatId: string + chatTitle: string | null + workspaceId: string | null + summary: string | null + } + | { kind: 'cancel'; toolCallId: string } + +/** A call this device now owns: the server's canonical arguments and the token that fences it. */ +export interface ClaimedDesktopCall { + toolCallId: string + toolName: string + args: Record + chatId: string + workspaceId: string | null + executionToken: string +} + +export type DesktopCompletionOutcome = 'recorded' | 'duplicate' | 'superseded' + +export interface DesktopCompletionRequest { + toolCallId: string + executionToken: string + completion: DesktopToolCompletion +} + +function positiveInteger(value: unknown): value is number { + return typeof value === 'number' && Number.isInteger(value) && value > 0 +} + +function nullableString(value: unknown): value is string | null { + return value === null || typeof value === 'string' +} + +export function parseRegistration(body: unknown): DesktopExecutorTiming | null { + if ( + !isRecordLike(body) || + typeof body.enabled !== 'boolean' || + !positiveInteger(body.protocolVersion) || + !positiveInteger(body.leaseMs) || + !positiveInteger(body.leaseRenewMs) || + !positiveInteger(body.reconcileMs) + ) { + return null + } + return { + enabled: body.enabled, + protocolVersion: body.protocolVersion, + leaseMs: body.leaseMs, + leaseRenewMs: body.leaseRenewMs, + reconcileMs: body.reconcileMs, + } +} + +function parseInboxItem(value: unknown): DesktopInboxItem | null { + if (!isRecordLike(value) || typeof value.toolCallId !== 'string' || !value.toolCallId) return null + const { toolCallId } = value + if (value.kind === 'cancel') return { kind: 'cancel', toolCallId } + if ( + typeof value.toolName !== 'string' || + typeof value.chatId !== 'string' || + !isDesktopScopeId(value.chatId) || + !nullableString(value.workspaceId) + ) { + return null + } + if (value.kind === 'call') { + return { + kind: 'call', + toolCallId, + toolName: value.toolName, + chatId: value.chatId, + workspaceId: value.workspaceId, + } + } + if ( + value.kind === 'approval_needed' && + nullableString(value.chatTitle) && + nullableString(value.summary) + ) { + return { + kind: 'approval_needed', + toolCallId, + toolName: value.toolName, + chatId: value.chatId, + chatTitle: value.chatTitle, + workspaceId: value.workspaceId, + summary: value.summary, + } + } + return null +} + +/** Unknown item kinds are skipped, so a newer Sim can add one without breaking this build. */ +export function parseInbox(body: unknown): DesktopInboxItem[] | null { + if (!isRecordLike(body) || !Array.isArray(body.items)) return null + return body.items.flatMap((item) => { + const parsed = parseInboxItem(item) + return parsed ? [parsed] : [] + }) +} + +export function parseClaim(toolCallId: string, body: unknown): ClaimedDesktopCall | null { + if ( + !isRecordLike(body) || + typeof body.toolName !== 'string' || + !isRecordLike(body.args) || + typeof body.chatId !== 'string' || + !isDesktopScopeId(body.chatId) || + !nullableString(body.workspaceId) || + typeof body.executionToken !== 'string' || + !body.executionToken + ) { + return null + } + return { + toolCallId, + toolName: body.toolName, + args: body.args, + chatId: body.chatId, + workspaceId: body.workspaceId, + executionToken: body.executionToken, + } +} + +export function parseCompletionOutcome(body: unknown): DesktopCompletionOutcome | null { + if (!isRecordLike(body)) return null + return body.outcome === 'recorded' || + body.outcome === 'duplicate' || + body.outcome === 'superseded' + ? body.outcome + : null +} diff --git a/apps/desktop/src/main/desktop-executor/runner.test.ts b/apps/desktop/src/main/desktop-executor/runner.test.ts new file mode 100644 index 00000000000..90fd0685890 --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/runner.test.ts @@ -0,0 +1,123 @@ +import type { TerminalToolResponse } from '@sim/terminal-protocol' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { ClaimedDesktopCall } from '@/main/desktop-executor/protocol' +import { createDesktopToolRunner, type DesktopToolRunnerDeps } from '@/main/desktop-executor/runner' + +function terminalCall(toolCallId: string, operation: string): ClaimedDesktopCall { + return { + toolCallId, + toolName: 'terminal', + args: { operation, args: {} }, + chatId: 'chat-b', + workspaceId: 'ws-1', + executionToken: `token-${toolCallId}`, + } +} + +function runner(overrides: Partial = {}) { + return createDesktopToolRunner({ + preferences: () => ({ browserEnabled: true, terminalEnabled: true }), + accountDataAvailable: () => true, + browser: { + executeTool: vi.fn(), + cancelTool: vi.fn(), + hasSession: () => true, + restoreScope: vi.fn(), + }, + terminal: { executeTool: vi.fn(), cancelTool: vi.fn(async () => true) }, + localFiles: { read: vi.fn() }, + localFilesystem: { handle: vi.fn(), vfsRoot: () => 'user-local/x--1' }, + ...overrides, + }) +} + +function runnerWithTerminal(executeTool: (toolCallId: string) => Promise) { + return runner({ + terminal: { + executeTool: (_scope, toolCallId) => executeTool(toolCallId), + cancelTool: vi.fn(async () => true), + }, + }) +} + +afterEach(() => { + vi.useRealTimers() +}) + +describe('background terminal calls', () => { + it('holds a chat terminal for an operation that outlived its deadline', async () => { + vi.useFakeTimers() + let releaseWedged: (response: TerminalToolResponse) => void = () => {} + const started: string[] = [] + const runner = runnerWithTerminal((toolCallId) => { + started.push(toolCallId) + if (toolCallId === 'wedged') + return new Promise((resolve) => { + releaseWedged = resolve + }) + return Promise.resolve({ ok: true, result: { output: '' } }) + }) + + const first = runner.run(terminalCall('wedged', 'input'), new AbortController().signal) + await vi.advanceTimersByTimeAsync(15_000) + const timedOut = await first + expect(timedOut.status).toBe('error') + expect(timedOut.data).toMatchObject({ outcomeUnknown: true, doNotRetry: true }) + + const second = runner.run(terminalCall('next', 'read'), new AbortController().signal) + await vi.advanceTimersByTimeAsync(0) + expect(started).toEqual(['wedged']) + + releaseWedged({ ok: true, result: {} }) + expect((await second).status).toBe('success') + expect(started).toEqual(['wedged', 'next']) + }) + + it('never starts a terminal operation stopped while it waited for the chat terminal', async () => { + vi.useFakeTimers() + let releaseWedged: (response: TerminalToolResponse) => void = () => {} + const started: string[] = [] + const runner = runnerWithTerminal((toolCallId) => { + started.push(toolCallId) + if (toolCallId === 'wedged') + return new Promise((resolve) => { + releaseWedged = resolve + }) + return Promise.resolve({ ok: true, result: { output: '' } }) + }) + const first = runner.run(terminalCall('wedged', 'input'), new AbortController().signal) + await vi.advanceTimersByTimeAsync(15_000) + await first + + const stop = new AbortController() + const second = runner.run(terminalCall('stopped', 'run'), stop.signal) + await vi.advanceTimersByTimeAsync(0) + stop.abort() + const completion = await second + releaseWedged({ ok: true, result: {} }) + await vi.advanceTimersByTimeAsync(0) + + expect(completion.status).toBe('error') + expect(started).toEqual(['wedged']) + }) +}) + +describe('local file calls', () => { + it('names a passing storage state, not a setting, when local files are out of reach', async () => { + const completion = await runner({ accountDataAvailable: () => false }).run( + { + toolCallId: 'read-1', + toolName: 'read_local_file', + args: { path: '~/notes.txt' }, + chatId: 'chat-b', + workspaceId: 'ws-1', + executionToken: 'token-read-1', + }, + new AbortController().signal + ) + + expect(completion.data).toMatchObject({ notStarted: true }) + expect(completion.message).toContain('cannot reach local files') + expect(completion.message).not.toContain('settings') + }) +}) diff --git a/apps/desktop/src/main/desktop-executor/runner.ts b/apps/desktop/src/main/desktop-executor/runner.ts new file mode 100644 index 00000000000..fc700251a3b --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/runner.ts @@ -0,0 +1,266 @@ +/** + * Runs one claimed call on this machine, in the scope of the chat Sim says it belongs to: that + * chat's built-in browser tabs, its terminals, or the user's granted folders. Arguments always + * come from Sim's record of the call. Every outcome, including a refusal, is a completion built + * by the same projections the chat view uses, so the model sees one shape either way. + */ +import { + type BrowserToolName, + browserToolRendererTimeoutMs, + isCurrentBrowserToolName, +} from '@sim/browser-protocol' +import type { + DesktopLocalFileResponse, + LocalFilesystemRequest, + LocalFilesystemResponse, +} from '@sim/desktop-bridge' +import { runUserLocalFilesystemTool } from '@sim/desktop-bridge/local-filesystem-tools' +import { + browserSessionClosedCompletion, + browserToolCompletion, + browserToolFailure, + browserToolNeedsLivePage, + browserToolTimeoutMessage, + type DesktopToolCompletion, + localFileReadCompletion, + localFilesystemToolCompletion, + terminalOperationTimeoutMs, + terminalToolCompletion, + terminalToolFailure, +} from '@sim/desktop-bridge/tool-results' +import { createLogger } from '@sim/logger' +import { + isTerminalOperation, + type TerminalOperation, + type TerminalToolArgs, + type TerminalToolResponse, +} from '@sim/terminal-protocol' +import { getErrorMessage } from '@sim/utils/errors' +import { interruptibleSleep } from '@sim/utils/helpers' +import { isRecordLike } from '@sim/utils/object' +import type { DesktopToolRunner } from '@/main/desktop-executor/executor' +import type { ClaimedDesktopCall } from '@/main/desktop-executor/protocol' + +const logger = createLogger('DesktopExecutorRunner') + +const USER_LOCAL_TOOLS: ReadonlySet = new Set(['read', 'grep', 'glob']) + +/** The model learns a call never ran because a surface is switched off on this machine. */ +function surfaceOff(surface: string): DesktopToolCompletion { + const message = `Not run: this action never started, because ${surface} is switched off in the Sim desktop app’s settings. Nothing happened on the user’s computer. Do not retry it in this turn; continue without it, or ask the user to switch it on.` + return { status: 'error', message, data: { error: message, notStarted: true } } +} + +/** Local files are out of reach while the app has no usable account storage, a passing state. */ +function localAccessUnavailable(): DesktopToolCompletion { + const message = + 'Not run: this action never started, because the Sim desktop app cannot reach local files right now (it is signing out, switching account, or its storage is unavailable). Nothing happened on the user’s computer. Do not retry it in this turn; tell the user, who can ask again once the app is ready.' + return { status: 'error', message, data: { error: message, notStarted: true } } +} + +function unsupported(toolName: string): DesktopToolCompletion { + const message = `Not run: this action never started, because this version of the Sim desktop app cannot run ${toolName} in the background. Nothing happened on the user’s computer. Do not retry it in this turn; tell the user to update the Sim desktop app.` + return { status: 'error', message, data: { error: message, notStarted: true } } +} + +export interface DesktopToolRunnerDeps { + preferences: () => { browserEnabled: boolean; terminalEnabled: boolean } + /** False while account-bearing local storage is unavailable; local tools then refuse. */ + accountDataAvailable: () => boolean + browser: { + executeTool( + scopeId: string, + tool: BrowserToolName, + params: Record, + toolCallId: string + ): Promise<{ ok: boolean; result?: unknown; error?: string }> + cancelTool(scopeId: string, toolCallId: string): boolean + hasSession(scopeId: string): boolean + restoreScope(scopeId: string): void + } + terminal: { + executeTool( + scope: string, + toolCallId: string, + operation: TerminalOperation, + args: TerminalToolArgs + ): Promise + cancelTool(scope: string, toolCallId: string): Promise + } + localFiles: { + read(call: ClaimedDesktopCall): Promise + } + localFilesystem: { + handle(request: LocalFilesystemRequest): Promise + vfsRoot(mount: { id: string; name: string }): string + } +} + +/** A terminal operation that outlived its deadline may still land, as a browser action may. */ +function terminalTimedOut(timeoutMs: number): DesktopToolCompletion { + const message = `The terminal did not respond within ${timeoutMs}ms. The operation may still be running and take effect: do not retry it automatically; check the terminal's state first.` + return { + status: 'error', + message, + data: { error: message, outcomeUnknown: true, doNotRetry: true }, + } +} + +/** Resolves once the signal aborts, which may be never. */ +function untilAborted(signal: AbortSignal): Promise { + if (signal.aborted) return Promise.resolve() + return new Promise((resolve) => signal.addEventListener('abort', () => resolve(), { once: true })) +} + +/** Resolves with the work's result, or with `onTimeout`'s once `timeoutMs` passes first. */ +async function withDeadline( + work: Promise, + timeoutMs: number | null, + onTimeout: () => T +): Promise { + if (timeoutMs === null) return work + // Aborted once the race settles, which cancels the sleep; a cancelled sleep never times out. + const settled = new AbortController() + const timeout = interruptibleSleep(timeoutMs, settled.signal).then(() => + settled.signal.aborted ? new Promise(() => {}) : onTimeout() + ) + try { + return await Promise.race([work, timeout]) + } finally { + settled.abort() + } +} + +export function createDesktopToolRunner(deps: DesktopToolRunnerDeps): DesktopToolRunner { + /** Each chat's last terminal operation, until it actually settles. */ + const unsettledTerminalWork = new Map>() + + async function runBrowser( + call: ClaimedDesktopCall, + tool: BrowserToolName + ): Promise { + if (!deps.preferences().browserEnabled) return surfaceOff('the built-in browser') + if (browserToolNeedsLivePage(tool) && !deps.browser.hasSession(call.chatId)) { + try { + deps.browser.restoreScope(call.chatId) + } catch (error) { + logger.warn('Could not restore a chat browser before a background call', { + toolCallId: call.toolCallId, + error: getErrorMessage(error), + }) + } + if (!deps.browser.hasSession(call.chatId)) return browserSessionClosedCompletion() + } + const timeoutMs = + tool === 'browser_request_takeover' ? null : browserToolRendererTimeoutMs(tool, call.args) + return withDeadline( + deps.browser.executeTool(call.chatId, tool, call.args, call.toolCallId).then((response) => + response.ok + ? browserToolCompletion(tool, response.result) + : browserToolFailure(response.error ?? 'The browser action failed.', { + sessionClosed: !deps.browser.hasSession(call.chatId), + }) + ), + timeoutMs, + () => { + deps.browser.cancelTool(call.chatId, call.toolCallId) + return browserToolFailure(browserToolTimeoutMessage(timeoutMs ?? 0), { + outcomeUnknown: true, + }) + } + ) + } + + async function runTerminal( + call: ClaimedDesktopCall, + signal: AbortSignal + ): Promise { + if (!deps.preferences().terminalEnabled) return surfaceOff('the terminal') + const { operation } = call.args + if (!isTerminalOperation(operation)) { + return terminalToolFailure( + `Unknown terminal operation: ${String(operation)}`, + 'INVALID_REQUEST' + ) + } + const args = isRecordLike(call.args.args) ? (call.args.args as TerminalToolArgs) : {} + const timeoutMs = terminalOperationTimeoutMs(operation) + // An operation reported as unresponsive may still land; the chat's next one waits for it, so + // two never act on the same terminals at once. + const previous = unsettledTerminalWork.get(call.chatId) + if (previous) await Promise.race([previous, untilAborted(signal)]) + // Stopped while it waited: the terminal never heard of it, so its own cancel cannot reach it. + if (signal.aborted) { + return terminalToolFailure('The terminal action was stopped before it started.', 'CANCELLED') + } + const operationDone = deps.terminal.executeTool(call.chatId, call.toolCallId, operation, args) + const settled = operationDone.then( + () => undefined, + () => undefined + ) + unsettledTerminalWork.set(call.chatId, settled) + void settled.then(() => { + if (unsettledTerminalWork.get(call.chatId) === settled) + unsettledTerminalWork.delete(call.chatId) + }) + return withDeadline(operationDone.then(terminalToolCompletion), timeoutMs, () => + terminalTimedOut(timeoutMs ?? 0) + ) + } + + async function runUserLocal( + call: ClaimedDesktopCall, + signal: AbortSignal + ): Promise { + try { + const data = await runUserLocalFilesystemTool(call.toolCallId, call.toolName, call.args, { + invoke: (request) => deps.localFilesystem.handle(request), + vfsRoot: deps.localFilesystem.vfsRoot, + signal, + }) + return localFilesystemToolCompletion({ ok: true, data }) + } catch (error) { + return localFilesystemToolCompletion({ ok: false, error: getErrorMessage(error) }) + } + } + + return { + async run(call, signal) { + try { + if (isCurrentBrowserToolName(call.toolName)) return await runBrowser(call, call.toolName) + if (call.toolName === 'terminal') return await runTerminal(call, signal) + if (!deps.accountDataAvailable()) return localAccessUnavailable() + if (call.toolName === 'read_local_file') { + return localFileReadCompletion(await deps.localFiles.read(call)) + } + if (USER_LOCAL_TOOLS.has(call.toolName)) return await runUserLocal(call, signal) + return unsupported(call.toolName) + } catch (error) { + logger.error('Background desktop call failed unexpectedly', { + toolCallId: call.toolCallId, + toolName: call.toolName, + error: getErrorMessage(error), + }) + const message = `The Sim desktop app could not finish this action: ${getErrorMessage(error)}` + return { + status: 'error', + message, + data: { error: message, outcomeUnknown: true, doNotRetry: true }, + } + } + }, + async cancel(call) { + if (isCurrentBrowserToolName(call.toolName)) { + deps.browser.cancelTool(call.chatId, call.toolCallId) + return + } + if (call.toolName === 'terminal') { + await deps.terminal.cancelTool(call.chatId, call.toolCallId) + return + } + if (USER_LOCAL_TOOLS.has(call.toolName)) { + await deps.localFilesystem.handle({ operation: 'cancel', requestId: call.toolCallId }) + } + }, + } +} diff --git a/apps/desktop/src/main/desktop-executor/service.test.ts b/apps/desktop/src/main/desktop-executor/service.test.ts new file mode 100644 index 00000000000..ce81c6cc1fb --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/service.test.ts @@ -0,0 +1,187 @@ +import { chmod, mkdir, mkdtemp } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { sleep } from '@sim/utils/helpers' +import { describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => import('@/test/electron-mock')) + +import { net } from 'electron' +import { createDesktopExecutorService, deviceName } from '@/main/desktop-executor/service' + +/** Sim's device routes, with registration answers held until the test releases them. */ +function fakeSim(protocolVersion = 1) { + const requests: string[] = [] + /** + * Answers to pending registrations: enabled or not, an HTTP status Sim fails with, or + * `'offline'` for a request that never reached Sim. + */ + const registrations: Array<(answer: boolean | number | 'offline') => void> = [] + const fetch = vi.fn(async (url: string, init: RequestInit): Promise => { + const path = new URL(url).pathname + requests.push(`${init.method} ${path}`) + if (path === '/api/desktop/devices') { + const answer = await new Promise((resolve) => + registrations.push(resolve) + ) + if (answer === 'offline') throw new TypeError('fetch failed') + if (typeof answer === 'number') { + return Response.json({ error: 'Not found' }, { status: answer }) + } + const enabled = answer + return Response.json({ + enabled, + protocolVersion, + leaseMs: 60_000, + leaseRenewMs: 20_000, + reconcileMs: 10_000, + }) + } + if (path === '/api/desktop/inbox') return Response.json({ items: [] }) + return new Promise((_resolve, reject) => + init.signal?.addEventListener('abort', () => reject(new Error('aborted'))) + ) + }) + return { fetch, requests, registrations } +} + +async function service(protocolVersion = 1, userDataPath?: string) { + const sim = fakeSim(protocolVersion) + const desktopExecutor = createDesktopExecutorService({ + userDataPath: userDataPath ?? (await mkdtemp(join(tmpdir(), 'sim-executor-service-'))), + origin: () => 'https://sim.test', + appSession: () => ({ fetch: sim.fetch }), + preferences: () => ({ browserEnabled: true, terminalEnabled: true }), + accountDataAvailable: () => true, + runner: { run: vi.fn(), cancel: vi.fn() }, + }) + return { sim, desktopExecutor } +} + +describe('desktop executor registration', () => { + it('offers the device for binding once Sim enables it', async () => { + const { sim, desktopExecutor } = await service() + desktopExecutor.start() + await vi.waitFor(() => expect(sim.registrations).toHaveLength(1)) + + sim.registrations[0]?.(true) + + await vi.waitFor(() => expect(desktopExecutor.getDevice()).not.toBeNull()) + await vi.waitFor(() => expect(sim.requests).toContain('GET /api/desktop/inbox')) + }) + + it('never resumes for a registration that Sim answers after sign-out', async () => { + const { sim, desktopExecutor } = await service() + desktopExecutor.start() + await vi.waitFor(() => expect(sim.registrations).toHaveLength(1)) + + await desktopExecutor.signOut() + sim.registrations[0]?.(true) + await sleep(50) + + expect(desktopExecutor.getDevice()).toBeNull() + expect(sim.requests).not.toContain('GET /api/desktop/inbox') + expect(sim.requests).not.toContain('GET /api/desktop/inbox/stream') + }) + + it('offers no binding to a Sim that speaks another protocol version', async () => { + const { sim, desktopExecutor } = await service(2) + desktopExecutor.start() + await vi.waitFor(() => expect(sim.registrations).toHaveLength(1)) + + sim.registrations[0]?.(true) + await sleep(100) + + expect(desktopExecutor.getDevice()).toBeNull() + expect(sim.requests.filter((request) => request.includes('/api/desktop/inbox'))).toEqual([]) + }) + + it('stays dormant while Sim has the executor off: no inbox and no doorbell', async () => { + const { sim, desktopExecutor } = await service() + desktopExecutor.start() + await vi.waitFor(() => expect(sim.registrations).toHaveLength(1)) + + sim.registrations[0]?.(false) + await sleep(100) + + expect(desktopExecutor.getDevice()).toBeNull() + expect(sim.requests.filter((request) => request.includes('/api/desktop/inbox'))).toEqual([]) + }) + + it('stays dormant against a Sim without the executor routes, checking back only slowly', async () => { + const { sim, desktopExecutor } = await service() + vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout', 'setInterval', 'clearInterval'] }) + try { + desktopExecutor.start() + await vi.waitFor(() => expect(sim.registrations).toHaveLength(1)) + + sim.registrations[0]?.(404) + await vi.advanceTimersByTimeAsync(60_000) + expect(sim.registrations).toHaveLength(1) + expect(sim.requests.filter((request) => request.includes('/api/desktop/inbox'))).toEqual([]) + + await vi.advanceTimersByTimeAsync(15 * 60_000) + expect(sim.registrations).toHaveLength(2) + } finally { + vi.useRealTimers() + } + }) + + it('registers again as soon as the network returns, not after its backoff', async () => { + const { sim, desktopExecutor } = await service() + vi.mocked(net.isOnline).mockReturnValue(false) + vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout', 'setInterval', 'clearInterval'] }) + try { + desktopExecutor.start() + // Four failures with no network leave the next attempt at least 12.8 s away. + for (let failure = 0; failure < 4; failure += 1) { + await vi.waitFor(() => expect(sim.registrations).toHaveLength(failure + 1)) + sim.registrations[failure]?.('offline') + await vi.advanceTimersByTimeAsync(failure < 3 ? 10_000 : 0) + } + expect(sim.registrations).toHaveLength(4) + + vi.mocked(net.isOnline).mockReturnValue(true) + await vi.advanceTimersByTimeAsync(2_500) + + expect(sim.registrations).toHaveLength(5) + } finally { + vi.useRealTimers() + vi.mocked(net.isOnline).mockReturnValue(true) + } + }) + + it('registers with no install id it could not save', async () => { + const userData = await mkdtemp(join(tmpdir(), 'sim-executor-service-')) + await chmod(userData, 0o500) + try { + const { sim, desktopExecutor } = await service(1, userData) + + desktopExecutor.start() + await sleep(200) + + expect(sim.requests.filter((request) => request.includes('/api/desktop/devices'))).toEqual([]) + } finally { + await chmod(userData, 0o700) + } + }) + + it('keeps its install id through a read failure instead of minting a new one', async () => { + const userData = await mkdtemp(join(tmpdir(), 'sim-executor-service-')) + // A path that exists but cannot be read as a file: not "no id yet". + await mkdir(join(userData, 'desktop-executor-device.json')) + const { sim, desktopExecutor } = await service(1, userData) + + desktopExecutor.start() + await sleep(200) + + expect(sim.requests.filter((request) => request.includes('/api/desktop/devices'))).toEqual([]) + }) +}) + +describe('device name', () => { + it('fits a long hostname within what Sim accepts at registration', () => { + expect(deviceName(`${'studio-'.repeat(40)}.local`).length).toBeLessThanOrEqual(128) + expect(deviceName('Studio-Mac.local')).toBe('Studio-Mac') + }) +}) diff --git a/apps/desktop/src/main/desktop-executor/service.ts b/apps/desktop/src/main/desktop-executor/service.ts new file mode 100644 index 00000000000..c0dd079d36c --- /dev/null +++ b/apps/desktop/src/main/desktop-executor/service.ts @@ -0,0 +1,446 @@ +/** + * Runs the background executor for as long as the app is signed in: registers this install as a + * device, keeps the inbox doorbell open, reads the inbox on every ring and on Sim's reconcile + * timer, and follows sleep, wake and network changes. One per process, which the single-instance + * lock makes one per machine. + */ +import { hostname } from 'node:os' +import { join } from 'node:path' +import type { DesktopExecutorDevice } from '@sim/desktop-bridge' +import { createLogger } from '@sim/logger' +import { getErrorMessage } from '@sim/utils/errors' +import { generateId } from '@sim/utils/id' +import { isRecordLike } from '@sim/utils/object' +import { randomFloat } from '@sim/utils/random' +import { backoffWithJitter } from '@sim/utils/retry' +import { truncate } from '@sim/utils/string' +import type { Session } from 'electron' +import { app, net, powerMonitor } from 'electron' +import { + readFileWithinLimit, + removeFileIfPresent, + writeJsonFileAtomically, +} from '@/main/atomic-json-file' +import { + createDesktopExecutorClient, + type DesktopExecutorClient, + DeviceRequestError, +} from '@/main/desktop-executor/client' +import { InboxDoorbell } from '@/main/desktop-executor/doorbell' +import { + type DesktopApprovalItem, + DesktopExecutor, + type DesktopToolRunner, +} from '@/main/desktop-executor/executor' +import { createExecutorJournal } from '@/main/desktop-executor/journal' +import { + DESKTOP_EXECUTOR_PROTOCOL_VERSION, + type DesktopExecutorTiming, +} from '@/main/desktop-executor/protocol' + +const logger = createLogger('DesktopExecutorService') + +/** Batches the burst of cookie and preference events a sign-in produces into one registration. */ +const REGISTRATION_DEBOUNCE_MS = 1_000 +const REGISTRATION_RETRY_MAX_MS = 5 * 60_000 +/** How often a device Sim has not enabled checks whether that changed. */ +const DORMANT_RECHECK_MS = 15 * 60_000 +const ONLINE_POLL_MS = 2_000 +/** Spreads reconcile reads so devices that woke together do not read together. */ +const RECONCILE_JITTER = 0.2 + +export interface DesktopExecutorServiceDeps { + userDataPath: string + origin: () => string + appSession: () => Pick + preferences: () => { browserEnabled: boolean; terminalEnabled: boolean } + accountDataAvailable: () => boolean + runner: DesktopToolRunner + onApprovals?: (items: DesktopApprovalItem[]) => void +} + +export interface DesktopExecutorService { + start(): void + /** Re-registers after a sign-in, a session change, or a change to what this device can run. */ + refreshRegistration(): void + getDevice(): DesktopExecutorDevice | null + /** Sign-out: stops every action, forgets every call, and retires this install id. */ + signOut(): Promise +} + +/** The install id, kept in userData; a new one is minted after sign-out or a conflict. */ +async function readInstallId(filePath: string): Promise { + let raw: string + try { + raw = (await readFileWithinLimit(filePath, 4096)).toString('utf8') + } catch (error) { + // Only a missing file means no id yet. Any other failure could be passing, and minting a new + // id over it would orphan the calls bound to this one, so it is left to the next attempt. + if (isRecordLike(error) && error.code === 'ENOENT') return null + throw error + } + try { + const parsed: unknown = JSON.parse(raw) + return isRecordLike(parsed) && typeof parsed.deviceId === 'string' ? parsed.deviceId : null + } catch { + // Unreadable contents can never yield the id; a fresh one replaces them. + return null + } +} + +/** Sim refuses a longer device name at registration. */ +const DEVICE_NAME_MAX_CHARS = 128 + +/** The machine's name as the user knows it, within what Sim accepts. */ +export function deviceName(host = hostname()): string { + const name = host.replace(/\.local$/, '').trim() || 'Sim desktop' + return truncate(name, DEVICE_NAME_MAX_CHARS - 3) +} + +export function createDesktopExecutorService( + deps: DesktopExecutorServiceDeps +): DesktopExecutorService { + const identityPath = join(deps.userDataPath, 'desktop-executor-device.json') + const journal = createExecutorJournal(join(deps.userDataPath, 'desktop-executor-journal.json')) + + let deviceId: string | null = null + let device: DesktopExecutorDevice | null = null + let timing: DesktopExecutorTiming | null = null + let client: DesktopExecutorClient | null = null + let executor: DesktopExecutor | null = null + let doorbell: InboxDoorbell | null = null + let reconcileTimer: ReturnType | null = null + let registrationTimer: ReturnType | null = null + let registering = false + let registerAgain = false + let registrationAttempt = 0 + /** Logged once per run of missing routes, not on every recheck. */ + let routesMissingNoted = false + /** The last registration failed with no answer from Sim at all. */ + let registrationFailedOffline = false + let suspended = false + let started = false + /** Bumped on sign-out, so work started for the previous session cannot resume it. */ + let generation = 0 + let executorDeviceId: string | null = null + + async function installId(): Promise { + if (deviceId) return deviceId + deviceId = (await readInstallId(identityPath)) ?? (await rotateInstallId()) + return deviceId + } + + /** + * Mints a new install id and makes it durable before anything uses it: an id the next launch + * would not find again could never resume the calls bound to it. Fails if it cannot be saved. + */ + async function rotateInstallId(): Promise { + const next = generateId() + deviceId = null + await writeJsonFileAtomically(identityPath, { deviceId: next }) + deviceId = next + return next + } + + /** + * Retires the current install id for good: a new one replaces it, or, if that cannot be saved, + * the old one is removed so no later launch or sign-in can pick it up again. False when neither + * worked and the old id is still on disk. + */ + async function retireInstallId(): Promise { + try { + await rotateInstallId() + return true + } catch (error) { + logger.warn('Could not save a new desktop install id', { error: getErrorMessage(error) }) + } + try { + await removeFileIfPresent(identityPath) + return true + } catch (error) { + logger.warn('Could not remove the retired desktop install id', { + error: getErrorMessage(error), + }) + return false + } + } + + function fetchWithAppSession(url: string, init: RequestInit): Promise { + return deps.appSession().fetch(url, init) + } + + function scheduleReconcile(): void { + if (reconcileTimer) clearTimeout(reconcileTimer) + // Stopped loops (a dormant or unrecognized device) stay stopped until registration restarts them. + if (!timing || !executor || !doorbell) return + const delay = timing.reconcileMs * (1 - RECONCILE_JITTER + randomFloat() * RECONCILE_JITTER) + reconcileTimer = setTimeout(() => { + reconcileTimer = null + void executor?.reconcile().finally(scheduleReconcile) + }, delay) + } + + function ring(): void { + if (suspended) return + void executor?.reconcile() + } + + /** + * Sim stopped recognizing this device. With the executor on, the session changed under it, so + * it registers again at once. With it off, Sim never recorded the device, so it stays quiet and + * checks back later in case the executor is switched on. + */ + function handleUnrecognized(): void { + if (device) { + scheduleRegistration() + return + } + stopLoops() + // Every refused request lands here; re-arming each time would push the recheck out forever. + if (!registrationTimer) scheduleRegistration(DORMANT_RECHECK_MS) + } + + function stopLoops(): void { + doorbell?.stop() + doorbell = null + if (reconcileTimer) clearTimeout(reconcileTimer) + reconcileTimer = null + } + + /** Builds the executor for this install the first time Sim recognizes it. */ + async function startExecutor( + id: string, + nextTiming: DesktopExecutorTiming, + registrationGeneration: number + ): Promise { + if (executor && executorDeviceId !== id) await resetExecutor() + timing = nextTiming + if (!executor || !client) { + executorDeviceId = id + client = createDesktopExecutorClient({ + origin: deps.origin, + fetch: fetchWithAppSession, + deviceId: id, + }) + executor = new DesktopExecutor({ + client, + journal, + runner: deps.runner, + leaseRenewMs: nextTiming.leaseRenewMs, + onUnregistered: handleUnrecognized, + ...(deps.onApprovals ? { onApprovals: deps.onApprovals } : {}), + }) + await executor.recover() + // Signed out while recovering: sign-out already disposed this executor. + if (registrationGeneration !== generation || !executor) return + } else { + executor.resumeParked() + } + if (!doorbell) { + doorbell = new InboxDoorbell({ + client, + onRing: ring, + onUnregistered: handleUnrecognized, + }) + if (!suspended) doorbell.start() + } + if (!suspended) void executor.reconcile() + scheduleReconcile() + } + + async function register(): Promise { + if (!deps.accountDataAvailable()) return + const registrationGeneration = generation + let id: string + try { + id = await installId() + } catch (error) { + registrationAttempt += 1 + logger.warn('Could not read the desktop install id', { error: getErrorMessage(error) }) + scheduleRegistration( + backoffWithJitter(registrationAttempt, null, { + baseMs: 2_000, + maxMs: REGISTRATION_RETRY_MAX_MS, + }) + ) + return + } + const preferences = deps.preferences() + const registrationClient = createDesktopExecutorClient({ + origin: deps.origin, + fetch: fetchWithAppSession, + deviceId: id, + }) + try { + const nextTiming = await registrationClient.register({ + deviceId: id, + name: deviceName(), + appVersion: app.getVersion(), + platform: `${process.platform}-${process.arch}`, + capabilities: { + executor: DESKTOP_EXECUTOR_PROTOCOL_VERSION, + browser: preferences.browserEnabled, + terminal: preferences.terminalEnabled, + localFiles: deps.accountDataAvailable(), + }, + }) + if (registrationGeneration !== generation || id !== deviceId) return + registrationAttempt = 0 + registrationFailedOffline = false + routesMissingNoted = false + // A Sim that speaks another protocol version gets no new turns bound to this device. + device = + nextTiming.enabled && nextTiming.protocolVersion === DESKTOP_EXECUTOR_PROTOCOL_VERSION + ? { deviceId: id, protocolVersion: DESKTOP_EXECUTOR_PROTOCOL_VERSION } + : null + logger.info('Desktop executor registered', { enabled: nextTiming.enabled }) + // Off for this user, and nothing here to finish: stay dormant. No inbox, no doorbell; only + // a slow recheck, so switching the executor on in Sim reaches this device without a relaunch. + if (!device && !executor && (await journal.load()).length === 0) { + if (registrationGeneration !== generation) return + scheduleRegistration(DORMANT_RECHECK_MS) + return + } + // Off for this user mid-session, or results from a previous run to deliver: a turn already + // bound here still finishes, so the executor serves the inbox; no new turn binds. + await startExecutor(id, nextTiming, registrationGeneration) + } catch (error) { + if (registrationGeneration !== generation) return + device = null + if (error instanceof DeviceRequestError && error.status === 409) { + // The id belongs to another account (a copied profile); this install takes a new one. + logger.warn('Desktop install id is registered to another account; minting a new one') + await resetExecutor() + if (await retireInstallId()) { + scheduleRegistration(0) + } else { + // The conflicting id is still on disk: retrying at once would only conflict again. + registrationAttempt += 1 + scheduleRegistration( + backoffWithJitter(registrationAttempt, null, { + baseMs: 2_000, + maxMs: REGISTRATION_RETRY_MAX_MS, + }) + ) + } + return + } + if (error instanceof DeviceRequestError && error.unregistered) { + // Signed out: the next sign-in's session change registers again. + stopLoops() + return + } + if (error instanceof DeviceRequestError && (error.status === 404 || error.status === 405)) { + // A Sim without the executor routes (older, or self-hosted): dormant, with only the slow + // recheck, so an upgrade of Sim reaches this device without a relaunch. + stopLoops() + if (!routesMissingNoted) { + routesMissingNoted = true + logger.info('Sim does not offer the desktop background executor; staying dormant') + } + scheduleRegistration(DORMANT_RECHECK_MS) + return + } + registrationFailedOffline = error instanceof DeviceRequestError && error.status === 0 + registrationAttempt += 1 + logger.warn('Desktop executor registration failed', { + attempt: registrationAttempt, + error: getErrorMessage(error), + }) + scheduleRegistration( + backoffWithJitter(registrationAttempt, null, { + baseMs: 2_000, + maxMs: REGISTRATION_RETRY_MAX_MS, + }) + ) + } + } + + function scheduleRegistration(delayMs = REGISTRATION_DEBOUNCE_MS): void { + if (!started) return + if (registrationTimer) clearTimeout(registrationTimer) + registrationTimer = setTimeout(() => { + registrationTimer = null + void runRegistration() + }, delayMs) + } + + /** One registration at a time; a trigger that lands mid-flight registers once more after it. */ + async function runRegistration(): Promise { + if (registering) { + registerAgain = true + return + } + registering = true + try { + do { + registerAgain = false + await register() + } while (registerAgain) + } finally { + registering = false + } + } + + async function resetExecutor(): Promise { + stopLoops() + const current = executor + executor = null + executorDeviceId = null + client = null + timing = null + device = null + await current?.dispose() + } + + function wake(): void { + suspended = false + executor?.setPaused(false) + doorbell?.wake() + void executor?.reconcile() + scheduleReconcile() + } + + function watchPowerAndNetwork(): void { + powerMonitor.on('suspend', () => { + suspended = true + executor?.setPaused(true) + doorbell?.stop() + }) + powerMonitor.on('resume', wake) + let online = net.isOnline() + setInterval(() => { + const now = net.isOnline() + if (now && !online) { + wake() + // A registration that failed for want of a network need not wait out its backoff. + if (registrationFailedOffline) scheduleRegistration(0) + } + online = now + }, ONLINE_POLL_MS).unref?.() + } + + return { + start() { + if (started) return + started = true + watchPowerAndNetwork() + scheduleRegistration(0) + }, + refreshRegistration() { + scheduleRegistration() + }, + getDevice() { + return device + }, + async signOut() { + // A registration in flight now answers for a session that is gone; it must not restart. + generation += 1 + if (registrationTimer) clearTimeout(registrationTimer) + registrationTimer = null + await resetExecutor() + await journal.clear() + await retireInstallId() + }, + } +} diff --git a/apps/desktop/src/main/index.ts b/apps/desktop/src/main/index.ts index 9b53f1815f3..990fa6724aa 100644 --- a/apps/desktop/src/main/index.ts +++ b/apps/desktop/src/main/index.ts @@ -2,7 +2,16 @@ import { join } from 'node:path' import { createLogger } from '@sim/logger' import { getErrorMessage } from '@sim/utils/errors' import type { OpenDialogOptions, Session, WebContents } from 'electron' -import { app, BrowserWindow, crashReporter, dialog, net, session, shell } from 'electron' +import { + app, + BrowserWindow, + crashReporter, + dialog, + Notification, + net, + session, + shell, +} from 'electron' import { beginAccountDataTeardown, completeDeploymentScopedTeardown, @@ -17,9 +26,13 @@ import { import { newChatRoute, settingsRoute } from '@/main/app-routes' import { activateBrowserScope as activateAgentBrowserScope, + cancelTool as cancelAgentBrowserTool, clearBrowserProfile as clearAgentBrowserProfile, closeBrowserSession as closeAgentBrowserSession, + executeTool as executeAgentBrowserTool, + hasBrowserScopeSession, initDriver as initBrowserAgentDriver, + restoreBrowserScope as restoreAgentBrowserScope, } from '@/main/browser-agent/driver' import { canReportPanelBounds, @@ -47,6 +60,9 @@ import { import { attachContextMenu } from '@/main/context-menu' import { attachCspFallback } from '@/main/csp' import { DesktopChatSessionStore } from '@/main/desktop-chat-session-store' +import { createApprovalNotifier } from '@/main/desktop-executor/approval-notifier' +import { createDesktopToolRunner } from '@/main/desktop-executor/runner' +import { createDesktopExecutorService } from '@/main/desktop-executor/service' import { createDesktopSettingsService } from '@/main/desktop-settings' import { attachDownloadHandling } from '@/main/downloads' import { createAuthFlow, createConnectFlow, createHandoffManager } from '@/main/handoff' @@ -56,7 +72,8 @@ import { } from '@/main/help-search' import { registerIpcHandlers } from '@/main/ipc' import { attachLoadHealth, type LoadHealthHandle } from '@/main/load-health' -import { LocalFilesystemService } from '@/main/local-filesystem' +import { executeLocalFileRequest } from '@/main/local-files' +import { LocalFilesystemService, mountVfsRoot } from '@/main/local-filesystem' import { createEncryptedLocalFilesystemGrantStore } from '@/main/local-filesystem-grant-store' import { attachLocalPageProtocol, @@ -75,6 +92,7 @@ import { createSessionLifecycleCoordinator, decideStartRoute, handleConnectIntercept, + isSessionCookieName, readSessionUserId, resolveStartRoute, } from '@/main/session-lifecycle' @@ -306,7 +324,22 @@ function main(): void { // Shells are account-scoped runtime state. Leaving them alive across // sign-out would stream the previous account's output into the next // renderer and keep its local processes running invisibly. - { label: 'terminal sessions', clear: () => terminal.dispose() }, + { + label: 'background executor', + clear: async () => { + approvalNotifier.clear() + await desktopExecutor.signOut() + }, + }, + { + label: 'terminal sessions', + // Agent commands are stopped by their own process groups first: closing a shell only + // hangs up its foreground, and a tmux run outlives the Sim terminal entirely. + clear: async () => { + await terminal.stopAgentCommands() + terminal.dispose() + }, + }, { label: 'task resource state', clear: clearDesktopChatSessions }, { label: 'local filesystem grants', clear: () => localFilesystem.forgetAll() }, ] @@ -501,9 +534,17 @@ function main(): void { // pinned strip survive, so switching back on resumes rather than restarts. setBrowserEnabled: (enabled) => { if (!enabled) closeAgentBrowserSession() + desktopExecutor.refreshRegistration() }, setTerminalEnabled: (enabled) => { - if (!enabled) terminal.dispose() + // Agent commands are stopped by their process groups first; a tmux run outlives its shell. + if (!enabled) { + void terminal.stopAgentCommands().then(() => { + // Switched back on while the commands stopped: the shells opened since are the user's. + if (!desktopSettings.getPreferences().terminalEnabled) terminal.dispose() + }) + } + desktopExecutor.refreshRegistration() }, setBrowserTheme: setAgentBrowserTheme, setBrowserDefaultZoom: setAgentBrowserDefaultZoom, @@ -531,6 +572,50 @@ function main(): void { }, }) + const approvalNotifier = createApprovalNotifier({ + preferences: () => desktopSettings.getPreferences(), + focusedChatId: () => { + const win = focusedAppWindow() + if (!win) return null + const route = routeFromAppUrl(win.webContents.getURL()) + return route ? (/\/chat\/([^/?#]+)/.exec(route)?.[1] ?? null) : null + }, + openRoute: (route) => void openMainWindowAt(route), + createNotification: (options) => + Notification.isSupported() ? new Notification(options) : null, + }) + + const desktopExecutor = createDesktopExecutorService({ + userDataPath, + origin: appOrigin, + appSession: ensureAppSession, + preferences: () => desktopSettings.getPreferences(), + accountDataAvailable, + onApprovals: (items) => approvalNotifier.update(items), + runner: createDesktopToolRunner({ + preferences: () => desktopSettings.getPreferences(), + accountDataAvailable, + browser: { + executeTool: executeAgentBrowserTool, + cancelTool: cancelAgentBrowserTool, + hasSession: hasBrowserScopeSession, + restoreScope: restoreAgentBrowserScope, + }, + terminal, + localFiles: { + read: (call) => + executeLocalFileRequest( + { operation: 'read', toolCallId: call.toolCallId }, + { toolName: call.toolName, args: call.args } + ), + }, + localFilesystem: { + handle: (request) => localFilesystem.handle(request), + vfsRoot: mountVfsRoot, + }, + }), + }) + const serverWindow = createServerWindow({ config, defaultOrigin: DEFAULT_ORIGIN, @@ -674,7 +759,14 @@ function main(): void { ...(kind === 'account' && origin ? [ { label: 'sign-in handoff state', clear: () => handoff.clear() }, - { label: 'terminal sessions', clear: () => terminal.dispose() }, + { label: 'background executor', clear: () => desktopExecutor.signOut() }, + { + label: 'terminal sessions', + clear: async () => { + await terminal.stopAgentCommands() + terminal.dispose() + }, + }, { label: 'task resource state', clear: clearDesktopChatSessions }, { label: 'app session storage', @@ -842,7 +934,15 @@ function main(): void { getConfiguration: () => serverWindow.getConfiguration(), setOrigin: (origin) => serverWindow.setOrigin(origin), }, + getExecutorDevice: () => desktopExecutor.getDevice(), }) + if (accountDataAvailable()) { + // A sign-in, or a session rotation, binds the device to the new session. + ensureAppSession().cookies.on('changed', (_event, cookie, _cause, removed) => { + if (!removed && isSessionCookieName(cookie.name)) desktopExecutor.refreshRegistration() + }) + desktopExecutor.start() + } await ensureMainWindow() installApplicationMenu({ config, diff --git a/apps/desktop/src/main/ipc.test.ts b/apps/desktop/src/main/ipc.test.ts index f322f03e454..4a596f77a65 100644 --- a/apps/desktop/src/main/ipc.test.ts +++ b/apps/desktop/src/main/ipc.test.ts @@ -289,6 +289,7 @@ describe('registerIpcHandlers', () => { mockCoordinator.fillCredential.mockClear() deps = { appOrigin: () => APP, + getExecutorDevice: () => null, allowHttpLocalhost: () => false, accountDataAvailable: () => true, isLocalPageUrl, diff --git a/apps/desktop/src/main/ipc.ts b/apps/desktop/src/main/ipc.ts index 423dca0b7e6..1ca3b569f59 100644 --- a/apps/desktop/src/main/ipc.ts +++ b/apps/desktop/src/main/ipc.ts @@ -9,6 +9,7 @@ import { isCurrentBrowserToolName, } from '@sim/browser-protocol' import { + type DesktopExecutorDevice, type DesktopNotificationPayload, type DesktopServerChangeResult, type DesktopServerConfiguration, @@ -374,6 +375,8 @@ export interface IpcDeps { getConfiguration: () => DesktopServerConfiguration setOrigin: (origin: string) => Promise } + /** The device a new chat turn may bind to, or null while Sim has not enabled one. */ + getExecutorDevice: () => DesktopExecutorDevice | null } /** @@ -771,6 +774,12 @@ export function registerIpcHandlers(deps: IpcDeps): void { }, handler: (request) => deps.localFilesystem.handle(request), }, + 'desktop-executor:get-device': { + kind: 'invoke', + gate: 'app-origin', + denied: null, + handler: () => deps.getExecutorDevice(), + }, 'desktop:settings:get': { kind: 'invoke', gate: 'app-origin', diff --git a/apps/desktop/src/main/local-filesystem.test.ts b/apps/desktop/src/main/local-filesystem.test.ts index 29b01d5e86d..2a98ddf8326 100644 --- a/apps/desktop/src/main/local-filesystem.test.ts +++ b/apps/desktop/src/main/local-filesystem.test.ts @@ -236,6 +236,27 @@ describe('LocalFilesystemService', () => { ).toBe(false) }) + it('hands back nothing from a folder forgotten while the read ran', async () => { + const granted = await mount(service) + + const reading = service.handle({ operation: 'read', uri: `${granted.uri}README.md` }) + const searching = service.handle({ operation: 'grep', uri: granted.uri, pattern: 'hello' }) + await service.handle({ operation: 'forget_mount', uri: granted.uri }) + + await expect(reading).resolves.toMatchObject({ ok: false, code: 'MOUNT_NOT_FOUND' }) + await expect(searching).resolves.toMatchObject({ ok: false, code: 'MOUNT_NOT_FOUND' }) + }) + + it('keeps a read whose folder the user selected again while it ran', async () => { + const granted = await mount(service) + + const reading = service.handle({ operation: 'read', uri: `${granted.uri}README.md` }) + const again = await mount(service) + + expect(again.uri).toBe(granted.uri) + await expect(reading).resolves.toMatchObject({ ok: true }) + }) + it('rejects unknown mounts and symlinks that escape the selected directory', async () => { const granted = await mount(service) const outside = await mkdtemp(join(tmpdir(), 'sim-localfs-outside-')) diff --git a/apps/desktop/src/main/local-filesystem.ts b/apps/desktop/src/main/local-filesystem.ts index bc3642c3ca7..5e12f7060ee 100644 --- a/apps/desktop/src/main/local-filesystem.ts +++ b/apps/desktop/src/main/local-filesystem.ts @@ -75,6 +75,22 @@ class LocalFilesystemError extends Error { } } +function mountNotFound(): LocalFilesystemError { + return new LocalFilesystemError( + 'MOUNT_NOT_FOUND', + 'That local folder is no longer available. Select it again.' + ) +} + +/** Operations that read inside one granted folder, named by the request's `uri`. */ +const GRANT_SCOPED_OPERATIONS: ReadonlySet = new Set([ + 'list', + 'glob', + 'read', + 'grep', + 'stat', +]) + interface GrantedMount extends LocalFilesystemMount { rootPath: string bookmark?: string @@ -157,7 +173,8 @@ function normalizeVfsDisplaySegment(segment: string): string { .replace(/\s+/g, ' ') } -function mountVfsRoot(mount: GrantedMount): string { +/** The `user-local/--` directory a granted folder appears under in the model's VFS. */ +export function mountVfsRoot(mount: Pick): string { return `user-local/${encodeURIComponent(normalizeVfsDisplaySegment(mount.name))}--${mount.id}` } @@ -420,7 +437,13 @@ export class LocalFilesystemService { } let data: LocalFilesystemData + // Reads and searches answer only while their folder is still granted: one forgotten while + // they ran must not hand back what they found in it. + let grant: GrantedMount | null = null try { + if (GRANT_SCOPED_OPERATIONS.has(request.operation)) { + grant = this.parseUri(this.requiredUri(request)).mount + } switch (request.operation) { case 'mount_directory': data = await this.mountDirectory() @@ -470,6 +493,7 @@ export class LocalFilesystemService { this.activeRequests.delete(requestId) } } + if (grant && this.mounts.get(grant.id)?.rootPath !== grant.rootPath) throw mountNotFound() return { ok: true, data } } catch (error) { const safe = safeError(error) @@ -889,12 +913,7 @@ export class LocalFilesystemService { } const mount = this.mounts.get(parsed.hostname) - if (!mount) { - throw new LocalFilesystemError( - 'MOUNT_NOT_FOUND', - 'That local folder is no longer available. Select it again.' - ) - } + if (!mount) throw mountNotFound() const encodedSegments = parsed.pathname.split('/').filter(Boolean) const segments = encodedSegments.map((segment) => { diff --git a/apps/desktop/src/main/terminal/index.ts b/apps/desktop/src/main/terminal/index.ts index 925330f72fb..56dfe8b8348 100644 --- a/apps/desktop/src/main/terminal/index.ts +++ b/apps/desktop/src/main/terminal/index.ts @@ -37,20 +37,24 @@ import { isResourceTabSelectionShortcut, resourceTabTargetIndex, } from '@/main/resource-shortcuts' +import { readForegroundProcessGroup, signalProcessGroup } from '@/main/terminal/process-group' import { elide, TerminalSession } from '@/main/terminal/session' import { activePane, awaitRun, capturePane, - closeRunWindow, + closeRunPane, isRunComplete, isTmuxUnavailable, killPane, listPanes, + pollRun, resolveAttachment, + runPaneState, sendKey, sendText, startRun, + stopRun, TMUX_KEY_NAMES, type TmuxAttachment, type TmuxRunHandle, @@ -150,8 +154,59 @@ export interface TerminalServiceOptions { */ loadCwd?(): string | undefined canSpawn?(): boolean + /** Reads and signals a terminal's foreground process group; the OS's by default. */ + processGroups?: TerminalProcessGroups } +interface TerminalProcessGroups { + foreground(shellPid: number): Promise + signal(pgid: number, signal: NodeJS.Signals): void +} + +const OS_PROCESS_GROUPS: TerminalProcessGroups = { + foreground: readForegroundProcessGroup, + signal: signalProcessGroup, +} + +/** + * One tool call's Stop: remembered if it arrives before the call's command starts, and handed to + * whatever is running once it does. + */ +interface StopLatch { + /** Aborts on Stop; work not yet started checks it, and input stops between keystrokes. */ + readonly signal: AbortSignal + stopRunning: (() => Promise) | null +} + +function stoppedBeforeStart(): TerminalError { + return new TerminalError( + 'CANCELLED', + 'Stopped before the command started, so nothing ran in the terminal.' + ) +} + +function stoppedPartWay(): TerminalError { + return new TerminalError( + 'CANCELLED', + 'Stopped part way through the input; some of it may already have reached the terminal.' + ) +} + +/** Operations that change a terminal; a Stop that lands before one starts means it never does. */ +const TERMINAL_CHANGING_OPERATIONS: ReadonlySet = new Set([ + 'close', + 'handoff', + 'input', + 'kill', + 'run', +]) + +/** + * How long a stopped command gets to exit after each escalation: Ctrl-C first, as the user would + * press it, then SIGTERM and finally SIGKILL to its process group. + */ +const STOP_ESCALATION_MS = 2_000 + export class TerminalService { /** Insertion-ordered, which is also the tab order the user sees. */ private readonly sessions = new Map() @@ -193,6 +248,17 @@ export class TerminalService { * Held here so the terminal's own lifecycle can reclaim them. */ private readonly pendingRuns = new Map() + /** + * Agent runs still going in tmux after their Sim terminal closed: the tmux session outlives the + * tab, but sign-out must still stop them. + */ + private readonly orphanedRuns = new Map() + /** Runs a `run` call is still waiting on; their files are read when the wait ends. */ + private readonly awaitedRuns = new Set() + /** Awaited runs released meanwhile (their terminal closed); their files go once the wait ends. */ + private readonly releasedAwaitedRuns = new Set() + /** How to stop each tool call still in flight, so Stop interrupts exactly what it started. */ + private readonly toolStops = new Map Promise>() constructor(private readonly options: TerminalServiceOptions = {}) {} @@ -412,33 +478,52 @@ export class TerminalService { } /** - * Removes the temp directories of tracked runs that have since finished. + * Removes the temp directories of tracked runs that have since finished, or whose pane is gone + * (the user closed it, or tmux restarted): nothing will ever write their status, and their ids + * may already belong to the user's own panes. * * Called when a new run starts on the same terminal, which is the one moment * the service is already doing run bookkeeping — a dedicated reaper timer * would be a subsystem to own for something this cheap. A run still going is * left alone: its `tee` is still appending to that directory. */ - private reapFinishedRuns(terminalId: string): void { - const pending = this.pendingRuns.get(terminalId) - if (!pending) return - const stillRunning: TmuxRunHandle[] = [] - for (const handle of pending) { - if (isRunComplete(handle)) handle.dispose() - else stillRunning.push(handle) + private async reapFinishedRuns(terminalId: string, env: NodeJS.ProcessEnv): Promise { + // A closed tab's run whose pane has since gone (its command ended) needs no stopping. + for (const [handle, orphanEnv] of this.orphanedRuns) { + if ((await runPaneState(handle, orphanEnv)) === 'gone') this.orphanedRuns.delete(handle) + } + for (const handle of this.pendingRuns.get(terminalId) ?? []) { + if (this.awaitedRuns.has(handle)) continue + if (isRunComplete(handle) || (await runPaneState(handle, env)) === 'gone') { + this.untrackRun(terminalId, handle) + handle.dispose() + } } - if (stillRunning.length === 0) this.pendingRuns.delete(terminalId) - else this.pendingRuns.set(terminalId, stillRunning) + } + + /** Removes a run's files now, or once the call still reading them is done with them. */ + private releaseRun(handle: TmuxRunHandle): void { + if (this.awaitedRuns.has(handle)) this.releasedAwaitedRuns.add(handle) + else handle.dispose() + } + + private untrackRun(terminalId: string, handle: TmuxRunHandle): void { + const remaining = (this.pendingRuns.get(terminalId) ?? []).filter((entry) => entry !== handle) + if (remaining.length === 0) this.pendingRuns.delete(terminalId) + else this.pendingRuns.set(terminalId, remaining) } /** * Releases every tracked run for a terminal, finished or not. The terminal is * going away, so nothing will ever read these files again. */ - private releasePendingRuns(terminalId: string): void { + private releasePendingRuns(terminalId: string, env?: NodeJS.ProcessEnv): void { const pending = this.pendingRuns.get(terminalId) if (!pending) return - for (const handle of pending) handle.dispose() + for (const handle of pending) { + if (env && !isRunComplete(handle)) this.orphanedRuns.set(handle, env) + this.releaseRun(handle) + } this.pendingRuns.delete(terminalId) } @@ -452,10 +537,11 @@ export class TerminalService { const closedCwd = session.currentCwd const order = [...this.sessions.keys()] const index = order.indexOf(terminalId) + const env = session.env session.dispose() this.sessions.delete(terminalId) this.tmuxCache.delete(terminalId) - this.releasePendingRuns(terminalId) + this.releasePendingRuns(terminalId, env) this.rememberClosed(closedCwd) // Nothing is left for the user to hold on to; the next shell the agent @@ -749,7 +835,7 @@ export class TerminalService { this.sessions.clear() this.tmuxCache.clear() for (const handles of this.pendingRuns.values()) { - for (const handle of handles) handle.dispose() + for (const handle of handles) this.releaseRun(handle) } this.pendingRuns.clear() this.activeId = null @@ -766,8 +852,17 @@ export class TerminalService { operation: TerminalOperation, args: TerminalToolArgs ): Promise { + // A Stop can arrive before the command exists (while the shell or tmux is still being + // resolved); the latch carries it to the moment the command would start. + const halt = new AbortController() + const latch: StopLatch = { signal: halt.signal, stopRunning: null } + const stop = async () => { + halt.abort() + await latch.stopRunning?.() + } + this.toolStops.set(toolCallId, stop) try { - const result = await this.dispatch(toolCallId, operation, args ?? {}) + const result = await this.dispatch(toolCallId, operation, args ?? {}, latch) return { ok: true, result } } catch (error) { if (error instanceof TerminalError) { @@ -777,13 +872,88 @@ export class TerminalService { const message = (error as Error).message logger.error('Terminal operation failed', { toolCallId, operation, error: message }) return { ok: false, error: message } + } finally { + if (this.toolStops.get(toolCallId) === stop) this.toolStops.delete(toolCallId) + } + } + + /** + * Stops a tool call still in flight: interrupts the command a `run` started (escalating to its + * process group if Ctrl-C does not end it) or ends a handoff. The call then returns its result + * as usual. False when this service is not running that call. + */ + async cancelTool(toolCallId: string): Promise { + const stop = this.toolStops.get(toolCallId) + if (!stop) return false + await stop() + return true + } + + /** + * Stops every command the agent started that is still running, for sign-out: a plain shell's + * agent command by its own process group, as Stop does, and every tmux run window still going. + * A command the user started is not the agent's and is left alone. + */ + async stopAgentCommands(): Promise { + const stops: Promise[] = [] + for (const session of this.sessions.values()) { + const toolCallId = session.agentCommandToolCallId + if (toolCallId) stops.push(this.stopCommand(session, toolCallId)) + for (const handle of this.pendingRuns.get(session.terminalId) ?? []) { + if (!isRunComplete(handle)) stops.push(stopRun(handle, session.env, STOP_ESCALATION_MS)) + } + } + for (const [handle, env] of this.orphanedRuns) { + stops.push(stopRun(handle, env, STOP_ESCALATION_MS)) + } + this.orphanedRuns.clear() + await Promise.allSettled(stops) + } + + /** Waits for the command a run started to end, up to `ms`. */ + private async commandEnds( + session: TerminalSession, + toolCallId: string, + ms: number + ): Promise { + const deadline = Date.now() + ms + while (session.agentCommandToolCallId === toolCallId) { + if (Date.now() >= deadline || !session.alive) { + return session.agentCommandToolCallId !== toolCallId + } + await sleep(50) + } + return true + } + + /** + * Interrupts the command a run started. Escalation is bound to that command's own process + * group, read while it still holds the foreground, and each signal is sent only while the same + * call and the same group still hold it, so a command the user starts meanwhile is never hit. + */ + private async stopCommand(session: TerminalSession, toolCallId: string): Promise { + if (session.agentCommandToolCallId !== toolCallId) return + const groups = this.options.processGroups ?? OS_PROCESS_GROUPS + const pgid = await groups.foreground(session.pid) + if (session.agentCommandToolCallId !== toolCallId) return + session.kill('SIGINT') + for (const escalation of ['SIGTERM', 'SIGKILL'] as const) { + if (await this.commandEnds(session, toolCallId, STOP_ESCALATION_MS)) return + if (pgid === null || (await groups.foreground(session.pid)) !== pgid) return + if (session.agentCommandToolCallId !== toolCallId) return + logger.info('Stopped command ignored the previous signal; escalating', { + toolCallId, + signal: escalation, + }) + groups.signal(pgid, escalation) } } private async dispatch( toolCallId: string, operation: TerminalOperation, - args: TerminalToolArgs + args: TerminalToolArgs, + latch: StopLatch ): Promise { switch (operation) { case 'list': @@ -807,6 +977,10 @@ export class TerminalService { // A tab either has tmux attached or it does not, and every operation below // behaves differently depending on which. const tmux = await this.resolveTmux(session) + // A Stop that landed while the session resolved: nothing that changes the terminal starts. + if (latch.signal.aborted && TERMINAL_CHANGING_OPERATIONS.has(operation)) { + throw stoppedBeforeStart() + } switch (operation) { case 'cwd': @@ -824,6 +998,7 @@ export class TerminalService { ) } const target = await this.resolvePane(tmux.session, args, session) + if (latch.signal.aborted) throw stoppedBeforeStart() const killed = await killPane(target, session.env) if (!killed.ok) { throw new TerminalError( @@ -838,6 +1013,8 @@ export class TerminalService { } } case 'handoff': + if (latch.signal.aborted) throw stoppedBeforeStart() + latch.stopRunning = async () => this.finishHandoff(session.terminalId) return this.handoff(session, args) case 'panes': { if (!tmux) { @@ -854,8 +1031,8 @@ export class TerminalService { } case 'run': return tmux - ? this.runInTmux(session, tmux.session, args) - : this.run(toolCallId, session, args) + ? this.runInTmux(session, tmux.session, args, latch) + : this.run(toolCallId, session, args, latch) case 'read': { const requested = Number(args.lines) const lines = Number.isFinite(requested) && requested > 0 ? requested : 200 @@ -879,8 +1056,8 @@ export class TerminalService { } case 'input': return tmux - ? this.inputToTmux(session, tmux.session, args) - : this.inputToShell(session, args) + ? this.inputToTmux(session, tmux.session, args, latch.signal) + : this.inputToShell(session, args, latch.signal) case 'kill': { const signal = args.signal === 'SIGTERM' || args.signal === 'SIGKILL' || args.signal === 'SIGINT' @@ -891,6 +1068,7 @@ export class TerminalService { // whole session rather than stopping the one thing they asked about. if (tmux) { const target = await this.resolvePane(tmux.session, args, session) + if (latch.signal.aborted) throw stoppedBeforeStart() await sendKey(target, signal === 'SIGKILL' ? 'C-\\' : 'C-c', session.env) return { signal, terminalId: session.terminalId, pane: target } } @@ -1006,15 +1184,18 @@ export class TerminalService { private async inputToTmux( terminal: TerminalSession, session: string, - args: TerminalToolArgs + args: TerminalToolArgs, + signal: AbortSignal ): Promise { const target = await this.resolvePane(session, args, terminal) + if (signal.aborted) throw stoppedBeforeStart() const keys = requestedKeys(args) if (keys.length > 0) { for (let index = 0; index < keys.length; index += 1) { // Paced like the pty path: a pane redraws between presses, so a batch // lands where the same keys pressed by hand would. if (index > 0) await sleep(TMUX_KEY_GAP_MS) + if (signal.aborted) throw stoppedPartWay() await sendKey(target, TMUX_KEY_NAMES[keys[index]] ?? keys[index], terminal.env) } } else if (typeof args.text === 'string') { @@ -1022,6 +1203,7 @@ export class TerminalService { // Enter is a separate send-keys for the same reason it is a separate pty // write: a program reading one chunk treats text plus a carriage return // as text, and the message sits unsubmitted. + if (signal.aborted) throw stoppedPartWay() if (/[\r\n]$/.test(args.text)) await sendKey(target, 'Enter', terminal.env) } else { throw new TerminalError('INVALID_REQUEST', 'input needs `text`, `key`, or `keys`.') @@ -1037,7 +1219,11 @@ export class TerminalService { } } - private async inputToShell(session: TerminalSession, args: TerminalToolArgs): Promise { + private async inputToShell( + session: TerminalSession, + args: TerminalToolArgs, + signal: AbortSignal + ): Promise { // Input is only ever delivered to a program that already holds the // foreground. At a bare shell prompt these bytes would be a command // line, and running commands that way would bypass the capture and @@ -1052,14 +1238,17 @@ export class TerminalService { // lets the model assume its message went through and start waiting on // a reply to text still sitting unsubmitted in a composer; the screen // is the evidence of what the program actually did with the input. + if (signal.aborted) throw stoppedBeforeStart() const keys = requestedKeys(args) if (keys.length > 0) { - await session.pressKeys(keys) + await session.pressKeys(keys, signal) + if (signal.aborted) throw stoppedPartWay() await sleep(INPUT_ECHO_MS) return { sent: keys.join(', '), ...(await session.readScrollback(INPUT_SCREEN_LINES)) } } if (typeof args.text === 'string') { - await session.type(args.text) + await session.type(args.text, signal) + if (signal.aborted) throw stoppedPartWay() await sleep(INPUT_ECHO_MS) return { sent: args.text, ...(await session.readScrollback(INPUT_SCREEN_LINES)) } } @@ -1078,28 +1267,51 @@ export class TerminalService { private async runInTmux( terminal: TerminalSession, session: string, - args: TerminalToolArgs + args: TerminalToolArgs, + latch: StopLatch ): Promise { const command = typeof args.command === 'string' ? args.command.trim() : '' if (!command) throw new TerminalError('INVALID_REQUEST', 'run needs a `command`.') + if (latch.signal.aborted) throw stoppedBeforeStart() const started = Date.now() - this.reapFinishedRuns(terminal.terminalId) + await this.reapFinishedRuns(terminal.terminalId, terminal.env) const handle = await startRun(session, command, terminal.currentCwd, terminal.env) if ('error' in handle) throw new TerminalError('SPAWN_FAILED', handle.error) + // Tracked from the moment its window exists, so sign-out can stop it even mid-wait. + const pending = this.pendingRuns.get(terminal.terminalId) + if (pending) pending.push(handle) + else this.pendingRuns.set(terminal.terminalId, [handle]) + this.awaitedRuns.add(handle) const waitMs = resolveRunWaitMs(args.waitSeconds) - const outcome = await awaitRun(handle, waitMs) + // Inside tmux a stop arrives as Ctrl-C in the run's own window; closing that window hangs up + // anything that ignored it. The wait ends with the stop, since a closed window never writes + // the run's exit status. + let endWait: () => void = () => {} + const stopped = new Promise((resolve) => { + endWait = resolve + }) + latch.stopRunning = async () => { + await stopRun(handle, terminal.env, STOP_ESCALATION_MS) + endWait() + } + // A Stop that landed while the run window opened applies now. + if (latch.signal.aborted) void latch.stopRunning() + const outcome = await Promise.race([ + awaitRun(handle, waitMs), + stopped.then(() => ({ ...pollRun(handle), done: true })), + ]).finally(() => { + this.awaitedRuns.delete(handle) + if (this.releasedAwaitedRuns.delete(handle)) handle.dispose() + }) if (outcome.done) { - await closeRunWindow(handle, terminal.env) + await closeRunPane(handle, terminal.env) + this.untrackRun(terminal.terminalId, handle) handle.dispose() - } else { - // Still going, and nothing polls the status file again — `read` captures - // the pane instead. - const pending = this.pendingRuns.get(terminal.terminalId) - if (pending) pending.push(handle) - else this.pendingRuns.set(terminal.terminalId, [handle]) } + // Still going, it stays tracked, and nothing polls the status file again: `read` captures + // the pane instead. const { text, truncated } = elideOutput(outcome.output) return { @@ -1110,7 +1322,7 @@ export class TerminalService { durationMs: Date.now() - started, cwd: terminal.currentCwd, terminalId: terminal.terminalId, - pane: handle.window, + pane: handle.pane, truncated, } } @@ -1118,7 +1330,8 @@ export class TerminalService { private async run( toolCallId: string, session: TerminalSession, - args: TerminalToolArgs + args: TerminalToolArgs, + latch: StopLatch ): Promise { const command = typeof args.command === 'string' ? args.command.trim() : '' if (!command) { @@ -1140,6 +1353,8 @@ export class TerminalService { ) } + if (latch.signal.aborted) throw stoppedBeforeStart() + latch.stopRunning = () => this.stopCommand(session, toolCallId) return session.runCommand(command, toolCallId, resolveRunWaitMs(args.waitSeconds)) } diff --git a/apps/desktop/src/main/terminal/pending-runs.test.ts b/apps/desktop/src/main/terminal/pending-runs.test.ts index ccb81aa3322..77a895b9252 100644 --- a/apps/desktop/src/main/terminal/pending-runs.test.ts +++ b/apps/desktop/src/main/terminal/pending-runs.test.ts @@ -20,17 +20,21 @@ vi.mock('@/main/terminal/tmux', () => ({ exitCode: tmuxStub.done ? 0 : null, })), capturePane: vi.fn(async () => ({ ok: true, stdout: '', stderr: '' })), - closeRunWindow: vi.fn(async () => undefined), + closeRunPane: vi.fn(async () => undefined), isRunComplete: vi.fn((handle: { window: string }) => tmuxStub.complete.has(handle.window)), + runPaneState: vi.fn(async () => 'ours'), isTmuxUnavailable: vi.fn(() => false), killPane: vi.fn(async () => ({ ok: true, stdout: '', stderr: '' })), listPanes: vi.fn(async () => []), resolveAttachment: vi.fn(async () => ({ session: 'sess' })), sendKey: vi.fn(async () => ({ ok: true, stdout: '', stderr: '' })), sendText: vi.fn(async () => ({ ok: true, stdout: '', stderr: '' })), + stopRun: vi.fn(async () => undefined), startRun: vi.fn(async () => { const handle = { window: `sess:${tmuxStub.handles.length}`, + pane: `%${tmuxStub.handles.length}`, + runId: `run-${tmuxStub.handles.length}`, outPath: '/tmp/fake/out', statusPath: '/tmp/fake/status', dispose: vi.fn(), diff --git a/apps/desktop/src/main/terminal/process-group.ts b/apps/desktop/src/main/terminal/process-group.ts new file mode 100644 index 00000000000..6e5a4e10a87 --- /dev/null +++ b/apps/desktop/src/main/terminal/process-group.ts @@ -0,0 +1,62 @@ +/** + * The process group a terminal's tty currently delivers keyboard signals to: the whole pipeline + * the user would stop with Ctrl-C, including its children, and never the shell itself. + */ +import { spawn } from 'node:child_process' +import { createLogger } from '@sim/logger' +import { getErrorMessage } from '@sim/utils/errors' + +const logger = createLogger('DesktopTerminalProcessGroup') + +const LOOKUP_TIMEOUT_MS = 2_000 + +/** + * The tty's foreground process group, read from `ps`. Null when it cannot be read, or when the + * shell itself holds the foreground (it is sitting at its prompt). + */ +export function readForegroundProcessGroup(shellPid: number): Promise { + if (!Number.isInteger(shellPid) || shellPid <= 0) return Promise.resolve(null) + return new Promise((resolve) => { + let child: ReturnType + try { + child = spawn('ps', ['-o', 'tpgid=', '-p', String(shellPid)], { + stdio: ['ignore', 'pipe', 'ignore'], + }) + } catch { + resolve(null) + return + } + let stdout = '' + let settled = false + const finish = (value: number | null) => { + if (settled) return + settled = true + clearTimeout(timer) + resolve(value) + } + const timer = setTimeout(() => { + child.kill('SIGKILL') + finish(null) + }, LOOKUP_TIMEOUT_MS) + child.stdout?.on('data', (chunk: Buffer) => { + stdout += chunk.toString() + }) + child.on('error', () => finish(null)) + child.on('close', () => { + const pgid = Number.parseInt(stdout.trim(), 10) + finish(Number.isInteger(pgid) && pgid > 0 && pgid !== shellPid ? pgid : null) + }) + }) +} + +/** Sends `signal` to every process in one group. */ +export function signalProcessGroup(pgid: number, signal: NodeJS.Signals): void { + try { + process.kill(-pgid, signal) + } catch (error) { + logger.warn('Could not signal a terminal process group', { + signal, + error: getErrorMessage(error), + }) + } +} diff --git a/apps/desktop/src/main/terminal/registry.ts b/apps/desktop/src/main/terminal/registry.ts index db737cad65f..0bdf7548537 100644 --- a/apps/desktop/src/main/terminal/registry.ts +++ b/apps/desktop/src/main/terminal/registry.ts @@ -234,6 +234,11 @@ export class TerminalRegistry { return entry.service.executeTool(toolCallId, operation, args) } + /** Stops one in-flight tool call in a chat's terminals; false when none is running it. */ + cancelTool(scope: string, toolCallId: string): Promise { + return this.entries.get(scope)?.service.cancelTool(toolCallId) ?? Promise.resolve(false) + } + /** * Moves a provisional chat's live service to its durable chat id. * @@ -364,6 +369,13 @@ export class TerminalRegistry { return true } + /** Stops every command the agent started in any chat's terminals; the user's own are untouched. */ + async stopAgentCommands(): Promise { + await Promise.allSettled( + [...this.entries.values()].map((entry) => entry.service.stopAgentCommands()) + ) + } + /** Tears down every shell owned by every chat scope. */ dispose(): void { const entries = [...this.entries.values()] diff --git a/apps/desktop/src/main/terminal/service.test.ts b/apps/desktop/src/main/terminal/service.test.ts index 155ec66668f..24256d6b30f 100644 --- a/apps/desktop/src/main/terminal/service.test.ts +++ b/apps/desktop/src/main/terminal/service.test.ts @@ -1,11 +1,83 @@ +import { existsSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import { describe, expect, it, vi } from 'vitest' import { TerminalService } from '@/main/terminal' +/** + * A tmux attachment the service sees only when a test turns it on. Runs get real status files; + * stopping a run ends its command the way Ctrl-C in its pane would. Which panes tmux still holds + * as the run's, and the identity checks behind that, are covered against a fake tmux binary in + * `tmux.test.ts`; here the service's bookkeeping is what is under test. + */ +const tmuxFake = vi.hoisted(() => ({ + on: false, + /** Panes a run was stopped in, in order. */ + stopped: [] as string[], + /** Panes no longer the run's (closed by the user, or reused after a tmux restart). */ + gone: new Set(), + statusPaths: new Map(), +})) + +vi.mock('@/main/terminal/tmux', async () => { + const actual = + await vi.importActual('@/main/terminal/tmux') + let nextPane = 1 + return { + ...actual, + isTmuxUnavailable: () => (tmuxFake.on ? false : actual.isTmuxUnavailable()), + resolveAttachment: async (pid: number, env: NodeJS.ProcessEnv) => + tmuxFake.on + ? { session: 'agent', clientTty: '/dev/ttys001' } + : actual.resolveAttachment(pid, env), + startRun: async (...args: Parameters) => { + if (!tmuxFake.on) return actual.startRun(...args) + const dir = mkdtempSync(join(tmpdir(), 'sim-tmux-fake-')) + const pane = `%${nextPane++}` + const statusPath = join(dir, 'status') + writeFileSync(join(dir, 'out'), 'partial output') + tmuxFake.statusPaths.set(pane, statusPath) + return { + window: `@${pane.slice(1)}`, + pane, + runId: `run-${pane}`, + outPath: join(dir, 'out'), + statusPath, + dispose: () => rmSync(dir, { recursive: true, force: true }), + } + }, + runPaneState: async (...args: Parameters) => { + if (!tmuxFake.on) return actual.runPaneState(...args) + return tmuxFake.gone.has(args[0].pane) ? 'gone' : 'ours' + }, + stopRun: async (...args: Parameters) => { + if (!tmuxFake.on) return actual.stopRun(...args) + const [handle] = args + if (tmuxFake.gone.has(handle.pane)) return + tmuxFake.stopped.push(handle.pane) + writeFileSync(handle.statusPath, '130') + }, + closeRunPane: async (...args: Parameters) => { + if (!tmuxFake.on) return actual.closeRunPane(...args) + }, + } +}) + /** Stub sessions by terminal id, populated by the mock below. */ const { stubSessions } = vi.hoisted(() => ({ stubSessions: new Map< string, - { setBusy(busy: boolean): void; exit(): void; clearScrollback(): void } + { + setBusy(busy: boolean): void + exit(): void + clearScrollback(): void + /** Whether Ctrl-C ends the running command, as it does for most programs. */ + setInterruptible(interruptible: boolean): void + /** Ends the running command the way its process exiting would. */ + finishRun(exitCode: number): void + kill: ReturnType + readonly runningToolCallId: string | null + } >(), })) @@ -32,11 +104,45 @@ vi.mock('@/main/terminal/session', async () => { terminalId: string callbacks: { onExit(terminalId: string): void } }) => { - const state = { cwd, disposed: false, busy: false } + const state = { + cwd, + disposed: false, + busy: false, + interruptible: true, + toolCallId: null as string | null, + resolveRun: null as ((result: Record) => void) | null, + } + const finishRun = (exitCode: number) => { + const resolve = state.resolveRun + state.busy = false + state.toolCallId = null + state.resolveRun = null + resolve?.({ status: 'completed', exitCode, terminalId }) + } const stub = { setBusy: (busy: boolean) => { state.busy = busy }, + setInterruptible: (interruptible: boolean) => { + state.interruptible = interruptible + }, + finishRun, + runCommand: (_command: string, toolCallId: string) => + new Promise((resolve) => { + state.busy = true + state.toolCallId = toolCallId + state.resolveRun = resolve + }), + get runningToolCallId() { + return state.toolCallId + }, + get agentCommandToolCallId() { + return state.toolCallId + }, + kill: vi.fn((signal: string) => { + if (signal === 'SIGINT' && state.interruptible) finishRun(130) + }), + waitForShellIntegration: async () => {}, /** Stands in for the user running `exit` or pressing Ctrl-D. */ exit: () => { state.disposed = true @@ -332,3 +438,226 @@ describe('a shell that ends by itself', () => { expect(after.activeTerminalId).toBeNull() }) }) + +describe('stopping a tool call', () => { + const COMMAND_GROUP = 4242 + + function cancellableService() { + /** The tty's foreground group as the OS would report it at each read. */ + let foreground: number | null = COMMAND_GROUP + const processGroups = { + foreground: vi.fn(async (_shellPid: number) => foreground), + signal: vi.fn((_pgid: number, _signal: NodeJS.Signals) => {}), + } + const terminal = new TerminalService({ loadCwd: () => '/tmp', processGroups }) + const { activeTerminalId } = terminal.start({ cols: 80, rows: 24 }) + const session = stubSessions.get(activeTerminalId as string) + if (!session) throw new Error('No stub session') + return { + terminal, + session, + processGroups, + setForeground: (pgid: number | null) => { + foreground = pgid + }, + } + } + + /** Starts a run and waits until its command holds the terminal's foreground. */ + async function startRun( + terminal: TerminalService, + session: { runningToolCallId: string | null }, + toolCallId: string + ) { + const running = terminal.executeTool(toolCallId, 'run', { command: 'sleep 60' }) + await vi.waitFor(() => expect(session.runningToolCallId).toBe(toolCallId)) + return { running } + } + + it('interrupts the command its run started, and the run reports how it ended', async () => { + const { terminal, session, processGroups } = cancellableService() + const { running } = await startRun(terminal, session, 'call-run') + + await expect(terminal.cancelTool('call-run')).resolves.toBe(true) + + expect(session.kill).toHaveBeenCalledWith('SIGINT') + await expect(running).resolves.toMatchObject({ ok: true, result: { exitCode: 130 } }) + expect(processGroups.signal).not.toHaveBeenCalled() + }) + + it("terminates the command's own process group when it ignores Ctrl-C", async () => { + const { terminal, session, processGroups } = cancellableService() + session.setInterruptible(false) + processGroups.signal.mockImplementation(() => session.finishRun(143)) + const { running } = await startRun(terminal, session, 'call-stubborn') + + await terminal.cancelTool('call-stubborn') + + expect(processGroups.signal).toHaveBeenCalledWith(COMMAND_GROUP, 'SIGTERM') + await expect(running).resolves.toMatchObject({ ok: true, result: { exitCode: 143 } }) + }) + + it('never signals a different command that took the foreground meanwhile', async () => { + const { terminal, session, processGroups, setForeground } = cancellableService() + session.setInterruptible(false) + const { running } = await startRun(terminal, session, 'call-replaced') + session.kill.mockImplementation(() => setForeground(9999)) + + await terminal.cancelTool('call-replaced') + + expect(processGroups.signal).not.toHaveBeenCalled() + session.finishRun(0) + await running + }) + + it('runs nothing when the Stop arrives before the command starts', async () => { + const { terminal, session } = cancellableService() + const running = terminal.executeTool('call-early', 'run', { command: 'rm -rf build' }) + await terminal.cancelTool('call-early') + + await expect(running).resolves.toMatchObject({ ok: false, code: 'CANCELLED' }) + expect(session.runningToolCallId).toBeNull() + }) + + it('sends no signal or keys for a kill or input stopped before it starts', async () => { + const { terminal, session } = cancellableService() + session.setBusy(true) + const killing = terminal.executeTool('call-kill', 'kill', { signal: 'SIGTERM' }) + const typing = terminal.executeTool('call-input', 'input', { text: 'yes\n' }) + await Promise.all([terminal.cancelTool('call-kill'), terminal.cancelTool('call-input')]) + + await expect(killing).resolves.toMatchObject({ ok: false, code: 'CANCELLED' }) + await expect(typing).resolves.toMatchObject({ ok: false, code: 'CANCELLED' }) + expect(session.kill).not.toHaveBeenCalled() + }) + + it("stops the agent's running command at sign-out", async () => { + const { terminal, session } = cancellableService() + const { running } = await startRun(terminal, session, 'call-left-running') + + await terminal.stopAgentCommands() + + expect(session.kill).toHaveBeenCalledWith('SIGINT') + await expect(running).resolves.toMatchObject({ ok: true, result: { exitCode: 130 } }) + }) + + it('leaves a command the user started alone at sign-out', async () => { + const { terminal, session, processGroups } = cancellableService() + session.setBusy(true) + + await terminal.stopAgentCommands() + + expect(session.kill).not.toHaveBeenCalled() + expect(processGroups.signal).not.toHaveBeenCalled() + }) + + it('leaves the terminal alone for a call it is not running', async () => { + const { terminal, session, processGroups } = cancellableService() + const { running } = await startRun(terminal, session, 'call-other') + + await expect(terminal.cancelTool('call-unknown')).resolves.toBe(false) + expect(session.kill).not.toHaveBeenCalled() + expect(processGroups.signal).not.toHaveBeenCalled() + session.finishRun(0) + await running + }) +}) + +describe('agent commands in tmux', () => { + it('stops a tmux run at sign-out while its call still waits on it', async () => { + tmuxFake.on = true + try { + const terminal = new TerminalService({ loadCwd: () => '/tmp' }) + terminal.start({ cols: 80, rows: 24 }) + const running = terminal.executeTool('call-tmux', 'run', { + command: 'sleep 600', + waitSeconds: 60, + }) + await vi.waitFor(() => expect(tmuxFake.statusPaths.size).toBe(1)) + + await terminal.stopAgentCommands() + + expect(tmuxFake.stopped).toEqual(['%1']) + await expect(running).resolves.toMatchObject({ + ok: true, + result: { status: 'completed', exitCode: 130 }, + }) + } finally { + tmuxFake.on = false + } + }) + + it("keeps a run's output readable for its call when the terminal goes away mid-wait", async () => { + tmuxFake.on = true + tmuxFake.statusPaths.clear() + try { + const terminal = new TerminalService({ loadCwd: () => '/tmp' }) + terminal.start({ cols: 80, rows: 24 }) + const running = terminal.executeTool('call-tmux-closed', 'run', { + command: 'make build', + waitSeconds: 1, + }) + await vi.waitFor(() => expect(tmuxFake.statusPaths.size).toBe(1)) + + terminal.dispose() + + await expect(running).resolves.toMatchObject({ + ok: true, + result: { status: 'running', output: 'partial output' }, + }) + } finally { + tmuxFake.on = false + } + }) + + it('releases a run whose pane is gone instead of ever stopping it', async () => { + tmuxFake.on = true + tmuxFake.statusPaths.clear() + tmuxFake.stopped.length = 0 + try { + const terminal = new TerminalService({ loadCwd: () => '/tmp' }) + terminal.start({ cols: 80, rows: 24 }) + const first = await terminal.executeTool('call-first', 'run', { + command: 'sleep 600', + waitSeconds: 1, + }) + expect(first).toMatchObject({ ok: true, result: { status: 'running' } }) + const [pane, statusPath] = [...tmuxFake.statusPaths][0] ?? [] + tmuxFake.gone.add(pane ?? '') + + await terminal.stopAgentCommands() + expect(tmuxFake.stopped).toEqual([]) + + // The next run's bookkeeping drops it for good. + void terminal.executeTool('call-next', 'run', { command: 'ls', waitSeconds: 1 }) + await vi.waitFor(() => expect(existsSync(statusPath ?? '')).toBe(false)) + await vi.waitFor(() => expect(tmuxFake.statusPaths.size).toBe(2)) + expect(existsSync(join(statusPath ?? '', '..'))).toBe(false) + } finally { + tmuxFake.on = false + tmuxFake.gone.clear() + } + }) + + it('still stops at sign-out a run whose Sim terminal was closed', async () => { + tmuxFake.on = true + tmuxFake.statusPaths.clear() + tmuxFake.stopped.length = 0 + try { + const terminal = new TerminalService({ loadCwd: () => '/tmp' }) + const { activeTerminalId } = terminal.start({ cols: 80, rows: 24 }) + const first = await terminal.executeTool('call-orphan', 'run', { + command: 'sleep 600', + waitSeconds: 1, + }) + expect(first).toMatchObject({ ok: true, result: { status: 'running' } }) + terminal.closeTerminal(activeTerminalId as string) + + await terminal.stopAgentCommands() + + expect(tmuxFake.stopped).toEqual([[...tmuxFake.statusPaths.keys()][0]]) + } finally { + tmuxFake.on = false + } + }) +}) diff --git a/apps/desktop/src/main/terminal/session.test.ts b/apps/desktop/src/main/terminal/session.test.ts index 879f2b6fcd6..45262298490 100644 --- a/apps/desktop/src/main/terminal/session.test.ts +++ b/apps/desktop/src/main/terminal/session.test.ts @@ -137,6 +137,9 @@ describe('TerminalSession command lifecycle', () => { const resultPromise = session.runCommand('vim', 'tool-call-1', 10_000) ptyStub.dataHandler?.('\u001b]633;C;test-nonce\u0007\u001b[?1049h') expect(await resultPromise).toMatchObject({ status: 'interactive' }) + // Detached from the tool call, the command is still the agent's until it exits. + expect(session.runningToolCallId).toBeNull() + expect(session.agentCommandToolCallId).toBe('tool-call-1') expect(commandEvents.at(-1)).toMatchObject({ terminalId: 'terminal-1', @@ -153,6 +156,7 @@ describe('TerminalSession command lifecycle', () => { command: 'vim', exitCode: 0, }) + expect(session.agentCommandToolCallId).toBeNull() expect(commandEvents.at(-1)?.toolCallId).toBeUndefined() } finally { session.dispose() @@ -160,4 +164,28 @@ describe('TerminalSession command lifecycle', () => { else process.env.SHELL = originalShell } }) + + it('stops pressing a batch of keys once the call is stopped', async () => { + vi.useFakeTimers() + const session = TerminalSession.create({ + terminalId: 'terminal-keys', + cwd: '/tmp', + cols: 80, + rows: 24, + callbacks: { onData: () => {}, onState: () => {}, onCommand: () => {}, onExit: () => {} }, + }) + try { + const writesBefore = ptyStub.writes.length + const stop = new AbortController() + const pressing = session.pressKeys(['down', 'down', 'down', 'enter'], stop.signal) + expect(ptyStub.writes.length - writesBefore).toBe(1) + stop.abort() + await vi.runAllTimersAsync() + await pressing + + expect(ptyStub.writes.length - writesBefore).toBe(1) + } finally { + session.dispose() + } + }) }) diff --git a/apps/desktop/src/main/terminal/session.ts b/apps/desktop/src/main/terminal/session.ts index beb64a403a3..4dd33ad7f43 100644 --- a/apps/desktop/src/main/terminal/session.ts +++ b/apps/desktop/src/main/terminal/session.ts @@ -316,6 +316,12 @@ export class TerminalSession { private altScreen = false private foregroundCommand: string | null = null private foregroundToolCallId: string | null = null + /** + * The agent tool call whose command is still running, until that command really ends. Unlike + * {@link runningToolCallId} it survives an interactive command detaching from the tool call: + * the command is still the agent's, so Stop and sign-out can still end it. + */ + private agentToolCallId: string | null = null private pendingCommand: PendingCommand | null = null /** Command line reported by the shell but not yet bracketed by output-start. */ private announcedCommand: string | null = null @@ -434,6 +440,16 @@ export class TerminalSession { return this.foregroundCommand } + /** The agent tool call whose command holds the foreground, if one does. */ + get runningToolCallId(): string | null { + return this.foregroundToolCallId + } + + /** The agent tool call whose command is still running in this shell, if one is. */ + get agentCommandToolCallId(): string | null { + return this.agentToolCallId + } + /** * Tab-strip view of this terminal. The label prefers the running command, * which is what the user is actually waiting on, and falls back to the @@ -508,10 +524,11 @@ export class TerminalSession { * The pause also lets a menu redraw between presses, which is what makes a * batch land on the row a person pressing the same keys would reach. */ - async pressKeys(keys: TerminalControlKey[]): Promise { + async pressKeys(keys: TerminalControlKey[], signal?: AbortSignal): Promise { for (let index = 0; index < keys.length; index += 1) { - if (this.disposed) return + if (this.disposed || signal?.aborted) return if (index > 0) await this.settleBetweenKeystrokes() + if (signal?.aborted) return this.sendKey(keys[index]) } } @@ -522,11 +539,12 @@ export class TerminalSession { * gets a chance to redraw between them. See {@link toInputChunks} for why * sending it all at once leaves the text unsubmitted. */ - async type(text: string): Promise { + async type(text: string, signal?: AbortSignal): Promise { const chunks = toInputChunks(text) for (let index = 0; index < chunks.length; index += 1) { - if (this.disposed) return + if (this.disposed || signal?.aborted) return if (index > 0) await this.settleBetweenKeystrokes() + if (signal?.aborted) return this.write(chunks[index]) } } @@ -597,6 +615,7 @@ export class TerminalSession { } this.foregroundCommand = command this.foregroundToolCallId = toolCallId + this.agentToolCallId = toolCallId this.emitState() this.callbacks.onCommand({ terminalId: this.terminalId, phase: 'start', command, toolCallId }) @@ -976,6 +995,7 @@ export class TerminalSession { this.foregroundCommand = null this.foregroundToolCallId = null + this.agentToolCallId = null this.announcedCommand = null this.altScreen = false this.emitState() diff --git a/apps/desktop/src/main/terminal/tmux.test.ts b/apps/desktop/src/main/terminal/tmux.test.ts index a13a77632a4..e2921380dca 100644 --- a/apps/desktop/src/main/terminal/tmux.test.ts +++ b/apps/desktop/src/main/terminal/tmux.test.ts @@ -1,12 +1,16 @@ -import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { chmodSync, existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' +import { sleep } from '@sim/utils/helpers' import { afterEach, describe, expect, it } from 'vitest' import { awaitRun, isDescendantOf, parseFormatLines, pollRun, + runPaneState, + startRun, + stopRun, type TmuxRunHandle, } from '@/main/terminal/tmux' @@ -43,6 +47,8 @@ describe('run status files', () => { const handleIn = (dir: string): TmuxRunHandle => ({ window: '@1', + pane: '%1', + runId: 'run-1', outPath: join(dir, 'out'), statusPath: join(dir, 'status'), dispose: () => {}, @@ -84,3 +90,210 @@ describe('run status files', () => { expect(outcome).toEqual({ output: 'still going\n', exitCode: null, done: false }) }) }) + +/** + * A stand-in `tmux` binary on PATH that keeps its panes in a JSON file, so the run helpers are + * exercised through the real `spawn` path. It knows the handful of commands runs use; a test + * reshapes the server directly (a split, a closed window, a restart) by rewriting that file. + */ +interface FakeTmuxState { + nextWindow: number + nextPane: number + panes: Record }> + /** Every command that reached a pane: `send-keys %1 C-c`, `kill-pane %1`. */ + log: string[] + /** Commands the fake fails, with the error tmux would print. */ + fail?: Record +} + +const FAKE_TMUX = ` +const fs = require('node:fs') +const file = process.env.FAKE_TMUX_STATE +const state = JSON.parse(fs.readFileSync(file, 'utf8')) +const args = process.argv.slice(2) +const save = () => fs.writeFileSync(file, JSON.stringify(state)) +const target = () => args[args.indexOf('-t') + 1] +const fail = (message) => { process.stderr.write(message); process.exit(1) } +if (state.fail && state.fail[args[0]]) fail(state.fail[args[0]]) +switch (args[0]) { + case 'new-window': { + const window = '@' + state.nextWindow++ + const pane = '%' + state.nextPane++ + state.panes[pane] = { window, options: {} } + save() + // Runs the pane's command for real, the way tmux would, when a test asks for it. + if (process.env.FAKE_TMUX_EXEC) { + require('node:child_process') + .spawn('sh', ['-c', args[args.length - 1]], { detached: true, stdio: 'ignore' }) + .unref() + } + process.stdout.write(window + ' ' + pane + '\\n') + break + } + case 'set-option': { + const pane = state.panes[target()] + if (!pane) fail("can't find pane") + pane.options[args[args.length - 2]] = args[args.length - 1] + save() + break + } + case 'display-message': { + const pane = state.panes[target()] + if (!pane) fail("can't find pane") + const name = args[args.length - 1].slice(2, -1) + process.stdout.write((pane.options[name] ?? '') + '\\n') + break + } + case 'send-keys': + case 'kill-pane': { + if (!state.panes[target()]) fail("can't find pane") + state.log.push(args[0] + ' ' + target() + (args[0] === 'send-keys' ? ' ' + args[args.length - 1] : '')) + if (args[0] === 'kill-pane') delete state.panes[target()] + save() + break + } + default: + fail('unsupported: ' + args[0]) +} +` + +function fakeTmux(options: { exec?: boolean } = {}) { + const dir = mkdtempSync(join(tmpdir(), 'fake-tmux-')) + const stateFile = join(dir, 'state.json') + const binary = join(dir, 'tmux') + writeFileSync(binary, `#!${process.execPath}\n${FAKE_TMUX}`) + chmodSync(binary, 0o755) + const write = (state: FakeTmuxState) => writeFileSync(stateFile, JSON.stringify(state)) + write({ nextWindow: 0, nextPane: 0, panes: {}, log: [] }) + const read = (): FakeTmuxState => JSON.parse(readFileSync(stateFile, 'utf8')) + return { + dir, + env: { + ...process.env, + PATH: `${dir}:${process.env.PATH ?? ''}`, + FAKE_TMUX_STATE: stateFile, + ...(options.exec ? { FAKE_TMUX_EXEC: '1' } : {}), + }, + read, + write, + /** The user splits a run's window: a new pane of theirs beside the run's. */ + split(window: string): string { + const state = read() + const pane = `%${state.nextPane++}` + state.panes[pane] = { window, options: {} } + write(state) + return pane + }, + /** tmux restarts: every pane is gone and ids start over. */ + restart() { + write({ ...read(), nextWindow: 0, nextPane: 0, panes: {} }) + }, + } +} + +describe('stopping a tmux run touches only its own pane', () => { + const dirs: string[] = [] + + afterEach(() => { + for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true, force: true }) + }) + + async function started(tmux: ReturnType): Promise { + dirs.push(tmux.dir) + const handle = await startRun('agent', 'sleep 600', null, tmux.env) + if ('error' in handle) throw new Error(handle.error) + return handle + } + + it('interrupts and then closes only the run pane, leaving a pane the user split beside it', async () => { + const tmux = fakeTmux() + const run = await started(tmux) + const users = tmux.split(run.window) + + await stopRun(run, tmux.env, 0) + + expect(tmux.read().log).toEqual([`send-keys ${run.pane} C-c`, `kill-pane ${run.pane}`]) + expect(Object.keys(tmux.read().panes)).toEqual([users]) + }) + + it('sends nothing to a pane that reused the run pane id after tmux restarted', async () => { + const tmux = fakeTmux() + const run = await started(tmux) + tmux.restart() + // The user's own new window gets the ids the run's pane had. + const state = tmux.read() + state.panes[run.pane] = { window: run.window, options: {} } + tmux.write(state) + + expect(await runPaneState(run, tmux.env)).toBe('gone') + await stopRun(run, tmux.env, 0) + + expect(tmux.read().log).toEqual([]) + expect(Object.keys(tmux.read().panes)).toEqual([run.pane]) + }) + + it('treats a run pane the user closed as gone', async () => { + const tmux = fakeTmux() + const run = await started(tmux) + const state = tmux.read() + delete state.panes[run.pane] + tmux.write(state) + + expect(await runPaneState(run, tmux.env)).toBe('gone') + await stopRun(run, tmux.env, 0) + expect(tmux.read().log).toEqual([]) + }) + + it('never lets a run it could not tag start, and closes no pane by id to stop it', async () => { + const tmux = fakeTmux() + dirs.push(tmux.dir) + tmux.write({ ...tmux.read(), fail: { 'set-option': 'invalid option: @sim-run-id' } }) + + const result = await startRun('agent', 'sleep 600', null, tmux.env) + + expect(result).toMatchObject({ error: expect.stringContaining('was not run') }) + // No pane is closed by an id that a restarted server might have handed to the user. + expect(tmux.read().log).toEqual([]) + }) + + it('lets a tagged run start only once its pane is tagged', async () => { + const tmux = fakeTmux() + const run = await started(tmux) + + expect(tmux.read().panes[run.pane]?.options['@sim-run-id']).toBe(run.runId) + expect(existsSync(join(run.statusPath, '..', 'go'))).toBe(true) + }) + + it('neither stops nor gives up on a run while tmux cannot be asked', async () => { + const tmux = fakeTmux() + const run = await started(tmux) + tmux.write({ ...tmux.read(), fail: { 'display-message': 'server exited unexpectedly' } }) + + expect(await runPaneState(run, tmux.env)).toBe('unknown') + await stopRun(run, tmux.env, 0) + + expect(tmux.read().log).toEqual([]) + }) + + it('runs a tagged command for real, and never runs one it could not tag', async () => { + const tagged = fakeTmux({ exec: true }) + dirs.push(tagged.dir) + const run = await startRun('agent', 'echo ran', null, tagged.env) + if ('error' in run) throw new Error(run.error) + await expect + .poll(() => pollRun(run), { timeout: 10_000 }) + .toMatchObject({ + done: true, + exitCode: 0, + }) + + const untagged = fakeTmux({ exec: true }) + dirs.push(untagged.dir) + untagged.write({ ...untagged.read(), fail: { 'set-option': 'invalid option' } }) + const marker = join(untagged.dir, 'ran') + await startRun('agent', `touch ${JSON.stringify(marker)}`, null, untagged.env) + await sleep(6_000) + + expect(existsSync(marker)).toBe(false) + }, 20_000) +}) diff --git a/apps/desktop/src/main/terminal/tmux.ts b/apps/desktop/src/main/terminal/tmux.ts index e97f31df116..109f82a2eee 100644 --- a/apps/desktop/src/main/terminal/tmux.ts +++ b/apps/desktop/src/main/terminal/tmux.ts @@ -17,12 +17,14 @@ * of the shell that launched it. */ import { spawn } from 'node:child_process' -import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { createLogger } from '@sim/logger' import type { TerminalPaneState } from '@sim/terminal-protocol' +import { getErrorMessage } from '@sim/utils/errors' import { sleep } from '@sim/utils/helpers' +import { generateId } from '@sim/utils/id' const logger = createLogger('DesktopTmux') @@ -300,11 +302,27 @@ export const TMUX_KEY_NAMES: Record = { */ export interface TmuxRunHandle { window: string + /** The run's own pane: input and stops go here, never to whatever pane is active. */ + pane: string + /** + * Tagged on the pane as the `@sim-run-id` user option. Window and pane ids restart from zero + * with the tmux server, so only the tag proves a pane is still this run's. + */ + runId: string outPath: string statusPath: string dispose(): void } +/** + * How many 50 ms polls a run's command waits for its go file: longer than the tagging call can + * take before it times out, so a tag that succeeds always lands inside the window. + */ +const RUN_GATE_POLLS = Math.ceil((2 * TMUX_TIMEOUT_MS + 5_000) / 50) + +/** The tmux user option that marks a pane as one run's own. */ +const RUN_ID_OPTION = '@sim-run-id' + /** * Starts a command in a dedicated tmux window. * @@ -326,6 +344,7 @@ export async function startRun( const dir = mkdtempSync(join(tmpdir(), 'sim-tmux-run-')) const outPath = join(dir, 'out') const statusPath = join(dir, 'status') + const goPath = join(dir, 'go') const dispose = () => { try { rmSync(dir, { recursive: true, force: true }) @@ -343,10 +362,40 @@ export async function startRun( // writing to the unlinked inode — but an unredirected `printf` would fail // into the pipeline and print `No such file or directory` into the user's own // tmux window, minutes after they closed the tab. - const script = `${command}\nprintf %s "\${PIPESTATUS[0]}" > ${JSON.stringify(statusPath)} 2>/dev/null` - const wrapper = `bash -lc ${JSON.stringify(`{ ${script}; } 2>&1 | tee ${JSON.stringify(outPath)}`)}` - - const args = ['new-window', '-d', '-P', '-F', '#{window_id}', '-t', session, '-n', 'sim-run'] + // + // The command waits for its pane to be tagged as this run's (the go file), so nothing runs that + // a later stop could not recognize. Untagged, the script gives up once the tagging call has + // surely failed, and its pane closes on its own; no one has to close a pane whose id might no + // longer be its own. + // + // The script is a file rather than a `bash -c` string: tmux hands its command to `sh -c`, which + // would expand `$` references meant for bash (the gate's counter, PIPESTATUS) before bash ran. + const scriptPath = join(dir, 'run.sh') + writeFileSync( + scriptPath, + [ + 'i=0', + `while [ ! -e ${JSON.stringify(goPath)} ] && [ "$i" -lt ${RUN_GATE_POLLS} ]; do sleep 0.05; i=$((i + 1)); done`, + `[ -e ${JSON.stringify(goPath)} ] || exit 0`, + `{ ${command}`, + `printf %s "\${PIPESTATUS[0]}" > ${JSON.stringify(statusPath)} 2>/dev/null; } 2>&1 | tee ${JSON.stringify(outPath)}`, + '', + ].join('\n'), + { mode: 0o600 } + ) + const wrapper = `bash -l ${JSON.stringify(scriptPath)}` + + const args = [ + 'new-window', + '-d', + '-P', + '-F', + '#{window_id} #{pane_id}', + '-t', + session, + '-n', + 'sim-run', + ] if (cwd) args.push('-c', cwd) args.push(wrapper) @@ -355,8 +404,63 @@ export async function startRun( dispose() return { error: created.stderr.trim() || 'tmux could not open a window for the command.' } } + const [window = '', pane = ''] = created.stdout.trim().split(' ') + const runId = generateId() + // An untagged pane is never treated as the run's: without the tag a stop could not tell it from + // a pane the user opened later under the same id, so it sends nothing at all. + const tagged = await runTmux(['set-option', '-p', '-t', pane, RUN_ID_OPTION, runId], env) + if (!tagged.ok) { + // Untagged, nothing could stop it safely later, so it never starts: without the go file the + // wrapper exits by itself. + dispose() + return { + error: `tmux could not mark the command's pane (${tagged.stderr.trim() || 'no detail'}), so the command was not run.`, + } + } + try { + writeFileSync(goPath, '') + } catch (error) { + dispose() + return { error: `The command could not be started: ${getErrorMessage(error)}` } + } - return { window: created.stdout.trim(), outPath, statusPath, dispose } + return { window, pane, runId, outPath, statusPath, dispose } +} + +/** + * Whether the run's pane is still the run's: `ours`, or `gone` when tmux has no such pane or the + * pane under that id is not tagged as this run's (the user closed it, or a restarted tmux server + * handed the id to one of the user's own panes). `unknown` when tmux could not be asked: such a + * pane is neither touched nor given up on. + */ +export async function runPaneState( + handle: TmuxRunHandle, + env: NodeJS.ProcessEnv +): Promise<'ours' | 'gone' | 'unknown'> { + if (!handle.pane) return 'gone' + const shown = await runTmux( + ['display-message', '-p', '-t', handle.pane, `#{${RUN_ID_OPTION}}`], + env + ) + if (shown.ok) return shown.stdout.trim() === handle.runId ? 'ours' : 'gone' + return /can't find|no server running/i.test(shown.stderr) ? 'gone' : 'unknown' +} + +/** + * Stops a run: Ctrl-C in its own pane, then closing that pane if the command ignored it. Every + * step first checks the pane is still the run's, and only that pane is ever closed, so a pane the + * user split off beside it, or a window that reused its ids, is never touched. + */ +export async function stopRun( + handle: TmuxRunHandle, + env: NodeJS.ProcessEnv, + graceMs: number +): Promise { + if ((await runPaneState(handle, env)) !== 'ours') return + await sendKey(handle.pane, 'C-c', env) + const deadline = Date.now() + graceMs + while (!isRunComplete(handle) && Date.now() < deadline) await sleep(100) + if (!isRunComplete(handle)) await closeRunPane(handle, env) } function readIfPresent(path: string): string | null { @@ -419,11 +523,14 @@ export async function killPane(target: string, env: NodeJS.ProcessEnv): Promise< return runTmux(['kill-pane', '-t', target], env) } -/** Closes a window opened by {@link startRun}. */ -export async function closeRunWindow(handle: TmuxRunHandle, env: NodeJS.ProcessEnv): Promise { - if (!handle.window) return - const killed = await runTmux(['kill-window', '-t', handle.window], env) +/** + * Closes the pane opened by {@link startRun}, and with it the window once that pane is the last + * one in it. Only the run's own pane, and only while it is still the run's. + */ +export async function closeRunPane(handle: TmuxRunHandle, env: NodeJS.ProcessEnv): Promise { + if ((await runPaneState(handle, env)) !== 'ours') return + const killed = await runTmux(['kill-pane', '-t', handle.pane], env) if (!killed.ok) { - logger.warn('Could not close the tmux run window', { error: killed.stderr.trim() }) + logger.warn('Could not close the tmux run pane', { error: killed.stderr.trim() }) } } diff --git a/apps/desktop/src/preload/index.ts b/apps/desktop/src/preload/index.ts index 4122466b9eb..6518ff00592 100644 --- a/apps/desktop/src/preload/index.ts +++ b/apps/desktop/src/preload/index.ts @@ -28,6 +28,7 @@ import type { BrowserToolbarCommand, DesktopAppearanceTheme, DesktopCommand, + DesktopExecutorDevice, DesktopLocalFileRequest, DesktopLocalFileResponse, DesktopNotificationPayload, @@ -193,6 +194,10 @@ const api: SimDesktopApi = { setTerminalDefaultZoom: (zoom: DesktopZoomPercent): Promise => ipcRenderer.invoke('desktop:settings:set-terminal-default-zoom', zoom), }, + desktopExecutor: { + getDevice: (): Promise => + ipcRenderer.invoke('desktop-executor:get-device'), + }, updates: { getState: (): Promise => ipcRenderer.invoke('desktop:updates:get-state'), check: (): void => { diff --git a/apps/desktop/src/test/electron-mock.ts b/apps/desktop/src/test/electron-mock.ts index 84155b510d7..c0f180897d0 100644 --- a/apps/desktop/src/test/electron-mock.ts +++ b/apps/desktop/src/test/electron-mock.ts @@ -99,6 +99,10 @@ export const net = { fetch: vi.fn(), } +export const powerMonitor = { + on: vi.fn(), +} + export const session = { fromPartition: vi.fn(), } @@ -393,6 +397,7 @@ export const electronMock = { screen, Menu, net, + powerMonitor, session, protocol, ipcMain, diff --git a/apps/sim/lib/browser-agent/transport.ts b/apps/sim/lib/browser-agent/transport.ts index bf368ce5942..14f11c677e5 100644 --- a/apps/sim/lib/browser-agent/transport.ts +++ b/apps/sim/lib/browser-agent/transport.ts @@ -36,6 +36,7 @@ import type { SimDesktopBrowserAgentApi, } from '@sim/desktop-bridge' import { isPendingDesktopScopeId } from '@sim/desktop-bridge' +import { browserToolTimeoutMessage } from '@sim/desktop-bridge/tool-results' import { createLogger } from '@sim/logger' import { toError } from '@sim/utils/errors' import { getDesktopBridge, isBrowserAgentEnabled } from '@/lib/desktop' @@ -243,11 +244,7 @@ export async function executeBrowserTool( error: toError(error).message, }) } - reject( - new BrowserOutcomeUnknownError( - `The browser did not respond within ${timeoutMs}ms. Its outcome is unknown and the action may already have taken effect. Do not retry it automatically; take a fresh browser snapshot before deciding what to do.` - ) - ) + reject(new BrowserOutcomeUnknownError(browserToolTimeoutMessage(timeoutMs))) }, timeoutMs) }), ]) diff --git a/apps/sim/lib/mothership/tools/client/browser-tool-execution.test.ts b/apps/sim/lib/mothership/tools/client/browser-tool-execution.test.ts index 7c45358c7c8..70b9898d0d7 100644 --- a/apps/sim/lib/mothership/tools/client/browser-tool-execution.test.ts +++ b/apps/sim/lib/mothership/tools/client/browser-tool-execution.test.ts @@ -332,6 +332,35 @@ describe('executeBrowserToolOnClient', () => { }) }) + it('resolves only once the browser action has settled and its result is handed to delivery', async () => { + let finishAction: (result: unknown) => void = () => {} + mockExecuteBrowserTool.mockReturnValue( + new Promise((resolve) => { + finishAction = resolve + }) + ) + const toolCallId = nextToolCallId() + let resolved = false + const running = executeBrowserToolOnClient(toolCallId, 'browser_snapshot', {}, CHAT_SCOPE).then( + () => { + resolved = true + } + ) + + await vi.waitFor(() => expect(mockExecuteBrowserTool).toHaveBeenCalled()) + await sleep(10) + expect(resolved).toBe(false) + expect(mockReportCompletion).not.toHaveBeenCalled() + + finishAction({ text: 'page content' }) + await running + // Delivery itself is owned by the retained-completion scheduler, which outlives the turn's + // Stop and the page; what the caller waits for is the action, with its report under way. + expect(mockReportCompletion).toHaveBeenCalledWith(toolCallId, 'success', expect.any(String), { + text: 'page content', + }) + }) + it('lets a running invocation own the genuine result when the same call is re-delivered', async () => { let finishExecution: (result: { text: string }) => void = () => {} mockExecuteBrowserTool.mockImplementation( diff --git a/apps/sim/lib/mothership/tools/client/browser-tool-execution.ts b/apps/sim/lib/mothership/tools/client/browser-tool-execution.ts index 95c908ec0af..df1250d5c81 100644 --- a/apps/sim/lib/mothership/tools/client/browser-tool-execution.ts +++ b/apps/sim/lib/mothership/tools/client/browser-tool-execution.ts @@ -8,9 +8,15 @@ * server-side waiter. */ import { type BrowserToolName, browserToolRendererTimeoutMs } from '@sim/browser-protocol' +import { + browserSessionClosedCompletion, + browserToolCompletion, + browserToolFailure, + browserToolNeedsLivePage, +} from '@sim/desktop-bridge/tool-results' import { createLogger } from '@sim/logger' import { toError } from '@sim/utils/errors' -import { isRecordLike, toRecord } from '@sim/utils/object' +import { toRecord } from '@sim/utils/object' import { truncate } from '@sim/utils/string' import { cancelBrowserTool, @@ -24,7 +30,6 @@ import { } from '@/lib/mothership/async-runs/lifecycle' import { COPILOT_CONFIRM_API_PATH } from '@/lib/mothership/constants' import { BrowserToolReplayLedger } from '@/lib/mothership/tools/client/browser-tool-replay-ledger' -import { sanitizeBrowserToolResultForModel } from '@/lib/mothership/tools/client/browser-tool-result' import { reportClientToolCompletion, reportClientToolCompletionOnPageExit, @@ -33,22 +38,6 @@ import { getBrowserSession, useBrowserSessionStore } from '@/stores/browser-sess const logger = createLogger('CopilotBrowserToolExecution') -/** - * Tools that do not require an existing live page. Most create a new page; - * `browser_list_sessions` reads the desktop's profile-level session registry. - * Everything else is rejected up front when a closed scope cannot be restored, - * instead of burning the full IPC timeout per call. - */ -const LIVE_PAGE_OPTIONAL_TOOLS: ReadonlySet = new Set([ - 'browser_navigate', - 'browser_open_url', - 'browser_open_tab', - 'browser_list_tabs', - 'browser_list_sessions', - 'browser_list_downloads', - 'browser_save_download', -]) - /** * Exhaustive replay policy for browser tools. Observation-only calls may run * when durable replay storage is unavailable because repeating them after a @@ -91,10 +80,6 @@ const OBSERVATION_ONLY_BROWSER_TOOLS = { browser_request_takeover: false, } as const satisfies Readonly> -const SESSION_CLOSED_MESSAGE = - 'The agent browser session is closed, so this browser tool cannot run. ' + - 'Call browser_open_url, browser_navigate, or browser_open_tab to start a new session, or report the situation to the user. ' + - 'Do not retry other browser tools until a new session is open.' /** Tool events older than this are replays, not live instructions — never act on them. */ const MAX_EVENT_AGE_MS = 120_000 const EXECUTED_STORAGE_PREFIX = 'sim:copilot:browser-tool-executed:' @@ -866,7 +851,7 @@ async function doExecuteBrowserTool( } try { - const needsLivePage = !LIVE_PAGE_OPTIONAL_TOOLS.has(toolName) + const needsLivePage = browserToolNeedsLivePage(toolName) if (needsLivePage && isSessionClosed(scopeId)) { try { await restoreBrowserScope(scopeId) @@ -886,11 +871,7 @@ async function doExecuteBrowserTool( }) if (cancelled) return reportTerminalCompletion( - { - status: ASYNC_TOOL_CONFIRMATION_STATUS.error, - message: SESSION_CLOSED_MESSAGE, - data: { error: SESSION_CLOSED_MESSAGE, sessionClosed: true }, - }, + browserSessionClosedCompletion(), 'Failed to report browser session-closed error', 'guard' ) @@ -918,50 +899,23 @@ async function doExecuteBrowserTool( nativeActionPending = false if (cancelled) return const sessionClosed = isSessionClosed(scopeId) - const outcomeUnknown = isOutcomeUnknownError(err) - const message = sessionClosed - ? `${toError(err).message} ${SESSION_CLOSED_MESSAGE}` - : toError(err).message - logger.warn('Browser tool failed', { toolCallId, toolName, error: message, sessionClosed }) - reportTerminalCompletion( - { - status: ASYNC_TOOL_CONFIRMATION_STATUS.error, - message, - data: { - error: message, - ...(outcomeUnknown ? { outcomeUnknown: true, doNotRetry: true } : {}), - ...(sessionClosed ? { sessionClosed: true } : {}), - }, - }, - 'Failed to report browser tool error' - ) + const failure = browserToolFailure(toError(err).message, { + outcomeUnknown: isOutcomeUnknownError(err), + sessionClosed, + }) + logger.warn('Browser tool failed', { + toolCallId, + toolName, + error: failure.message, + sessionClosed, + }) + reportTerminalCompletion(failure, 'Failed to report browser tool error') return } nativeActionPending = false if (cancelled) return - const outcomeUnknown = isRecordLike(result) && result.outcomeUnknown === true - const effectUnconfirmed = isRecordLike(result) && result.effectObserved === false - const stoppedMessage = - toolName === 'browser_fill_form' && isRecordLike(result) && result.completed === false - ? 'Form filling stopped; inspect the partial result' - : toolName === 'browser_batch' && isRecordLike(result) && result.stoppedBy === 'failure' - ? 'A batched browser action failed; inspect the partial result' - : undefined reportTerminalCompletion( - { - status: - stoppedMessage || outcomeUnknown - ? ASYNC_TOOL_CONFIRMATION_STATUS.error - : ASYNC_TOOL_CONFIRMATION_STATUS.success, - message: - stoppedMessage ?? - (outcomeUnknown - ? 'Browser action outcome is unconfirmed; inspect the page before repeating it.' - : effectUnconfirmed - ? 'Browser input completed; its effect is unconfirmed. Inspect the current state before retrying.' - : 'Browser action completed'), - data: sanitizeBrowserToolResultForModel(toolName, result), - }, + browserToolCompletion(toolName, result), 'Failed to report browser tool completion' ) } finally { diff --git a/apps/sim/lib/mothership/tools/client/browser-tool-result.ts b/apps/sim/lib/mothership/tools/client/browser-tool-result.ts deleted file mode 100644 index 2a63053313a..00000000000 --- a/apps/sim/lib/mothership/tools/client/browser-tool-result.ts +++ /dev/null @@ -1,81 +0,0 @@ -import type { BrowserToolName } from '@sim/browser-protocol' -import { isRecordLike, toRecordOrNull } from '@sim/utils/object' - -function finiteNumber(value: unknown): value is number { - return typeof value === 'number' && Number.isFinite(value) -} - -function imageDimensions(value: unknown): { width: number; height: number } | null { - if ( - !isRecordLike(value) || - !finiteNumber(value.width) || - !finiteNumber(value.height) || - value.width <= 0 || - value.height <= 0 - ) { - return null - } - return { width: value.width, height: value.height } -} - -/** Projects image bytes and coordinate metadata into the model's image-content contract. */ -export function sanitizeBrowserToolResultForModel( - toolName: BrowserToolName, - result: unknown -): Record | undefined { - if (!isRecordLike(result)) { - return result === undefined ? undefined : { value: result } - } - if (toolName !== 'browser_screenshot' || typeof result.dataUrl !== 'string') return result - - const { dataUrl, ...rest } = result - const image = /^data:([^;,]+);base64,(.+)$/s.exec(dataUrl) - if (!image) { - return { - ...rest, - note: 'The screenshot could not be encoded. Use browser_snapshot or browser_read_text instead.', - } - } - const viewport = toRecordOrNull(rest.viewport) - const screenshotUrl = - typeof rest.url === 'string' && rest.url - ? rest.url - : viewport && typeof viewport.url === 'string' - ? viewport.url - : '' - const location = screenshotUrl ? ` of ${screenshotUrl}` : '' - const clip = toRecordOrNull(rest.clip) - const cropSize = imageDimensions(clip) - const viewportSize = imageDimensions(viewport) - const imageSize = imageDimensions(rest.imageSize) - const capturedSize = clip ? cropSize : viewportSize - const scaleX = imageSize && capturedSize ? imageSize.width / capturedSize.width : null - const scaleY = imageSize && capturedSize ? imageSize.height / capturedSize.height : null - const hasScale = finiteNumber(scaleX) && scaleX > 0 && finiteNumber(scaleY) && scaleY > 0 - const origin = clip - ? finiteNumber(clip.x) && finiteNumber(clip.y) - ? { x: clip.x, y: clip.y } - : null - : { x: 0, y: 0 } - const content = [ - `Screenshot${location}. This is the rendered ${clip ? 'element' : 'viewport'} only — it carries no element ids. Use browser_snapshot for element-ref actions and the mapping below for coordinate actions.`, - viewportSize && `Viewport: ${viewportSize.width} × ${viewportSize.height} CSS pixels.`, - imageSize && `Encoded image: ${imageSize.width} × ${imageSize.height} pixels.`, - hasScale && `Image scale: X=${scaleX}, Y=${scaleY} encoded image pixels per CSS pixel.`, - cropSize && `Crop size: ${cropSize.width} × ${cropSize.height} CSS pixels.`, - clip && origin && `Crop origin: (${origin.x}, ${origin.y}) in viewport CSS pixels.`, - hasScale && origin - ? `Coordinate actions use viewport CSS pixels: cssX = ${origin.x} + imageX / ${scaleX}; cssY = ${origin.y} + imageY / ${scaleY}. imageX/imageY refer to the encoded image before any display resizing.` - : 'Screenshot coordinate mapping is unavailable; use browser_snapshot element references or take a new viewport screenshot before coordinate actions.', - ] - .filter(Boolean) - .join(' ') - return { - ...rest, - content, - attachment: { - type: 'image', - source: { type: 'base64', media_type: image[1], data: image[2] }, - }, - } -} diff --git a/apps/sim/lib/mothership/tools/client/local-filesystem.test.ts b/apps/sim/lib/mothership/tools/client/local-filesystem.test.ts index 7d4940498e8..a1ebc1f210f 100644 --- a/apps/sim/lib/mothership/tools/client/local-filesystem.test.ts +++ b/apps/sim/lib/mothership/tools/client/local-filesystem.test.ts @@ -1,6 +1,7 @@ /** * @vitest-environment jsdom */ +import { sleep } from '@sim/utils/helpers' import { beforeEach, describe, expect, it, vi } from 'vitest' const { mockReportCompletion } = vi.hoisted(() => ({ @@ -231,4 +232,45 @@ describe('executeLocalFilesystemTool', () => { }) expect(mockReportCompletion).not.toHaveBeenCalled() }) + + it('resolves only once the tool has settled and its result is reported', async () => { + let finishRead: (response: unknown) => void = () => {} + let finishReport: () => void = () => {} + localFilesystem.mockImplementation(async (request: { operation: string }) => { + if (request.operation === 'list_mounts') return { ok: true, data: { mounts: [mount] } } + return new Promise((resolve) => { + finishRead = resolve + }) + }) + mockReportCompletion.mockImplementation( + () => + new Promise((resolve) => { + finishReport = resolve + }) + ) + let resolved = false + const running = executeLocalFilesystemTool( + 'tool-held', + 'read', + { path: `${vfsRoot}/README.md` }, + { workspaceId: 'ws-1' } + ).then(() => { + resolved = true + }) + + await vi.waitFor(() => + expect(localFilesystem).toHaveBeenCalledWith(expect.objectContaining({ operation: 'read' })) + ) + await sleep(10) + expect(resolved).toBe(false) + + finishRead({ ok: true, data: { content: 'hello', totalLines: 1 } }) + await vi.waitFor(() => expect(mockReportCompletion).toHaveBeenCalled()) + await sleep(10) + expect(resolved).toBe(false) + + finishReport() + await running + expect(resolved).toBe(true) + }) }) diff --git a/apps/sim/lib/mothership/tools/client/local-filesystem.ts b/apps/sim/lib/mothership/tools/client/local-filesystem.ts index a78eb85d19e..a8a7dc6f208 100644 --- a/apps/sim/lib/mothership/tools/client/local-filesystem.ts +++ b/apps/sim/lib/mothership/tools/client/local-filesystem.ts @@ -1,42 +1,15 @@ -import type { - LocalFilesystemData, - LocalFilesystemMount, - LocalFilesystemRequest, - LocalFilesystemResponse, -} from '@sim/desktop-bridge' -import { - DEFAULT_GREP_CONTEXT, - DEFAULT_GREP_RESULTS, - DEFAULT_READ_LINES, - MAX_GREP_CONTEXT, - MAX_GREP_RESULTS, - MAX_READ_LINES, -} from '@sim/desktop-bridge/local-filesystem-limits' +import type { LocalFilesystemMount } from '@sim/desktop-bridge' +import { runUserLocalFilesystemTool } from '@sim/desktop-bridge/local-filesystem-tools' +import { localFilesystemToolCompletion } from '@sim/desktop-bridge/tool-results' import { createLogger } from '@sim/logger' import { toError } from '@sim/utils/errors' -import micromatch from 'micromatch' import { getDesktopBridge } from '@/lib/desktop' -import { ASYNC_TOOL_CONFIRMATION_STATUS } from '@/lib/mothership/async-runs/lifecycle' import { reportClientToolCompletion } from '@/lib/mothership/tools/client/completion' import { executeNativeFileTool } from '@/lib/mothership/tools/client/native-files' import { isNativeFileTool, USER_LOCAL_VFS_ROOT } from '@/lib/mothership/tools/local-filesystem' import { encodeVfsSegment } from '@/lib/mothership/vfs/path-utils' const logger = createLogger('CopilotLocalFilesystemTool') -/** - * This glob runs entirely in the renderer against already-listed mounts, so - * unlike the grep and read limits it is nobody else's business — the shell - * never authorizes against it. Deliberately local. - */ -const MAX_USER_LOCAL_GLOB_RESULTS = 500 - -const VFS_GLOB_OPTIONS: micromatch.Options = { - bash: false, - dot: false, - windows: false, - nobrace: true, - noext: true, -} interface LocalFilesystemExecutionContext { workspaceId?: string @@ -44,18 +17,6 @@ interface LocalFilesystemExecutionContext { signal?: AbortSignal } -function requiredString(args: Record, name: string): string { - const value = args[name] - if (typeof value !== 'string' || value.length === 0) { - throw new Error(`${name} is required`) - } - return value -} - -function optionalNumber(value: unknown): number | undefined { - return typeof value === 'number' && Number.isFinite(value) ? value : undefined -} - function bridge(): NonNullable { const desktop = getDesktopBridge() if (!desktop?.localFilesystem) { @@ -64,44 +25,6 @@ function bridge(): NonNullable { return desktop } -function successfulData(response: LocalFilesystemResponse): LocalFilesystemData { - if (!response.ok) { - throw new Error(response.error) - } - return response.data -} - -function abortError(signal: AbortSignal): Error { - const error = new Error(signal.reason ? String(signal.reason) : 'Operation aborted') - error.name = 'AbortError' - return error -} - -function requestIdForToolCall(toolCallId: string): string { - return toolCallId -} - -async function invokeBridge( - request: LocalFilesystemRequest, - signal?: AbortSignal -): Promise { - if (signal?.aborted) throw abortError(signal) - const requestId = 'requestId' in request ? request.requestId : undefined - const onAbort = () => { - if (requestId) { - void bridge().localFilesystem({ operation: 'cancel', requestId }) - } - } - signal?.addEventListener('abort', onAbort, { once: true }) - try { - const data = successfulData(await bridge().localFilesystem(request)) - if (signal?.aborted) throw abortError(signal) - return data - } finally { - signal?.removeEventListener('abort', onAbort) - } -} - function encodeMountName(name: string): string { try { return encodeVfsSegment(name) @@ -114,227 +37,26 @@ function mountVfsRoot(mount: LocalFilesystemMount): string { return `${USER_LOCAL_VFS_ROOT}/${encodeMountName(mount.name)}--${mount.id}` } -function vfsPathForUri(mount: LocalFilesystemMount, uri: string): string { - const parsed = new URL(uri) - if (parsed.protocol !== 'localfs:' || parsed.hostname !== mount.id) { - throw new Error('The desktop app returned a local path outside the selected folder.') - } - const relativePath = parsed.pathname.replace(/^\/+/, '') - return relativePath ? `${mountVfsRoot(mount)}/${relativePath}` : mountVfsRoot(mount) -} - -function localUriForVfsPath(mount: LocalFilesystemMount, path: string): string { - const root = mountVfsRoot(mount) - if (path === root) return mount.uri - if (!path.startsWith(`${root}/`)) { - throw new Error(`Path is not inside a granted user-local folder: ${path}`) - } - return `${mount.uri}${path.slice(root.length + 1)}` -} - -async function listMounts(): Promise { - const data = await invokeBridge({ operation: 'list_mounts' }) - if (!('mounts' in data)) { - throw new Error('The desktop app returned an invalid mount list.') - } - return data.mounts -} - -function mountForPath(mounts: LocalFilesystemMount[], path: string): LocalFilesystemMount { - const match = mounts.find((mount) => { - const root = mountVfsRoot(mount) - return path === root || path.startsWith(`${root}/`) - }) - if (!match) { - throw new Error( - `No granted user-local folder contains "${path}". Use glob({pattern:"user-local/**"}) to discover canonical paths.` - ) - } - return match -} - -async function executeUserLocalGlob( - toolCallId: string, - args: Record, - signal?: AbortSignal -): Promise<{ files: string[] }> { - const pattern = requiredString(args, 'pattern') - const mounts = await listMounts() - const files = new Set() - const requestId = requestIdForToolCall(toolCallId) - - for (const mount of mounts) { - if (signal?.aborted) throw abortError(signal) - const root = mountVfsRoot(mount) - if (micromatch.isMatch(root, pattern, VFS_GLOB_OPTIONS)) { - files.add(root) - } - - const data = await invokeBridge( - { - operation: 'glob', - uri: mount.uri, - pattern, - pathPrefix: root, - requestId, - }, - signal - ) - if (!('entries' in data)) { - throw new Error('The desktop app returned an invalid glob result.') - } - for (const entry of data.entries) { - const path = vfsPathForUri(mount, entry.uri) - if (micromatch.isMatch(path, pattern, VFS_GLOB_OPTIONS)) { - files.add(path) - if (files.size >= MAX_USER_LOCAL_GLOB_RESULTS) break - } - } - if (files.size >= MAX_USER_LOCAL_GLOB_RESULTS) break - } - - return { files: [...files].sort() } -} - -async function executeUserLocalRead( - toolCallId: string, - args: Record, - signal?: AbortSignal -): Promise<{ content: string; totalLines: number }> { - const path = requiredString(args, 'path') - const mounts = await listMounts() - const mount = mountForPath(mounts, path) - const offset = Math.max(0, Math.trunc(optionalNumber(args.offset) ?? 0)) - const requestedLimit = optionalNumber(args.limit) - const lineCount = Math.min( - MAX_READ_LINES, - Math.max(1, Math.trunc(requestedLimit ?? DEFAULT_READ_LINES)) - ) - const data = await invokeBridge( - { - operation: 'read', - uri: localUriForVfsPath(mount, path), - startLine: offset + 1, - lineCount, - requestId: requestIdForToolCall(toolCallId), - }, - signal - ) - if (!('content' in data) || !('totalLines' in data)) { - throw new Error('The desktop app returned an invalid read result.') - } - return { content: data.content, totalLines: data.totalLines } -} - -async function executeUserLocalGrep( +async function reportCompletion( toolCallId: string, - args: Record, - signal?: AbortSignal -): Promise> { - const pattern = requiredString(args, 'pattern') - const path = requiredString(args, 'path').replace(/\/+$/, '') - const outputMode = - args.output_mode === 'files_with_matches' || args.output_mode === 'count' - ? args.output_mode - : 'content' - const maxResults = Math.min( - MAX_GREP_RESULTS, - Math.max(1, Math.trunc(optionalNumber(args.maxResults) ?? DEFAULT_GREP_RESULTS)) - ) - const mounts = await listMounts() - const targets = - path === USER_LOCAL_VFS_ROOT - ? mounts.map((mount) => ({ mount, uri: mount.uri })) - : [ - { - mount: mountForPath(mounts, path), - uri: '', - }, - ] - if (targets.length === 1 && !targets[0].uri) { - targets[0].uri = localUriForVfsPath(targets[0].mount, path) - } - - const contentMatches: Array<{ path: string; line: number; content: string }> = [] - const matchingFiles = new Set() - const counts = new Map() - const requestId = requestIdForToolCall(toolCallId) - - for (const target of targets) { - const data = await invokeBridge( - { - operation: 'grep', - uri: target.uri, - pattern, - caseSensitive: args.ignoreCase !== true, - maxResults, - outputMode, - lineNumbers: args.lineNumbers !== false, - context: Math.min( - MAX_GREP_CONTEXT, - Math.max(0, Math.trunc(optionalNumber(args.context) ?? DEFAULT_GREP_CONTEXT)) - ), - requestId, - }, - signal + toolName: string, + outcome: Parameters[0] +): Promise { + const completion = localFilesystemToolCompletion(outcome) + try { + await reportClientToolCompletion( + toolCallId, + completion.status, + completion.message, + completion.data ) - - if ('matches' in data) { - for (const match of data.matches) { - contentMatches.push({ - path: vfsPathForUri(target.mount, match.uri), - line: match.line, - content: match.text, - }) - if (contentMatches.length >= maxResults) break - } - } else if ('files' in data) { - for (const uri of data.files) { - matchingFiles.add(vfsPathForUri(target.mount, uri)) - if (matchingFiles.size >= maxResults) break - } - } else if ('counts' in data) { - for (const count of data.counts) { - counts.set(vfsPathForUri(target.mount, count.uri), count.count) - if (counts.size >= maxResults) break - } - } else { - throw new Error('The desktop app returned an invalid grep result.') - } - - const currentCount = - outputMode === 'files_with_matches' - ? matchingFiles.size - : outputMode === 'count' - ? counts.size - : contentMatches.length - if (currentCount >= maxResults) break + } catch (reportError) { + logger.error('Failed to report local filesystem tool result', { + toolCallId, + toolName, + error: toError(reportError).message, + }) } - - if (outputMode === 'files_with_matches') { - return { files: [...matchingFiles].sort() } - } - if (outputMode === 'count') { - return { - counts: [...counts.entries()] - .map(([countPath, count]) => ({ path: countPath, count })) - .sort((a, b) => a.path.localeCompare(b.path)), - } - } - return { - matches: contentMatches.sort((a, b) => a.path.localeCompare(b.path) || a.line - b.line), - } -} - -async function execute( - toolCallId: string, - toolName: string, - args: Record, - context: LocalFilesystemExecutionContext -): Promise { - if (toolName === 'glob') return executeUserLocalGlob(toolCallId, args, context.signal) - if (toolName === 'grep') return executeUserLocalGrep(toolCallId, args, context.signal) - return executeUserLocalRead(toolCallId, args, context.signal) } export async function executeLocalFilesystemTool( @@ -347,44 +69,27 @@ export async function executeLocalFilesystemTool( await executeNativeFileTool(toolCallId, toolName, context.signal) return } - await execute(toolCallId, toolName, args, context).then( - async (data) => { - if (context.signal?.aborted) return - try { - await reportClientToolCompletion( - toolCallId, - ASYNC_TOOL_CONFIRMATION_STATUS.success, - 'Local filesystem tool completed.', - data - ) - } catch (reportError) { - logger.error('Failed to report local filesystem tool completion', { - toolCallId, - toolName, - error: toError(reportError).message, - }) - } - }, - async (error) => { - if (context.signal?.aborted || (error instanceof Error && error.name === 'AbortError')) { - return - } - const message = toError(error).message - logger.warn('Local filesystem tool failed', { toolCallId, toolName, error: message }) - try { - await reportClientToolCompletion( - toolCallId, - ASYNC_TOOL_CONFIRMATION_STATUS.error, - message, - { error: message } - ) - } catch (reportError) { - logger.error('Failed to report local filesystem tool error', { - toolCallId, - toolName, - error: toError(reportError).message, - }) + // Awaited to the end: the caller holds the turn's Stop lease until the tool has reported. + await Promise.resolve() + .then(() => + runUserLocalFilesystemTool(toolCallId, toolName, args, { + invoke: (request) => bridge().localFilesystem(request), + vfsRoot: mountVfsRoot, + signal: context.signal, + }) + ) + .then( + async (data) => { + if (context.signal?.aborted) return + await reportCompletion(toolCallId, toolName, { ok: true, data }) + }, + async (error) => { + if (context.signal?.aborted || (error instanceof Error && error.name === 'AbortError')) { + return + } + const message = toError(error).message + logger.warn('Local filesystem tool failed', { toolCallId, toolName, error: message }) + await reportCompletion(toolCallId, toolName, { ok: false, error: message }) } - } - ) + ) } diff --git a/apps/sim/lib/mothership/tools/client/terminal-tool-execution.test.ts b/apps/sim/lib/mothership/tools/client/terminal-tool-execution.test.ts index 7bf7968924b..79d7c0fb7db 100644 --- a/apps/sim/lib/mothership/tools/client/terminal-tool-execution.test.ts +++ b/apps/sim/lib/mothership/tools/client/terminal-tool-execution.test.ts @@ -71,4 +71,22 @@ describe('terminal client execution', () => { expect.objectContaining({ notStarted: true }) ) }) + + it('tells the model the code of a generic terminal failure', async () => { + const reported: Array<{ status: string; data: unknown }> = [] + reportClientToolCompletion.mockImplementation( + async (_id: string, status: string, _message: string, data: unknown) => { + reported.push({ status, data }) + } + ) + executeTerminalTool.mockRejectedValue(new Error('The terminal went away')) + + executeTerminalToolOnClient('terminal-generic', { operation: 'read', args: {} }, 'chat-1') + + await vi.waitFor(() => expect(reported).toHaveLength(1)) + expect(reported[0]).toEqual({ + status: 'error', + data: { error: 'The terminal went away', code: 'Error' }, + }) + }) }) diff --git a/apps/sim/lib/mothership/tools/client/terminal-tool-execution.ts b/apps/sim/lib/mothership/tools/client/terminal-tool-execution.ts index 71bfc1a6489..75f6ed499e8 100644 --- a/apps/sim/lib/mothership/tools/client/terminal-tool-execution.ts +++ b/apps/sim/lib/mothership/tools/client/terminal-tool-execution.ts @@ -8,6 +8,11 @@ * server-side waiter. */ +import { + terminalOperationTimeoutMs, + terminalToolCompletion, + terminalToolFailure, +} from '@sim/desktop-bridge/tool-results' import { createLogger } from '@sim/logger' import { isTerminalOperation, @@ -64,20 +69,6 @@ function eventAgeMs(eventTs: string | undefined): number | null { return Number.isNaN(emitted) ? null : Date.now() - emitted } -/** - * `terminal_run` has no client-side deadline: it can sit on an approval chip - * for as long as the user takes, and the desktop side already bounds the - * command itself. The rest are near-instant, so a short timeout keeps a wedged - * bridge from stalling the turn. - */ -const QUICK_TOOL_TIMEOUT_MS = 15_000 - -function timeoutForOperation(operation: TerminalOperation): number | null { - // `run` waits on a command and `handoff` waits on a person; neither has a - // deadline this side can usefully impose. - return operation === 'run' || operation === 'handoff' ? null : QUICK_TOOL_TIMEOUT_MS -} - /** * Splits a `terminal` tool call into its operation and arguments. The model * supplies both inside one params object, and an unrecognized operation is @@ -183,7 +174,7 @@ async function doExecuteTerminalTool( logger.info('Executing terminal operation via the desktop terminal', { toolCallId, operation }) try { - const timeoutMs = timeoutForOperation(operation) + const timeoutMs = terminalOperationTimeoutMs(operation) const invocation = executeTerminalTool(toolCallId, operation, args, scopeId) const result = timeoutMs === null @@ -197,25 +188,24 @@ async function doExecuteTerminalTool( ) }), ]) + const completion = terminalToolCompletion({ ok: true, result }) await reportClientToolCompletion( toolCallId, - ASYNC_TOOL_CONFIRMATION_STATUS.success, - 'Terminal action completed', - result as Record | undefined + completion.status, + completion.message, + completion.data ) } catch (err) { const error = toError(err) - // A declined command is a normal outcome, not a fault: reporting it as - // cancelled lets the model adapt instead of retrying the same command. - const status = - error.name === 'REJECTED' - ? ASYNC_TOOL_CONFIRMATION_STATUS.cancelled - : ASYNC_TOOL_CONFIRMATION_STATUS.error logger.warn('Terminal operation failed', { toolCallId, operation, error: error.message }) - await reportClientToolCompletion(toolCallId, status, error.message, { - error: error.message, - ...(error.name ? { code: error.name } : {}), - }).catch((reportErr) => { + // The error's name goes to the model as its code, `Error` included, as it always has. + const completion = terminalToolFailure(error.message, error.name || undefined) + await reportClientToolCompletion( + toolCallId, + completion.status, + completion.message, + completion.data + ).catch((reportErr) => { logger.error('Failed to report terminal tool error', { toolCallId, error: toError(reportErr).message, diff --git a/bun.lock b/bun.lock index 206323b095a..1b2e0623614 100644 --- a/bun.lock +++ b/bun.lock @@ -519,10 +519,13 @@ "@sim/browser-protocol": "workspace:*", "@sim/terminal-protocol": "workspace:*", "@sim/utils": "workspace:*", + "micromatch": "4.0.8", }, "devDependencies": { "@sim/tsconfig": "workspace:*", + "@types/micromatch": "4.0.10", "typescript": "^7.0.2", + "vitest": "^5.0.1", }, }, "packages/emcn": { diff --git a/packages/desktop-bridge/package.json b/packages/desktop-bridge/package.json index 05a90f37c52..02c11ca4af9 100644 --- a/packages/desktop-bridge/package.json +++ b/packages/desktop-bridge/package.json @@ -17,10 +17,19 @@ "./local-filesystem-limits": { "types": "./src/local-filesystem-limits.ts", "default": "./src/local-filesystem-limits.ts" + }, + "./local-filesystem-tools": { + "types": "./src/local-filesystem-tools.ts", + "default": "./src/local-filesystem-tools.ts" + }, + "./tool-results": { + "types": "./src/tool-results.ts", + "default": "./src/tool-results.ts" } }, "scripts": { "type-check": "tsc --noEmit", + "test": "vitest run", "lint": "biome check --write --unsafe .", "lint:check": "biome check .", "format": "biome format --write .", @@ -29,10 +38,13 @@ "dependencies": { "@sim/browser-protocol": "workspace:*", "@sim/terminal-protocol": "workspace:*", - "@sim/utils": "workspace:*" + "@sim/utils": "workspace:*", + "micromatch": "4.0.8" }, "devDependencies": { "@sim/tsconfig": "workspace:*", - "typescript": "^7.0.2" + "@types/micromatch": "4.0.10", + "typescript": "^7.0.2", + "vitest": "^5.0.1" } } diff --git a/packages/desktop-bridge/src/index.ts b/packages/desktop-bridge/src/index.ts index bca524088ab..89c89f28e2b 100644 --- a/packages/desktop-bridge/src/index.ts +++ b/packages/desktop-bridge/src/index.ts @@ -1073,7 +1073,7 @@ interface SimDesktopServerApi { } /** The device the desktop app registered for its background executor. */ -interface DesktopExecutorDevice { +export interface DesktopExecutorDevice { /** The install id Sim binds a turn to. */ deviceId: string /** The executor protocol version Sim accepted at registration. */ diff --git a/packages/desktop-bridge/src/local-filesystem-tools.test.ts b/packages/desktop-bridge/src/local-filesystem-tools.test.ts new file mode 100644 index 00000000000..122ec3e6bad --- /dev/null +++ b/packages/desktop-bridge/src/local-filesystem-tools.test.ts @@ -0,0 +1,81 @@ +import { describe, expect, it } from 'vitest' +import type { LocalFilesystemRequest, LocalFilesystemResponse } from './index' +import { runUserLocalFilesystemTool } from './local-filesystem-tools' + +const mount = { id: 'mount-1', name: 'Project', uri: 'localfs://mount-1/', remembered: true } + +function context(globTruncated: boolean) { + return { + vfsRoot: () => 'user-local/Project--mount-1', + invoke: async (request: LocalFilesystemRequest): Promise => { + if (request.operation === 'list_mounts') return { ok: true, data: { mounts: [mount] } } + if (request.operation === 'grep') + return { + ok: true, + data: { + matches: [{ uri: 'localfs://mount-1/a.ts', line: 1, text: 'const a = 1' }], + truncated: globTruncated, + }, + } + return { + ok: true, + data: { + entries: [ + { + name: 'a.ts', + uri: 'localfs://mount-1/a.ts', + kind: 'file', + size: 1, + modifiedAt: '2026-01-01T00:00:00.000Z', + }, + ], + truncated: globTruncated, + }, + } + }, + } +} + +describe('user-local glob', () => { + it('says when the folder scan stopped early, so a missing file is not read as absent', async () => { + await expect( + runUserLocalFilesystemTool('call-1', 'glob', { pattern: 'user-local/**/*.ts' }, context(true)) + ).resolves.toEqual({ files: ['user-local/Project--mount-1/a.ts'], truncated: true }) + }) + + it('reports a complete listing without the flag', async () => { + await expect( + runUserLocalFilesystemTool( + 'call-1', + 'glob', + { pattern: 'user-local/**/*.ts' }, + context(false) + ) + ).resolves.toEqual({ files: ['user-local/Project--mount-1/a.ts'] }) + }) +}) + +describe('user-local grep', () => { + const args = { pattern: 'const', path: 'user-local' } + + it('says when the search stopped early, so a missing match is not read as absent', async () => { + await expect( + runUserLocalFilesystemTool('call-1', 'grep', args, context(true)) + ).resolves.toEqual({ + matches: [{ path: 'user-local/Project--mount-1/a.ts', line: 1, content: 'const a = 1' }], + truncated: true, + }) + }) + + it('says when the result cap cut the listing short', async () => { + await expect( + runUserLocalFilesystemTool('call-1', 'grep', { ...args, maxResults: 1 }, context(false)) + ).resolves.toMatchObject({ truncated: true }) + }) + + it('reports a complete search without the flag', async () => { + await expect( + runUserLocalFilesystemTool('call-1', 'grep', args, context(false)) + ).resolves.not.toHaveProperty('truncated') + }) +}) diff --git a/packages/desktop-bridge/src/local-filesystem-tools.ts b/packages/desktop-bridge/src/local-filesystem-tools.ts new file mode 100644 index 00000000000..084d3dc0a95 --- /dev/null +++ b/packages/desktop-bridge/src/local-filesystem-tools.ts @@ -0,0 +1,324 @@ +/** + * The model's `read`, `grep` and `glob` over user-local folders, run against the desktop's + * local filesystem service. The chat view runs them through the preload bridge; the desktop's + * background executor runs them in-process. Both get the same paths and result shapes. + */ +import micromatch from 'micromatch' +import type { + LocalFilesystemData, + LocalFilesystemMount, + LocalFilesystemRequest, + LocalFilesystemResponse, +} from './index' +import { + DEFAULT_GREP_CONTEXT, + DEFAULT_GREP_RESULTS, + DEFAULT_READ_LINES, + MAX_GREP_CONTEXT, + MAX_GREP_RESULTS, + MAX_READ_LINES, +} from './local-filesystem-limits' + +/** The VFS directory every granted folder is mounted under. */ +const USER_LOCAL_VFS_ROOT = 'user-local' + +/** + * This glob runs against already-listed mounts, so unlike the grep and read limits it is nobody + * else's business: the shell never authorizes against it. + */ +const MAX_USER_LOCAL_GLOB_RESULTS = 500 + +const VFS_GLOB_OPTIONS: micromatch.Options = { + bash: false, + dot: false, + windows: false, + nobrace: true, + noext: true, +} + +export interface UserLocalFilesystemToolContext { + /** Sends one request to the local filesystem service. */ + invoke: (request: LocalFilesystemRequest) => Promise + /** The `user-local/--` directory a mount appears under. */ + vfsRoot: (mount: LocalFilesystemMount) => string + signal?: AbortSignal +} + +function requiredString(args: Record, name: string): string { + const value = args[name] + if (typeof value !== 'string' || value.length === 0) { + throw new Error(`${name} is required`) + } + return value +} + +function optionalNumber(value: unknown): number | undefined { + return typeof value === 'number' && Number.isFinite(value) ? value : undefined +} + +function abortError(signal: AbortSignal): Error { + const error = new Error(signal.reason ? String(signal.reason) : 'Operation aborted') + error.name = 'AbortError' + return error +} + +/** + * One request, cancelled through the service when the signal aborts. A request id names the + * tool call so the service can refuse a duplicate and cancel exactly this one. + */ +async function invoke( + context: UserLocalFilesystemToolContext, + request: LocalFilesystemRequest +): Promise { + const { signal } = context + if (signal?.aborted) throw abortError(signal) + const requestId = 'requestId' in request ? request.requestId : undefined + const onAbort = () => { + if (requestId) void context.invoke({ operation: 'cancel', requestId }) + } + signal?.addEventListener('abort', onAbort, { once: true }) + try { + const response = await context.invoke(request) + if (!response.ok) throw new Error(response.error) + if (signal?.aborted) throw abortError(signal) + return response.data + } finally { + signal?.removeEventListener('abort', onAbort) + } +} + +function vfsPathForUri( + context: UserLocalFilesystemToolContext, + mount: LocalFilesystemMount, + uri: string +): string { + const parsed = new URL(uri) + if (parsed.protocol !== 'localfs:' || parsed.hostname !== mount.id) { + throw new Error('The desktop app returned a local path outside the selected folder.') + } + const relativePath = parsed.pathname.replace(/^\/+/, '') + const root = context.vfsRoot(mount) + return relativePath ? `${root}/${relativePath}` : root +} + +function localUriForVfsPath( + context: UserLocalFilesystemToolContext, + mount: LocalFilesystemMount, + path: string +): string { + const root = context.vfsRoot(mount) + if (path === root) return mount.uri + if (!path.startsWith(`${root}/`)) { + throw new Error(`Path is not inside a granted user-local folder: ${path}`) + } + return `${mount.uri}${path.slice(root.length + 1)}` +} + +async function listMounts( + context: UserLocalFilesystemToolContext +): Promise { + const data = await invoke(context, { operation: 'list_mounts' }) + if (!('mounts' in data)) { + throw new Error('The desktop app returned an invalid mount list.') + } + return data.mounts +} + +function mountForPath( + context: UserLocalFilesystemToolContext, + mounts: LocalFilesystemMount[], + path: string +): LocalFilesystemMount { + const match = mounts.find((mount) => { + const root = context.vfsRoot(mount) + return path === root || path.startsWith(`${root}/`) + }) + if (!match) { + throw new Error( + `No granted user-local folder contains "${path}". Use glob({pattern:"user-local/**"}) to discover canonical paths.` + ) + } + return match +} + +async function glob( + context: UserLocalFilesystemToolContext, + requestId: string, + args: Record +): Promise<{ files: string[]; truncated?: true }> { + const pattern = requiredString(args, 'pattern') + const mounts = await listMounts(context) + const files = new Set() + let scanIncomplete = false + + for (const mount of mounts) { + if (context.signal?.aborted) throw abortError(context.signal) + const root = context.vfsRoot(mount) + if (micromatch.isMatch(root, pattern, VFS_GLOB_OPTIONS)) { + files.add(root) + } + + const data = await invoke(context, { + operation: 'glob', + uri: mount.uri, + pattern, + pathPrefix: root, + requestId, + }) + if (!('entries' in data)) { + throw new Error('The desktop app returned an invalid glob result.') + } + if (data.truncated) scanIncomplete = true + for (const entry of data.entries) { + const path = vfsPathForUri(context, mount, entry.uri) + if (micromatch.isMatch(path, pattern, VFS_GLOB_OPTIONS)) { + files.add(path) + if (files.size >= MAX_USER_LOCAL_GLOB_RESULTS) break + } + } + if (files.size >= MAX_USER_LOCAL_GLOB_RESULTS) break + } + + // A partial listing says so, so the model does not treat a missing file as absent. + const truncated = scanIncomplete || files.size >= MAX_USER_LOCAL_GLOB_RESULTS + return { files: [...files].sort(), ...(truncated ? { truncated: true as const } : {}) } +} + +async function read( + context: UserLocalFilesystemToolContext, + requestId: string, + args: Record +): Promise<{ content: string; totalLines: number }> { + const path = requiredString(args, 'path') + const mounts = await listMounts(context) + const mount = mountForPath(context, mounts, path) + const offset = Math.max(0, Math.trunc(optionalNumber(args.offset) ?? 0)) + const requestedLimit = optionalNumber(args.limit) + const lineCount = Math.min( + MAX_READ_LINES, + Math.max(1, Math.trunc(requestedLimit ?? DEFAULT_READ_LINES)) + ) + const data = await invoke(context, { + operation: 'read', + uri: localUriForVfsPath(context, mount, path), + startLine: offset + 1, + lineCount, + requestId, + }) + if (!('content' in data) || !('totalLines' in data)) { + throw new Error('The desktop app returned an invalid read result.') + } + return { content: data.content, totalLines: data.totalLines } +} + +async function grep( + context: UserLocalFilesystemToolContext, + requestId: string, + args: Record +): Promise> { + const pattern = requiredString(args, 'pattern') + const path = requiredString(args, 'path').replace(/\/+$/, '') + const outputMode = + args.output_mode === 'files_with_matches' || args.output_mode === 'count' + ? args.output_mode + : 'content' + const maxResults = Math.min( + MAX_GREP_RESULTS, + Math.max(1, Math.trunc(optionalNumber(args.maxResults) ?? DEFAULT_GREP_RESULTS)) + ) + const mounts = await listMounts(context) + const targets = + path === USER_LOCAL_VFS_ROOT + ? mounts.map((mount) => ({ mount, uri: mount.uri })) + : (() => { + const mount = mountForPath(context, mounts, path) + return [{ mount, uri: localUriForVfsPath(context, mount, path) }] + })() + + const contentMatches: Array<{ path: string; line: number; content: string }> = [] + const matchingFiles = new Set() + const counts = new Map() + let scanIncomplete = false + const found = () => + outputMode === 'files_with_matches' + ? matchingFiles.size + : outputMode === 'count' + ? counts.size + : contentMatches.length + + for (const target of targets) { + const data = await invoke(context, { + operation: 'grep', + uri: target.uri, + pattern, + caseSensitive: args.ignoreCase !== true, + maxResults, + outputMode, + lineNumbers: args.lineNumbers !== false, + context: Math.min( + MAX_GREP_CONTEXT, + Math.max(0, Math.trunc(optionalNumber(args.context) ?? DEFAULT_GREP_CONTEXT)) + ), + requestId, + }) + + if ('truncated' in data && data.truncated) scanIncomplete = true + if ('matches' in data) { + for (const match of data.matches) { + contentMatches.push({ + path: vfsPathForUri(context, target.mount, match.uri), + line: match.line, + content: match.text, + }) + if (contentMatches.length >= maxResults) break + } + } else if ('files' in data) { + for (const uri of data.files) { + matchingFiles.add(vfsPathForUri(context, target.mount, uri)) + if (matchingFiles.size >= maxResults) break + } + } else if ('counts' in data) { + for (const count of data.counts) { + counts.set(vfsPathForUri(context, target.mount, count.uri), count.count) + if (counts.size >= maxResults) break + } + } else { + throw new Error('The desktop app returned an invalid grep result.') + } + + if (found() >= maxResults) break + } + + // Stopped at the cap or by the desktop's own scan limit: more may match than listed. + const truncated = scanIncomplete || found() >= maxResults ? { truncated: true as const } : {} + if (outputMode === 'files_with_matches') { + return { files: [...matchingFiles].sort(), ...truncated } + } + if (outputMode === 'count') { + return { + counts: [...counts.entries()] + .map(([countPath, count]) => ({ path: countPath, count })) + .sort((a, b) => a.path.localeCompare(b.path)), + ...truncated, + } + } + return { + matches: contentMatches.sort((a, b) => a.path.localeCompare(b.path) || a.line - b.line), + ...truncated, + } +} + +/** + * Runs one user-local `read`, `grep` or `glob` tool call. The tool call id is the request id, so + * the service can cancel it and refuse to run it twice at once. + */ +export async function runUserLocalFilesystemTool( + toolCallId: string, + toolName: string, + args: Record, + context: UserLocalFilesystemToolContext +): Promise> { + if (toolName === 'glob') return glob(context, toolCallId, args) + if (toolName === 'grep') return grep(context, toolCallId, args) + return read(context, toolCallId, args) +} diff --git a/apps/sim/lib/mothership/tools/client/browser-tool-result.test.ts b/packages/desktop-bridge/src/tool-results.test.ts similarity index 97% rename from apps/sim/lib/mothership/tools/client/browser-tool-result.test.ts rename to packages/desktop-bridge/src/tool-results.test.ts index a0bac1d6f2c..ba45614060f 100644 --- a/apps/sim/lib/mothership/tools/client/browser-tool-result.test.ts +++ b/packages/desktop-bridge/src/tool-results.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { sanitizeBrowserToolResultForModel } from '@/lib/mothership/tools/client/browser-tool-result' +import { sanitizeBrowserToolResultForModel } from './tool-results' describe('browser screenshot model projection', () => { it('keeps an image usable when an older desktop omits coordinate metadata', () => { diff --git a/packages/desktop-bridge/src/tool-results.ts b/packages/desktop-bridge/src/tool-results.ts new file mode 100644 index 00000000000..ff6f864995b --- /dev/null +++ b/packages/desktop-bridge/src/tool-results.ts @@ -0,0 +1,234 @@ +/** + * What a desktop tool call reports to Sim once it finishes, built the same way whichever process + * ran it: the chat view (for a turn it serves) or the desktop app's background executor (for a + * turn bound to it). One projection per surface keeps the model's view of a result identical on + * both paths. + */ +import type { BrowserToolName } from '@sim/browser-protocol' +import type { TerminalOperation, TerminalToolResponse } from '@sim/terminal-protocol' +import { isRecordLike, toRecordOrNull } from '@sim/utils/object' +import type { DesktopLocalFileResponse } from './local-files' + +type DesktopToolCompletionStatus = 'success' | 'error' | 'cancelled' + +export interface DesktopToolCompletion { + status: DesktopToolCompletionStatus + message: string + data?: Record +} + +/** + * Tools that do not need an existing live page: most create one, and `browser_list_sessions` + * reads the profile-level session registry. Every other tool fails fast on a closed session + * instead of waiting out its timeout. + */ +const LIVE_PAGE_OPTIONAL_BROWSER_TOOLS: ReadonlySet = new Set([ + 'browser_navigate', + 'browser_open_url', + 'browser_open_tab', + 'browser_list_tabs', + 'browser_list_sessions', + 'browser_list_downloads', + 'browser_save_download', +]) + +export function browserToolNeedsLivePage(toolName: BrowserToolName): boolean { + return !LIVE_PAGE_OPTIONAL_BROWSER_TOOLS.has(toolName) +} + +const BROWSER_SESSION_CLOSED_MESSAGE = + 'The agent browser session is closed, so this browser tool cannot run. ' + + 'Call browser_open_url, browser_navigate, or browser_open_tab to start a new session, or report the situation to the user. ' + + 'Do not retry other browser tools until a new session is open.' + +/** A browser tool refused because its chat has no live page to act on. */ +export function browserSessionClosedCompletion(): DesktopToolCompletion { + return { + status: 'error', + message: BROWSER_SESSION_CLOSED_MESSAGE, + data: { error: BROWSER_SESSION_CLOSED_MESSAGE, sessionClosed: true }, + } +} + +/** What the model learns when the browser did not answer within the tool's deadline. */ +export function browserToolTimeoutMessage(timeoutMs: number): string { + return `The browser did not respond within ${timeoutMs}ms. Its outcome is unknown and the action may already have taken effect. Do not retry it automatically; take a fresh browser snapshot before deciding what to do.` +} + +/** A browser tool that failed; `sessionClosed` adds the guidance for reopening a session. */ +export function browserToolFailure( + error: string, + options: { outcomeUnknown?: boolean; sessionClosed?: boolean } = {} +): DesktopToolCompletion { + const message = options.sessionClosed ? `${error} ${BROWSER_SESSION_CLOSED_MESSAGE}` : error + return { + status: 'error', + message, + data: { + error: message, + ...(options.outcomeUnknown ? { outcomeUnknown: true, doNotRetry: true } : {}), + ...(options.sessionClosed ? { sessionClosed: true } : {}), + }, + } +} + +function finiteNumber(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) +} + +function imageDimensions(value: unknown): { width: number; height: number } | null { + if ( + !isRecordLike(value) || + !finiteNumber(value.width) || + !finiteNumber(value.height) || + value.width <= 0 || + value.height <= 0 + ) { + return null + } + return { width: value.width, height: value.height } +} + +/** Projects image bytes and coordinate metadata into the model's image-content contract. */ +export function sanitizeBrowserToolResultForModel( + toolName: BrowserToolName, + result: unknown +): Record | undefined { + if (!isRecordLike(result)) { + return result === undefined ? undefined : { value: result } + } + if (toolName !== 'browser_screenshot' || typeof result.dataUrl !== 'string') return result + + const { dataUrl, ...rest } = result + const image = /^data:([^;,]+);base64,(.+)$/s.exec(dataUrl) + if (!image) { + return { + ...rest, + note: 'The screenshot could not be encoded. Use browser_snapshot or browser_read_text instead.', + } + } + const viewport = toRecordOrNull(rest.viewport) + const screenshotUrl = + typeof rest.url === 'string' && rest.url + ? rest.url + : viewport && typeof viewport.url === 'string' + ? viewport.url + : '' + const location = screenshotUrl ? ` of ${screenshotUrl}` : '' + const clip = toRecordOrNull(rest.clip) + const cropSize = imageDimensions(clip) + const viewportSize = imageDimensions(viewport) + const imageSize = imageDimensions(rest.imageSize) + const capturedSize = clip ? cropSize : viewportSize + const scaleX = imageSize && capturedSize ? imageSize.width / capturedSize.width : null + const scaleY = imageSize && capturedSize ? imageSize.height / capturedSize.height : null + const hasScale = finiteNumber(scaleX) && scaleX > 0 && finiteNumber(scaleY) && scaleY > 0 + const origin = clip + ? finiteNumber(clip.x) && finiteNumber(clip.y) + ? { x: clip.x, y: clip.y } + : null + : { x: 0, y: 0 } + const content = [ + `Screenshot${location}. This is the rendered ${clip ? 'element' : 'viewport'} only — it carries no element ids. Use browser_snapshot for element-ref actions and the mapping below for coordinate actions.`, + viewportSize && `Viewport: ${viewportSize.width} × ${viewportSize.height} CSS pixels.`, + imageSize && `Encoded image: ${imageSize.width} × ${imageSize.height} pixels.`, + hasScale && `Image scale: X=${scaleX}, Y=${scaleY} encoded image pixels per CSS pixel.`, + cropSize && `Crop size: ${cropSize.width} × ${cropSize.height} CSS pixels.`, + clip && origin && `Crop origin: (${origin.x}, ${origin.y}) in viewport CSS pixels.`, + hasScale && origin + ? `Coordinate actions use viewport CSS pixels: cssX = ${origin.x} + imageX / ${scaleX}; cssY = ${origin.y} + imageY / ${scaleY}. imageX/imageY refer to the encoded image before any display resizing.` + : 'Screenshot coordinate mapping is unavailable; use browser_snapshot element references or take a new viewport screenshot before coordinate actions.', + ] + .filter(Boolean) + .join(' ') + return { + ...rest, + content, + attachment: { + type: 'image', + source: { type: 'base64', media_type: image[1], data: image[2] }, + }, + } +} + +/** + * A browser tool the desktop ran to the end. A partial form fill, a failed batch step, or an + * action whose outcome the driver could not confirm is reported as an error so the model + * inspects the page before acting on it. + */ +export function browserToolCompletion( + toolName: BrowserToolName, + result: unknown +): DesktopToolCompletion { + const record = isRecordLike(result) ? result : null + const outcomeUnknown = record?.outcomeUnknown === true + const stoppedMessage = + toolName === 'browser_fill_form' && record?.completed === false + ? 'Form filling stopped; inspect the partial result' + : toolName === 'browser_batch' && record?.stoppedBy === 'failure' + ? 'A batched browser action failed; inspect the partial result' + : undefined + return { + status: stoppedMessage || outcomeUnknown ? 'error' : 'success', + message: + stoppedMessage ?? + (outcomeUnknown + ? 'Browser action outcome is unconfirmed; inspect the page before repeating it.' + : record?.effectObserved === false + ? 'Browser input completed; its effect is unconfirmed. Inspect the current state before retrying.' + : 'Browser action completed'), + data: sanitizeBrowserToolResultForModel(toolName, result), + } +} + +/** A terminal operation that failed, with the terminal's error code when it gave one. */ +export function terminalToolFailure(error: string, code?: string): DesktopToolCompletion { + return { status: 'error', message: error, data: { error, ...(code ? { code } : {}) } } +} + +/** A terminal operation's outcome as the terminal service returned it. */ +export function terminalToolCompletion(response: TerminalToolResponse): DesktopToolCompletion { + if (!response.ok) { + return terminalToolFailure(response.error || 'The terminal reported an error', response.code) + } + return { + status: 'success', + message: 'Terminal action completed', + ...(isRecordLike(response.result) ? { data: response.result } : {}), + } +} + +const QUICK_TERMINAL_OPERATION_TIMEOUT_MS = 15_000 + +/** + * How long a terminal operation may take before it is reported as unresponsive. `run` waits on a + * command and `handoff` on a person, and both bound themselves; everything else is near-instant, + * so a short deadline keeps a wedged terminal from stalling the turn. + */ +export function terminalOperationTimeoutMs(operation: TerminalOperation): number | null { + return operation === 'run' || operation === 'handoff' ? null : QUICK_TERMINAL_OPERATION_TIMEOUT_MS +} + +/** A `read_local_file` call's outcome; the read itself never changes anything on the machine. */ +export function localFileReadCompletion(response: DesktopLocalFileResponse): DesktopToolCompletion { + if (!response.ok) + return { status: 'error', message: response.error, data: { error: response.error } } + if (response.data.kind !== 'read') { + const error = 'The desktop app returned an unexpected local file result.' + return { status: 'error', message: error, data: { error } } + } + return { + status: 'success', + message: 'Local file operation completed.', + data: { ...response.data }, + } +} + +/** A user-local folder read (`read`, `grep`, `glob`) the desktop finished or failed. */ +export function localFilesystemToolCompletion( + outcome: { ok: true; data: Record } | { ok: false; error: string } +): DesktopToolCompletion { + return outcome.ok + ? { status: 'success', message: 'Local filesystem tool completed.', data: outcome.data } + : { status: 'error', message: outcome.error, data: { error: outcome.error } } +} diff --git a/packages/desktop-bridge/vitest.config.ts b/packages/desktop-bridge/vitest.config.ts new file mode 100644 index 00000000000..6f729af4c84 --- /dev/null +++ b/packages/desktop-bridge/vitest.config.ts @@ -0,0 +1,11 @@ +import { defineConfig, mergeConfig } from 'vitest/config' +import sharedConfig from '../../vitest.shared' + +export default mergeConfig( + sharedConfig, + defineConfig({ + test: { + include: ['src/**/*.test.ts'], + }, + }) +) diff --git a/packages/terminal-protocol/src/index.ts b/packages/terminal-protocol/src/index.ts index 69b57877722..7356750ca7c 100644 --- a/packages/terminal-protocol/src/index.ts +++ b/packages/terminal-protocol/src/index.ts @@ -454,6 +454,8 @@ export type TerminalErrorCode = /** No pane with that target — the targets come from the `panes` operation. */ | 'NO_SUCH_PANE' | 'INVALID_REQUEST' + /** The call was stopped before its command started; nothing ran. */ + | 'CANCELLED' export interface TerminalStartOptions { cols: number diff --git a/scripts/check-unused-exports.baseline.json b/scripts/check-unused-exports.baseline.json index e9cff4acdb6..cf291b4465e 100644 --- a/scripts/check-unused-exports.baseline.json +++ b/scripts/check-unused-exports.baseline.json @@ -40,7 +40,6 @@ "apps/desktop/src/main/observability.ts#DesktopEventName", "apps/desktop/src/main/session-lifecycle.ts#SessionLifecycleDeps", "apps/desktop/src/main/session-lifecycle.ts#isLogoutNavigation", - "apps/desktop/src/main/session-lifecycle.ts#isSessionCookieName", "apps/desktop/src/main/session-lifecycle.ts#revokeAppSession", "apps/desktop/src/main/telemetry-policy.ts#BLOCKED_ANALYTICS_HOSTS", "apps/desktop/src/main/terminal-themes.ts#parseTerminalThemeProfiles",