From 767fb500346591c4402ec0c461d7f3bd9218412c Mon Sep 17 00:00:00 2001 From: Peter Steinberger Date: Tue, 28 Jul 2026 21:30:01 -0400 Subject: [PATCH] refactor(code-mode): split focused runtime stress suites (#115462) --- config/max-lines-baseline.txt | 1 - src/agents/code-mode.bridge.test.ts | 616 +++++ src/agents/code-mode.config.test.ts | 75 + src/agents/code-mode.guest.test.ts | 713 +++++ src/agents/code-mode.limits.test.ts | 413 +++ src/agents/code-mode.mcp.test.ts | 372 +++ src/agents/code-mode.replay.test.ts | 227 ++ src/agents/code-mode.skills.test.ts | 178 ++ src/agents/code-mode.test-support.ts | 154 +- src/agents/code-mode.test.ts | 3308 +---------------------- src/agents/code-mode.typescript.test.ts | 180 ++ src/agents/code-mode.wait.test.ts | 583 ++++ 12 files changed, 3521 insertions(+), 3299 deletions(-) create mode 100644 src/agents/code-mode.bridge.test.ts create mode 100644 src/agents/code-mode.config.test.ts create mode 100644 src/agents/code-mode.guest.test.ts create mode 100644 src/agents/code-mode.limits.test.ts create mode 100644 src/agents/code-mode.mcp.test.ts create mode 100644 src/agents/code-mode.replay.test.ts create mode 100644 src/agents/code-mode.skills.test.ts create mode 100644 src/agents/code-mode.typescript.test.ts create mode 100644 src/agents/code-mode.wait.test.ts diff --git a/config/max-lines-baseline.txt b/config/max-lines-baseline.txt index 75619b32e51a..233b8720b4ec 100644 --- a/config/max-lines-baseline.txt +++ b/config/max-lines-baseline.txt @@ -382,7 +382,6 @@ src/agents/cli-runner/claude-live-session.ts src/agents/cli-runner/execute.supervisor-capture.test.ts src/agents/cli-runner/prepare.test.ts src/agents/cli-runner/prepare.ts -src/agents/code-mode.test.ts src/agents/command/attempt-execution.cli.test.ts src/agents/command/attempt-execution.test.ts src/agents/command/attempt-execution.ts diff --git a/src/agents/code-mode.bridge.test.ts b/src/agents/code-mode.bridge.test.ts new file mode 100644 index 000000000000..f5aecbe7bb9a --- /dev/null +++ b/src/agents/code-mode.bridge.test.ts @@ -0,0 +1,616 @@ +/** Tests Code Mode bridge settlement and cancellation. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { Type } from "typebox"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { buildBlockedToolResult } from "./agent-tools.before-tool-call.js"; +import { applyCodeModeCatalog, createCodeModeTools } from "./code-mode.js"; +import { + resetCodeModeTestState, + pluginTool, + pluginToolWithExecute, + resultDetails, + createCodeModeHarness, + runUntilCompleted, + testing, +} from "./code-mode.test-support.js"; +import { createToolSearchCatalogRef } from "./tool-search.js"; +import { jsonResult } from "./tools/common.js"; + +describe("Code Mode bridge settlement and cancellation", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("drains a nested combinator after its outer race wins", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + let nestedAborted = false; + const never = pluginToolWithExecute( + "fake_nested_race_never", + "Never-settling nested race helper", + async (_toolCallId, _input, signal) => { + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, 25); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + nestedAborted = true; + reject(new Error("aborted")); + }, + { once: true }, + ); + }); + return jsonResult({ winner: "nested" }); + }, + ); + const fast = pluginToolWithExecute( + "fake_nested_race_fast", + "Fast nested race helper", + async () => jsonResult({ winner: "fast" }), + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, never, fast], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-nested-combinator-race", + { + code: `return await Promise.race([ + Promise.all([tools.callValue("fake_nested_race_never", {})]), + tools.callValue("fake_nested_race_fast", {}), + ]);`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); + expect(never.execute).toHaveBeenCalledOnce(); + expect(fast.execute).toHaveBeenCalledOnce(); + expect(nestedAborted).toBe(false); + expect(testing.activeRuns.size).toBe(0); + }); + + it("resolves sequential bridge tool calls inline within one exec instead of a wait per call", async () => { + const catalogRef = createToolSearchCatalogRef(); + // maxPendingToolCalls stays a per-batch concurrency cap; five sequential + // awaits must drain inline even with a cap of 2. + const config = { + tools: { codeMode: { enabled: true, maxPendingToolCalls: 2 } }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const codeModeTools = createCodeModeTools(ctx); + const ticket = pluginTool("fake_create_ticket", "Create a fake ticket"); + applyCodeModeCatalog({ + tools: [...codeModeTools, ticket], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + // Five separate awaits would each suspend to the model under a wait-per-call + // design; inline resumption collapses them into a single completed exec so + // the model spends one turn instead of six. + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-inline", + { + code: ` + const ids = []; + for (let index = 0; index < 5; index += 1) { + const called = await tools.callValue("fake_create_ticket", { value: index }); + ids.push(called.input.value); + } + return ids; + `, + }, + ), + ); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual([0, 1, 2, 3, 4]); + expect(ticket.execute).toHaveBeenCalledTimes(5); + expect(testing.activeRuns.size).toBe(0); + }); + + it("keeps the actual winner when the later-started nested tool settles first", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + let firstAborted = false; + const first = pluginToolWithExecute( + "fake_first", + "Earlier slow helper", + async (_toolCallId, _input, signal) => { + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, 25); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + firstAborted = true; + reject(new Error("aborted")); + }, + { once: true }, + ); + }); + return jsonResult({ winner: "first" }); + }, + ); + const second = pluginToolWithExecute("fake_second", "Later fast helper", async () => + jsonResult({ winner: "second" }), + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, first, second], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-later-winner", + { + code: `return await Promise.race([ + tools.callValue("fake_first", {}), + tools.callValue("fake_second", {}), + ]);`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: { winner: "second" } }); + expect(first.execute).toHaveBeenCalledOnce(); + expect(second.execute).toHaveBeenCalledOnce(); + expect(firstAborted).toBe(false); + expect(testing.activeRuns.size).toBe(0); + }); + + it.each([ + { + label: "directly", + auditCode: 'void tools.callValue("fake_early_audit", {});', + }, + { + label: "in a detached already-settled Promise.race", + auditCode: 'void Promise.race([tools.callValue("fake_early_audit", {}), Promise.resolve()]);', + }, + { + label: "in a detached Promise.all", + auditCode: 'void Promise.all([tools.callValue("fake_early_audit", {})]);', + }, + { + label: "in a detached Promise.allSettled", + auditCode: 'void Promise.allSettled([tools.callValue("fake_early_audit", {})]);', + }, + { + label: "in a detached Promise.any", + auditCode: 'void Promise.any([tools.callValue("fake_early_audit", {})]);', + }, + { + label: "in a detached Promise.race", + auditCode: 'void Promise.race([tools.callValue("fake_early_audit", {})]);', + }, + ])( + "drains a detached audit started $label before an awaited nested call", + async ({ auditCode }) => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + let auditCompleted = false; + let auditAborted = false; + const audit = pluginToolWithExecute( + "fake_early_audit", + "Early detached audit", + async (_toolCallId, _input, signal) => { + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, 250); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + auditAborted = true; + reject(new Error("aborted")); + }, + { once: true }, + ); + }); + auditCompleted = true; + return jsonResult({ recorded: true }); + }, + ); + const fast = pluginToolWithExecute("fake_awaited_fast", "Awaited fast helper", async () => + jsonResult({ winner: "fast" }), + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, audit, fast], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-early-detached-audit", + { + code: `${auditCode} + return await tools.callValue("fake_awaited_fast", {});`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); + expect(audit.execute).toHaveBeenCalledOnce(); + expect(fast.execute).toHaveBeenCalledOnce(); + expect(auditCompleted).toBe(true); + expect(auditAborted).toBe(false); + expect(testing.activeRuns.size).toBe(0); + }, + ); + + it("drains a race winner's detached audit and its slower race branch", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + let loserAborted = false; + const winner = pluginToolWithExecute("fake_race_winner", "Race winner", async () => + jsonResult({ winner: "fast" }), + ); + const loser = pluginToolWithExecute( + "fake_race_loser", + "Race loser", + async (_toolCallId, _input, signal) => { + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, 25); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + loserAborted = true; + reject(new Error("aborted")); + }, + { once: true }, + ); + }); + return jsonResult({ winner: "slow" }); + }, + ); + const audit = pluginToolWithExecute("fake_race_audit", "Detached audit", async () => + jsonResult({ recorded: true }), + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, winner, loser, audit], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-race-detached-audit", + { + code: `return Promise.race([ + tools.callValue("fake_race_winner", {}), + tools.callValue("fake_race_loser", {}), + ]).then((value) => { + void tools.callValue("fake_race_audit", {}); + return value; + });`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); + expect(winner.execute).toHaveBeenCalledOnce(); + expect(loser.execute).toHaveBeenCalledOnce(); + expect(audit.execute).toHaveBeenCalledOnce(); + expect(loserAborted).toBe(false); + expect(testing.activeRuns.size).toBe(0); + }); + + it("drains every detached nested tool before completing the guest", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const first = pluginToolWithExecute("fake_detached_first", "First detached helper", async () => + jsonResult({ name: "first" }), + ); + const second = pluginToolWithExecute( + "fake_detached_second", + "Second detached helper", + async () => jsonResult({ name: "second" }), + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, first, second], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-detached", + { + code: `void tools.callValue("fake_detached_first", {}); + void tools.callValue("fake_detached_second", {}); + return "done";`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: "done" }); + expect(first.execute).toHaveBeenCalledOnce(); + expect(second.execute).toHaveBeenCalledOnce(); + expect(testing.activeRuns.size).toBe(0); + }); + + it.each(["race", "any"] as const)( + "preserves the Promise.%s winner while draining the slower nested tool", + async (combinator) => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + let slowAborted = false; + let slowCompleted = false; + const fast = pluginToolWithExecute("fake_fast", "Fast helper", async () => + jsonResult({ winner: "fast" }), + ); + const slow = pluginToolWithExecute( + "fake_slow", + "Slow helper", + async (_toolCallId, _input, signal) => { + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, 25); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + slowAborted = true; + reject(new Error("aborted")); + }, + { once: true }, + ); + }); + slowCompleted = true; + return jsonResult({ winner: "slow" }); + }, + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, fast, slow], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + `code-call-${combinator}-fast`, + { + code: `return await Promise.${combinator}([ + tools.callValue("fake_fast", {}), + tools.callValue("fake_slow", {}), + ]);`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); + expect(fast.execute).toHaveBeenCalledOnce(); + expect(slow.execute).toHaveBeenCalledOnce(); + expect(slowCompleted).toBe(true); + expect(slowAborted).toBe(false); + expect(testing.activeRuns.size).toBe(0); + }, + ); + + it("preserves fail-fast Promise.all while draining the slower nested tool", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + let slowAborted = false; + let slowCompleted = false; + const failed = pluginToolWithExecute("fake_failed", "Failed helper", async () => { + throw new Error("fast failure"); + }); + const slow = pluginToolWithExecute( + "fake_slow", + "Slow helper", + async (_toolCallId, _input, signal) => { + await new Promise((resolve, reject) => { + const timer = setTimeout(resolve, 25); + signal?.addEventListener( + "abort", + () => { + clearTimeout(timer); + slowAborted = true; + reject(new Error("aborted")); + }, + { once: true }, + ); + }); + slowCompleted = true; + return jsonResult({ winner: "slow" }); + }, + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, failed, slow], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-fail-fast", + { + code: `try { + await Promise.all([ + tools.callValue("fake_failed", {}), + tools.callValue("fake_slow", {}), + ]); + return "unexpected success"; + } catch (error) { + return error.message; + }`, + }, + ), + ); + + expect(details).toMatchObject({ status: "completed", value: "fast failure" }); + expect(failed.execute).toHaveBeenCalledOnce(); + expect(slow.execute).toHaveBeenCalledOnce(); + expect(slowCompleted).toBe(true); + expect(slowAborted).toBe(false); + expect(testing.activeRuns.size).toBe(0); + }); + + it("fails fast without parking a suspended run when the exec call is aborted", async () => { + const catalogRef = createToolSearchCatalogRef(); + // Long timeout so a missing abort short-circuit would block the whole test. + const config = { + tools: { codeMode: { enabled: true, timeoutMs: 30_000 } }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const codeModeTools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [ + ...codeModeTools, + // A tool that never settles and ignores its abort signal; only the + // host-level abort race can free the cancelled exec. + pluginToolWithExecute("fake_stuck", "Stuck helper", async () => { + await new Promise(() => {}); + return null as never; + }), + ], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const controller = new AbortController(); + controller.abort(); + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-abort", + { code: "await tools.fake_stuck({}); return 'done';" }, + controller.signal, + ), + ); + + // Abort drops the run instead of parking it; a cancelled call must not pin + // one of the process-global suspended-run slots until TTL expiry. + expect(details.status).toBe("failed"); + expect(details.error).toBe("code mode execution aborted"); + expect(details.code).toBe("aborted"); + expect(testing.activeRuns.size).toBe(0); + }); + + it("terminates a running guest promptly when the exec call is aborted", async () => { + const catalogRef = createToolSearchCatalogRef(); + // Long timeout so only the abort race can end the hostile loop quickly. + const config = { + tools: { codeMode: { enabled: true, timeoutMs: 30_000 } }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const codeModeTools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const controller = new AbortController(); + const abortTimer = setTimeout(() => controller.abort(), 200); + const startedAt = Date.now(); + try { + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-abort-live", + { code: "while (true) {}" }, + controller.signal, + ), + ); + expect(details.status).toBe("failed"); + expect(details.error).toBe("code mode execution aborted"); + expect(details.code).toBe("aborted"); + } finally { + clearTimeout(abortTimer); + } + expect(Date.now() - startedAt).toBeLessThan(10_000); + expect(testing.activeRuns.size).toBe(0); + }); + + it("surfaces policy blocks as guest call errors for declared outputs", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const target = pluginTool("fake_policy_block", "Return policy-controlled rows"); + target.outputSchema = Type.Array( + Type.Object({ id: Type.String() }, { additionalProperties: false }), + ); + target.execute = vi.fn(async () => + buildBlockedToolResult({ reason: "blocked by orchard policy" }), + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, target], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + try { + const rows = await tools.callValue("fake_policy_block", {}); + return rows.map((row) => row.id); + } catch (error) { + return error.message; + } + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toContain("was blocked before execution: blocked by orchard policy"); + }); +}); diff --git a/src/agents/code-mode.config.test.ts b/src/agents/code-mode.config.test.ts new file mode 100644 index 000000000000..aef693f75781 --- /dev/null +++ b/src/agents/code-mode.config.test.ts @@ -0,0 +1,75 @@ +/** Tests pure Code Mode config without loading the guest or test runtime. */ + +import { describe, expect, it } from "vitest"; +import { resolveCodeModeConfig } from "./code-mode-runtime.js"; + +describe("Code Mode configuration", () => { + it("resolves object config defaults", () => { + expect(resolveCodeModeConfig({ tools: { codeMode: true } } as never).enabled).toBe(true); + const resolved = resolveCodeModeConfig({ + tools: { + codeMode: { + timeoutMs: 1234, + languages: ["typescript"], + }, + }, + } as never); + expect(resolved.enabled).toBe("auto"); + expect(resolveCodeModeConfig({ tools: { codeMode: { enabled: true } } } as never).enabled).toBe( + true, + ); + expect(resolved.runtime).toBe("quickjs-wasi"); + expect(resolved.mode).toBe("only"); + expect(resolved.timeoutMs).toBe(1234); + expect(resolved.languages).toEqual(["typescript"]); + const limitedSearch = resolveCodeModeConfig({ + tools: { + codeMode: { + enabled: true, + maxSearchLimit: 3, + }, + }, + } as never); + expect(limitedSearch.searchDefaultLimit).toBe(3); + expect(limitedSearch.maxSearchLimit).toBe(3); + }); + + it("resolves active-agent code mode over the runtime default", () => { + const config = { + tools: { + codeMode: { + enabled: false, + timeoutMs: 1234, + searchDefaultLimit: 6, + }, + }, + agents: { + list: [ + { + id: "ops", + tools: { + codeMode: { + enabled: true, + searchDefaultLimit: 4, + }, + }, + }, + { + id: "chat", + tools: { + codeMode: false, + }, + }, + ], + }, + } as never; + + const ops = resolveCodeModeConfig(config, "ops"); + expect(ops.enabled).toBe(true); + expect(ops.timeoutMs).toBe(1234); + expect(ops.searchDefaultLimit).toBe(4); + + expect(resolveCodeModeConfig(config, "chat").enabled).toBe(false); + expect(resolveCodeModeConfig(config, "missing").enabled).toBe(false); + }); +}); diff --git a/src/agents/code-mode.guest.test.ts b/src/agents/code-mode.guest.test.ts new file mode 100644 index 000000000000..b1dce25dfbe0 --- /dev/null +++ b/src/agents/code-mode.guest.test.ts @@ -0,0 +1,713 @@ +/** Tests Code Mode guest execution. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { applyCodeModeCatalog, createCodeModeTools } from "./code-mode.js"; +import { + resetCodeModeTestState, + pluginTool, + resultDetails, + createCodeModeHarness, + runUntilCompleted, + testing, +} from "./code-mode.test-support.js"; +import { createToolSearchCatalogRef } from "./tool-search.js"; + +describe("Code Mode guest execution", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("accepts command as an exec-compatible code alias", async () => { + const { config, catalogRef, tools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + const result = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-command-alias", { + command: "return 7;", + }), + ); + + expect(result.status).toBe("completed"); + expect(result.value).toBe(7); + }); + + it("rejects divergent code and command aliases", async () => { + const { config, catalogRef, tools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + await expect( + expectDefined(tools[0], "tools[0] test invariant").execute("code-call-divergent-alias", { + code: "return 1;", + command: "return 2;", + }), + ).rejects.toThrow("code and command must match when both are provided"); + }); + + it.each([ + { code: "ls -la /workspace/" }, + { code: "ls -1" }, + { command: "ls -la /workspace/" }, + { code: "pwd", command: "pwd" }, + { command: "pwd;" }, + { command: "pwd; // inspect the workspace" }, + { code: "# inspect the workspace\npwd" }, + { code: "#!/bin/sh\npwd" }, + { code: "pwd\nls -la /workspace" }, + { command: "pwd;ls -la /workspace" }, + { command: "/bin/ls /workspace/" }, + { command: "./gradlew test" }, + { code: ".\\gradlew.bat test" }, + { command: ".\\script.ps1" }, + { code: "C:\\workspace\\run.cmd /q" }, + { code: "/workspace/run.sh --verbose" }, + { command: "sh -c 'ls /workspace/'" }, + { command: "git status" }, + { command: 'git status; const note = "git";' }, + { command: "ls -1; const metadata = { ls: true };" }, + { code: "ls -1; const note = 'function ls';" }, + { command: "ls -1; let ls = 7;" }, + { command: "npm test" }, + { command: "NODE_ENV=test npm test" }, + { code: "NODE_ENV=test\nnpm test" }, + { code: "FOO=bar ./gradlew test" }, + { command: 'GREETING="hello world" npm test' }, + { command: "whoami" }, + { code: "set -euo pipefail" }, + { command: "exit" }, + { command: "if [ -d /workspace ]; then pwd; fi" }, + { code: "while test -d /workspace; do pwd; done" }, + { command: 'for ((i=0; i<3; i++)); do echo "$i"; done' }, + { code: "function task { pwd; }" }, + { command: "source ./env" }, + { code: "command ls" }, + { command: "go test ./..." }, + { code: "cargo test" }, + { command: "sort /workspace/file" }, + { code: "wc -l file" }, + { command: "jq . file.json" }, + { code: "exec ls" }, + { command: "custom-tool --format=json" }, + { command: "ls > output" }, + { code: "ls>output" }, + { command: "ls >output" }, + { code: "ls >> output" }, + { command: "cat { + const { config, catalogRef, tools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute( + "code-call-shell-source", + args, + ), + ); + + expect(details.status).toBe("failed"); + expect(details.code).toBe("invalid_input"); + expect(details.error).toMatch(/JavaScript or TypeScript, not shell commands/); + expect(testing.activeRuns.size).toBe(0); + }); + + it.each([ + { code: "true;", value: null }, + { code: "false;", value: null }, + { code: "return true || false;", value: true }, + { code: "return -1;", value: -1 }, + { code: "return /foo/.test('foo');", value: true }, + { code: "Infinity -1; return 42;", value: 42 }, + { code: "eval; return typeof eval;", value: "function" }, + { code: "if (true) { return -1; }", value: -1 }, + { code: "for (let i = 0; i < 3; i++) { if (i === 2) { return i; } }", value: 2 }, + { code: "function task() { return 7; } return task();", value: 7 }, + { code: "// explain the guest program\nreturn 7;", value: 7 }, + { code: "const ls = 7; return ls;", value: 7 }, + { code: "const echo = (value) => value; return echo('hello');", value: "hello" }, + { code: "test instanceof Function; function test() {}", value: null }, + { code: "ls -1; function ls() {}", value: null }, + { code: "ls -1; function/**/ls() {}", value: null }, + { code: "ls > limit; function ls() {} var limit = 1;", value: null }, + { code: "echo `hello`; function echo(parts) { return parts[0]; }", value: null }, + { code: "pwd; var { pwd } = { pwd: 7 }; return pwd;", value: 7 }, + { code: "pwd; var [pwd] = [7]; return pwd;", value: 7 }, + { code: "pwd; for (var pwd of [7]) {} return pwd;", value: 7 }, + { code: "pwd; var other = 1, pwd = 7; return pwd;", value: 7 }, + { code: "pwd; function* pwd() { yield 7; } return pwd().next().value;", value: 7 }, + { code: "pwd; function/**/pwd() { return 7; } return pwd();", value: 7 }, + { code: "pwd; var/**/{ pwd } = { pwd: 7 }; return pwd;", value: 7 }, + { code: "node -version; function/**/node() {}; var version = 1;", value: null }, + ])( + "executes valid shell-like JavaScript without false rejection: %j", + async ({ code, value }) => { + const { config, catalogRef, tools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute( + "code-call-valid-shell-like-source", + { code }, + ), + ); + + expect(details.status).toBe("completed"); + expect(details.value).toBe(value); + }, + ); + + it("runs JavaScript through QuickJS-WASI and resumes nested tool calls with wait", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const ticket = pluginTool("fake_create_ticket", "Create a fake ticket"); + applyCodeModeCatalog({ + tools: [...codeModeTools, ticket], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const hits = await tools.search("ticket", { limit: 1 }); + const called = await tools.callValue(hits[0].id, { value: "ship" }); + text("created"); + return called; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + name: "fake_create_ticket", + input: { value: "ship" }, + }); + expect(details.output).toEqual([{ type: "text", text: "created" }]); + expect(details.telemetry).toMatchObject({ searchCount: 1, describeCount: 0, callCount: 1 }); + expect(ticket.execute).toHaveBeenCalledTimes(1); + }); + + it("returns structured values from named tools while preserving the raw call envelope", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const ticket = pluginTool("fake_create_ticket", "Create a fake ticket"); + applyCodeModeCatalog({ + tools: [...codeModeTools, ticket], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const id = "openclaw:fake-code-mode:fake_create_ticket"; + const input = { value: "ship" }; + return { + named: await tools.fake_create_ticket(input), + value: await tools.callValue(id, input), + envelope: await tools.call(id, input), + }; + `, + }); + + const expectedValue = { + name: "fake_create_ticket", + input: { value: "ship" }, + }; + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + named: expectedValue, + value: expectedValue, + envelope: { + tool: expect.objectContaining({ + id: "openclaw:fake-code-mode:fake_create_ticket", + name: "fake_create_ticket", + }), + result: expect.objectContaining({ details: expectedValue }), + }, + }); + expect(details.telemetry).toMatchObject({ callCount: 3 }); + expect(ticket.execute).toHaveBeenCalledTimes(3); + }); + + it("uses tools recovery guidance for guessed tool ids", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const writeTool = pluginTool("write", "Write a file to the workspace"); + applyCodeModeCatalog({ + tools: [...codeModeTools, writeTool], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + try { + await tools.call("file_write", { + path: "memory/2026-05-22.md", + content: "remember this", + }); + return "unexpected success"; + } catch (error) { + return error.message; + } + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toBe( + "Unknown tool id: file_write. Did you mean: write? Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", + ); + expect(writeTool.execute).not.toHaveBeenCalled(); + }); + + it("uses tools recovery guidance when no generic Code Mode suggestion matches", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: codeModeTools, + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + try { + await tools.call("missing_tool", {}); + return "unexpected success"; + } catch (error) { + return error.message; + } + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toBe( + "Unknown tool id: missing_tool. Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", + ); + }); + + it("does not load TypeScript for plain JavaScript code mode runs", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: "return 42;", + }); + + expect(details.status).toBe("completed"); + expect(details.value).toBe(42); + expect(testing.getTypescriptRuntimePromise()).toBeNull(); + }); + + it("allows identifiers and strings that contain import without module access", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const important = 41; + const message = "import docs later"; + return important + (message.includes("import") ? 1 : 0); + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toBe(42); + }); + + it.each([ + { + name: "template-literal import text", + code: "return `import('node:fs')`;", + value: "import('node:fs')", + }, + { + name: "template-literal require text", + code: "return `require('node:fs')`;", + value: "require('node:fs')", + }, + { + name: "nested template-literal module text", + code: "return `outer ${`require('node:fs')`}`;", + value: "outer require('node:fs')", + }, + { + name: "regular-expression module text", + code: 'return /import.meta/.test("import.meta");', + value: true, + }, + { + name: "regular-expression module text inside interpolation", + code: 'return `${/import.meta/.test("import.meta")}`;', + value: "true", + }, + { + name: "ordinary import method", + code: "const api = { import(value) { return value; } }; return api.import(42);", + value: 42, + }, + { + name: "ordinary require method", + code: "const api = { require(value) { return value; } }; return api.require(42);", + value: 42, + }, + { + name: "optional ordinary import method", + code: "const api = { import(value) { return value; } }; return api?.import?.(42);", + value: 42, + }, + { + name: "computed ordinary require method", + code: 'const api = { require(value) { return value; } }; return api["require"](42);', + value: 42, + }, + { + name: "ordinary import metadata property", + code: "const api = { import: { meta: 42 } }; return api.import.meta;", + value: 42, + }, + ])("executes harmless $name in the real guest worker", async ({ code, value }) => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code, + }); + + expect(details).toMatchObject({ status: "completed", value }); + expect(testing.activeRuns.size).toBe(0); + }); + + it("never exposes Node module-loader globals to the real guest worker", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: "return [typeof process, typeof module, typeof require];", + }); + + expect(details).toMatchObject({ + status: "completed", + value: ["undefined", "undefined", "undefined"], + }); + expect(testing.activeRuns.size).toBe(0); + }); + + it("isolates and cleans up 12 concurrent real guest workers", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + const execTool = expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"); + + const results = await Promise.all( + Array.from({ length: 12 }, async (_, index) => + resultDetails( + await execTool.execute(`code-call-concurrent-worker-${index}`, { + code: `return { index: ${index}, message: \`require('node:fs')\` };`, + }), + ), + ), + ); + + expect(results).toEqual( + Array.from({ length: 12 }, (_, index) => + expect.objectContaining({ + status: "completed", + value: { index, message: "require('node:fs')" }, + }), + ), + ); + expect(testing.activeRuns.size).toBe(0); + expect(testing.resumingRunIds.size).toBe(0); + }); + + it("fails pending promises that have no host bridge work", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const beforeRunCount = testing.activeRuns.size; + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-empty-wait", + { + code: "await new Promise(() => undefined); return 'never';", + }, + ), + ); + + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("pending without host work"); + expect(testing.activeRuns.size).toBe(beforeRunCount); + }); + + it("surfaces the QuickJS error name and message for guest syntax errors", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-syntax", + { code: "const x = ;" }, + ), + ); + + expect(details.status).toBe("failed"); + const error = String(details.error); + // Regression guard: QuickJS stacks are frames only, so the error used to + // collapse to a bare "at openclaw-code-mode:user.js:..." location with the + // actual cause dropped. The model now sees the name and message. + expect(error).toContain("SyntaxError"); + expect(error).toContain("unexpected token"); + expect(error.startsWith("at ")).toBe(false); + }); + + it("surfaces the QuickJS error name and message for guest runtime errors", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-runtime", + { code: "return missingFn();" }, + ), + ); + + expect(details.status).toBe("failed"); + const error = String(details.error); + expect(error).toContain("ReferenceError"); + expect(error).toContain("missingFn is not defined"); + expect(error.startsWith("at ")).toBe(false); + }); + + it("does not expose the raw host request callback", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-hidden-host-request", + { code: "return typeof globalThis.__openclawHostRequest;" }, + ), + ); + + expect(details).toMatchObject({ + status: "completed", + value: "undefined", + }); + }); + + it("clamps omitted code-mode catalog search limits to maxSearchLimit", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + maxSearchLimit: 3, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const codeModeTools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [ + ...codeModeTools, + pluginTool("fake_ticket_one", "ticket helper"), + pluginTool("fake_ticket_two", "ticket helper"), + pluginTool("fake_ticket_three", "ticket helper"), + pluginTool("fake_ticket_four", "ticket helper"), + pluginTool("fake_ticket_five", "ticket helper"), + ], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: 'const hits = await tools.search("ticket"); return hits.length;', + }); + + expect(details.status).toBe("completed"); + expect(details.value).toBe(3); + }); + + it.each([ + "const fs = require('node:fs'); return fs;", + String.raw`return r\u0065quire('node:fs');`, + "return require?.('node:fs');", + "return (require)('node:fs');", + "return (0, require)('node:fs');", + "const load = require; return load('node:fs');", + "return module.require('node:fs');", + "return process.getBuiltinModule('node:fs');", + "return import('node:fs');", + "return import.meta.url;", + "return `${import('node:fs')}`;", + "return `${require('node:fs')}`;", + "return `${`nested ${import('node:fs')}`}`;", + "return `${`nested ${require('node:fs')}`}`;", + "return `${({ value: import('node:fs') }).value}`;", + "const message = `import('node:fs')`; return require('node:fs');", + "const pattern = /import.meta/; return import('node:fs');", + "let value = 1; return value++ / import('node:fs');", + "let value = 1; return value-- / import('node:fs');", + "const value = { of: 1 }; return value.of / import('node:fs');", + "const value = { return: 1 }; return value.return / import('node:fs');", + "const value = { if() { return 1; } }; return value.if() / import('node:fs');", + "const value = { return: 1 }; return value?.return / import('node:fs') / 1;", + "const value = { return: 1 }; return value?.return / require('node:fs') / 1;", + "const value = { if() { return 1; } }; return value?.if() / import('node:fs');", + "function run() { const await = 1; return await / (globalThis.pending = import('node:fs')); } run(); return globalThis.pending;", + "class Guest { #return = 1; run() { return this.#return / (globalThis.pending = import('node:fs')); } } new Guest().run(); return globalThis.pending;", + ])("rejects module access: %s", async (code) => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-import", + { + code, + }, + ), + ); + + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("module access is disabled"); + }); +}); diff --git a/src/agents/code-mode.limits.test.ts b/src/agents/code-mode.limits.test.ts new file mode 100644 index 000000000000..695bc926ac27 --- /dev/null +++ b/src/agents/code-mode.limits.test.ts @@ -0,0 +1,413 @@ +/** Tests Code Mode runtime and output limits. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { applyCodeModeCatalog, createCodeModeTools, resolveCodeModeConfig } from "./code-mode.js"; +import { + resetCodeModeTestState, + pluginTool, + mcpTool, + resultDetails, + createCodeModeHarness, + testing, +} from "./code-mode.test-support.js"; +import { createToolSearchCatalogRef } from "./tool-search.js"; +import { jsonResult } from "./tools/common.js"; + +describe("Code Mode runtime and output limits", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("enforces output limits on completed exec calls", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + maxOutputBytes: 1024, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const tools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-large", { + code: "return 'x'.repeat(2048);", + }), + ); + + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("output limit exceeded"); + expect(details.code).toBe("output_limit_exceeded"); + }); + + it("enforces output limits before suspending runs", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + maxOutputBytes: 1024, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const tools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const beforeRunCount = testing.activeRuns.size; + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-large-suspend", { + code: "text('x'.repeat(2048)); await yield_control('pause'); return 1;", + }), + ); + + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("output limit exceeded"); + expect(details.code).toBe("output_limit_exceeded"); + expect(testing.activeRuns.size).toBe(beforeRunCount); + }); + + it("enforces the cumulative output limit across yielded waits", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + maxOutputBytes: 1024, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const tools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(tools[0], "Code Mode exec test invariant").execute( + "code-call-cumulative-output", + { + code: ` + text("a".repeat(600)); + await yield_control("pause"); + text("b".repeat(600)); + return "done"; + `, + }, + ), + ); + + expect(first.status).toBe("waiting"); + expect(first.output).toEqual([{ type: "text", text: "a".repeat(600) }]); + + const second = resultDetails( + await expectDefined(tools[1], "Code Mode wait test invariant").execute( + "code-wait-cumulative-output", + { runId: first.runId }, + ), + ); + + expect(second.status).toBe("failed"); + expect(second.code).toBe("output_limit_exceeded"); + expect(testing.activeRuns.has(first.runId as string)).toBe(false); + }); + + it("enforces output limits before auto-draining namespace calls", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + maxOutputBytes: 1024, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const tools = createCodeModeTools(ctx); + const executeListIssues = vi.fn(async () => jsonResult({ ok: true })); + const listIssues = mcpTool({ + name: "tickets__list", + serverName: "tickets", + toolName: "list", + execute: executeListIssues, + }); + applyCodeModeCatalog({ + tools: [...tools, listIssues], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute( + "code-call-large-namespace", + { + code: 'text("x".repeat(2048)); await MCP.tickets.list({ state: "open" }); return 1;', + }, + ), + ); + + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("output limit exceeded"); + expect(details.code).toBe("output_limit_exceeded"); + expect(executeListIssues).not.toHaveBeenCalled(); + }); + + it("preserves guest output when a run fails", async () => { + const { config, catalogRef, tools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute( + "code-call-output-before-error", + { + code: 'text("before"); throw new Error("boom");', + }, + ), + ); + + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("Error: boom"); + expect(details.output).toEqual([{ type: "text", text: "before" }]); + }); + + it("classifies snapshot limit failures", async () => { + const config = resolveCodeModeConfig({ + tools: { codeMode: { enabled: true, maxSnapshotBytes: 1024 } }, + } as never); + + const result = await testing.runCodeModeWorker( + { + kind: "exec", + source: 'const value = "x".repeat(100000); await yield_control("pause"); return value;', + config, + catalog: [], + }, + 5000, + ); + + expect(result.status).toBe("failed"); + expect(result).toMatchObject({ + code: "snapshot_limit_exceeded", + error: "code mode snapshot limit exceeded", + }); + }); + + it("terminates hostile infinite loops outside the main event loop", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + timeoutMs: 100, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const tools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [...tools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const heartbeat = Promise.resolve("main-event-loop-alive"); + const details = resultDetails( + await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-loop", { + code: "while (true) {}", + }), + ); + + await expect(heartbeat).resolves.toBe("main-event-loop-alive"); + expect(details.status).toBe("failed"); + expect(String(details.error)).toContain("timeout exceeded"); + expect(details.code).toBe("timeout"); + }); + + it("normalizes QuickJS interrupt timeout errors", () => { + expect( + testing.normalizeCodeModeWorkerResult({ + status: "failed", + code: "timeout", + error: "interrupted", + output: [], + }), + ).toMatchObject({ + code: "timeout", + error: "code mode timeout exceeded", + }); + + expect( + testing.normalizeCodeModeWorkerResult({ + status: "failed", + code: "internal_error", + error: "interrupted", + output: [], + }), + ).toMatchObject({ + code: "internal_error", + error: "interrupted", + }); + }); + + it("classifies missing worker runtime as unavailable", async () => { + const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); + const missingWorkerUrl = new URL("./missing-code-mode.worker.js", import.meta.url); + + const result = await testing.runCodeModeWorker( + { + kind: "exec", + source: "return 1;", + config, + catalog: [], + }, + 500, + missingWorkerUrl, + ); + + expect(result.status).toBe("failed"); + expect(result).toMatchObject({ + code: "runtime_unavailable", + }); + }); + + it("classifies nonzero worker exits as unavailable", async () => { + const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); + const exitingWorkerUrl = new URL("data:text/javascript,process.exit(1)"); + + const result = await testing.runCodeModeWorker( + { + kind: "exec", + source: "return 1;", + config, + catalog: [], + }, + 500, + exitingWorkerUrl, + ); + + expect(result.status).toBe("failed"); + expect(result).toMatchObject({ + code: "runtime_unavailable", + }); + }); + + it("classifies clean worker exits without a result as unavailable", async () => { + const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); + const exitingWorkerUrl = new URL("data:text/javascript,"); + + const result = await testing.runCodeModeWorker( + { + kind: "exec", + source: "return 1;", + config, + catalog: [], + }, + 5_000, + exitingWorkerUrl, + ); + + expect(result).toMatchObject({ + status: "failed", + code: "runtime_unavailable", + error: "code mode worker exited with code 0 before returning a result", + }); + }); + + it("does not classify guest interrupted errors as timeouts", async () => { + const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); + + const result = await testing.runCodeModeWorker( + { + kind: "exec", + source: 'throw new Error("interrupted");', + config, + catalog: [], + }, + 10_000, + ); + + expect(result.status).toBe("failed"); + // A guest error whose message happens to be "interrupted" must stay + // internal_error and not be misclassified as a QuickJS interrupt/timeout. + expect(result).toMatchObject({ code: "internal_error" }); + if (result.status === "failed") { + expect(result.error).toContain("interrupted"); + } + }); +}); diff --git a/src/agents/code-mode.mcp.test.ts b/src/agents/code-mode.mcp.test.ts new file mode 100644 index 000000000000..94590231c6b2 --- /dev/null +++ b/src/agents/code-mode.mcp.test.ts @@ -0,0 +1,372 @@ +/** Tests Code Mode MCP namespace. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { applyCodeModeCatalog } from "./code-mode.js"; +import { + resetCodeModeTestState, + mcpTool, + createCodeModeHarness, + runUntilCompleted, +} from "./code-mode.test-support.js"; + +describe("Code Mode MCP namespace", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("exposes MCP tools only through the MCP namespace", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const githubCreate = mcpTool({ + name: "github__create_issue", + serverName: "github", + toolName: "create_issue", + parameters: { + type: "object", + properties: { + owner: { type: "string" }, + repo: { type: "string", description: "Repository 名称" }, + title: { type: "string", description: "Issue title\nShown in tracker" }, + body: { type: "string", default: "" }, + }, + required: ["owner", "repo", "title"], + }, + }); + const compacted = applyCodeModeCatalog({ + tools: [...codeModeTools, githubCreate], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + expect(compacted.tools[0]?.description).toContain("MCP: MCP server tools grouped by server."); + expect(compacted.tools[0]?.description).toContain("visible servers: github"); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const rootApi = await MCP.$api(); + const api = await MCP.github.$api("createIssue", { schema: true }); + const apiFiles = await API.list("mcp"); + const apiFilesTrailingSlash = await API.list("mcp/"); + const rootFile = await API.read("mcp/index.d.ts"); + const serverFile = await API.read("mcp/github.d.ts"); + const created = await MCP.github.createIssue({ + owner: "openclaw", + repo: "openclaw", + title: "Ship it", + }); + const createdPayload = JSON.parse(created.content[0].text); + const searchHits = await tools.search("github create issue", { limit: 5 }); + const allHasMcp = ALL_TOOLS.some((tool) => tool.source === "mcp"); + let directCall; + let directDescribe; + try { + await tools.describe("github__create_issue"); + directDescribe = "unexpected"; + } catch (error) { + directDescribe = error.message; + } + try { + await tools.call("github__create_issue", { owner: "x", repo: "y", title: "blocked" }); + directCall = "unexpected"; + } catch (error) { + directCall = error.message; + } + return { + apiHeader: api.header, + apiFilePaths: apiFiles.files.map((file) => file.path), + apiFilePathsTrailingSlash: apiFilesTrailingSlash.files.map((file) => file.path), + listedServerFileBytes: apiFiles.files.find((file) => file.path === "mcp/github.d.ts").bytes, + serverFileBytes: serverFile.bytes, + serverFileContent: serverFile.content, + rootFileHasReference: rootFile.content.includes('./github.d.ts'), + serverFileHasCreateIssue: serverFile.content.includes('function createIssue('), + serverFileHasTitleDoc: serverFile.content.includes('@param title Issue title Shown in tracker'), + apiSchemaTitle: api.schemas.createIssue.type, + rootServers: rootApi.servers, + createdPayload, + createdDetails: created.details, + searchHits, + allHasMcp, + directDescribe, + directCall, + hasMcp: "MCP" in namespaces, + }; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + createdPayload: { + serverName: "github", + toolName: "create_issue", + input: { + owner: "openclaw", + repo: "openclaw", + title: "Ship it", + body: "", + }, + }, + createdDetails: { + serverName: "github", + toolName: "create_issue", + input: { + owner: "openclaw", + repo: "openclaw", + title: "Ship it", + body: "", + }, + }, + searchHits: [], + allHasMcp: false, + directDescribe: + "Unknown tool id: github__create_issue. Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", + directCall: + "Unknown tool id: github__create_issue. Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", + hasMcp: true, + apiSchemaTitle: "object", + apiHeader: expect.stringContaining("function createIssue("), + apiFilePaths: ["mcp/index.d.ts", "mcp/github.d.ts"], + apiFilePathsTrailingSlash: ["mcp/index.d.ts", "mcp/github.d.ts"], + listedServerFileBytes: expect.any(Number), + serverFileBytes: expect.any(Number), + serverFileContent: expect.stringContaining("Repository 名称"), + rootFileHasReference: true, + serverFileHasCreateIssue: true, + serverFileHasTitleDoc: true, + rootServers: [{ identifier: "github", serverName: "github", toolCount: 1 }], + }); + const value = details.value as { + apiHeader: string; + listedServerFileBytes: number; + serverFileBytes: number; + serverFileContent: string; + }; + expect(value.listedServerFileBytes).toBe(value.serverFileBytes); + expect(value.serverFileBytes).toBe(Buffer.byteLength(value.serverFileContent, "utf8")); + expect(value.serverFileBytes).toBeGreaterThan(value.serverFileContent.length); + expect(value.apiHeader).toContain("@param title Issue title Shown in tracker"); + expect(value.apiHeader).not.toContain("@param title Issue title\n"); + expect(value.apiHeader).toContain("title: string;"); + expect(githubCreate.execute).toHaveBeenCalledTimes(1); + }); + + it("lets agents inspect MCP declaration files before calling MCP tools", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const githubCreate = mcpTool({ + name: "github__create_issue", + serverName: "github", + toolName: "create_issue", + parameters: { + type: "object", + properties: { + owner: { type: "string" }, + repo: { type: "string" }, + title: { type: "string", description: "Issue title" }, + }, + required: ["owner", "repo", "title"], + }, + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, githubCreate], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const files = await API.list("mcp"); + const api = await API.read("mcp/github.d.ts"); + const created = await MCP.github.createIssue({ + owner: "openclaw", + repo: "openclaw", + title: "From file docs", + }); + return { + fileCount: files.files.length, + headerHasSignature: api.content.includes("function createIssue("), + usedApiCall: api.content.includes("function $api("), + created: JSON.parse(created.content[0].text), + }; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + fileCount: 2, + headerHasSignature: true, + usedApiCall: true, + created: { + serverName: "github", + toolName: "create_issue", + input: { + owner: "openclaw", + repo: "openclaw", + title: "From file docs", + }, + }, + }); + expect(githubCreate.execute).toHaveBeenCalledTimes(1); + }); + + it("groups MCP resources and prompts under server namespaces", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const resourceRead = mcpTool({ + name: "docs__resources_read", + serverName: "docs", + toolName: "resources_read", + operation: "resources_read", + parameters: { + type: "object", + properties: { uri: { type: "string" } }, + required: ["uri"], + }, + }); + const promptGet = mcpTool({ + name: "docs__prompts_get", + serverName: "docs", + toolName: "prompts_get", + operation: "prompts_get", + parameters: { + type: "object", + properties: { + name: { type: "string" }, + arguments: { type: "object" }, + }, + required: ["name"], + }, + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, resourceRead, promptGet], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const api = await MCP.docs.$api(); + const resource = await MCP.docs.resources.read({ uri: "memo://one" }); + const prompt = await MCP.docs.prompts.get({ name: "brief", arguments: { topic: "mcp" } }); + return { header: api.header, resource: resource.details, prompt: prompt.details }; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + resource: { + serverName: "docs", + toolName: "resources_read", + input: { uri: "memo://one" }, + }, + prompt: { + serverName: "docs", + toolName: "prompts_get", + input: { name: "brief", arguments: { topic: "mcp" } }, + }, + header: expect.stringContaining("namespace resources"), + }); + }); + + it("renames MCP namespace identifiers that would be unsafe path segments", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const dangerous = mcpTool({ + name: "constructor__prototype", + serverName: "constructor", + toolName: "prototype", + parameters: { + type: "object", + properties: { value: { type: "string" } }, + required: ["value"], + }, + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, dangerous], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: 'return (await MCP.constructor2.prototype2({ value: "safe" })).details;', + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + serverName: "constructor", + toolName: "prototype", + input: { value: "safe" }, + }); + }); + + it.each(["delete", "default", "return", "enum", "class"])( + "renders and executes reserved MCP tool name %s safely", + async (toolName) => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const target = mcpTool({ + name: `github__${toolName}`, + serverName: "github", + toolName, + parameters: { + type: "object", + properties: { value: { type: "string" } }, + required: ["value"], + }, + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, target], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const safeName = `${toolName}2`; + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "Code Mode exec test invariant"), + waitTool: expectDefined(codeModeTools[1], "Code Mode wait test invariant"), + code: ` + const file = await API.read("mcp/github.d.ts"); + const api = await MCP.github.$api("${safeName}"); + const result = await MCP.github.${safeName}({ value: "safe" }); + return { file: file.content, header: api.header, result: result.details }; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + file: expect.stringContaining(`function ${safeName}(`), + header: expect.stringContaining(`function ${safeName}(`), + result: { + serverName: "github", + toolName, + input: { value: "safe" }, + }, + }); + expect(target.execute).toHaveBeenCalledTimes(1); + }, + ); +}); diff --git a/src/agents/code-mode.replay.test.ts b/src/agents/code-mode.replay.test.ts new file mode 100644 index 000000000000..eab788bedd4a --- /dev/null +++ b/src/agents/code-mode.replay.test.ts @@ -0,0 +1,227 @@ +/** Tests Code Mode restart-safe replay. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { setPluginToolMeta } from "../plugins/tools.js"; +import { applyCodeModeCatalog } from "./code-mode.js"; +import { + resetCodeModeTestState, + fakeTool, + pluginTool, + mcpTool, + resultDetails, + createCodeModeHarness, + runUntilCompleted, +} from "./code-mode.test-support.js"; + +describe("Code Mode restart-safe replay", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("keeps restart-safe mode across audited core reads", async () => { + const targetTool = fakeTool("read", "Read"); + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, targetTool], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-replay-safety", + { + restartSafe: true, + code: ` + const matches = await tools.search(${JSON.stringify(targetTool.name)}); + return await tools.call(matches[0].id, {}); + `, + }, + ), + ); + expect(first.status).toBe("waiting"); + expect(first.replaySafe).toBe(true); + + const second = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-replay-safety", + { runId: first.runId }, + ), + ); + expect(second.status).toBe("waiting"); + expect(second.replaySafe).toBe(true); + + const completed = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-replay-safety-complete", + { + runId: second.runId, + }, + ), + ); + expect(completed.status).toBe("completed"); + }); + + it("allows explicitly replay-safe plugin tools by exact catalog id", async () => { + const targetTool = pluginTool("fake_plugin_read", "Plugin read"); + setPluginToolMeta(targetTool, { + pluginId: "fake-code-mode", + optional: true, + replaySafe: true, + }); + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, targetTool], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const completed = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + restartSafe: true, + code: ` + const matches = await tools.search("fake_plugin_read"); + return await tools.call(matches[0].id, {}); + `, + }); + + expect(completed.status).toBe("completed"); + expect(completed.replaySafe).toBe(true); + expect(targetTool.execute).toHaveBeenCalledTimes(1); + }); + + it("rejects MCP tools even when their metadata claims replay safety", async () => { + const targetTool = mcpTool({ + name: "mcp_github_read_file", + serverName: "github", + toolName: "read_file", + }); + setPluginToolMeta(targetTool, { + pluginId: "bundle-mcp", + optional: false, + replaySafe: true, + mcp: { + serverName: "github", + safeServerName: "github", + toolName: "read_file", + operation: "tool", + }, + }); + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, targetTool], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const completed = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + restartSafe: true, + code: 'return await MCP.github.readFile({ path: "README.md" });', + }); + + expect(completed.status).toBe("failed"); + expect(completed.replaySafe).toBe(true); + expect(completed.error).toContain("cannot call namespace tools"); + expect(targetTool.execute).not.toHaveBeenCalled(); + }); + + it("rejects side-effecting calls before executing them in restart-safe mode", async () => { + const targetTool = pluginTool("fake_write", "Write"); + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, targetTool], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-unsafe-restart", + { + restartSafe: true, + code: ` + const matches = await tools.search("fake_write"); + return await tools.call(matches[0].id, {}); + `, + }, + ), + ); + expect(first.status).toBe("waiting"); + expect(first.replaySafe).toBe(true); + + const failed = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-unsafe-restart", + { runId: first.runId }, + ), + ); + expect(failed.status).toBe("failed"); + expect(failed.error).toContain("cannot call side-effecting tools"); + expect(targetTool.execute).not.toHaveBeenCalled(); + }); + + it("keeps host-forced restart safety when the model clears the exec flag", async () => { + const targetTool = pluginTool("fake_forced_write", "Write"); + const { + config, + catalogRef, + tools: codeModeTools, + } = createCodeModeHarness({ + forceRestartSafeTools: true, + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, targetTool], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-forced-restart", + { + restartSafe: false, + code: ` + const matches = await tools.search("fake_forced_write"); + return await tools.call(matches[0].id, {}); + `, + }, + ), + ); + expect(first.status).toBe("waiting"); + expect(first.replaySafe).toBe(true); + + const failed = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-forced-restart", + { runId: first.runId }, + ), + ); + expect(failed.status).toBe("failed"); + expect(failed.error).toContain("cannot call side-effecting tools"); + expect(targetTool.execute).not.toHaveBeenCalled(); + }); +}); diff --git a/src/agents/code-mode.skills.test.ts b/src/agents/code-mode.skills.test.ts new file mode 100644 index 000000000000..a0b80e3016fd --- /dev/null +++ b/src/agents/code-mode.skills.test.ts @@ -0,0 +1,178 @@ +/** Tests Code Mode skills and read tools. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import type { Skill } from "../skills/loading/skill-contract.js"; +import { resolveSkillsPromptForRun } from "../skills/loading/workspace.js"; +import { createFixtureSkillEntry } from "../skills/test-support/test-helpers.js"; +import { createOpenClawReadTool } from "./agent-tools.read.js"; +import { resolveCodeModeSkills } from "./code-mode-skills.js"; +import { applyCodeModeCatalog } from "./code-mode.js"; +import { + resetCodeModeTestState, + pluginTool, + createCodeModeHarness, + runUntilCompleted, +} from "./code-mode.test-support.js"; +import { createReadTool } from "./sessions/index.js"; + +function skillCandidate(params: { + name: string; + description: string; + filePath: string; + readContent?: string; +}): Skill { + return { + ...params, + baseDir: params.filePath.replace(/\/[^/]+$/u, ""), + sourceInfo: { + path: params.filePath, + source: "test", + scope: "temporary", + origin: "top-level", + }, + disableModelInvocation: false, + source: "test", + }; +} + +describe("Code Mode skills and read tools", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("keeps Code Mode skill parsing aligned with the production prompt renderer", () => { + const entries = [createFixtureSkillEntry("alpha"), createFixtureSkillEntry("beta")]; + const skillsPrompt = resolveSkillsPromptForRun({ + entries, + workspaceDir: "/workspace", + }); + + expect( + resolveCodeModeSkills({ + skillsPrompt, + candidates: entries.map((entry) => entry.skill), + }).map(({ name, location }) => ({ name, location })), + ).toEqual([ + { name: "alpha", location: "/skills/alpha/SKILL.md" }, + { name: "beta", location: "/skills/beta/SKILL.md" }, + ]); + }); + + it("lists and reads only prompt-eligible skills through the worker bridge", async () => { + const demo = skillCandidate({ + name: "demo", + description: "Full demo description", + filePath: "/host/skills/demo/SKILL.md", + }); + const hidden = skillCandidate({ + name: "hidden", + description: "Hidden skill", + filePath: "/host/skills/hidden/SKILL.md", + }); + const reader = vi.fn(async ({ location }: { location: string }) => + location === "/guest/skills/demo/SKILL.md" + ? "---\nname: demo\n---\n\n# Complete demo instructions\n" + : "# Hidden\n", + ); + const codeModeSkills = resolveCodeModeSkills({ + skillsPrompt: [ + "", + " ", + " demo", + " Short prompt description", + " /guest/skills/demo/SKILL.md", + " ", + "", + ].join("\n"), + candidates: [demo, hidden], + reader, + }); + const { + config, + catalogRef, + tools: codeModeTools, + } = createCodeModeHarness({ + codeModeSkills, + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + codeModeSkills, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: ` + const listed = await skills.list(); + const body = await skills.read("demo"); + let unknown; + try { + await skills.read("missing"); + } catch (error) { + unknown = error.message; + } + return { listed, body, unknown }; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ + listed: [ + { + name: "demo", + description: "Full demo description", + location: "/guest/skills/demo/SKILL.md", + }, + ], + body: "---\nname: demo\n---\n\n# Complete demo instructions\n", + unknown: 'Unknown skill "missing". Available skills: demo', + }); + expect(codeModeTools[0]?.description).toContain("`await skills.read(name)`"); + expect(reader).toHaveBeenCalledOnce(); + expect(reader).toHaveBeenCalledWith({ + location: "/guest/skills/demo/SKILL.md", + signal: expect.any(AbortSignal), + }); + }); + + it("returns ordinary read content through tools.callValue", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const read = createOpenClawReadTool( + createReadTool("/workspace", { + operations: { + access: async () => {}, + detectImageMimeType: async () => null, + readFile: async () => Buffer.from("ordinary file content"), + }, + }) as unknown as Parameters[0], + ); + applyCodeModeCatalog({ + tools: [...codeModeTools, read], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + code: `return await tools.callValue("openclaw:core:read", { path: "notes.txt" });`, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ kind: "text", content: "ordinary file content" }); + }); +}); diff --git a/src/agents/code-mode.test-support.ts b/src/agents/code-mode.test-support.ts index 2570abf6947d..4f374f47d4ab 100644 --- a/src/agents/code-mode.test-support.ts +++ b/src/agents/code-mode.test-support.ts @@ -1,5 +1,13 @@ -import "./code-mode.js"; -import type { ToolSearchToolContext } from "./tool-search.js"; +import { expect, vi } from "vitest"; +import { setPluginToolMeta } from "../plugins/tools.js"; +import type { CodeModeSkill } from "./code-mode-skills.js"; +import { createCodeModeTools } from "./code-mode.js"; +import { + createToolSearchCatalogRef, + type ToolSearchCatalogRef, + type ToolSearchToolContext, +} from "./tool-search.js"; +import { jsonResult, type AnyAgentTool } from "./tools/common.js"; type CodeModeConfig = { enabled: boolean; @@ -111,3 +119,145 @@ function getTestApi(): CodeModeTestApi { } export const testing = getTestApi(); + +export function resetCodeModeTestState(): void { + testing.activeRuns.clear(); + testing.resumingRunIds.clear(); + testing.setTypescriptRuntimeForTest(null); +} + +export function fakeTool(name: string, description: string): AnyAgentTool { + // Minimal tool shape keeps Code Mode catalog tests runtime-free. + return { + name, + label: name, + description, + parameters: { + type: "object", + properties: { + value: { type: "string" }, + }, + }, + execute: vi.fn(async (_toolCallId, input) => jsonResult({ name, input })), + }; +} + +export function pluginTool( + name: string, + description: string, + pluginId = "fake-code-mode", +): AnyAgentTool { + const tool = fakeTool(name, description); + setPluginToolMeta(tool, { + pluginId, + optional: true, + }); + return tool; +} + +export function pluginToolWithExecute( + name: string, + description: string, + execute: AnyAgentTool["execute"], +): AnyAgentTool { + const tool = pluginTool(name, description); + tool.execute = vi.fn(execute) as AnyAgentTool["execute"]; + return tool; +} + +export function mcpTool(params: { + name: string; + serverName: string; + safeServerName?: string; + toolName: string; + description?: string; + parameters?: AnyAgentTool["parameters"]; + operation?: "tool" | "resources_list" | "resources_read" | "prompts_list" | "prompts_get"; + execute?: AnyAgentTool["execute"]; +}): AnyAgentTool { + // MCP metadata drives Code Mode grouping and raw tool routing. + const tool: AnyAgentTool = { + name: params.name, + label: params.toolName, + description: params.description ?? `MCP ${params.toolName}`, + parameters: params.parameters ?? { + type: "object", + properties: {}, + }, + execute: + params.execute ?? + vi.fn(async (_toolCallId, input) => + jsonResult({ + serverName: params.serverName, + toolName: params.toolName, + input, + }), + ), + }; + setPluginToolMeta(tool, { + pluginId: "bundle-mcp", + optional: false, + mcp: { + serverName: params.serverName, + safeServerName: params.safeServerName ?? params.serverName, + toolName: params.toolName, + operation: params.operation ?? "tool", + }, + }); + return tool; +} + +export function resultDetails(result: { details?: unknown }): Record { + expect(result.details).toBeDefined(); + expect(typeof result.details).toBe("object"); + return result.details as Record; +} + +export function createCodeModeHarness( + params: { + agentId?: string; + catalogRef?: ToolSearchCatalogRef; + codeModeSkills?: readonly CodeModeSkill[]; + forceRestartSafeTools?: boolean; + } = {}, +) { + const catalogRef = params.catalogRef ?? createToolSearchCatalogRef(); + const config = { tools: { codeMode: true } } as never; + const ctx = { + config, + runtimeConfig: config, + ...(params.agentId ? { agentId: params.agentId } : {}), + sessionId: "session-code-mode", + sessionKey: params.agentId ? `agent:${params.agentId}:main` : "agent:main:main", + runId: "run-code-mode", + catalogRef, + forceRestartSafeTools: params.forceRestartSafeTools, + codeModeSkills: params.codeModeSkills, + }; + const tools = createCodeModeTools(ctx); + return { catalogRef, config, ctx, tools }; +} + +export async function runUntilCompleted(params: { + execTool: AnyAgentTool; + waitTool: AnyAgentTool; + code: string; + language?: "javascript" | "typescript"; + restartSafe?: boolean; +}) { + // Code Mode may return a waiting state before completion; tests poll through + // the public wait tool instead of reaching into activeRuns. + let details = resultDetails( + await params.execTool.execute("code-call-1", { + code: params.code, + language: params.language, + restartSafe: params.restartSafe, + }), + ); + for (let index = 0; index < 8 && details.status === "waiting"; index += 1) { + const runId = details.runId; + expect(typeof runId).toBe("string"); + details = resultDetails(await params.waitTool.execute(`code-wait-${index}`, { runId })); + } + return details; +} diff --git a/src/agents/code-mode.test.ts b/src/agents/code-mode.test.ts index 58ab1bc585f9..69e4291bbb0d 100644 --- a/src/agents/code-mode.test.ts +++ b/src/agents/code-mode.test.ts @@ -1,265 +1,38 @@ -/** Tests Code Mode tool registration, namespace filtering, and run lifecycle. */ +/** Tests Code Mode catalog and model-visible surface. */ import { expectDefined } from "@openclaw/normalization-core"; import { Type } from "typebox"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; -import { runWithAgentToolExecutionContext } from "../../packages/agent-core/src/tool-execution-context.js"; -import { setPluginToolMeta } from "../plugins/tools.js"; -import type { Skill } from "../skills/loading/skill-contract.js"; -import { resolveSkillsPromptForRun } from "../skills/loading/workspace.js"; -import { createFixtureSkillEntry } from "../skills/test-support/test-helpers.js"; -import { buildBlockedToolResult } from "./agent-tools.before-tool-call.js"; -import { createOpenClawReadTool } from "./agent-tools.read.js"; -import { resolveCodeModeSkills, type CodeModeSkill } from "./code-mode-skills.js"; import { applyCodeModeCatalog, CODE_MODE_EXEC_TOOL_NAME, CODE_MODE_WAIT_TOOL_NAME, createCodeModeTools, - resolveCodeModeConfig, } from "./code-mode.js"; -import { testing } from "./code-mode.test-support.js"; -import { createReadTool } from "./sessions/index.js"; -import { createToolSearchCatalogRef, type ToolSearchCatalogRef } from "./tool-search.js"; import { + resetCodeModeTestState, + fakeTool, + pluginTool, + mcpTool, + createCodeModeHarness, + testing, +} from "./code-mode.test-support.js"; +import { + createToolSearchCatalogRef, TOOL_CALL_RAW_TOOL_NAME, TOOL_DESCRIBE_RAW_TOOL_NAME, TOOL_SEARCH_CODE_MODE_TOOL_NAME, TOOL_SEARCH_RAW_TOOL_NAME, } from "./tool-search.js"; -import { jsonResult, type AnyAgentTool } from "./tools/common.js"; -function skillCandidate(params: { - name: string; - description: string; - filePath: string; - readContent?: string; -}): Skill { - return { - ...params, - baseDir: params.filePath.replace(/\/[^/]+$/u, ""), - sourceInfo: { - path: params.filePath, - source: "test", - scope: "temporary", - origin: "top-level", - }, - disableModelInvocation: false, - source: "test", - }; -} - -function fakeTool(name: string, description: string): AnyAgentTool { - // Minimal tool shape keeps Code Mode catalog tests runtime-free. - return { - name, - label: name, - description, - parameters: { - type: "object", - properties: { - value: { type: "string" }, - }, - }, - execute: vi.fn(async (_toolCallId, input) => jsonResult({ name, input })), - }; -} - -function pluginTool(name: string, description: string, pluginId = "fake-code-mode"): AnyAgentTool { - const tool = fakeTool(name, description); - setPluginToolMeta(tool, { - pluginId, - optional: true, - }); - return tool; -} - -function pluginToolWithExecute( - name: string, - description: string, - execute: AnyAgentTool["execute"], -): AnyAgentTool { - const tool = pluginTool(name, description); - tool.execute = vi.fn(execute) as AnyAgentTool["execute"]; - return tool; -} - -function mcpTool(params: { - name: string; - serverName: string; - safeServerName?: string; - toolName: string; - description?: string; - parameters?: AnyAgentTool["parameters"]; - operation?: "tool" | "resources_list" | "resources_read" | "prompts_list" | "prompts_get"; - execute?: AnyAgentTool["execute"]; -}): AnyAgentTool { - // MCP metadata drives Code Mode grouping and raw tool routing. - const tool: AnyAgentTool = { - name: params.name, - label: params.toolName, - description: params.description ?? `MCP ${params.toolName}`, - parameters: params.parameters ?? { - type: "object", - properties: {}, - }, - execute: - params.execute ?? - vi.fn(async (_toolCallId, input) => - jsonResult({ - serverName: params.serverName, - toolName: params.toolName, - input, - }), - ), - }; - setPluginToolMeta(tool, { - pluginId: "bundle-mcp", - optional: false, - mcp: { - serverName: params.serverName, - safeServerName: params.safeServerName ?? params.serverName, - toolName: params.toolName, - operation: params.operation ?? "tool", - }, - }); - return tool; -} - -function resultDetails(result: { details?: unknown }): Record { - expect(result.details).toBeDefined(); - expect(typeof result.details).toBe("object"); - return result.details as Record; -} - -function createCodeModeHarness( - params: { - agentId?: string; - catalogRef?: ToolSearchCatalogRef; - codeModeSkills?: readonly CodeModeSkill[]; - forceRestartSafeTools?: boolean; - } = {}, -) { - const catalogRef = params.catalogRef ?? createToolSearchCatalogRef(); - const config = { tools: { codeMode: true } } as never; - const ctx = { - config, - runtimeConfig: config, - ...(params.agentId ? { agentId: params.agentId } : {}), - sessionId: "session-code-mode", - sessionKey: params.agentId ? `agent:${params.agentId}:main` : "agent:main:main", - runId: "run-code-mode", - catalogRef, - forceRestartSafeTools: params.forceRestartSafeTools, - codeModeSkills: params.codeModeSkills, - }; - const tools = createCodeModeTools(ctx); - return { catalogRef, config, ctx, tools }; -} - -async function runUntilCompleted(params: { - execTool: AnyAgentTool; - waitTool: AnyAgentTool; - code: string; - language?: "javascript" | "typescript"; - restartSafe?: boolean; -}) { - // Code Mode may return a waiting state before completion; tests poll through - // the public wait tool instead of reaching into activeRuns. - let details = resultDetails( - await params.execTool.execute("code-call-1", { - code: params.code, - language: params.language, - restartSafe: params.restartSafe, - }), - ); - for (let index = 0; index < 8 && details.status === "waiting"; index += 1) { - const runId = details.runId; - expect(typeof runId).toBe("string"); - details = resultDetails(await params.waitTool.execute(`code-wait-${index}`, { runId })); - } - return details; -} - -describe("Code Mode", () => { +describe("Code Mode catalog and model-visible surface", () => { beforeEach(() => { vi.useRealTimers(); }); afterEach(() => { vi.useRealTimers(); - testing.activeRuns.clear(); - testing.resumingRunIds.clear(); - testing.setTypescriptRuntimeForTest(null); - }); - - it("resolves object config defaults", () => { - expect(resolveCodeModeConfig({ tools: { codeMode: true } } as never).enabled).toBe(true); - const resolved = resolveCodeModeConfig({ - tools: { - codeMode: { - timeoutMs: 1234, - languages: ["typescript"], - }, - }, - } as never); - expect(resolved.enabled).toBe("auto"); - expect(resolveCodeModeConfig({ tools: { codeMode: { enabled: true } } } as never).enabled).toBe( - true, - ); - expect(resolved.runtime).toBe("quickjs-wasi"); - expect(resolved.mode).toBe("only"); - expect(resolved.timeoutMs).toBe(1234); - expect(resolved.languages).toEqual(["typescript"]); - const limitedSearch = resolveCodeModeConfig({ - tools: { - codeMode: { - enabled: true, - maxSearchLimit: 3, - }, - }, - } as never); - expect(limitedSearch.searchDefaultLimit).toBe(3); - expect(limitedSearch.maxSearchLimit).toBe(3); - }); - - it("resolves active-agent code mode over the runtime default", () => { - const config = { - tools: { - codeMode: { - enabled: false, - timeoutMs: 1234, - searchDefaultLimit: 6, - }, - }, - agents: { - list: [ - { - id: "ops", - tools: { - codeMode: { - enabled: true, - searchDefaultLimit: 4, - }, - }, - }, - { - id: "chat", - tools: { - codeMode: false, - }, - }, - ], - }, - } as never; - - const ops = resolveCodeModeConfig(config, "ops"); - expect(ops.enabled).toBe(true); - expect(ops.timeoutMs).toBe(1234); - expect(ops.searchDefaultLimit).toBe(4); - - expect(resolveCodeModeConfig(config, "chat").enabled).toBe(false); - expect(resolveCodeModeConfig(config, "missing").enabled).toBe(false); + resetCodeModeTestState(); }); it("resolves the packaged worker URL from stable and hashed dist modules", () => { @@ -780,3061 +553,4 @@ describe("Code Mode", () => { ]); expect(compacted.catalogToolCount).toBe(1); }); - - it("accepts command as an exec-compatible code alias", async () => { - const { config, catalogRef, tools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - const result = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-command-alias", { - command: "return 7;", - }), - ); - - expect(result.status).toBe("completed"); - expect(result.value).toBe(7); - }); - - it("rejects divergent code and command aliases", async () => { - const { config, catalogRef, tools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - await expect( - expectDefined(tools[0], "tools[0] test invariant").execute("code-call-divergent-alias", { - code: "return 1;", - command: "return 2;", - }), - ).rejects.toThrow("code and command must match when both are provided"); - }); - - it("drains a nested combinator after its outer race wins", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - let nestedAborted = false; - const never = pluginToolWithExecute( - "fake_nested_race_never", - "Never-settling nested race helper", - async (_toolCallId, _input, signal) => { - await new Promise((resolve, reject) => { - const timer = setTimeout(resolve, 25); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - nestedAborted = true; - reject(new Error("aborted")); - }, - { once: true }, - ); - }); - return jsonResult({ winner: "nested" }); - }, - ); - const fast = pluginToolWithExecute( - "fake_nested_race_fast", - "Fast nested race helper", - async () => jsonResult({ winner: "fast" }), - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, never, fast], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-nested-combinator-race", - { - code: `return await Promise.race([ - Promise.all([tools.callValue("fake_nested_race_never", {})]), - tools.callValue("fake_nested_race_fast", {}), - ]);`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); - expect(never.execute).toHaveBeenCalledOnce(); - expect(fast.execute).toHaveBeenCalledOnce(); - expect(nestedAborted).toBe(false); - expect(testing.activeRuns.size).toBe(0); - }); - - it.each([ - { code: "ls -la /workspace/" }, - { code: "ls -1" }, - { command: "ls -la /workspace/" }, - { code: "pwd", command: "pwd" }, - { command: "pwd;" }, - { command: "pwd; // inspect the workspace" }, - { code: "# inspect the workspace\npwd" }, - { code: "#!/bin/sh\npwd" }, - { code: "pwd\nls -la /workspace" }, - { command: "pwd;ls -la /workspace" }, - { command: "/bin/ls /workspace/" }, - { command: "./gradlew test" }, - { code: ".\\gradlew.bat test" }, - { command: ".\\script.ps1" }, - { code: "C:\\workspace\\run.cmd /q" }, - { code: "/workspace/run.sh --verbose" }, - { command: "sh -c 'ls /workspace/'" }, - { command: "git status" }, - { command: 'git status; const note = "git";' }, - { command: "ls -1; const metadata = { ls: true };" }, - { code: "ls -1; const note = 'function ls';" }, - { command: "ls -1; let ls = 7;" }, - { command: "npm test" }, - { command: "NODE_ENV=test npm test" }, - { code: "NODE_ENV=test\nnpm test" }, - { code: "FOO=bar ./gradlew test" }, - { command: 'GREETING="hello world" npm test' }, - { command: "whoami" }, - { code: "set -euo pipefail" }, - { command: "exit" }, - { command: "if [ -d /workspace ]; then pwd; fi" }, - { code: "while test -d /workspace; do pwd; done" }, - { command: 'for ((i=0; i<3; i++)); do echo "$i"; done' }, - { code: "function task { pwd; }" }, - { command: "source ./env" }, - { code: "command ls" }, - { command: "go test ./..." }, - { code: "cargo test" }, - { command: "sort /workspace/file" }, - { code: "wc -l file" }, - { command: "jq . file.json" }, - { code: "exec ls" }, - { command: "custom-tool --format=json" }, - { command: "ls > output" }, - { code: "ls>output" }, - { command: "ls >output" }, - { code: "ls >> output" }, - { command: "cat { - const { config, catalogRef, tools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute( - "code-call-shell-source", - args, - ), - ); - - expect(details.status).toBe("failed"); - expect(details.code).toBe("invalid_input"); - expect(details.error).toMatch(/JavaScript or TypeScript, not shell commands/); - expect(testing.activeRuns.size).toBe(0); - }); - - it.each([ - { code: "true;", value: null }, - { code: "false;", value: null }, - { code: "return true || false;", value: true }, - { code: "return -1;", value: -1 }, - { code: "return /foo/.test('foo');", value: true }, - { code: "Infinity -1; return 42;", value: 42 }, - { code: "eval; return typeof eval;", value: "function" }, - { code: "if (true) { return -1; }", value: -1 }, - { code: "for (let i = 0; i < 3; i++) { if (i === 2) { return i; } }", value: 2 }, - { code: "function task() { return 7; } return task();", value: 7 }, - { code: "// explain the guest program\nreturn 7;", value: 7 }, - { code: "const ls = 7; return ls;", value: 7 }, - { code: "const echo = (value) => value; return echo('hello');", value: "hello" }, - { code: "test instanceof Function; function test() {}", value: null }, - { code: "ls -1; function ls() {}", value: null }, - { code: "ls -1; function/**/ls() {}", value: null }, - { code: "ls > limit; function ls() {} var limit = 1;", value: null }, - { code: "echo `hello`; function echo(parts) { return parts[0]; }", value: null }, - { code: "pwd; var { pwd } = { pwd: 7 }; return pwd;", value: 7 }, - { code: "pwd; var [pwd] = [7]; return pwd;", value: 7 }, - { code: "pwd; for (var pwd of [7]) {} return pwd;", value: 7 }, - { code: "pwd; var other = 1, pwd = 7; return pwd;", value: 7 }, - { code: "pwd; function* pwd() { yield 7; } return pwd().next().value;", value: 7 }, - { code: "pwd; function/**/pwd() { return 7; } return pwd();", value: 7 }, - { code: "pwd; var/**/{ pwd } = { pwd: 7 }; return pwd;", value: 7 }, - { code: "node -version; function/**/node() {}; var version = 1;", value: null }, - ])( - "executes valid shell-like JavaScript without false rejection: %j", - async ({ code, value }) => { - const { config, catalogRef, tools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute( - "code-call-valid-shell-like-source", - { code }, - ), - ); - - expect(details.status).toBe("completed"); - expect(details.value).toBe(value); - }, - ); - - it("runs JavaScript through QuickJS-WASI and resumes nested tool calls with wait", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const ticket = pluginTool("fake_create_ticket", "Create a fake ticket"); - applyCodeModeCatalog({ - tools: [...codeModeTools, ticket], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const hits = await tools.search("ticket", { limit: 1 }); - const called = await tools.callValue(hits[0].id, { value: "ship" }); - text("created"); - return called; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - name: "fake_create_ticket", - input: { value: "ship" }, - }); - expect(details.output).toEqual([{ type: "text", text: "created" }]); - expect(details.telemetry).toMatchObject({ searchCount: 1, describeCount: 0, callCount: 1 }); - expect(ticket.execute).toHaveBeenCalledTimes(1); - }); - - it("returns structured values from named tools while preserving the raw call envelope", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const ticket = pluginTool("fake_create_ticket", "Create a fake ticket"); - applyCodeModeCatalog({ - tools: [...codeModeTools, ticket], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const id = "openclaw:fake-code-mode:fake_create_ticket"; - const input = { value: "ship" }; - return { - named: await tools.fake_create_ticket(input), - value: await tools.callValue(id, input), - envelope: await tools.call(id, input), - }; - `, - }); - - const expectedValue = { - name: "fake_create_ticket", - input: { value: "ship" }, - }; - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - named: expectedValue, - value: expectedValue, - envelope: { - tool: expect.objectContaining({ - id: "openclaw:fake-code-mode:fake_create_ticket", - name: "fake_create_ticket", - }), - result: expect.objectContaining({ details: expectedValue }), - }, - }); - expect(details.telemetry).toMatchObject({ callCount: 3 }); - expect(ticket.execute).toHaveBeenCalledTimes(3); - }); - - it("keeps Code Mode skill parsing aligned with the production prompt renderer", () => { - const entries = [createFixtureSkillEntry("alpha"), createFixtureSkillEntry("beta")]; - const skillsPrompt = resolveSkillsPromptForRun({ - entries, - workspaceDir: "/workspace", - }); - - expect( - resolveCodeModeSkills({ - skillsPrompt, - candidates: entries.map((entry) => entry.skill), - }).map(({ name, location }) => ({ name, location })), - ).toEqual([ - { name: "alpha", location: "/skills/alpha/SKILL.md" }, - { name: "beta", location: "/skills/beta/SKILL.md" }, - ]); - }); - - it("lists and reads only prompt-eligible skills through the worker bridge", async () => { - const demo = skillCandidate({ - name: "demo", - description: "Full demo description", - filePath: "/host/skills/demo/SKILL.md", - }); - const hidden = skillCandidate({ - name: "hidden", - description: "Hidden skill", - filePath: "/host/skills/hidden/SKILL.md", - }); - const reader = vi.fn(async ({ location }: { location: string }) => - location === "/guest/skills/demo/SKILL.md" - ? "---\nname: demo\n---\n\n# Complete demo instructions\n" - : "# Hidden\n", - ); - const codeModeSkills = resolveCodeModeSkills({ - skillsPrompt: [ - "", - " ", - " demo", - " Short prompt description", - " /guest/skills/demo/SKILL.md", - " ", - "", - ].join("\n"), - candidates: [demo, hidden], - reader, - }); - const { - config, - catalogRef, - tools: codeModeTools, - } = createCodeModeHarness({ - codeModeSkills, - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - codeModeSkills, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const listed = await skills.list(); - const body = await skills.read("demo"); - let unknown; - try { - await skills.read("missing"); - } catch (error) { - unknown = error.message; - } - return { listed, body, unknown }; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - listed: [ - { - name: "demo", - description: "Full demo description", - location: "/guest/skills/demo/SKILL.md", - }, - ], - body: "---\nname: demo\n---\n\n# Complete demo instructions\n", - unknown: 'Unknown skill "missing". Available skills: demo', - }); - expect(codeModeTools[0]?.description).toContain("`await skills.read(name)`"); - expect(reader).toHaveBeenCalledOnce(); - expect(reader).toHaveBeenCalledWith({ - location: "/guest/skills/demo/SKILL.md", - signal: expect.any(AbortSignal), - }); - }); - - it("returns ordinary read content through tools.callValue", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const read = createOpenClawReadTool( - createReadTool("/workspace", { - operations: { - access: async () => {}, - detectImageMimeType: async () => null, - readFile: async () => Buffer.from("ordinary file content"), - }, - }) as unknown as Parameters[0], - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, read], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: `return await tools.callValue("openclaw:core:read", { path: "notes.txt" });`, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ kind: "text", content: "ordinary file content" }); - }); - - it("resolves sequential bridge tool calls inline within one exec instead of a wait per call", async () => { - const catalogRef = createToolSearchCatalogRef(); - // maxPendingToolCalls stays a per-batch concurrency cap; five sequential - // awaits must drain inline even with a cap of 2. - const config = { - tools: { codeMode: { enabled: true, maxPendingToolCalls: 2 } }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const codeModeTools = createCodeModeTools(ctx); - const ticket = pluginTool("fake_create_ticket", "Create a fake ticket"); - applyCodeModeCatalog({ - tools: [...codeModeTools, ticket], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - // Five separate awaits would each suspend to the model under a wait-per-call - // design; inline resumption collapses them into a single completed exec so - // the model spends one turn instead of six. - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-inline", - { - code: ` - const ids = []; - for (let index = 0; index < 5; index += 1) { - const called = await tools.callValue("fake_create_ticket", { value: index }); - ids.push(called.input.value); - } - return ids; - `, - }, - ), - ); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual([0, 1, 2, 3, 4]); - expect(ticket.execute).toHaveBeenCalledTimes(5); - expect(testing.activeRuns.size).toBe(0); - }); - - it("keeps the actual winner when the later-started nested tool settles first", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - let firstAborted = false; - const first = pluginToolWithExecute( - "fake_first", - "Earlier slow helper", - async (_toolCallId, _input, signal) => { - await new Promise((resolve, reject) => { - const timer = setTimeout(resolve, 25); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - firstAborted = true; - reject(new Error("aborted")); - }, - { once: true }, - ); - }); - return jsonResult({ winner: "first" }); - }, - ); - const second = pluginToolWithExecute("fake_second", "Later fast helper", async () => - jsonResult({ winner: "second" }), - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, first, second], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-later-winner", - { - code: `return await Promise.race([ - tools.callValue("fake_first", {}), - tools.callValue("fake_second", {}), - ]);`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: { winner: "second" } }); - expect(first.execute).toHaveBeenCalledOnce(); - expect(second.execute).toHaveBeenCalledOnce(); - expect(firstAborted).toBe(false); - expect(testing.activeRuns.size).toBe(0); - }); - - it.each([ - { - label: "directly", - auditCode: 'void tools.callValue("fake_early_audit", {});', - }, - { - label: "in a detached already-settled Promise.race", - auditCode: 'void Promise.race([tools.callValue("fake_early_audit", {}), Promise.resolve()]);', - }, - { - label: "in a detached Promise.all", - auditCode: 'void Promise.all([tools.callValue("fake_early_audit", {})]);', - }, - { - label: "in a detached Promise.allSettled", - auditCode: 'void Promise.allSettled([tools.callValue("fake_early_audit", {})]);', - }, - { - label: "in a detached Promise.any", - auditCode: 'void Promise.any([tools.callValue("fake_early_audit", {})]);', - }, - { - label: "in a detached Promise.race", - auditCode: 'void Promise.race([tools.callValue("fake_early_audit", {})]);', - }, - ])( - "drains a detached audit started $label before an awaited nested call", - async ({ auditCode }) => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - let auditCompleted = false; - let auditAborted = false; - const audit = pluginToolWithExecute( - "fake_early_audit", - "Early detached audit", - async (_toolCallId, _input, signal) => { - await new Promise((resolve, reject) => { - const timer = setTimeout(resolve, 250); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - auditAborted = true; - reject(new Error("aborted")); - }, - { once: true }, - ); - }); - auditCompleted = true; - return jsonResult({ recorded: true }); - }, - ); - const fast = pluginToolWithExecute("fake_awaited_fast", "Awaited fast helper", async () => - jsonResult({ winner: "fast" }), - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, audit, fast], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-early-detached-audit", - { - code: `${auditCode} - return await tools.callValue("fake_awaited_fast", {});`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); - expect(audit.execute).toHaveBeenCalledOnce(); - expect(fast.execute).toHaveBeenCalledOnce(); - expect(auditCompleted).toBe(true); - expect(auditAborted).toBe(false); - expect(testing.activeRuns.size).toBe(0); - }, - ); - - it("drains a race winner's detached audit and its slower race branch", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - let loserAborted = false; - const winner = pluginToolWithExecute("fake_race_winner", "Race winner", async () => - jsonResult({ winner: "fast" }), - ); - const loser = pluginToolWithExecute( - "fake_race_loser", - "Race loser", - async (_toolCallId, _input, signal) => { - await new Promise((resolve, reject) => { - const timer = setTimeout(resolve, 25); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - loserAborted = true; - reject(new Error("aborted")); - }, - { once: true }, - ); - }); - return jsonResult({ winner: "slow" }); - }, - ); - const audit = pluginToolWithExecute("fake_race_audit", "Detached audit", async () => - jsonResult({ recorded: true }), - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, winner, loser, audit], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-race-detached-audit", - { - code: `return Promise.race([ - tools.callValue("fake_race_winner", {}), - tools.callValue("fake_race_loser", {}), - ]).then((value) => { - void tools.callValue("fake_race_audit", {}); - return value; - });`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); - expect(winner.execute).toHaveBeenCalledOnce(); - expect(loser.execute).toHaveBeenCalledOnce(); - expect(audit.execute).toHaveBeenCalledOnce(); - expect(loserAborted).toBe(false); - expect(testing.activeRuns.size).toBe(0); - }); - - it("drains every detached nested tool before completing the guest", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const first = pluginToolWithExecute("fake_detached_first", "First detached helper", async () => - jsonResult({ name: "first" }), - ); - const second = pluginToolWithExecute( - "fake_detached_second", - "Second detached helper", - async () => jsonResult({ name: "second" }), - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, first, second], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-detached", - { - code: `void tools.callValue("fake_detached_first", {}); - void tools.callValue("fake_detached_second", {}); - return "done";`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: "done" }); - expect(first.execute).toHaveBeenCalledOnce(); - expect(second.execute).toHaveBeenCalledOnce(); - expect(testing.activeRuns.size).toBe(0); - }); - - it.each(["race", "any"] as const)( - "preserves the Promise.%s winner while draining the slower nested tool", - async (combinator) => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - let slowAborted = false; - let slowCompleted = false; - const fast = pluginToolWithExecute("fake_fast", "Fast helper", async () => - jsonResult({ winner: "fast" }), - ); - const slow = pluginToolWithExecute( - "fake_slow", - "Slow helper", - async (_toolCallId, _input, signal) => { - await new Promise((resolve, reject) => { - const timer = setTimeout(resolve, 25); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - slowAborted = true; - reject(new Error("aborted")); - }, - { once: true }, - ); - }); - slowCompleted = true; - return jsonResult({ winner: "slow" }); - }, - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, fast, slow], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - `code-call-${combinator}-fast`, - { - code: `return await Promise.${combinator}([ - tools.callValue("fake_fast", {}), - tools.callValue("fake_slow", {}), - ]);`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: { winner: "fast" } }); - expect(fast.execute).toHaveBeenCalledOnce(); - expect(slow.execute).toHaveBeenCalledOnce(); - expect(slowCompleted).toBe(true); - expect(slowAborted).toBe(false); - expect(testing.activeRuns.size).toBe(0); - }, - ); - - it("preserves fail-fast Promise.all while draining the slower nested tool", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - let slowAborted = false; - let slowCompleted = false; - const failed = pluginToolWithExecute("fake_failed", "Failed helper", async () => { - throw new Error("fast failure"); - }); - const slow = pluginToolWithExecute( - "fake_slow", - "Slow helper", - async (_toolCallId, _input, signal) => { - await new Promise((resolve, reject) => { - const timer = setTimeout(resolve, 25); - signal?.addEventListener( - "abort", - () => { - clearTimeout(timer); - slowAborted = true; - reject(new Error("aborted")); - }, - { once: true }, - ); - }); - slowCompleted = true; - return jsonResult({ winner: "slow" }); - }, - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, failed, slow], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-fail-fast", - { - code: `try { - await Promise.all([ - tools.callValue("fake_failed", {}), - tools.callValue("fake_slow", {}), - ]); - return "unexpected success"; - } catch (error) { - return error.message; - }`, - }, - ), - ); - - expect(details).toMatchObject({ status: "completed", value: "fast failure" }); - expect(failed.execute).toHaveBeenCalledOnce(); - expect(slow.execute).toHaveBeenCalledOnce(); - expect(slowCompleted).toBe(true); - expect(slowAborted).toBe(false); - expect(testing.activeRuns.size).toBe(0); - }); - - it("fails fast without parking a suspended run when the exec call is aborted", async () => { - const catalogRef = createToolSearchCatalogRef(); - // Long timeout so a missing abort short-circuit would block the whole test. - const config = { - tools: { codeMode: { enabled: true, timeoutMs: 30_000 } }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const codeModeTools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [ - ...codeModeTools, - // A tool that never settles and ignores its abort signal; only the - // host-level abort race can free the cancelled exec. - pluginToolWithExecute("fake_stuck", "Stuck helper", async () => { - await new Promise(() => {}); - return null as never; - }), - ], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const controller = new AbortController(); - controller.abort(); - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-abort", - { code: "await tools.fake_stuck({}); return 'done';" }, - controller.signal, - ), - ); - - // Abort drops the run instead of parking it; a cancelled call must not pin - // one of the process-global suspended-run slots until TTL expiry. - expect(details.status).toBe("failed"); - expect(details.error).toBe("code mode execution aborted"); - expect(details.code).toBe("aborted"); - expect(testing.activeRuns.size).toBe(0); - }); - - it("terminates a running guest promptly when the exec call is aborted", async () => { - const catalogRef = createToolSearchCatalogRef(); - // Long timeout so only the abort race can end the hostile loop quickly. - const config = { - tools: { codeMode: { enabled: true, timeoutMs: 30_000 } }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const codeModeTools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const controller = new AbortController(); - const abortTimer = setTimeout(() => controller.abort(), 200); - const startedAt = Date.now(); - try { - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-abort-live", - { code: "while (true) {}" }, - controller.signal, - ), - ); - expect(details.status).toBe("failed"); - expect(details.error).toBe("code mode execution aborted"); - expect(details.code).toBe("aborted"); - } finally { - clearTimeout(abortTimer); - } - expect(Date.now() - startedAt).toBeLessThan(10_000); - expect(testing.activeRuns.size).toBe(0); - }); - - it("uses tools recovery guidance for guessed tool ids", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const writeTool = pluginTool("write", "Write a file to the workspace"); - applyCodeModeCatalog({ - tools: [...codeModeTools, writeTool], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - try { - await tools.call("file_write", { - path: "memory/2026-05-22.md", - content: "remember this", - }); - return "unexpected success"; - } catch (error) { - return error.message; - } - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toBe( - "Unknown tool id: file_write. Did you mean: write? Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", - ); - expect(writeTool.execute).not.toHaveBeenCalled(); - }); - - it("uses tools recovery guidance when no generic Code Mode suggestion matches", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: codeModeTools, - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - try { - await tools.call("missing_tool", {}); - return "unexpected success"; - } catch (error) { - return error.message; - } - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toBe( - "Unknown tool id: missing_tool. Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", - ); - }); - - it("surfaces policy blocks as guest call errors for declared outputs", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const target = pluginTool("fake_policy_block", "Return policy-controlled rows"); - target.outputSchema = Type.Array( - Type.Object({ id: Type.String() }, { additionalProperties: false }), - ); - target.execute = vi.fn(async () => - buildBlockedToolResult({ reason: "blocked by orchard policy" }), - ); - applyCodeModeCatalog({ - tools: [...codeModeTools, target], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - try { - const rows = await tools.callValue("fake_policy_block", {}); - return rows.map((row) => row.id); - } catch (error) { - return error.message; - } - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toContain("was blocked before execution: blocked by orchard policy"); - }); - - it("exposes MCP tools only through the MCP namespace", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const githubCreate = mcpTool({ - name: "github__create_issue", - serverName: "github", - toolName: "create_issue", - parameters: { - type: "object", - properties: { - owner: { type: "string" }, - repo: { type: "string", description: "Repository 名称" }, - title: { type: "string", description: "Issue title\nShown in tracker" }, - body: { type: "string", default: "" }, - }, - required: ["owner", "repo", "title"], - }, - }); - const compacted = applyCodeModeCatalog({ - tools: [...codeModeTools, githubCreate], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - expect(compacted.tools[0]?.description).toContain("MCP: MCP server tools grouped by server."); - expect(compacted.tools[0]?.description).toContain("visible servers: github"); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const rootApi = await MCP.$api(); - const api = await MCP.github.$api("createIssue", { schema: true }); - const apiFiles = await API.list("mcp"); - const apiFilesTrailingSlash = await API.list("mcp/"); - const rootFile = await API.read("mcp/index.d.ts"); - const serverFile = await API.read("mcp/github.d.ts"); - const created = await MCP.github.createIssue({ - owner: "openclaw", - repo: "openclaw", - title: "Ship it", - }); - const createdPayload = JSON.parse(created.content[0].text); - const searchHits = await tools.search("github create issue", { limit: 5 }); - const allHasMcp = ALL_TOOLS.some((tool) => tool.source === "mcp"); - let directCall; - let directDescribe; - try { - await tools.describe("github__create_issue"); - directDescribe = "unexpected"; - } catch (error) { - directDescribe = error.message; - } - try { - await tools.call("github__create_issue", { owner: "x", repo: "y", title: "blocked" }); - directCall = "unexpected"; - } catch (error) { - directCall = error.message; - } - return { - apiHeader: api.header, - apiFilePaths: apiFiles.files.map((file) => file.path), - apiFilePathsTrailingSlash: apiFilesTrailingSlash.files.map((file) => file.path), - listedServerFileBytes: apiFiles.files.find((file) => file.path === "mcp/github.d.ts").bytes, - serverFileBytes: serverFile.bytes, - serverFileContent: serverFile.content, - rootFileHasReference: rootFile.content.includes('./github.d.ts'), - serverFileHasCreateIssue: serverFile.content.includes('function createIssue('), - serverFileHasTitleDoc: serverFile.content.includes('@param title Issue title Shown in tracker'), - apiSchemaTitle: api.schemas.createIssue.type, - rootServers: rootApi.servers, - createdPayload, - createdDetails: created.details, - searchHits, - allHasMcp, - directDescribe, - directCall, - hasMcp: "MCP" in namespaces, - }; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - createdPayload: { - serverName: "github", - toolName: "create_issue", - input: { - owner: "openclaw", - repo: "openclaw", - title: "Ship it", - body: "", - }, - }, - createdDetails: { - serverName: "github", - toolName: "create_issue", - input: { - owner: "openclaw", - repo: "openclaw", - title: "Ship it", - body: "", - }, - }, - searchHits: [], - allHasMcp: false, - directDescribe: - "Unknown tool id: github__create_issue. Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", - directCall: - "Unknown tool id: github__create_issue. Use tools.search to find a tool, tools.describe to inspect it, then tools.call with the exact id or name.", - hasMcp: true, - apiSchemaTitle: "object", - apiHeader: expect.stringContaining("function createIssue("), - apiFilePaths: ["mcp/index.d.ts", "mcp/github.d.ts"], - apiFilePathsTrailingSlash: ["mcp/index.d.ts", "mcp/github.d.ts"], - listedServerFileBytes: expect.any(Number), - serverFileBytes: expect.any(Number), - serverFileContent: expect.stringContaining("Repository 名称"), - rootFileHasReference: true, - serverFileHasCreateIssue: true, - serverFileHasTitleDoc: true, - rootServers: [{ identifier: "github", serverName: "github", toolCount: 1 }], - }); - const value = details.value as { - apiHeader: string; - listedServerFileBytes: number; - serverFileBytes: number; - serverFileContent: string; - }; - expect(value.listedServerFileBytes).toBe(value.serverFileBytes); - expect(value.serverFileBytes).toBe(Buffer.byteLength(value.serverFileContent, "utf8")); - expect(value.serverFileBytes).toBeGreaterThan(value.serverFileContent.length); - expect(value.apiHeader).toContain("@param title Issue title Shown in tracker"); - expect(value.apiHeader).not.toContain("@param title Issue title\n"); - expect(value.apiHeader).toContain("title: string;"); - expect(githubCreate.execute).toHaveBeenCalledTimes(1); - }); - - it("lets agents inspect MCP declaration files before calling MCP tools", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const githubCreate = mcpTool({ - name: "github__create_issue", - serverName: "github", - toolName: "create_issue", - parameters: { - type: "object", - properties: { - owner: { type: "string" }, - repo: { type: "string" }, - title: { type: "string", description: "Issue title" }, - }, - required: ["owner", "repo", "title"], - }, - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, githubCreate], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const files = await API.list("mcp"); - const api = await API.read("mcp/github.d.ts"); - const created = await MCP.github.createIssue({ - owner: "openclaw", - repo: "openclaw", - title: "From file docs", - }); - return { - fileCount: files.files.length, - headerHasSignature: api.content.includes("function createIssue("), - usedApiCall: api.content.includes("function $api("), - created: JSON.parse(created.content[0].text), - }; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - fileCount: 2, - headerHasSignature: true, - usedApiCall: true, - created: { - serverName: "github", - toolName: "create_issue", - input: { - owner: "openclaw", - repo: "openclaw", - title: "From file docs", - }, - }, - }); - expect(githubCreate.execute).toHaveBeenCalledTimes(1); - }); - - it("groups MCP resources and prompts under server namespaces", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const resourceRead = mcpTool({ - name: "docs__resources_read", - serverName: "docs", - toolName: "resources_read", - operation: "resources_read", - parameters: { - type: "object", - properties: { uri: { type: "string" } }, - required: ["uri"], - }, - }); - const promptGet = mcpTool({ - name: "docs__prompts_get", - serverName: "docs", - toolName: "prompts_get", - operation: "prompts_get", - parameters: { - type: "object", - properties: { - name: { type: "string" }, - arguments: { type: "object" }, - }, - required: ["name"], - }, - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, resourceRead, promptGet], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const api = await MCP.docs.$api(); - const resource = await MCP.docs.resources.read({ uri: "memo://one" }); - const prompt = await MCP.docs.prompts.get({ name: "brief", arguments: { topic: "mcp" } }); - return { header: api.header, resource: resource.details, prompt: prompt.details }; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - resource: { - serverName: "docs", - toolName: "resources_read", - input: { uri: "memo://one" }, - }, - prompt: { - serverName: "docs", - toolName: "prompts_get", - input: { name: "brief", arguments: { topic: "mcp" } }, - }, - header: expect.stringContaining("namespace resources"), - }); - }); - - it("renames MCP namespace identifiers that would be unsafe path segments", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const dangerous = mcpTool({ - name: "constructor__prototype", - serverName: "constructor", - toolName: "prototype", - parameters: { - type: "object", - properties: { value: { type: "string" } }, - required: ["value"], - }, - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, dangerous], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: 'return (await MCP.constructor2.prototype2({ value: "safe" })).details;', - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - serverName: "constructor", - toolName: "prototype", - input: { value: "safe" }, - }); - }); - - it.each(["delete", "default", "return", "enum", "class"])( - "renders and executes reserved MCP tool name %s safely", - async (toolName) => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const target = mcpTool({ - name: `github__${toolName}`, - serverName: "github", - toolName, - parameters: { - type: "object", - properties: { value: { type: "string" } }, - required: ["value"], - }, - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, target], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const safeName = `${toolName}2`; - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "Code Mode exec test invariant"), - waitTool: expectDefined(codeModeTools[1], "Code Mode wait test invariant"), - code: ` - const file = await API.read("mcp/github.d.ts"); - const api = await MCP.github.$api("${safeName}"); - const result = await MCP.github.${safeName}({ value: "safe" }); - return { file: file.content, header: api.header, result: result.details }; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ - file: expect.stringContaining(`function ${safeName}(`), - header: expect.stringContaining(`function ${safeName}(`), - result: { - serverName: "github", - toolName, - input: { value: "safe" }, - }, - }); - expect(target.execute).toHaveBeenCalledTimes(1); - }, - ); - - it("marks yield suspensions and resumes the snapshot with wait", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-yield", - { - restartSafe: true, - code: ` - text("before"); - await yield_control("pause"); - text("after"); - return "done"; - `, - }, - ), - ); - - expect(first.status).toBe("waiting"); - expect(first.reason).toBe("yield"); - expect(first.replaySafe).toBe(true); - expect(first.output).toEqual([{ type: "text", text: "before" }]); - - const runId = first.runId; - expect(typeof runId).toBe("string"); - const resumed = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-yield", - { runId }, - ), - ); - - expect(resumed.status).toBe("completed"); - expect(resumed.value).toBe("done"); - expect(resumed.output).toEqual([{ type: "text", text: "after" }]); - }); - - it("delivers each yielded output block exactly once across repeated waits", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const execTool = expectDefined(codeModeTools[0], "Code Mode exec test invariant"); - const waitTool = expectDefined(codeModeTools[1], "Code Mode wait test invariant"); - const first = resultDetails( - await execTool.execute("code-call-incremental-output", { - code: ` - text("phase 1"); - await yield_control("first pause"); - text("phase 2"); - await yield_control("second pause"); - text("phase 3"); - return "done"; - `, - }), - ); - - expect(first.status).toBe("waiting"); - expect(first.output).toEqual([{ type: "text", text: "phase 1" }]); - - const second = resultDetails( - await waitTool.execute("code-wait-incremental-output-1", { runId: first.runId }), - ); - - expect(second.status).toBe("waiting"); - expect(second.output).toEqual([{ type: "text", text: "phase 2" }]); - - const third = resultDetails( - await waitTool.execute("code-wait-incremental-output-2", { runId: second.runId }), - ); - - expect(third.status).toBe("completed"); - expect(third.value).toBe("done"); - expect(third.output).toEqual([{ type: "text", text: "phase 3" }]); - }); - - it("returns only newly emitted output when a resumed guest fails", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-incremental-failure", - { - code: ` - text("before pause"); - await yield_control("pause"); - text("before failure"); - throw new Error("resumed failure"); - `, - }, - ), - ); - - expect(first.status).toBe("waiting"); - expect(first.output).toEqual([{ type: "text", text: "before pause" }]); - - const second = resultDetails( - await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( - "code-wait-incremental-failure", - { runId: first.runId }, - ), - ); - - expect(second.status).toBe("failed"); - expect(second.error).toContain("resumed failure"); - expect(second.output).toEqual([{ type: "text", text: "before failure" }]); - expect(testing.activeRuns.has(first.runId as string)).toBe(false); - }); - - it("preserves the original exec identity for tool calls after yield and wait", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - const target = pluginTool("fake_resumed_identity", "Resumed identity helper"); - applyCodeModeCatalog({ - tools: [...codeModeTools, target], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const suspended = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-original-parent", - { - code: 'await yield_control("pause"); return await tools.callValue("fake_resumed_identity", {});', - }, - ), - ); - expect(suspended.status).toBe("waiting"); - - const resumed = resultDetails( - await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( - "code-wait-different-parent", - { runId: suspended.runId }, - ), - ); - - expect(resumed.status).toBe("completed"); - expect(target.execute).toHaveBeenCalledOnce(); - expect(vi.mocked(target.execute).mock.calls[0]?.[0]).toContain("code-call-original-parent"); - expect(vi.mocked(target.execute).mock.calls[0]?.[0]).not.toContain( - "code-wait-different-parent", - ); - }); - - it("allocates distinct replay identities when a later turn reuses a tool-call id", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - const execTool = expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"); - const input = { code: 'await yield_control("pause"); return "done";' }; - const executionContext = (turnId: string) => - ({ - assistantMessage: { responseId: " ", turnId }, - toolCall: { type: "toolCall", id: "reused-call-id", name: "exec", arguments: input }, - }) as never; - - const first = resultDetails( - await runWithAgentToolExecutionContext(executionContext("response-turn-1"), () => - execTool.execute("reused-call-id", input), - ), - ); - const second = resultDetails( - await runWithAgentToolExecutionContext(executionContext("response-turn-2"), () => - execTool.execute("reused-call-id", input), - ), - ); - - expect(first.status).toBe("waiting"); - expect(second.status).toBe("waiting"); - expect(second.runId).not.toBe(first.runId); - expect(testing.activeRuns.size).toBe(2); - expect(new Set([...testing.activeRuns.values()].map((state) => state.replayId)).size).toBe(2); - }); - - it("keeps restart-safe mode across audited core reads", async () => { - const targetTool = fakeTool("read", "Read"); - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, targetTool], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-replay-safety", - { - restartSafe: true, - code: ` - const matches = await tools.search(${JSON.stringify(targetTool.name)}); - return await tools.call(matches[0].id, {}); - `, - }, - ), - ); - expect(first.status).toBe("waiting"); - expect(first.replaySafe).toBe(true); - - const second = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-replay-safety", - { runId: first.runId }, - ), - ); - expect(second.status).toBe("waiting"); - expect(second.replaySafe).toBe(true); - - const completed = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-replay-safety-complete", - { - runId: second.runId, - }, - ), - ); - expect(completed.status).toBe("completed"); - }); - - it("allows explicitly replay-safe plugin tools by exact catalog id", async () => { - const targetTool = pluginTool("fake_plugin_read", "Plugin read"); - setPluginToolMeta(targetTool, { - pluginId: "fake-code-mode", - optional: true, - replaySafe: true, - }); - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, targetTool], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const completed = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - restartSafe: true, - code: ` - const matches = await tools.search("fake_plugin_read"); - return await tools.call(matches[0].id, {}); - `, - }); - - expect(completed.status).toBe("completed"); - expect(completed.replaySafe).toBe(true); - expect(targetTool.execute).toHaveBeenCalledTimes(1); - }); - - it("rejects MCP tools even when their metadata claims replay safety", async () => { - const targetTool = mcpTool({ - name: "mcp_github_read_file", - serverName: "github", - toolName: "read_file", - }); - setPluginToolMeta(targetTool, { - pluginId: "bundle-mcp", - optional: false, - replaySafe: true, - mcp: { - serverName: "github", - safeServerName: "github", - toolName: "read_file", - operation: "tool", - }, - }); - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, targetTool], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const completed = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - restartSafe: true, - code: 'return await MCP.github.readFile({ path: "README.md" });', - }); - - expect(completed.status).toBe("failed"); - expect(completed.replaySafe).toBe(true); - expect(completed.error).toContain("cannot call namespace tools"); - expect(targetTool.execute).not.toHaveBeenCalled(); - }); - - it("rejects side-effecting calls before executing them in restart-safe mode", async () => { - const targetTool = pluginTool("fake_write", "Write"); - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, targetTool], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-unsafe-restart", - { - restartSafe: true, - code: ` - const matches = await tools.search("fake_write"); - return await tools.call(matches[0].id, {}); - `, - }, - ), - ); - expect(first.status).toBe("waiting"); - expect(first.replaySafe).toBe(true); - - const failed = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-unsafe-restart", - { runId: first.runId }, - ), - ); - expect(failed.status).toBe("failed"); - expect(failed.error).toContain("cannot call side-effecting tools"); - expect(targetTool.execute).not.toHaveBeenCalled(); - }); - - it("keeps host-forced restart safety when the model clears the exec flag", async () => { - const targetTool = pluginTool("fake_forced_write", "Write"); - const { - config, - catalogRef, - tools: codeModeTools, - } = createCodeModeHarness({ - forceRestartSafeTools: true, - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, targetTool], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-forced-restart", - { - restartSafe: false, - code: ` - const matches = await tools.search("fake_forced_write"); - return await tools.call(matches[0].id, {}); - `, - }, - ), - ); - expect(first.status).toBe("waiting"); - expect(first.replaySafe).toBe(true); - - const failed = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-forced-restart", - { runId: first.runId }, - ), - ); - expect(failed.status).toBe("failed"); - expect(failed.error).toContain("cannot call side-effecting tools"); - expect(targetTool.execute).not.toHaveBeenCalled(); - }); - - it("fails yield suspension when snapshot expiry would exceed the Date range", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - const nowSpy = vi.spyOn(Date, "now").mockReturnValue(8_640_000_000_000_000); - let details: Record; - try { - details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-yield-overflow", - { - code: 'await yield_control("pause"); return "done";', - }, - ), - ); - } finally { - nowSpy.mockRestore(); - } - - expect(details.status).toBe("failed"); - expect(details.error).toBe("code mode run expiry is unavailable."); - expect(testing.activeRuns.size).toBe(0); - }); - - it("expires suspended runs with invalid expiry timestamps", async () => { - const { tools: codeModeTools } = createCodeModeHarness(); - testing.activeRuns.set("invalid-expiry-run", { - expiresAt: 8_640_000_000_000_001, - } as never); - - await expect( - expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-invalid-expiry", - { runId: "invalid-expiry-run" }, - ), - ).rejects.toThrow("code mode run is unavailable or expired"); - expect(testing.activeRuns.has("invalid-expiry-run")).toBe(false); - }); - - it("rejects wait calls from a different session scope", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-wrong-session", - { - code: 'await yield_control("pause"); return "done";', - }, - ), - ); - expect(first.status).toBe("waiting"); - const otherWaitTool = expectDefined( - createCodeModeTools({ - config, - runtimeConfig: config, - sessionId: "other-session", - sessionKey: "agent:other:main", - runId: "run-code-mode", - catalogRef, - })[1], - 'createCodeModeTools({ config, runtimeConfig: config, sessionId: "othe... test invariant', - ); - - await expect( - otherWaitTool.execute("code-wait-wrong-session", { runId: first.runId }), - ).rejects.toThrow("different session"); - }); - - it.each(["runId", "sessionId", "sessionKey", "agentId"] as const)( - "rejects suspended-run callers missing the owner %s", - async (missingIdentity) => { - const { - config, - catalogRef, - ctx, - tools: codeModeTools, - } = createCodeModeHarness({ - agentId: "owner", - }); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: ctx.sessionId, - sessionKey: ctx.sessionKey, - agentId: ctx.agentId, - runId: ctx.runId, - catalogRef, - }); - - const suspended = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-scoped-owner", - { code: 'await yield_control("pause"); return "owner-secret";' }, - ), - ); - expect(suspended.status).toBe("waiting"); - - const missingIdentityWait = expectDefined( - createCodeModeTools({ - config, - runtimeConfig: config, - catalogRef, - ...(missingIdentity === "runId" ? {} : { runId: ctx.runId }), - ...(missingIdentity === "sessionId" ? {} : { sessionId: ctx.sessionId }), - ...(missingIdentity === "sessionKey" ? {} : { sessionKey: ctx.sessionKey }), - ...(missingIdentity === "agentId" ? {} : { agentId: ctx.agentId }), - })[1], - "Unscoped Code Mode wait test invariant", - ); - - await expect( - missingIdentityWait.execute("code-wait-missing-owner", { runId: suspended.runId }), - ).rejects.toThrow(missingIdentity === "runId" ? "different agent run" : "different session"); - expect(testing.activeRuns.has(suspended.runId as string)).toBe(true); - - const rightfulResult = resultDetails( - await expectDefined(codeModeTools[1], "Owner Code Mode wait test invariant").execute( - "code-wait-rightful-owner", - { runId: suspended.runId }, - ), - ); - expect(rightfulResult.status).toBe("completed"); - expect(rightfulResult.value).toBe("owner-secret"); - }, - ); - - it("rejects concurrent waits for the same suspended run", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - timeoutMs: 500, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const codeModeTools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [ - ...codeModeTools, - pluginToolWithExecute( - "fake_slow", - "Slow helper", - async () => await new Promise(() => {}), - ), - ], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-concurrent-wait", - { - code: "await tools.fake_slow({}); return 'done';", - }, - ), - ); - expect(first.status).toBe("waiting"); - - const firstWait = expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-concurrent-a", - { - runId: first.runId, - }, - ); - await expect( - expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-concurrent-b", - { runId: first.runId }, - ), - ).rejects.toThrow("already being resumed"); - const stillWaiting = resultDetails(await firstWait); - - expect(stillWaiting.status).toBe("waiting"); - expect(stillWaiting.runId).toBe(first.runId); - }); - - it("resumes and reparks a yielding run at the suspended-run capacity limit", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-at-capacity", - { - code: ` - await yield_control("first"); - await yield_control("second"); - return "done"; - `, - }, - ), - ); - expect(first.status).toBe("waiting"); - const firstRunId = first.runId; - expect(typeof firstRunId).toBe("string"); - if (typeof firstRunId !== "string") { - throw new Error("expected a parked Code Mode run"); - } - const parked = testing.activeRuns.get(firstRunId); - expect(parked).toBeDefined(); - if (!parked) { - throw new Error("expected a parked Code Mode snapshot"); - } - - // Inert snapshots occupy real capacity without starting 63 extra workers. - for (let index = 0; index < 63; index += 1) { - const runId = `cm_code_mode_capacity_${index}`; - testing.activeRuns.set(runId, { ...parked, runId, pending: [] }); - } - - const second = resultDetails( - await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( - "code-wait-at-capacity", - { runId: firstRunId }, - ), - ); - - expect(second.status).toBe("waiting"); - expect(second.reason).toBe("yield"); - expect(testing.activeRuns.size).toBe(64); - - const completed = resultDetails( - await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( - "code-wait-after-capacity", - { runId: second.runId }, - ), - ); - - expect(completed).toMatchObject({ status: "completed", value: "done" }); - expect(testing.activeRuns.size).toBe(63); - }); - - it("reports only unsettled pending tool calls when wait times out", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - timeoutMs: 500, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const codeModeTools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [ - ...codeModeTools, - pluginTool("fake_fast", "Fast helper"), - pluginToolWithExecute( - "fake_slow", - "Slow helper", - async () => await new Promise(() => {}), - ), - ], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-timeout", - { - code: ` - text("before timeout"); - const fast = tools.fake_fast({}); - const slow = tools.fake_slow({}); - await fast; - await slow; - return "done"; - `, - }, - ), - ); - expect(first.status).toBe("waiting"); - expect(first.output).toEqual([{ type: "text", text: "before timeout" }]); - expect(first.pendingToolCalls).toEqual([expect.objectContaining({ method: "callValue" })]); - const runId = first.runId; - expect(typeof runId).toBe("string"); - if (typeof runId !== "string") { - throw new Error("expected code mode run id"); - } - - const activeRun = testing.activeRuns.get(runId); - expect(activeRun).toBeDefined(); - activeRun!.config.timeoutMs = 100; - - const second = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-timeout", - { runId }, - ), - ); - - expect(second.status).toBe("waiting"); - expect(second.output).toEqual([]); - expect(second.pendingToolCalls).toEqual([expect.objectContaining({ method: "callValue" })]); - - const third = resultDetails( - await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( - "code-wait-timeout-again", - { runId }, - ), - ); - - expect(third.status).toBe("waiting"); - expect(third.output).toEqual([]); - expect(third.pendingToolCalls).toEqual([expect.objectContaining({ method: "callValue" })]); - }); - - it("does not load TypeScript for plain JavaScript code mode runs", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: "return 42;", - }); - - expect(details.status).toBe("completed"); - expect(details.value).toBe(42); - expect(testing.getTypescriptRuntimePromise()).toBeNull(); - }); - - it("allows identifiers and strings that contain import without module access", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: ` - const important = 41; - const message = "import docs later"; - return important + (message.includes("import") ? 1 : 0); - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toBe(42); - }); - - it.each([ - { - name: "template-literal import text", - code: "return `import('node:fs')`;", - value: "import('node:fs')", - }, - { - name: "template-literal require text", - code: "return `require('node:fs')`;", - value: "require('node:fs')", - }, - { - name: "nested template-literal module text", - code: "return `outer ${`require('node:fs')`}`;", - value: "outer require('node:fs')", - }, - { - name: "regular-expression module text", - code: 'return /import.meta/.test("import.meta");', - value: true, - }, - { - name: "regular-expression module text inside interpolation", - code: 'return `${/import.meta/.test("import.meta")}`;', - value: "true", - }, - { - name: "ordinary import method", - code: "const api = { import(value) { return value; } }; return api.import(42);", - value: 42, - }, - { - name: "ordinary require method", - code: "const api = { require(value) { return value; } }; return api.require(42);", - value: 42, - }, - { - name: "optional ordinary import method", - code: "const api = { import(value) { return value; } }; return api?.import?.(42);", - value: 42, - }, - { - name: "computed ordinary require method", - code: 'const api = { require(value) { return value; } }; return api["require"](42);', - value: 42, - }, - { - name: "ordinary import metadata property", - code: "const api = { import: { meta: 42 } }; return api.import.meta;", - value: 42, - }, - ])("executes harmless $name in the real guest worker", async ({ code, value }) => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code, - }); - - expect(details).toMatchObject({ status: "completed", value }); - expect(testing.activeRuns.size).toBe(0); - }); - - it("never exposes Node module-loader globals to the real guest worker", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: "return [typeof process, typeof module, typeof require];", - }); - - expect(details).toMatchObject({ - status: "completed", - value: ["undefined", "undefined", "undefined"], - }); - expect(testing.activeRuns.size).toBe(0); - }); - - it("isolates and cleans up 12 concurrent real guest workers", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - const execTool = expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"); - - const results = await Promise.all( - Array.from({ length: 12 }, async (_, index) => - resultDetails( - await execTool.execute(`code-call-concurrent-worker-${index}`, { - code: `return { index: ${index}, message: \`require('node:fs')\` };`, - }), - ), - ), - ); - - expect(results).toEqual( - Array.from({ length: 12 }, (_, index) => - expect.objectContaining({ - status: "completed", - value: { index, message: "require('node:fs')" }, - }), - ), - ); - expect(testing.activeRuns.size).toBe(0); - expect(testing.resumingRunIds.size).toBe(0); - }); - - it("fails pending promises that have no host bridge work", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const beforeRunCount = testing.activeRuns.size; - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-empty-wait", - { - code: "await new Promise(() => undefined); return 'never';", - }, - ), - ); - - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("pending without host work"); - expect(testing.activeRuns.size).toBe(beforeRunCount); - }); - - it("surfaces the QuickJS error name and message for guest syntax errors", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-syntax", - { code: "const x = ;" }, - ), - ); - - expect(details.status).toBe("failed"); - const error = String(details.error); - // Regression guard: QuickJS stacks are frames only, so the error used to - // collapse to a bare "at openclaw-code-mode:user.js:..." location with the - // actual cause dropped. The model now sees the name and message. - expect(error).toContain("SyntaxError"); - expect(error).toContain("unexpected token"); - expect(error.startsWith("at ")).toBe(false); - }); - - it("surfaces the QuickJS error name and message for guest runtime errors", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-runtime", - { code: "return missingFn();" }, - ), - ); - - expect(details.status).toBe("failed"); - const error = String(details.error); - expect(error).toContain("ReferenceError"); - expect(error).toContain("missingFn is not defined"); - expect(error.startsWith("at ")).toBe(false); - }); - - it("does not expose the raw host request callback", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-hidden-host-request", - { code: "return typeof globalThis.__openclawHostRequest;" }, - ), - ); - - expect(details).toMatchObject({ - status: "completed", - value: "undefined", - }); - }); - - it("clamps omitted code-mode catalog search limits to maxSearchLimit", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - maxSearchLimit: 3, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const codeModeTools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [ - ...codeModeTools, - pluginTool("fake_ticket_one", "ticket helper"), - pluginTool("fake_ticket_two", "ticket helper"), - pluginTool("fake_ticket_three", "ticket helper"), - pluginTool("fake_ticket_four", "ticket helper"), - pluginTool("fake_ticket_five", "ticket helper"), - ], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - code: 'const hits = await tools.search("ticket"); return hits.length;', - }); - - expect(details.status).toBe("completed"); - expect(details.value).toBe(3); - }); - - it("supports TypeScript source transform", async () => { - testing.setTypescriptRuntimeForTest({ - ...(await import("typescript")), - transpileModule: vi.fn((code: string) => ({ - outputText: code.replace(": number", ""), - diagnostics: [], - })), - ScriptTarget: { ES2022: 9 }, - ModuleKind: { ESNext: 99 }, - ImportsNotUsedAsValues: { Remove: 0 }, - DiagnosticCategory: { Error: 1 }, - flattenDiagnosticMessageText: (message: unknown) => String(message), - } as never); - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - language: "typescript", - code: ` - const value: number = 40 + 2; - return { value }; - `, - }); - - expect(details.status).toBe("completed"); - expect(details.value).toEqual({ value: 42 }); - - const moduleShapedTypeScript = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - language: "typescript", - code: "const value: number = 42; return `import('node:fs') ${value}`;", - }); - - expect(moduleShapedTypeScript.status).toBe("completed"); - expect(moduleShapedTypeScript.value).toBe("import('node:fs') 42"); - - const moduleShapedTypeScriptRegex = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - language: "typescript", - code: 'const value: number = 42; return /import.meta/.test("import.meta");', - }); - - expect(moduleShapedTypeScriptRegex.status).toBe("completed"); - expect(moduleShapedTypeScriptRegex.value).toBe(true); - - for (const moduleAccess of ["import('node:fs')", "require('node:fs')"]) { - const unicodeModuleAccess = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - `code-call-typescript-unicode-${moduleAccess.startsWith("import") ? "import" : "require"}`, - { - language: "typescript", - code: `const value: number = 1; const padding = "${"😀".repeat(96)}"; return ${moduleAccess};`, - }, - ), - ); - - expect(unicodeModuleAccess.status).toBe("failed"); - expect(unicodeModuleAccess.code).toBe("invalid_input"); - expect(unicodeModuleAccess.error).toContain("module access is disabled"); - } - - const commandLikeTypeScript = await runUntilCompleted({ - execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), - waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), - language: "typescript", - code: "node -1; var node: number = 7; return node;", - }); - - expect(commandLikeTypeScript.status).toBe("completed"); - expect(commandLikeTypeScript.value).toBe(7); - - const typedShell = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-typescript-shell-source", - { code: "pwd", language: "typescript" }, - ), - ); - - expect(typedShell.status).toBe("failed"); - expect(typedShell.code).toBe("invalid_input"); - expect(typedShell.error).toMatch(/JavaScript or TypeScript, not shell commands/); - expect(testing.activeRuns.size).toBe(0); - }); - - it("times out an unfinished TypeScript runtime load without starting a worker", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - (config as { tools: { codeMode: unknown } }).tools.codeMode = { - enabled: true, - timeoutMs: 100, - }; - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - testing.setTypescriptRuntimeForTest(new Promise(() => {})); - - const result = resultDetails( - await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-typescript-load-timeout", - { code: "return 42;", language: "typescript" }, - ), - ); - - expect(result).toMatchObject({ - status: "failed", - code: "timeout", - error: "code mode timeout exceeded", - output: [], - }); - expect(testing.activeRuns.size).toBe(0); - }); - - it("aborts an unfinished TypeScript runtime load without starting a worker", async () => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - testing.setTypescriptRuntimeForTest(new Promise(() => {})); - const controller = new AbortController(); - const resultPromise = expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( - "code-call-typescript-load-abort", - { code: "return 42;", language: "typescript" }, - controller.signal, - ); - - controller.abort(); - - expect(resultDetails(await resultPromise)).toMatchObject({ - status: "failed", - code: "aborted", - error: "code mode execution aborted", - output: [], - }); - expect(testing.activeRuns.size).toBe(0); - }); - - it.each([ - "const fs = require('node:fs'); return fs;", - String.raw`return r\u0065quire('node:fs');`, - "return require?.('node:fs');", - "return (require)('node:fs');", - "return (0, require)('node:fs');", - "const load = require; return load('node:fs');", - "return module.require('node:fs');", - "return process.getBuiltinModule('node:fs');", - "return import('node:fs');", - "return import.meta.url;", - "return `${import('node:fs')}`;", - "return `${require('node:fs')}`;", - "return `${`nested ${import('node:fs')}`}`;", - "return `${`nested ${require('node:fs')}`}`;", - "return `${({ value: import('node:fs') }).value}`;", - "const message = `import('node:fs')`; return require('node:fs');", - "const pattern = /import.meta/; return import('node:fs');", - "let value = 1; return value++ / import('node:fs');", - "let value = 1; return value-- / import('node:fs');", - "const value = { of: 1 }; return value.of / import('node:fs');", - "const value = { return: 1 }; return value.return / import('node:fs');", - "const value = { if() { return 1; } }; return value.if() / import('node:fs');", - "const value = { return: 1 }; return value?.return / import('node:fs') / 1;", - "const value = { return: 1 }; return value?.return / require('node:fs') / 1;", - "const value = { if() { return 1; } }; return value?.if() / import('node:fs');", - "function run() { const await = 1; return await / (globalThis.pending = import('node:fs')); } run(); return globalThis.pending;", - "class Guest { #return = 1; run() { return this.#return / (globalThis.pending = import('node:fs')); } } new Guest().run(); return globalThis.pending;", - ])("rejects module access: %s", async (code) => { - const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( - "code-call-import", - { - code, - }, - ), - ); - - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("module access is disabled"); - }); - - it("enforces output limits on completed exec calls", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - maxOutputBytes: 1024, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const tools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-large", { - code: "return 'x'.repeat(2048);", - }), - ); - - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("output limit exceeded"); - expect(details.code).toBe("output_limit_exceeded"); - }); - - it("enforces output limits before suspending runs", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - maxOutputBytes: 1024, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const tools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const beforeRunCount = testing.activeRuns.size; - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-large-suspend", { - code: "text('x'.repeat(2048)); await yield_control('pause'); return 1;", - }), - ); - - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("output limit exceeded"); - expect(details.code).toBe("output_limit_exceeded"); - expect(testing.activeRuns.size).toBe(beforeRunCount); - }); - - it("enforces the cumulative output limit across yielded waits", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - maxOutputBytes: 1024, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const tools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const first = resultDetails( - await expectDefined(tools[0], "Code Mode exec test invariant").execute( - "code-call-cumulative-output", - { - code: ` - text("a".repeat(600)); - await yield_control("pause"); - text("b".repeat(600)); - return "done"; - `, - }, - ), - ); - - expect(first.status).toBe("waiting"); - expect(first.output).toEqual([{ type: "text", text: "a".repeat(600) }]); - - const second = resultDetails( - await expectDefined(tools[1], "Code Mode wait test invariant").execute( - "code-wait-cumulative-output", - { runId: first.runId }, - ), - ); - - expect(second.status).toBe("failed"); - expect(second.code).toBe("output_limit_exceeded"); - expect(testing.activeRuns.has(first.runId as string)).toBe(false); - }); - - it("enforces output limits before auto-draining namespace calls", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - maxOutputBytes: 1024, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const tools = createCodeModeTools(ctx); - const executeListIssues = vi.fn(async () => jsonResult({ ok: true })); - const listIssues = mcpTool({ - name: "tickets__list", - serverName: "tickets", - toolName: "list", - execute: executeListIssues, - }); - applyCodeModeCatalog({ - tools: [...tools, listIssues], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute( - "code-call-large-namespace", - { - code: 'text("x".repeat(2048)); await MCP.tickets.list({ state: "open" }); return 1;', - }, - ), - ); - - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("output limit exceeded"); - expect(details.code).toBe("output_limit_exceeded"); - expect(executeListIssues).not.toHaveBeenCalled(); - }); - - it("preserves guest output when a run fails", async () => { - const { config, catalogRef, tools } = createCodeModeHarness(); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute( - "code-call-output-before-error", - { - code: 'text("before"); throw new Error("boom");', - }, - ), - ); - - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("Error: boom"); - expect(details.output).toEqual([{ type: "text", text: "before" }]); - }); - - it("classifies snapshot limit failures", async () => { - const config = resolveCodeModeConfig({ - tools: { codeMode: { enabled: true, maxSnapshotBytes: 1024 } }, - } as never); - - const result = await testing.runCodeModeWorker( - { - kind: "exec", - source: 'const value = "x".repeat(100000); await yield_control("pause"); return value;', - config, - catalog: [], - }, - 5000, - ); - - expect(result.status).toBe("failed"); - expect(result).toMatchObject({ - code: "snapshot_limit_exceeded", - error: "code mode snapshot limit exceeded", - }); - }); - - it("terminates hostile infinite loops outside the main event loop", async () => { - const catalogRef = createToolSearchCatalogRef(); - const config = { - tools: { - codeMode: { - enabled: true, - timeoutMs: 100, - }, - }, - } as never; - const ctx = { - config, - runtimeConfig: config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }; - const tools = createCodeModeTools(ctx); - applyCodeModeCatalog({ - tools: [...tools, pluginTool("fake_noop", "Noop")], - config, - sessionId: "session-code-mode", - sessionKey: "agent:main:main", - runId: "run-code-mode", - catalogRef, - }); - - const heartbeat = Promise.resolve("main-event-loop-alive"); - const details = resultDetails( - await expectDefined(tools[0], "tools[0] test invariant").execute("code-call-loop", { - code: "while (true) {}", - }), - ); - - await expect(heartbeat).resolves.toBe("main-event-loop-alive"); - expect(details.status).toBe("failed"); - expect(String(details.error)).toContain("timeout exceeded"); - expect(details.code).toBe("timeout"); - }); - - it("normalizes QuickJS interrupt timeout errors", () => { - expect( - testing.normalizeCodeModeWorkerResult({ - status: "failed", - code: "timeout", - error: "interrupted", - output: [], - }), - ).toMatchObject({ - code: "timeout", - error: "code mode timeout exceeded", - }); - - expect( - testing.normalizeCodeModeWorkerResult({ - status: "failed", - code: "internal_error", - error: "interrupted", - output: [], - }), - ).toMatchObject({ - code: "internal_error", - error: "interrupted", - }); - }); - - it("classifies missing worker runtime as unavailable", async () => { - const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); - const missingWorkerUrl = new URL("./missing-code-mode.worker.js", import.meta.url); - - const result = await testing.runCodeModeWorker( - { - kind: "exec", - source: "return 1;", - config, - catalog: [], - }, - 500, - missingWorkerUrl, - ); - - expect(result.status).toBe("failed"); - expect(result).toMatchObject({ - code: "runtime_unavailable", - }); - }); - - it("classifies nonzero worker exits as unavailable", async () => { - const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); - const exitingWorkerUrl = new URL("data:text/javascript,process.exit(1)"); - - const result = await testing.runCodeModeWorker( - { - kind: "exec", - source: "return 1;", - config, - catalog: [], - }, - 500, - exitingWorkerUrl, - ); - - expect(result.status).toBe("failed"); - expect(result).toMatchObject({ - code: "runtime_unavailable", - }); - }); - - it("classifies clean worker exits without a result as unavailable", async () => { - const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); - const exitingWorkerUrl = new URL("data:text/javascript,"); - - const result = await testing.runCodeModeWorker( - { - kind: "exec", - source: "return 1;", - config, - catalog: [], - }, - 5_000, - exitingWorkerUrl, - ); - - expect(result).toMatchObject({ - status: "failed", - code: "runtime_unavailable", - error: "code mode worker exited with code 0 before returning a result", - }); - }); - - it("does not classify guest interrupted errors as timeouts", async () => { - const config = resolveCodeModeConfig({ tools: { codeMode: true } } as never); - - const result = await testing.runCodeModeWorker( - { - kind: "exec", - source: 'throw new Error("interrupted");', - config, - catalog: [], - }, - 10_000, - ); - - expect(result.status).toBe("failed"); - // A guest error whose message happens to be "interrupted" must stay - // internal_error and not be misclassified as a QuickJS interrupt/timeout. - expect(result).toMatchObject({ code: "internal_error" }); - if (result.status === "failed") { - expect(result.error).toContain("interrupted"); - } - }); }); -/* oxlint-disable max-lines -- TODO: split this grandfathered oversized file. */ diff --git a/src/agents/code-mode.typescript.test.ts b/src/agents/code-mode.typescript.test.ts new file mode 100644 index 000000000000..3abd6252934b --- /dev/null +++ b/src/agents/code-mode.typescript.test.ts @@ -0,0 +1,180 @@ +/** Tests Code Mode TypeScript execution. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { applyCodeModeCatalog } from "./code-mode.js"; +import { + resetCodeModeTestState, + pluginTool, + resultDetails, + createCodeModeHarness, + runUntilCompleted, + testing, +} from "./code-mode.test-support.js"; + +describe("Code Mode TypeScript execution", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("supports TypeScript source transform", async () => { + testing.setTypescriptRuntimeForTest({ + ...(await import("typescript")), + transpileModule: vi.fn((code: string) => ({ + outputText: code.replace(": number", ""), + diagnostics: [], + })), + ScriptTarget: { ES2022: 9 }, + ModuleKind: { ESNext: 99 }, + ImportsNotUsedAsValues: { Remove: 0 }, + DiagnosticCategory: { Error: 1 }, + flattenDiagnosticMessageText: (message: unknown) => String(message), + } as never); + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const details = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + language: "typescript", + code: ` + const value: number = 40 + 2; + return { value }; + `, + }); + + expect(details.status).toBe("completed"); + expect(details.value).toEqual({ value: 42 }); + + const moduleShapedTypeScript = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + language: "typescript", + code: "const value: number = 42; return `import('node:fs') ${value}`;", + }); + + expect(moduleShapedTypeScript.status).toBe("completed"); + expect(moduleShapedTypeScript.value).toBe("import('node:fs') 42"); + + const moduleShapedTypeScriptRegex = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + language: "typescript", + code: 'const value: number = 42; return /import.meta/.test("import.meta");', + }); + + expect(moduleShapedTypeScriptRegex.status).toBe("completed"); + expect(moduleShapedTypeScriptRegex.value).toBe(true); + + for (const moduleAccess of ["import('node:fs')", "require('node:fs')"]) { + const unicodeModuleAccess = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + `code-call-typescript-unicode-${moduleAccess.startsWith("import") ? "import" : "require"}`, + { + language: "typescript", + code: `const value: number = 1; const padding = "${"😀".repeat(96)}"; return ${moduleAccess};`, + }, + ), + ); + + expect(unicodeModuleAccess.status).toBe("failed"); + expect(unicodeModuleAccess.code).toBe("invalid_input"); + expect(unicodeModuleAccess.error).toContain("module access is disabled"); + } + + const commandLikeTypeScript = await runUntilCompleted({ + execTool: expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"), + waitTool: expectDefined(codeModeTools[1], "codeModeTools[1] test invariant"), + language: "typescript", + code: "node -1; var node: number = 7; return node;", + }); + + expect(commandLikeTypeScript.status).toBe("completed"); + expect(commandLikeTypeScript.value).toBe(7); + + const typedShell = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-typescript-shell-source", + { code: "pwd", language: "typescript" }, + ), + ); + + expect(typedShell.status).toBe("failed"); + expect(typedShell.code).toBe("invalid_input"); + expect(typedShell.error).toMatch(/JavaScript or TypeScript, not shell commands/); + expect(testing.activeRuns.size).toBe(0); + }); + + it("times out an unfinished TypeScript runtime load without starting a worker", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + (config as { tools: { codeMode: unknown } }).tools.codeMode = { + enabled: true, + timeoutMs: 100, + }; + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + testing.setTypescriptRuntimeForTest(new Promise(() => {})); + + const result = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-typescript-load-timeout", + { code: "return 42;", language: "typescript" }, + ), + ); + + expect(result).toMatchObject({ + status: "failed", + code: "timeout", + error: "code mode timeout exceeded", + output: [], + }); + expect(testing.activeRuns.size).toBe(0); + }); + + it("aborts an unfinished TypeScript runtime load without starting a worker", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + testing.setTypescriptRuntimeForTest(new Promise(() => {})); + const controller = new AbortController(); + const resultPromise = expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-typescript-load-abort", + { code: "return 42;", language: "typescript" }, + controller.signal, + ); + + controller.abort(); + + expect(resultDetails(await resultPromise)).toMatchObject({ + status: "failed", + code: "aborted", + error: "code mode execution aborted", + output: [], + }); + expect(testing.activeRuns.size).toBe(0); + }); +}); diff --git a/src/agents/code-mode.wait.test.ts b/src/agents/code-mode.wait.test.ts new file mode 100644 index 000000000000..7ee79b287d33 --- /dev/null +++ b/src/agents/code-mode.wait.test.ts @@ -0,0 +1,583 @@ +/** Tests Code Mode wait, scope, and suspended runs. */ + +import { expectDefined } from "@openclaw/normalization-core"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { runWithAgentToolExecutionContext } from "../../packages/agent-core/src/tool-execution-context.js"; +import { applyCodeModeCatalog, createCodeModeTools } from "./code-mode.js"; +import { + resetCodeModeTestState, + pluginTool, + pluginToolWithExecute, + resultDetails, + createCodeModeHarness, + testing, +} from "./code-mode.test-support.js"; +import { createToolSearchCatalogRef } from "./tool-search.js"; + +describe("Code Mode wait, scope, and suspended runs", () => { + beforeEach(() => { + vi.useRealTimers(); + }); + + afterEach(() => { + vi.useRealTimers(); + resetCodeModeTestState(); + }); + + it("marks yield suspensions and resumes the snapshot with wait", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-yield", + { + restartSafe: true, + code: ` + text("before"); + await yield_control("pause"); + text("after"); + return "done"; + `, + }, + ), + ); + + expect(first.status).toBe("waiting"); + expect(first.reason).toBe("yield"); + expect(first.replaySafe).toBe(true); + expect(first.output).toEqual([{ type: "text", text: "before" }]); + + const runId = first.runId; + expect(typeof runId).toBe("string"); + const resumed = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-yield", + { runId }, + ), + ); + + expect(resumed.status).toBe("completed"); + expect(resumed.value).toBe("done"); + expect(resumed.output).toEqual([{ type: "text", text: "after" }]); + }); + + it("delivers each yielded output block exactly once across repeated waits", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const execTool = expectDefined(codeModeTools[0], "Code Mode exec test invariant"); + const waitTool = expectDefined(codeModeTools[1], "Code Mode wait test invariant"); + const first = resultDetails( + await execTool.execute("code-call-incremental-output", { + code: ` + text("phase 1"); + await yield_control("first pause"); + text("phase 2"); + await yield_control("second pause"); + text("phase 3"); + return "done"; + `, + }), + ); + + expect(first.status).toBe("waiting"); + expect(first.output).toEqual([{ type: "text", text: "phase 1" }]); + + const second = resultDetails( + await waitTool.execute("code-wait-incremental-output-1", { runId: first.runId }), + ); + + expect(second.status).toBe("waiting"); + expect(second.output).toEqual([{ type: "text", text: "phase 2" }]); + + const third = resultDetails( + await waitTool.execute("code-wait-incremental-output-2", { runId: second.runId }), + ); + + expect(third.status).toBe("completed"); + expect(third.value).toBe("done"); + expect(third.output).toEqual([{ type: "text", text: "phase 3" }]); + }); + + it("returns only newly emitted output when a resumed guest fails", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-incremental-failure", + { + code: ` + text("before pause"); + await yield_control("pause"); + text("before failure"); + throw new Error("resumed failure"); + `, + }, + ), + ); + + expect(first.status).toBe("waiting"); + expect(first.output).toEqual([{ type: "text", text: "before pause" }]); + + const second = resultDetails( + await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( + "code-wait-incremental-failure", + { runId: first.runId }, + ), + ); + + expect(second.status).toBe("failed"); + expect(second.error).toContain("resumed failure"); + expect(second.output).toEqual([{ type: "text", text: "before failure" }]); + expect(testing.activeRuns.has(first.runId as string)).toBe(false); + }); + + it("preserves the original exec identity for tool calls after yield and wait", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + const target = pluginTool("fake_resumed_identity", "Resumed identity helper"); + applyCodeModeCatalog({ + tools: [...codeModeTools, target], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const suspended = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-original-parent", + { + code: 'await yield_control("pause"); return await tools.callValue("fake_resumed_identity", {});', + }, + ), + ); + expect(suspended.status).toBe("waiting"); + + const resumed = resultDetails( + await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( + "code-wait-different-parent", + { runId: suspended.runId }, + ), + ); + + expect(resumed.status).toBe("completed"); + expect(target.execute).toHaveBeenCalledOnce(); + expect(vi.mocked(target.execute).mock.calls[0]?.[0]).toContain("code-call-original-parent"); + expect(vi.mocked(target.execute).mock.calls[0]?.[0]).not.toContain( + "code-wait-different-parent", + ); + }); + + it("allocates distinct replay identities when a later turn reuses a tool-call id", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + const execTool = expectDefined(codeModeTools[0], "codeModeTools[0] test invariant"); + const input = { code: 'await yield_control("pause"); return "done";' }; + const executionContext = (turnId: string) => + ({ + assistantMessage: { responseId: " ", turnId }, + toolCall: { type: "toolCall", id: "reused-call-id", name: "exec", arguments: input }, + }) as never; + + const first = resultDetails( + await runWithAgentToolExecutionContext(executionContext("response-turn-1"), () => + execTool.execute("reused-call-id", input), + ), + ); + const second = resultDetails( + await runWithAgentToolExecutionContext(executionContext("response-turn-2"), () => + execTool.execute("reused-call-id", input), + ), + ); + + expect(first.status).toBe("waiting"); + expect(second.status).toBe("waiting"); + expect(second.runId).not.toBe(first.runId); + expect(testing.activeRuns.size).toBe(2); + expect(new Set([...testing.activeRuns.values()].map((state) => state.replayId)).size).toBe(2); + }); + + it("fails yield suspension when snapshot expiry would exceed the Date range", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + const nowSpy = vi.spyOn(Date, "now").mockReturnValue(8_640_000_000_000_000); + let details: Record; + try { + details = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-yield-overflow", + { + code: 'await yield_control("pause"); return "done";', + }, + ), + ); + } finally { + nowSpy.mockRestore(); + } + + expect(details.status).toBe("failed"); + expect(details.error).toBe("code mode run expiry is unavailable."); + expect(testing.activeRuns.size).toBe(0); + }); + + it("expires suspended runs with invalid expiry timestamps", async () => { + const { tools: codeModeTools } = createCodeModeHarness(); + testing.activeRuns.set("invalid-expiry-run", { + expiresAt: 8_640_000_000_000_001, + } as never); + + await expect( + expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-invalid-expiry", + { runId: "invalid-expiry-run" }, + ), + ).rejects.toThrow("code mode run is unavailable or expired"); + expect(testing.activeRuns.has("invalid-expiry-run")).toBe(false); + }); + + it("rejects wait calls from a different session scope", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-wrong-session", + { + code: 'await yield_control("pause"); return "done";', + }, + ), + ); + expect(first.status).toBe("waiting"); + const otherWaitTool = expectDefined( + createCodeModeTools({ + config, + runtimeConfig: config, + sessionId: "other-session", + sessionKey: "agent:other:main", + runId: "run-code-mode", + catalogRef, + })[1], + 'createCodeModeTools({ config, runtimeConfig: config, sessionId: "othe... test invariant', + ); + + await expect( + otherWaitTool.execute("code-wait-wrong-session", { runId: first.runId }), + ).rejects.toThrow("different session"); + }); + + it.each(["runId", "sessionId", "sessionKey", "agentId"] as const)( + "rejects suspended-run callers missing the owner %s", + async (missingIdentity) => { + const { + config, + catalogRef, + ctx, + tools: codeModeTools, + } = createCodeModeHarness({ + agentId: "owner", + }); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: ctx.sessionId, + sessionKey: ctx.sessionKey, + agentId: ctx.agentId, + runId: ctx.runId, + catalogRef, + }); + + const suspended = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-scoped-owner", + { code: 'await yield_control("pause"); return "owner-secret";' }, + ), + ); + expect(suspended.status).toBe("waiting"); + + const missingIdentityWait = expectDefined( + createCodeModeTools({ + config, + runtimeConfig: config, + catalogRef, + ...(missingIdentity === "runId" ? {} : { runId: ctx.runId }), + ...(missingIdentity === "sessionId" ? {} : { sessionId: ctx.sessionId }), + ...(missingIdentity === "sessionKey" ? {} : { sessionKey: ctx.sessionKey }), + ...(missingIdentity === "agentId" ? {} : { agentId: ctx.agentId }), + })[1], + "Unscoped Code Mode wait test invariant", + ); + + await expect( + missingIdentityWait.execute("code-wait-missing-owner", { runId: suspended.runId }), + ).rejects.toThrow(missingIdentity === "runId" ? "different agent run" : "different session"); + expect(testing.activeRuns.has(suspended.runId as string)).toBe(true); + + const rightfulResult = resultDetails( + await expectDefined(codeModeTools[1], "Owner Code Mode wait test invariant").execute( + "code-wait-rightful-owner", + { runId: suspended.runId }, + ), + ); + expect(rightfulResult.status).toBe("completed"); + expect(rightfulResult.value).toBe("owner-secret"); + }, + ); + + it("rejects concurrent waits for the same suspended run", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + timeoutMs: 500, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const codeModeTools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [ + ...codeModeTools, + pluginToolWithExecute( + "fake_slow", + "Slow helper", + async () => await new Promise(() => {}), + ), + ], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-concurrent-wait", + { + code: "await tools.fake_slow({}); return 'done';", + }, + ), + ); + expect(first.status).toBe("waiting"); + + const firstWait = expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-concurrent-a", + { + runId: first.runId, + }, + ); + await expect( + expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-concurrent-b", + { runId: first.runId }, + ), + ).rejects.toThrow("already being resumed"); + const stillWaiting = resultDetails(await firstWait); + + expect(stillWaiting.status).toBe("waiting"); + expect(stillWaiting.runId).toBe(first.runId); + }); + + it("resumes and reparks a yielding run at the suspended-run capacity limit", async () => { + const { config, catalogRef, tools: codeModeTools } = createCodeModeHarness(); + applyCodeModeCatalog({ + tools: [...codeModeTools, pluginTool("fake_noop", "Noop")], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "Code Mode exec test invariant").execute( + "code-call-at-capacity", + { + code: ` + await yield_control("first"); + await yield_control("second"); + return "done"; + `, + }, + ), + ); + expect(first.status).toBe("waiting"); + const firstRunId = first.runId; + expect(typeof firstRunId).toBe("string"); + if (typeof firstRunId !== "string") { + throw new Error("expected a parked Code Mode run"); + } + const parked = testing.activeRuns.get(firstRunId); + expect(parked).toBeDefined(); + if (!parked) { + throw new Error("expected a parked Code Mode snapshot"); + } + + // Inert snapshots occupy real capacity without starting 63 extra workers. + for (let index = 0; index < 63; index += 1) { + const runId = `cm_code_mode_capacity_${index}`; + testing.activeRuns.set(runId, { ...parked, runId, pending: [] }); + } + + const second = resultDetails( + await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( + "code-wait-at-capacity", + { runId: firstRunId }, + ), + ); + + expect(second.status).toBe("waiting"); + expect(second.reason).toBe("yield"); + expect(testing.activeRuns.size).toBe(64); + + const completed = resultDetails( + await expectDefined(codeModeTools[1], "Code Mode wait test invariant").execute( + "code-wait-after-capacity", + { runId: second.runId }, + ), + ); + + expect(completed).toMatchObject({ status: "completed", value: "done" }); + expect(testing.activeRuns.size).toBe(63); + }); + + it("reports only unsettled pending tool calls when wait times out", async () => { + const catalogRef = createToolSearchCatalogRef(); + const config = { + tools: { + codeMode: { + enabled: true, + timeoutMs: 500, + }, + }, + } as never; + const ctx = { + config, + runtimeConfig: config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }; + const codeModeTools = createCodeModeTools(ctx); + applyCodeModeCatalog({ + tools: [ + ...codeModeTools, + pluginTool("fake_fast", "Fast helper"), + pluginToolWithExecute( + "fake_slow", + "Slow helper", + async () => await new Promise(() => {}), + ), + ], + config, + sessionId: "session-code-mode", + sessionKey: "agent:main:main", + runId: "run-code-mode", + catalogRef, + }); + + const first = resultDetails( + await expectDefined(codeModeTools[0], "codeModeTools[0] test invariant").execute( + "code-call-timeout", + { + code: ` + text("before timeout"); + const fast = tools.fake_fast({}); + const slow = tools.fake_slow({}); + await fast; + await slow; + return "done"; + `, + }, + ), + ); + expect(first.status).toBe("waiting"); + expect(first.output).toEqual([{ type: "text", text: "before timeout" }]); + expect(first.pendingToolCalls).toEqual([expect.objectContaining({ method: "callValue" })]); + const runId = first.runId; + expect(typeof runId).toBe("string"); + if (typeof runId !== "string") { + throw new Error("expected code mode run id"); + } + + const activeRun = testing.activeRuns.get(runId); + expect(activeRun).toBeDefined(); + activeRun!.config.timeoutMs = 100; + + const second = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-timeout", + { runId }, + ), + ); + + expect(second.status).toBe("waiting"); + expect(second.output).toEqual([]); + expect(second.pendingToolCalls).toEqual([expect.objectContaining({ method: "callValue" })]); + + const third = resultDetails( + await expectDefined(codeModeTools[1], "codeModeTools[1] test invariant").execute( + "code-wait-timeout-again", + { runId }, + ), + ); + + expect(third.status).toBe("waiting"); + expect(third.output).toEqual([]); + expect(third.pendingToolCalls).toEqual([expect.objectContaining({ method: "callValue" })]); + }); +});