Files
openclaw/src/agents/cli-runner.spawn.test.ts
T
Feng 7dbe21b9cf fix(agents): hydrate CLI images from agent workspaces (#122684)
* fix(agents): hydrate CLI images from agent workspace

* fix(agents): preserve resolved CLI workspace owner

* fix(agents): keep workspace owner in prepared params

* fix(agents): resolve CLI owner before preparation

* fix(agents): preserve CLI runtime policy owner

---------

Co-authored-by: Adkid-Zephyr <169631528+Adkid-Zephyr@users.noreply.github.com>
Co-authored-by: FullerStackDev <263060202+fuller-stack-dev@users.noreply.github.com>
2026-08-12 20:18:15 -07:00

2686 lines
90 KiB
TypeScript

/** Tests CLI runner process spawning, logging, diagnostics, and live-session paths. */
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { expectDefined } from "@openclaw/normalization-core";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { createSolidPngBuffer } from "../../test/helpers/image-fixtures.js";
import { useAutoCleanupTempDirTracker } from "../../test/helpers/temp-dir.js";
import {
markMcpLoopbackToolCallFinished,
markMcpLoopbackToolCallStarted,
recordMcpLoopbackToolCallResult,
} from "../gateway/mcp-http.loopback-runtime.js";
import { invokeNodeClaudeCliRun } from "../gateway/node-agent-cli-runtime.js";
import { onAgentEvent, resetAgentEventsForTest } from "../infra/agent-events.js";
import {
onTrustedToolExecutionEvent,
setDiagnosticsEnabledForProcess,
waitForDiagnosticEventsDrained,
} from "../infra/diagnostic-events.js";
import {
resetDiagnosticRunActivityForTest,
startDiagnosticRunActivityTracking,
} from "../logging/diagnostic-run-activity.js";
import type { getProcessSupervisor } from "../process/supervisor/index.js";
import { createTestAdmittedRunContext } from "./admitted-run-context.test-support.js";
import {
buildClaudeLiveRunContext,
buildPreparedCliRunContext,
captureModelCallDiagnostics,
createClaudeInputStartedEvent,
expectPathMissing,
expectRejectsWithFields,
expectModelCallTypes,
mockCallArg,
mockClaudeLiveRun,
requireArgAfter,
requireRecord,
requireRegexMatch,
} from "./cli-runner.test-helpers.js";
import { resetClaudeLiveSessionsForTest } from "./cli-runner/claude-live-session.test-support.js";
import {
attachCliMessagingDeliveryEvidence,
getCliMessagingDeliveryEvidence,
} from "./cli-runner/delivery-evidence.js";
import { executePreparedCliRun } from "./cli-runner/execute.js";
import {
buildCliEnvAuthLog,
buildCliExecLogLine,
createManagedRun,
setCliRunnerExecuteTestDeps,
supervisorSpawnMock,
} from "./cli-runner/execute.test-support.js";
import { buildCliAgentSystemPrompt, writeCliSystemPromptFile } from "./cli-runner/helpers.js";
import { cliBackendLog, formatCliBackendOutputDigest } from "./cli-runner/log.js";
import type { PreparedCliRunContext } from "./cli-runner/types.js";
// Approval behavior is injected below; loading its gateway/tool graph here is incidental.
vi.mock("./bash-tools.exec-approval-request.js", () => ({
registerExecApprovalRequestForHostOrThrow: vi.fn(),
resolveRegisteredExecApprovalDecision: vi.fn(),
}));
// Gateway unit coverage owns quiet-admission timing. These spawn cases only
// need to drain calls already in flight, so skip the repeated 250 ms quiet window.
vi.mock("../gateway/mcp-http.loopback-runtime.js", async (importOriginal) => {
const actual = await importOriginal<typeof import("../gateway/mcp-http.loopback-runtime.js")>();
return {
...actual,
waitForMcpLoopbackToolCallCaptureIdle: (
captureKey: string,
options: Parameters<typeof actual.waitForMcpLoopbackToolCallCaptureIdle>[1],
) =>
actual.waitForMcpLoopbackToolCallCaptureIdle(captureKey, {
...options,
admissionGraceMs: 0,
}),
};
});
function emitClaudeInputStarted(stdout: ((chunk: string) => void) | undefined, data: string): void {
const event = createClaudeInputStartedEvent(data);
if (event) {
stdout?.(`${JSON.stringify(event)}\n`);
}
}
beforeEach(() => {
setDiagnosticsEnabledForProcess(true);
resetAgentEventsForTest();
resetDiagnosticRunActivityForTest();
startDiagnosticRunActivityTracking();
resetClaudeLiveSessionsForTest();
setCliRunnerExecuteTestDeps({
writeCliSystemPromptFile,
invokeNodeClaudeCliRun,
registerExecApprovalRequestForHostOrThrow: async () => {
throw new Error("unexpected exec approval registration");
},
resolveRegisteredExecApprovalDecision: async () => {
throw new Error("unexpected exec approval resolution");
},
});
supervisorSpawnMock.mockClear();
});
afterEach(() => {
vi.restoreAllMocks();
vi.useRealTimers();
resetDiagnosticRunActivityForTest();
resetClaudeLiveSessionsForTest();
});
const CLAUDE_OK_JSONL = `${JSON.stringify({ type: "result", result: "ok" })}\n`;
const GEMINI_OK_JSONL = `${[
JSON.stringify({ type: "message", role: "assistant", content: "ok", delta: true }),
JSON.stringify({ type: "result", status: "success" }),
].join("\n")}\n`;
const tempDirs = useAutoCleanupTempDirTracker(afterEach);
function mockSuccessfulCliRun(stdout = "ok") {
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
}
async function createCliPackageFixture(version: string): Promise<{
root: string;
entrypoint: string;
}> {
const root = tempDirs.make("openclaw-cli-version-gate-");
const entrypoint = path.join(root, "bin", "cli.js");
await fs.mkdir(path.dirname(entrypoint), { recursive: true });
await fs.writeFile(
path.join(root, "package.json"),
`${JSON.stringify({ name: "@fixture/versioned-cli", version })}\n`,
);
await fs.writeFile(entrypoint, `#!${process.execPath}\n`, { mode: 0o755 });
await fs.chmod(entrypoint, 0o755);
return { root, entrypoint };
}
describe("runCliAgent spawn path", () => {
it("hydrates a session-key-owned agent workspace image before spawning the CLI", async () => {
const stateDir = tempDirs.make("openclaw-cli-agent-image-");
const workspaceDir = path.join(stateDir, "workspace-arthur");
const imagePath = path.join(workspaceDir, "media", "inbound", "photo.png");
const image = createSolidPngBuffer(1, 1, { r: 255, g: 0, b: 0 });
await fs.mkdir(path.dirname(imagePath), { recursive: true });
await fs.writeFile(imagePath, image);
vi.stubEnv("OPENCLAW_STATE_DIR", stateDir);
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
const context = buildPreparedCliRunContext({
sessionKey: "agent:arthur:main",
agentId: "arthur",
workspaceDir,
config: {
agents: { entries: { arthur: { default: true, workspace: workspaceDir } } },
},
backend: { imageArg: "--image" },
});
context.params.media = [{ path: imagePath, contentType: "image/png" }];
await expect(executePreparedCliRun(context)).resolves.toMatchObject({ text: "ok" });
const spawn = requireRecord(mockCallArg(supervisorSpawnMock), "CLI spawn");
const hydratedPath = requireArgAfter(spawn.argv as string[], "--image");
await expect(fs.readFile(hydratedPath)).resolves.toEqual(image);
});
it("formats output digests without logging response content", () => {
expect(formatCliBackendOutputDigest("one")).toBe("outBytes=3 outHash=7692c3ad3540");
expect(formatCliBackendOutputDigest("∑")).toBe("outBytes=3 outHash=be27c7179a61");
});
it("formats redacted CLI resume diagnostics without exposing raw session ids", () => {
const logLine = buildCliExecLogLine({
provider: "claude-cli",
model: "claude-opus-4-7",
promptChars: 42,
trigger: "heartbeat",
useResume: true,
cliSessionId: "claude-session-secret",
resolvedSessionId: "claude-session-secret",
reusableSession: { mode: "reuse", sessionId: "claude-session-secret" },
hasHistoryPrompt: false,
});
expect(logLine).toContain("trigger=heartbeat");
expect(logLine).toContain("useResume=true");
expect(logLine).toContain("session=present");
expect(logLine).toContain("reuse=reusable");
expect(logLine).toContain("historyPrompt=none");
expect(logLine).not.toContain("claude-session-secret");
});
it("formats soft-resume drift in CLI resume diagnostics", () => {
const logLine = buildCliExecLogLine({
provider: "claude-cli",
model: "claude-opus-4-7",
promptChars: 42,
trigger: "user",
useResume: true,
cliSessionId: "claude-session-secret",
resolvedSessionId: "claude-session-secret",
reusableSession: {
mode: "reuse-with-drift",
sessionId: "claude-session-secret",
drift: { reasons: ["system-prompt"] },
},
hasHistoryPrompt: false,
});
expect(logLine).toContain("reuse=reusable-drift:system-prompt");
expect(logLine).not.toContain("claude-session-secret");
});
it("streams a node-placed Claude resume through the normal JSONL parser", async () => {
const writeSystemPrompt = vi.fn(writeCliSystemPromptFile);
let toolAvailability: unknown = "unset";
const invokeNode = vi.fn(async (params: Parameters<typeof invokeNodeClaudeCliRun>[0]) => {
const jsonl = [
JSON.stringify({ type: "system", subtype: "init", session_id: "forked-node-session" }),
JSON.stringify({
type: "result",
session_id: "forked-node-session",
result: "node answer",
}),
"",
].join("\n");
params.onProgress(jsonl.slice(0, 40));
params.onProgress(jsonl.slice(40));
return {
ok: true,
payloadJSON: JSON.stringify({ exitCode: 0, stderrTail: "", truncated: false }),
};
});
setCliRunnerExecuteTestDeps({
writeCliSystemPromptFile: writeSystemPrompt,
invokeNodeClaudeCliRun: invokeNode,
});
const context = buildClaudeLiveRunContext({
model: "claude-opus-4-8",
runId: "run-node-claude",
prompt: "current turn",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
execCwd: "/work/on-node",
},
backend: {
args: [
"-p",
"--output-format",
"stream-json",
"--permission-mode",
"bypassPermissions",
"--strict-mcp-config",
"--mcp-config",
"/tmp/gateway-mcp.json",
"--allowedTools",
"mcp__openclaw__*",
],
resumeArgs: [
"-p",
"--output-format",
"stream-json",
"--permission-mode",
"bypassPermissions",
"--strict-mcp-config",
"--mcp-config",
"/tmp/gateway-mcp.json",
"--allowedTools",
"mcp__openclaw__*",
"--resume",
"{sessionId}",
],
forkArg: "--fork-session",
env: { ANTHROPIC_API_KEY: "configured-backend-key" },
clearEnv: ["ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"],
systemPromptWhen: "always",
},
preparedEnv: { CLAUDE_CODE_OAUTH_TOKEN_FILE_DESCRIPTOR: "3" },
resolveExecutionArgs: (execution) => {
toolAvailability = execution.toolAvailability;
return [...execution.baseArgs];
},
cliToolAvailability: { native: [], openClaw: ["message"] },
});
context.preparedBackend.secretInput = {
fd: 3,
fingerprint: "selected-node-token-fingerprint",
createData: () => Buffer.from("selected-node-token"),
};
context.openClawHistoryPrompt = "gateway transcript reseed";
context.claudeSkillsPluginArgs = ["--plugin-dir", "/tmp/gateway-skills"];
context.params.forkCliSessionOnResume = true;
context.params.claimCliSessionFork = vi.fn(async () => true);
context.params.persistCliSessionForkSuccessor = vi.fn(async () => {});
const output = await executePreparedCliRun(context, "source-node-session");
expect(output).toMatchObject({ text: "node answer", sessionId: "forked-node-session" });
// Node runs keep the gateway's native tool policy; loopback MCP tools do
// not exist on the node so the OpenClaw list is projected empty.
expect(toolAvailability).toEqual({ native: [], openClaw: [], mcp: [] });
expect(writeSystemPrompt).not.toHaveBeenCalled();
expect(supervisorSpawnMock).not.toHaveBeenCalled();
expect(invokeNode).toHaveBeenCalledWith(
expect.objectContaining({
nodeId: "node-a",
cwd: "/work/on-node",
stdin: "current turn",
argv: expect.arrayContaining(["--resume", "source-node-session", "--fork-session"]),
systemPrompt: "You are a helpful assistant.",
env: { CLAUDE_CODE_OAUTH_TOKEN: "selected-node-token" },
clearEnv: ["ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"],
}),
);
expect(invokeNode.mock.calls[0]?.[0].env).not.toHaveProperty("ANTHROPIC_API_KEY");
expect(invokeNode.mock.calls[0]?.[0].env).not.toHaveProperty(
"CLAUDE_CODE_SUBPROCESS_ENV_SCRUB",
);
const argv = invokeNode.mock.calls[0]?.[0].argv ?? [];
expect(argv).not.toContain("--mcp-config");
expect(argv).not.toContain("--permission-mode");
expect(argv).not.toContain("bypassPermissions");
expect(argv).not.toContain("--strict-mcp-config");
expect(argv).not.toContain("--allowedTools");
expect(argv).not.toContain("--plugin-dir");
expect(argv).not.toContain("--append-system-prompt");
expect(argv).not.toContain("--append-system-prompt-file");
expect(invokeNode.mock.calls[0]?.[0].stdin).not.toContain("gateway transcript reseed");
expect(context.params.persistCliSessionForkSuccessor).toHaveBeenCalledWith(
"forked-node-session",
);
});
it("surfaces a node-placed Claude synthetic empty terminal through the shared parser", async () => {
const invokeNode = vi.fn(async (params: Parameters<typeof invokeNodeClaudeCliRun>[0]) => {
params.onProgress(
[
JSON.stringify({
type: "assistant",
message: {
model: "<synthetic>",
role: "assistant",
content: [{ type: "text", text: "No response requested." }],
},
}),
JSON.stringify({
type: "result",
subtype: "success",
session_id: "node-synthetic-empty",
result: "",
}),
"",
].join("\n"),
);
return {
ok: true,
payloadJSON: JSON.stringify({ exitCode: 0, stderrTail: "", truncated: false }),
};
});
setCliRunnerExecuteTestDeps({ invokeNodeClaudeCliRun: invokeNode });
const context = buildClaudeLiveRunContext({
model: "claude-opus-4-8",
runId: "run-node-synthetic-empty",
prompt: "current turn",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
});
await expect(executePreparedCliRun(context)).rejects.toMatchObject({
name: "FailoverError",
reason: "format",
code: "cli_synthetic_no_response",
});
expect(invokeNode).toHaveBeenCalledOnce();
expect(supervisorSpawnMock).not.toHaveBeenCalled();
});
it("rejects a truncated node stream that lost the terminal result", async () => {
const invokeNode = vi.fn(async (params: Parameters<typeof invokeNodeClaudeCliRun>[0]) => {
params.onProgress(
`${JSON.stringify({ type: "system", subtype: "init", session_id: "trunc-node-session" })}\n`,
);
params.onProgress('{"type":"assistant","message":{"content":[{"type":"te');
return {
ok: true,
payloadJSON: JSON.stringify({ exitCode: 0, stderrTail: "", truncated: true }),
};
});
setCliRunnerExecuteTestDeps({ invokeNodeClaudeCliRun: invokeNode });
const context = buildClaudeLiveRunContext({
model: "claude-opus-4-8",
prompt: "current turn",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
backend: {
args: ["-p", "--output-format", "stream-json"],
resumeArgs: ["-p", "--output-format", "stream-json", "--resume", "{sessionId}"],
forkArg: "--fork-session",
env: { ANTHROPIC_API_KEY: "gateway-backend-key" },
systemPromptWhen: "always",
},
});
await expect(executePreparedCliRun(context, undefined)).rejects.toThrow(
/truncated the Claude CLI stream before the terminal result/,
);
expect(invokeNode.mock.calls[0]?.[0].env).toBeUndefined();
expect(invokeNode.mock.calls[0]?.[0].clearEnv).toBeUndefined();
});
it("cancels a node-placed Claude process when the run aborts", async () => {
const controller = new AbortController();
const invokeNode = vi.fn(
async (params: Parameters<typeof invokeNodeClaudeCliRun>[0]) =>
await new Promise<Awaited<ReturnType<typeof invokeNodeClaudeCliRun>>>((resolve) => {
params.signal?.addEventListener(
"abort",
() =>
resolve({
ok: false,
error: { code: "ABORTED", message: "node invoke cancelled" },
}),
{ once: true },
);
}),
);
setCliRunnerExecuteTestDeps({ invokeNodeClaudeCliRun: invokeNode });
const context = buildPreparedCliRunContext({
model: "claude-opus-4-8",
runId: "run-node-abort",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
});
context.params.abortSignal = controller.signal;
const diagnostics = captureModelCallDiagnostics("run-node-abort");
try {
const run = executePreparedCliRun(context);
await vi.waitFor(() => expect(invokeNode).toHaveBeenCalledOnce());
controller.abort();
await expect(run).rejects.toMatchObject({ name: "AbortError" });
await waitForDiagnosticEventsDrained();
expect(invokeNode.mock.calls[0]?.[0].signal?.aborted).toBe(true);
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.error"]);
expect(diagnostics.events[1]?.event).toMatchObject({
transport: "paired-node-cli",
observationUnit: "turn",
failureKind: "aborted",
});
} finally {
diagnostics.stop();
}
});
it("uses the canonical exec approval flow before retrying a node Claude run", async () => {
const plan = {
argv: ["/trusted/claude", "-p"],
cwd: "/work/on-node",
commandText: "/trusted/claude -p",
agentId: "main",
sessionKey: "agent:main:catalog-adopt:claude:node",
};
const invokeNode = vi.fn(async (input: Parameters<typeof invokeNodeClaudeCliRun>[0]) => {
if (invokeNode.mock.calls.length === 1) {
return {
ok: true,
payloadJSON: JSON.stringify({
approvalRequired: true,
systemRunPlan: plan,
security: "allowlist",
ask: "on-miss",
}),
};
}
input.onProgress(
`${JSON.stringify({ type: "result", session_id: "approved-node-session", result: "ok" })}\n`,
);
return {
ok: true,
payloadJSON: JSON.stringify({ exitCode: 0, stderrTail: "", truncated: false }),
};
});
const registerApproval = vi.fn(async () => ({
id: "approval-1",
expiresAtMs: Date.now() + 1_000,
}));
const resolveApproval = vi.fn(async () => {
await new Promise((resolve) => {
setTimeout(resolve, 20);
});
return "allow-once";
});
setCliRunnerExecuteTestDeps({
invokeNodeClaudeCliRun: invokeNode,
registerExecApprovalRequestForHostOrThrow: registerApproval,
resolveRegisteredExecApprovalDecision: resolveApproval,
});
const context = buildPreparedCliRunContext({
model: "claude-opus-4-8",
runId: "run-node-approval",
sessionKey: plan.sessionKey,
agentId: "main",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
execCwd: plan.cwd,
},
timeoutMs: 500,
});
await expect(executePreparedCliRun(context)).resolves.toMatchObject({
text: "ok",
sessionId: "approved-node-session",
});
expect(registerApproval).toHaveBeenCalledWith(
expect.objectContaining({
systemRunPlan: plan,
host: "node",
nodeId: "node-a",
security: "allowlist",
ask: "on-miss",
}),
);
expect(resolveApproval).toHaveBeenCalledWith(
expect.objectContaining({ approvalId: "approval-1" }),
);
expect(invokeNode).toHaveBeenCalledTimes(2);
expect(invokeNode.mock.calls[1]?.[0]).toMatchObject({
approvalDecision: "allow-once",
systemRunPlan: plan,
});
expect(invokeNode.mock.calls[1]?.[0].timeoutMs).toBeLessThan(
invokeNode.mock.calls[0]?.[0].timeoutMs ?? 0,
);
});
it("keeps the node Claude hard deadline while waiting for approval", async () => {
const plan = {
argv: ["/trusted/claude", "-p"],
commandText: "/trusted/claude -p",
};
const invokeNode = vi.fn(async () => ({
ok: true,
payloadJSON: JSON.stringify({
approvalRequired: true,
systemRunPlan: plan,
security: "allowlist",
ask: "on-miss",
}),
}));
setCliRunnerExecuteTestDeps({
invokeNodeClaudeCliRun: invokeNode,
registerExecApprovalRequestForHostOrThrow: vi.fn(async () => ({
id: "approval-timeout",
expiresAtMs: Date.now() + 60_000,
})),
resolveRegisteredExecApprovalDecision: vi.fn(
async () => await new Promise<string | null>(() => {}),
),
});
const context = buildPreparedCliRunContext({
model: "claude-opus-4-8",
timeoutMs: 25,
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
});
await expect(executePreparedCliRun(context)).rejects.toMatchObject({
code: "cli_overall_timeout",
});
expect(invokeNode).toHaveBeenCalledOnce();
});
it("keeps the node Claude hard deadline while registering approval", async () => {
const invokeNode = vi.fn(async () => ({
ok: true,
payloadJSON: JSON.stringify({
approvalRequired: true,
systemRunPlan: {
argv: ["/trusted/claude", "-p"],
commandText: "/trusted/claude -p",
},
security: "allowlist",
ask: "on-miss",
}),
}));
const resolveApproval = vi.fn();
setCliRunnerExecuteTestDeps({
invokeNodeClaudeCliRun: invokeNode,
registerExecApprovalRequestForHostOrThrow: vi.fn(
async () => await new Promise<never>(() => {}),
),
resolveRegisteredExecApprovalDecision: resolveApproval,
});
const context = buildPreparedCliRunContext({
model: "claude-opus-4-8",
timeoutMs: 25,
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
});
await expect(executePreparedCliRun(context)).rejects.toMatchObject({
code: "cli_overall_timeout",
});
expect(invokeNode).toHaveBeenCalledOnce();
expect(resolveApproval).not.toHaveBeenCalled();
});
it("rejects images before invoking a node-placed Claude session", async () => {
const invokeNode = vi.fn();
setCliRunnerExecuteTestDeps({ invokeNodeClaudeCliRun: invokeNode });
const context = buildPreparedCliRunContext({
model: "claude-opus-4-8",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
});
context.params.images = [{ type: "image", data: "aGVsbG8=", mimeType: "image/png" }];
await expect(executePreparedCliRun(context)).rejects.toThrow(
"paired-node Claude CLI sessions do not support attachments or images",
);
context.params.images = undefined;
context.params.imagePrompt = "[image: /tmp/gateway-only.png]";
await expect(executePreparedCliRun(context)).rejects.toThrow(
"paired-node Claude CLI sessions do not support attachments or images",
);
context.params.imagePrompt = undefined;
context.params.media = [{ path: "/tmp/hydratable.png", kind: "image" }];
await expect(executePreparedCliRun(context)).rejects.toThrow(
"paired-node Claude CLI sessions do not support attachments or images",
);
expect(invokeNode).not.toHaveBeenCalled();
});
it("allows non-hydratable image facts on a text-only node turn", async () => {
const invokeNode = vi.fn(async (params: Parameters<typeof invokeNodeClaudeCliRun>[0]) => {
params.onProgress(
[
JSON.stringify({ type: "system", subtype: "init", session_id: "node-text-only" }),
JSON.stringify({ type: "result", session_id: "node-text-only", result: "ok" }),
"",
].join("\n"),
);
return {
ok: true,
payloadJSON: JSON.stringify({ exitCode: 0, stderrTail: "", truncated: false }),
};
});
setCliRunnerExecuteTestDeps({ invokeNodeClaudeCliRun: invokeNode });
const context = buildPreparedCliRunContext({
provider: "claude-cli",
model: "claude-opus-4-8",
runId: "run-node-text-only-media-facts",
prompt: "already described",
sessionEntry: {
sessionId: "openclaw-session",
updatedAt: 1,
execHost: "node",
execNode: "node-a",
},
});
context.params.media = [
{ kind: "image" },
{ kind: "image", url: "https://example.test/described.png" },
];
await expect(executePreparedCliRun(context)).resolves.toMatchObject({ text: "ok" });
expect(invokeNode).toHaveBeenCalledOnce();
});
it("does not inject hardcoded 'Tools are disabled' text into CLI arguments", async () => {
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: CLAUDE_OK_JSONL,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
const backendConfig = {
command: "claude",
args: ["-p", "--output-format", "stream-json"],
output: "jsonl" as const,
input: "stdin" as const,
modelArg: "--model",
sessionArgs: ["--session-id", "{sessionId}"],
systemPromptArg: "--append-system-prompt",
systemPromptWhen: "first" as const,
serialize: true,
};
const context: PreparedCliRunContext = {
params: {
admittedRunContext: createTestAdmittedRunContext("run-no-tools-disabled"),
sessionId: "s1",
sessionFile: "/tmp/session.jsonl",
workspaceDir: "/tmp",
prompt: "Run: node script.mjs",
provider: "claude-cli",
model: "sonnet",
timeoutMs: 1_000,
runId: "run-no-tools-disabled",
extraSystemPrompt: "You are a helpful assistant.",
},
started: Date.now(),
workspaceDir: "/tmp",
backendResolved: {
id: "claude-cli",
config: backendConfig,
bundleMcp: true,
pluginId: "anthropic",
},
preparedBackend: {
backend: backendConfig,
env: {},
},
reusableCliSession: { mode: "none" },
hadSessionFile: false,
contextEngineConfig: {},
modelId: "sonnet",
normalizedModel: "sonnet",
systemPrompt: "You are a helpful assistant.",
systemPromptReport: {} as PreparedCliRunContext["systemPromptReport"],
bootstrapPromptWarningLines: [],
authEpochVersion: 2,
};
await executePreparedCliRun(context);
const input = mockCallArg(supervisorSpawnMock) as { argv?: string[] };
const allArgs = (input.argv ?? []).join("\n");
expect(allArgs).not.toContain("Tools are disabled in this session");
expect(allArgs).toContain("You are a helpful assistant.");
});
it("includes the OpenClaw skills prompt in CLI system prompts", () => {
const systemPrompt = buildCliAgentSystemPrompt({
workspaceDir: "/tmp",
modelDisplay: "claude-cli/sonnet",
tools: [],
skillsPrompt: [
"<available_skills>",
" <skill>",
" <name>weather</name>",
" <description>Use weather tools.</description>",
" <location>/tmp/skills/weather/SKILL.md</location>",
" </skill>",
"</available_skills>",
].join("\n"),
});
expect(systemPrompt).toContain("## Skills");
expect(systemPrompt).toContain("<name>weather</name>");
expect(systemPrompt).toContain("/tmp/skills/weather/SKILL.md");
});
it("pipes Claude prompts over stdin instead of argv", async () => {
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: CLAUDE_OK_JSONL,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
await executePreparedCliRun(
buildPreparedCliRunContext({
prompt: "Explain this diff",
}),
);
const input = mockCallArg(supervisorSpawnMock) as {
argv?: string[];
input?: string;
};
expect(input.input).toContain("Explain this diff");
expect(input.argv).not.toContain("Explain this diff");
});
it("emits metadata-only one-shot Claude model-call diagnostics with aggregate usage", async () => {
const prompt = "Trace this turn";
const stdout =
[
JSON.stringify({ type: "system", subtype: "init", session_id: "cli-trace-1" }),
JSON.stringify({
type: "assistant",
message: {
role: "assistant",
content: [{ type: "text", text: "traced reply" }],
usage: {
input_tokens: 11,
output_tokens: 6,
cache_read_input_tokens: 125,
cache_creation_input_tokens: 7,
},
},
}),
JSON.stringify({
type: "result",
subtype: "success",
session_id: "cli-trace-1",
result: "traced reply",
usage: {
input_tokens: 30,
output_tokens: 15,
cache_read_input_tokens: 300,
cache_creation_input_tokens: 12,
total_tokens: 357,
},
}),
].join("\n") + "\n";
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
const diagnostics = captureModelCallDiagnostics("run-claude-model-call-metadata");
try {
const output = await executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
runId: "run-claude-model-call-metadata",
prompt,
}),
);
await waitForDiagnosticEventsDrained();
expect(output.usage).toEqual({
input: 11,
output: 6,
cacheRead: 125,
cacheWrite: 7,
total: undefined,
});
expect(output.diagnosticUsage).toEqual({
input: 30,
output: 15,
cacheRead: 300,
cacheWrite: 12,
total: 357,
});
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.completed"]);
const started = diagnostics.events[0];
const completed = diagnostics.events[1];
expect(started?.event).toMatchObject({
provider: "anthropic",
model: "claude-sonnet-4-6",
api: "claude-code",
transport: "stdio",
observationUnit: "turn",
promptStats: {
inputMessagesCount: 1,
inputMessagesChars: prompt.length,
systemPromptChars: "You are a helpful assistant.".length,
totalChars: prompt.length + "You are a helpful assistant.".length,
},
});
expect(completed?.event).toMatchObject({
provider: "anthropic",
model: "claude-sonnet-4-6",
api: "claude-code",
transport: "stdio",
requestPayloadBytes: Buffer.byteLength(prompt),
responseStreamBytes: Buffer.byteLength(stdout),
timeToFirstByteMs: expect.any(Number),
usage: {
input: 30,
output: 15,
cacheRead: 300,
cacheWrite: 12,
total: 357,
},
});
expect(completed?.event.callId).toBe(started?.event.callId);
expect(completed?.event).not.toHaveProperty("upstreamRequestIdHash");
expect(started?.privateData.modelContent).toBeUndefined();
expect(completed?.privateData.modelContent).toBeUndefined();
} finally {
diagnostics.stop();
}
});
it("captures only representable Claude prompt and assistant content when opted in", async () => {
const prompt = "Explain the trace";
const stdout =
[
JSON.stringify({
type: "assistant",
message: {
role: "assistant",
stop_reason: "end_turn",
content: [
{ type: "text", text: "visible answer" },
{ type: "thinking", thinking: "visible reasoning", signature: "opaque-signature" },
{
type: "tool_use",
id: "tool-1",
name: "Read",
input: { path: "/private/path" },
},
],
},
}),
JSON.stringify({ type: "result", result: "visible answer" }),
].join("\n") + "\n";
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
const diagnostics = captureModelCallDiagnostics("run-claude-model-call-content");
try {
await executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
runId: "run-claude-model-call-content",
prompt,
config: {
diagnostics: {
enabled: true,
otel: {
enabled: true,
traces: true,
captureContent: true,
},
},
},
}),
);
await waitForDiagnosticEventsDrained();
const completed = diagnostics.events.find(
({ event }) => event.type === "model.call.completed",
);
expect(completed?.privateData.modelContent).toEqual({
inputMessages: [{ role: "user", content: [{ type: "text", text: prompt }] }],
outputMessages: [
{
role: "assistant",
stopReason: "end_turn",
content: [
{ type: "text", text: "visible answer" },
{ type: "thinking", thinking: "visible reasoning" },
{ type: "tool_call", id: "tool-1", name: "Read" },
],
},
],
});
expect(completed?.privateData.modelContent?.toolDefinitions).toBeUndefined();
expect(JSON.stringify(completed?.privateData.modelContent)).not.toContain("/private/path");
expect(JSON.stringify(completed?.privateData.modelContent)).not.toContain("opaque-signature");
} finally {
diagnostics.stop();
}
});
it("emits one Claude model-call error when one-shot process startup fails", async () => {
supervisorSpawnMock.mockRejectedValueOnce(new Error("claude process spawn failed"));
const diagnostics = captureModelCallDiagnostics("run-claude-model-call-spawn-error");
try {
await expect(
executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
runId: "run-claude-model-call-spawn-error",
prompt: "fail now",
}),
),
).rejects.toThrow("claude process spawn failed");
await waitForDiagnosticEventsDrained();
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.error"]);
expect(diagnostics.events[1]?.event).toMatchObject({
errorCategory: "Error",
requestPayloadBytes: Buffer.byteLength("fail now"),
});
expect(diagnostics.events[1]?.privateData.errorMessage).toBe("claude process spawn failed");
} finally {
diagnostics.stop();
}
});
it.each([
{
label: "timeout",
runId: "run-claude-model-call-timeout",
exit: {
reason: "overall-timeout" as const,
exitCode: null,
exitSignal: null,
durationMs: 50,
stdout: "",
stderr: "",
timedOut: true,
noOutputTimedOut: false,
},
errorCategory: "timeout",
failureKind: "timeout",
},
{
label: "parse failure",
runId: "run-claude-model-call-parse-error",
exit: {
reason: "exit" as const,
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: `${JSON.stringify({ type: "system", subtype: "unexpected" })}\n`,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
},
errorCategory: "unknown",
failureKind: undefined,
},
])("emits one Claude model-call error for $label", async (testCase) => {
supervisorSpawnMock.mockResolvedValueOnce(createManagedRun(testCase.exit));
const diagnostics = captureModelCallDiagnostics(testCase.runId);
try {
await expect(
executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
runId: testCase.runId,
}),
),
).rejects.toThrow();
await waitForDiagnosticEventsDrained();
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.error"]);
expect(diagnostics.events[1]?.event).toMatchObject({
errorCategory: testCase.errorCategory,
});
if (testCase.failureKind) {
expect(diagnostics.events[1]?.event).toMatchObject({
failureKind: testCase.failureKind,
});
} else {
expect(diagnostics.events[1]?.event).not.toHaveProperty("failureKind");
}
} finally {
diagnostics.stop();
}
});
it("passes Claude system prompts through a file instead of argv", async () => {
let systemPromptPath = "";
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { argv?: string[] };
systemPromptPath = requireArgAfter(input.argv, "--append-system-prompt-file");
expect(systemPromptPath).toContain("openclaw-cli-system-prompt-");
await expect(fs.readFile(systemPromptPath, "utf-8")).resolves.toBe(
"You are a helpful assistant.",
);
expect(input.argv).not.toContain("You are a helpful assistant.");
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: CLAUDE_OK_JSONL,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
await executePreparedCliRun(buildPreparedCliRunContext({}));
await expectPathMissing(systemPromptPath);
});
it("resends system prompts through a file for soft-resumed prompt-tool drift", async () => {
const writeSoftResumeSystemPromptFile = vi.fn(async () => ({
filePath: "/tmp/openclaw-soft-resume-system-prompt.md",
cleanup: async () => {},
}));
setCliRunnerExecuteTestDeps({
writeCliSystemPromptFile: writeSoftResumeSystemPromptFile,
});
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { argv?: string[] };
expect(input.argv).toContain("resume");
expect(input.argv).toContain("soft-cli-session");
expect(input.argv?.join(" ")).toContain("/tmp/openclaw-soft-resume-system-prompt.md");
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "ok",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
const context = buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
});
context.reusableCliSession = {
mode: "reuse-with-drift",
sessionId: "soft-cli-session",
drift: { reasons: ["prompt-tools"] },
};
await executePreparedCliRun(context, "soft-cli-session");
expect(writeSoftResumeSystemPromptFile).toHaveBeenCalledWith({
backend: context.preparedBackend.backend,
systemPrompt: "You are a helpful assistant.",
});
});
it("passes --session-id for new Claude sessions", async () => {
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
await executePreparedCliRun(buildPreparedCliRunContext({}));
const input = mockCallArg(supervisorSpawnMock) as {
argv?: string[];
input?: string;
mode?: string;
};
expect(input.mode).toBe("child");
expect(input.argv).toContain("claude");
expect(requireArgAfter(input.argv, "--session-id")).not.toBe("");
expect(input.input).toContain("hi");
expect(input.argv).not.toContain("hi");
});
it("does not pass a Claude session id for side-question runs", async () => {
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
const resolveExecutionArgs = vi.fn(({ baseArgs }) => [...baseArgs, "--max-turns", "1"]);
await executePreparedCliRun(
buildPreparedCliRunContext({
runId: "run-claude-side-question",
executionMode: "side-question",
backend: { sessionMode: "none" },
resolveExecutionArgs,
}),
);
const resolveArgsInput = requireRecord(mockCallArg(resolveExecutionArgs), "resolved args");
expect(resolveArgsInput.executionMode).toBe("side-question");
expect(resolveArgsInput.useResume).toBe(false);
const input = mockCallArg(supervisorSpawnMock) as { argv?: string[]; input?: string };
expect(input.argv).not.toContain("--session-id");
expect(input.argv).toContain("--max-turns");
expect(input.input).toContain("hi");
});
it("applies backend-owned per-run args before spawning", async () => {
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
const resolveExecutionArgs = vi.fn(({ baseArgs }) => [...baseArgs, "--effort", "high"]);
await executePreparedCliRun(
buildPreparedCliRunContext({
thinkLevel: "high",
resolveExecutionArgs,
}),
);
const resolveArgsInput = requireRecord(mockCallArg(resolveExecutionArgs), "resolved args");
expect(resolveArgsInput.provider).toBe("claude-cli");
expect(resolveArgsInput.modelId).toBe("sonnet");
expect(resolveArgsInput.thinkingLevel).toBe("high");
expect(resolveArgsInput.useResume).toBe(false);
expect(resolveArgsInput.baseArgs).toEqual(["-p", "--output-format", "stream-json"]);
const input = mockCallArg(supervisorSpawnMock) as { argv?: string[] };
expect(requireArgAfter(input.argv, "--effort")).toBe("high");
});
it("preserves exact tool availability through execution-time argument resolution", async () => {
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
const toolAvailability: NonNullable<PreparedCliRunContext["params"]["cliToolAvailability"]> = {
native: [],
openClaw: ["openclaw"],
};
const resolveExecutionArgs = vi.fn(({ baseArgs }) => baseArgs);
await executePreparedCliRun(
buildPreparedCliRunContext({
runId: "run-claude-tool-policy",
cliToolAvailability: toolAvailability,
resolveExecutionArgs,
}),
);
expect(resolveExecutionArgs).toHaveBeenCalledWith(
expect.objectContaining({
toolAvailability: {
...toolAvailability,
mcp: ["mcp__openclaw__openclaw"],
},
}),
);
});
it("fails closed when a selectable backend does not enforce exact tool availability", async () => {
const resolveExecutionArgs = vi.fn(() => undefined);
await expect(
executePreparedCliRun(
buildPreparedCliRunContext({
cliToolAvailability: {
native: [],
openClaw: ["openclaw"],
},
resolveExecutionArgs,
}),
),
).rejects.toThrow("did not enforce exact per-run tool availability");
expect(supervisorSpawnMock).not.toHaveBeenCalled();
});
it("does not require an argv rewrite after prepared-execution enforcement", async () => {
mockSuccessfulCliRun(GEMINI_OK_JSONL);
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "google-gemini-cli",
model: "gemini-3.1-pro-preview",
cliToolAvailability: { native: [], openClaw: ["openclaw"] },
toolAvailabilityEnforcement: "prepare-execution",
}),
);
expect(supervisorSpawnMock).toHaveBeenCalledOnce();
});
it("binds and admits the exact package artifact at the tool-availability version floor", async () => {
const fixture = await createCliPackageFixture("0.39.1");
try {
mockSuccessfulCliRun(GEMINI_OK_JSONL);
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "google-gemini-cli",
model: "gemini-3.1-pro-preview",
backend: { command: fixture.entrypoint },
cliToolAvailability: { native: [], openClaw: [] },
runtimeArtifact: {
kind: "bundled-package-tree",
packageName: "@fixture/versioned-cli",
entrypoint: "command",
exactToolAvailabilityVersionPolicy: { stableMinimum: "0.39.1" },
},
}),
);
const input = mockCallArg(supervisorSpawnMock) as { argv?: string[] };
expect(input.argv?.slice(0, 2)).toEqual([
await fs.realpath(process.execPath),
await fs.realpath(fixture.entrypoint),
]);
} finally {
await fs.rm(fixture.root, { recursive: true, force: true });
}
});
it("rejects an exact tool-availability run below the package version floor before spawn", async () => {
const fixture = await createCliPackageFixture("0.39.0");
try {
const context = buildPreparedCliRunContext({
provider: "google-gemini-cli",
model: "gemini-3.1-pro-preview",
backend: { command: fixture.entrypoint },
cliToolAvailability: { native: [], openClaw: [] },
runtimeArtifact: {
kind: "bundled-package-tree",
packageName: "@fixture/versioned-cli",
entrypoint: "command",
exactToolAvailabilityVersionPolicy: { stableMinimum: "0.39.1" },
},
});
context.params.isolatedCompletion = true;
await expect(executePreparedCliRun(context)).rejects.toMatchObject({
code: "unsupported",
message: expect.stringContaining("requires >=0.39.1; found 0.39.0"),
});
expect(supervisorSpawnMock).not.toHaveBeenCalled();
} finally {
await fs.rm(fixture.root, { recursive: true, force: true });
}
});
it.each([
{
version: "0.40.0-preview.2",
admitted: false,
expectedError: "requires >=0.40.0-preview.3; found 0.40.0-preview.2",
},
{
version: "0.41.0-nightly.20260427.g42587de73",
admitted: true,
stableMinimum: "99.0.0",
expectedError: undefined,
},
{
version: "0.53.0-beta.0",
admitted: false,
expectedError: "unsupported release line; found 0.53.0-beta.0",
},
])(
"applies the exact tool-availability policy to $version",
async ({ version, admitted, stableMinimum = "0.39.1", expectedError }) => {
const fixture = await createCliPackageFixture(version);
const run = () =>
executePreparedCliRun(
buildPreparedCliRunContext({
provider: "google-gemini-cli",
model: "gemini-3.1-pro-preview",
backend: { command: fixture.entrypoint },
cliToolAvailability: { native: [], openClaw: [] },
runtimeArtifact: {
kind: "bundled-package-tree",
packageName: "@fixture/versioned-cli",
entrypoint: "command",
exactToolAvailabilityVersionPolicy: {
stableMinimum,
prereleaseMinimums: {
preview: "0.40.0-preview.3",
nightly: "0.41.0-nightly.20260427.g42587de73",
},
},
},
}),
);
try {
if (admitted) {
mockSuccessfulCliRun(GEMINI_OK_JSONL);
await expect(run()).resolves.toBeDefined();
expect(supervisorSpawnMock).toHaveBeenCalledOnce();
} else {
await expect(run()).rejects.toThrow(
expectDefined(expectedError, "rejected version error"),
);
expect(supervisorSpawnMock).not.toHaveBeenCalled();
}
} finally {
await fs.rm(fixture.root, { recursive: true, force: true });
}
},
);
it("does not apply the exact tool-availability version floor to normal agent turns", async () => {
const fixture = await createCliPackageFixture("0.39.0");
try {
mockSuccessfulCliRun(GEMINI_OK_JSONL);
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "google-gemini-cli",
model: "gemini-3.1-pro-preview",
backend: { command: fixture.entrypoint },
runtimeArtifact: {
kind: "bundled-package-tree",
packageName: "@fixture/versioned-cli",
entrypoint: "command",
exactToolAvailabilityVersionPolicy: { stableMinimum: "0.39.1" },
},
}),
);
const input = mockCallArg(supervisorSpawnMock) as { argv?: string[] };
expect(input.argv?.[0]).toBe(fixture.entrypoint);
} finally {
await fs.rm(fixture.root, { recursive: true, force: true });
}
});
it("maps Ultra to the strongest generic CLI backend level", async () => {
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
const resolveExecutionArgs = vi.fn(({ baseArgs }) => baseArgs);
await executePreparedCliRun(
buildPreparedCliRunContext({
thinkLevel: "ultra",
resolveExecutionArgs,
}),
);
const resolveArgsInput = requireRecord(mockCallArg(resolveExecutionArgs), "resolved args");
expect(resolveArgsInput.thinkingLevel).toBe("max");
});
it("passes prepared backend env to the spawned CLI process", async () => {
mockSuccessfulCliRun();
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.5",
backend: {
env: {
GEMINI_CLI_HOME: "/ignored/static-home",
STATIC_BACKEND_FLAG: "set",
},
},
preparedEnv: {
GEMINI_CLI_HOME: "/tmp/openclaw-gemini-profile-home",
GEMINI_CLI_SYSTEM_SETTINGS_PATH: "/tmp/openclaw-gemini-system-settings.json",
},
}),
);
const input = mockCallArg(supervisorSpawnMock) as { env?: Record<string, string> };
expect(input.env?.STATIC_BACKEND_FLAG).toBe("set");
expect(input.env?.GEMINI_CLI_HOME).toBe("/tmp/openclaw-gemini-profile-home");
expect(input.env?.GEMINI_CLI_SYSTEM_SETTINGS_PATH).toBe(
"/tmp/openclaw-gemini-system-settings.json",
);
});
it("captures a runtime artifact for a strict CLI credential", async () => {
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-cli-strict-artifact-"));
const executable = path.join(dir, "claude-fixture");
try {
await fs.copyFile(process.execPath, executable);
await fs.chmod(executable, 0o755);
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
const context = buildPreparedCliRunContext({
backend: { command: executable },
onSuccessfulAuthBinding: () => {},
runtimeArtifact: {
kind: "bundled-package-tree",
packageName: "@fixture/native-cli",
entrypoint: "command",
nativeExecutableNames: ["claude-fixture"],
},
});
context.authBindingFingerprint = "strict-credential-owner";
await executePreparedCliRun(context);
expect(context.runtimeArtifactFingerprint).toMatch(/^[a-f0-9]{64}$/u);
expect(context.runtimeOwnerFingerprint).toBeUndefined();
const input = mockCallArg(supervisorSpawnMock) as { argv?: string[] };
expect(input.argv?.[0]).toBe(await fs.realpath(executable));
} finally {
await fs.rm(dir, { recursive: true, force: true });
}
});
it("passes OpenClaw skills to Claude as a session plugin", async () => {
const workspaceDir = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-cli-skills-"));
const skillDir = path.join(workspaceDir, "skills", "weather");
await fs.mkdir(skillDir, { recursive: true });
await fs.writeFile(
path.join(skillDir, "SKILL.md"),
[
"---",
"name: weather",
"description: Use weather tools for forecasts.",
"---",
"",
"Read forecast data before replying.",
].join("\n"),
"utf-8",
);
let pluginDir = "";
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { argv?: string[] };
pluginDir = requireArgAfter(input.argv, "--plugin-dir");
const manifest = JSON.parse(
await fs.readFile(path.join(pluginDir, ".claude-plugin", "plugin.json"), "utf-8"),
) as { name?: string; skills?: string };
expect(manifest.name).toBe("openclaw-skills");
expect(manifest.skills).toBe("./skills");
await expect(
fs.readFile(path.join(pluginDir, "skills", "weather", "SKILL.md"), "utf-8"),
).resolves.toContain("Read forecast data before replying.");
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: CLAUDE_OK_JSONL,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
try {
await executePreparedCliRun(
buildPreparedCliRunContext({
workspaceDir,
skillsSnapshot: {
prompt: "",
skills: [{ name: "weather" }],
resolvedSkills: [
{
name: "weather",
description: "Use weather tools for forecasts.",
filePath: path.join(skillDir, "SKILL.md"),
baseDir: skillDir,
source: "test",
sourceInfo: {
path: skillDir,
source: "test",
scope: "project",
origin: "top-level",
baseDir: skillDir,
},
disableModelInvocation: false,
},
],
},
}),
);
let accessError: unknown;
try {
await fs.access(pluginDir);
} catch (error) {
accessError = error;
}
expect((accessError as NodeJS.ErrnoException | undefined)?.code).toBe("ENOENT");
} finally {
await fs.rm(workspaceDir, { recursive: true, force: true });
}
});
it("injects skill env overrides into CLI child env and restores host env", async () => {
const previousEnvValue = process.env.CLI_SKILL_API_KEY;
delete process.env.CLI_SKILL_API_KEY;
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { env?: Record<string, string> };
expect(input.env?.CLI_SKILL_API_KEY).toBe("skill-secret");
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: CLAUDE_OK_JSONL,
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
try {
await executePreparedCliRun(
buildPreparedCliRunContext({
config: {
skills: {
entries: {
envskill: { apiKey: "skill-secret" }, // pragma: allowlist secret
},
},
},
skillsSnapshot: {
prompt: "",
skills: [{ name: "envskill", primaryEnv: "CLI_SKILL_API_KEY" }],
},
}),
);
expect(process.env.CLI_SKILL_API_KEY).toBeUndefined();
} finally {
if (previousEnvValue === undefined) {
delete process.env.CLI_SKILL_API_KEY;
} else {
process.env.CLI_SKILL_API_KEY = previousEnvValue;
}
}
});
it("runs CLI through supervisor and returns payload", async () => {
const logInfoSpy = vi.spyOn(cliBackendLog, "info").mockImplementation(() => undefined);
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "ok",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
const context = buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
});
context.reusableCliSession = { mode: "reuse", sessionId: "thread-123" };
try {
const result = await executePreparedCliRun(context, "thread-123");
expect(result.text).toBe("ok");
const input = mockCallArg(supervisorSpawnMock) as {
argv?: string[];
mode?: string;
timeoutMs?: number;
noOutputTimeoutMs?: number;
replaceExistingScope?: boolean;
scopeKey?: string;
};
expect(input.mode).toBe("child");
expect(input.argv).toEqual([
"codex",
"exec",
"resume",
"thread-123",
"--skip-git-repo-check",
"--model",
"gpt-5.4",
"hi",
]);
expect(input.timeoutMs).toBe(1_000);
expect(input.noOutputTimeoutMs).toBeGreaterThanOrEqual(1_000);
expect(input.replaceExistingScope).toBe(true);
expect(input.scopeKey).toContain("thread-123");
const turnLog = logInfoSpy.mock.calls
.map(([message]) => message)
.find((message) => message.startsWith("cli turn:"));
expect(turnLog).toContain("provider=codex-cli");
expect(turnLog).toContain("model=gpt-5.4");
expect(turnLog).toContain("outBytes=2 outHash=2689367b205c");
expect(turnLog).not.toContain("ok");
} finally {
logInfoSpy.mockRestore();
}
});
it("returns process diagnostics with byte counts and bounded output hashes", async () => {
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 75,
stdout: "ok",
stderr: "warn\n",
timedOut: false,
noOutputTimedOut: false,
}),
);
const result = await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
}),
);
expect(result.diagnostics?.process).toEqual({
backendId: "codex-cli",
processReason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 75,
stdoutBytes: 2,
stdoutHash: "2689367b205c",
stderrBytes: 5,
stderrHash: "7597e6b3a377",
useResume: false,
});
});
it("rejects Gemini stream-json error results emitted with a zero exit code", async () => {
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout:
[
JSON.stringify({
type: "message",
role: "assistant",
content: "partial text",
delta: true,
}),
JSON.stringify({
type: "result",
status: "error",
error: {
message: "Gemini stream failed",
},
}),
].join("\n") + "\n",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
await expectRejectsWithFields(
executePreparedCliRun(
buildPreparedCliRunContext({
provider: "google-gemini-cli",
model: "gemini-3.1-pro-preview",
}),
),
{
name: "FailoverError",
message: "Gemini stream failed",
reason: "unknown",
},
);
});
it("passes Codex system prompts through model_instructions_file", async () => {
let promptFileText = "";
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { argv?: string[] };
const configArg = requireArgAfter(input.argv, "-c");
const match = requireRegexMatch(configArg, /^model_instructions_file="(.+)"$/);
promptFileText = await fs.readFile(
expectDefined(match[1], "match[1] test invariant"),
"utf-8",
);
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "ok",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
}),
);
expect(promptFileText).toBe("You are a helpful assistant.");
});
it("cancels the managed CLI run when the abort signal fires", async () => {
const abortController = new AbortController();
let resolveWait:
| ((value: {
reason:
| "manual-cancel"
| "overall-timeout"
| "no-output-timeout"
| "spawn-error"
| "signal"
| "exit";
exitCode: number | null;
exitSignal: NodeJS.Signals | number | null;
durationMs: number;
stdout: string;
stderr: string;
timedOut: boolean;
noOutputTimedOut: boolean;
}) => void)
| undefined;
const cancel = vi.fn((reason?: string) => {
if (!resolveWait) {
throw new Error("Expected managed CLI wait resolver to be initialized");
}
resolveWait({
reason: reason === "manual-cancel" ? "manual-cancel" : "signal",
exitCode: null,
exitSignal: null,
durationMs: 50,
stdout: "",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
supervisorSpawnMock.mockResolvedValueOnce({
pid: 1234,
startedAtMs: Date.now(),
stdin: undefined,
wait: vi.fn(
async () =>
await new Promise((resolve) => {
resolveWait = resolve;
}),
),
cancel,
});
const context = buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
});
context.params.abortSignal = abortController.signal;
const runPromise = executePreparedCliRun(context);
await vi.waitFor(() => {
expect(supervisorSpawnMock).toHaveBeenCalledTimes(1);
});
abortController.abort();
await expectRejectsWithFields(runPromise, { name: "AbortError" });
expect(cancel).toHaveBeenCalledWith("manual-cancel");
});
it("streams Claude text deltas from stream-json stdout", async () => {
const agentEvents: Array<{ stream: string; text?: string; delta?: string }> = [];
const stop = onAgentEvent((evt) => {
agentEvents.push({
stream: evt.stream,
text: typeof evt.data.text === "string" ? evt.data.text : undefined,
delta: typeof evt.data.delta === "string" ? evt.data.delta : undefined,
});
});
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { onStdout?: (chunk: string) => void };
input.onStdout?.(
[
JSON.stringify({ type: "init", session_id: "session-123" }),
JSON.stringify({
type: "stream_event",
event: { type: "content_block_delta", delta: { type: "text_delta", text: "Hello" } },
}),
].join("\n") + "\n",
);
input.onStdout?.(
JSON.stringify({
type: "stream_event",
event: { type: "content_block_delta", delta: { type: "text_delta", text: " world" } },
}) + "\n",
);
input.onStdout?.(
JSON.stringify({
type: "result",
session_id: "session-123",
result: "Hello world",
}) + "\n",
);
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
try {
const result = await executePreparedCliRun(buildPreparedCliRunContext({}));
expect(result.text).toBe("Hello world");
expect(agentEvents).toEqual([
{ stream: "assistant", text: "Hello", delta: "Hello" },
{ stream: "assistant", text: "Hello world", delta: " world" },
]);
} finally {
stop();
}
});
it("suppresses Claude text delta events for side-question runs", async () => {
const agentEvents: Array<{ stream: string; text?: string; delta?: string }> = [];
const stop = onAgentEvent((evt) => {
agentEvents.push({
stream: evt.stream,
text: typeof evt.data.text === "string" ? evt.data.text : undefined,
delta: typeof evt.data.delta === "string" ? evt.data.delta : undefined,
});
});
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { onStdout?: (chunk: string) => void };
input.onStdout?.(
[
JSON.stringify({ type: "init", session_id: "session-123" }),
JSON.stringify({
type: "stream_event",
event: { type: "content_block_delta", delta: { type: "text_delta", text: "Hello" } },
}),
JSON.stringify({
type: "result",
session_id: "session-123",
result: "Hello",
}),
].join("\n") + "\n",
);
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
try {
const result = await executePreparedCliRun(
buildPreparedCliRunContext({
executionMode: "side-question",
backend: { sessionMode: "none" },
}),
);
expect(result.text).toBe("Hello");
expect(agentEvents).toEqual([]);
} finally {
stop();
}
});
it("keeps one managed Claude model call open until background task results drain", async () => {
let stdoutListener: ((chunk: string) => void) | undefined;
const writes: string[] = [];
const cancel = vi.fn();
const interimChunk =
[
JSON.stringify({ type: "system", subtype: "init", session_id: "live-trace" }),
JSON.stringify({
type: "assistant",
session_id: "live-trace",
message: {
role: "assistant",
content: [{ type: "text", text: "working" }],
usage: { input_tokens: 4, output_tokens: 1, cache_read_input_tokens: 20 },
},
}),
JSON.stringify({
type: "system",
subtype: "background_tasks_changed",
tasks: [{ task_id: "task-1", task_type: "local_agent", description: "research" }],
}),
JSON.stringify({
type: "result",
subtype: "success",
session_id: "live-trace",
result: "working",
usage: { input_tokens: 5, output_tokens: 1, cache_read_input_tokens: 25 },
}),
].join("\n") + "\n";
const finalChunk =
[
JSON.stringify({ type: "system", subtype: "background_tasks_changed", tasks: [] }),
JSON.stringify({
type: "assistant",
session_id: "live-trace",
message: {
role: "assistant",
content: [{ type: "text", text: "finished" }],
usage: { input_tokens: 6, output_tokens: 2, cache_read_input_tokens: 30 },
},
}),
JSON.stringify({
type: "result",
subtype: "success",
session_id: "live-trace",
result: "finished",
usage: {
input_tokens: 10,
output_tokens: 3,
cache_read_input_tokens: 50,
cache_creation_input_tokens: 2,
},
}),
].join("\n") + "\n";
const stdin = {
write: vi.fn((data: string, cb?: (err?: Error | null) => void) => {
writes.push(data);
emitClaudeInputStarted(stdoutListener, data);
stdoutListener?.(interimChunk);
cb?.();
}),
end: vi.fn(),
};
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { onStdout?: (chunk: string) => void };
stdoutListener = input.onStdout;
return {
runId: "live-model-call",
pid: 2345,
startedAtMs: Date.now(),
stdin,
wait: vi.fn(() => new Promise(() => {})),
cancel,
};
});
const diagnostics = captureModelCallDiagnostics("run-live-model-call-background");
try {
const run = executePreparedCliRun(
buildClaudeLiveRunContext({
model: "claude-sonnet-4-6",
runId: "run-live-model-call-background",
prompt: "research this",
config: {
diagnostics: {
enabled: true,
otel: {
enabled: true,
traces: true,
captureContent: true,
},
},
},
}),
);
await vi.waitFor(() => expect(writes).toHaveLength(1));
await waitForDiagnosticEventsDrained();
expect(diagnostics.events.map(({ event }) => event.type)).toEqual(["model.call.started"]);
stdoutListener?.(finalChunk);
const output = await run;
await waitForDiagnosticEventsDrained();
expect(output.text).toContain("working");
expect(output.text).toContain("finished");
expect(output.usage).toEqual({
input: 6,
output: 2,
cacheRead: 30,
cacheWrite: undefined,
total: undefined,
});
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.completed"]);
const completed = diagnostics.events[1];
const inputUuid = (JSON.parse(writes[0] ?? "{}") as { uuid?: string }).uuid;
const lifecycleChunk = `${JSON.stringify({
type: "command_lifecycle",
command_uuid: inputUuid,
state: "started",
})}\n`;
expect(completed?.event).toMatchObject({
api: "claude-code",
transport: "stdio-live",
observationUnit: "turn",
requestPayloadBytes: Buffer.byteLength(writes[0] ?? ""),
responseStreamBytes:
Buffer.byteLength(lifecycleChunk) +
Buffer.byteLength(interimChunk) +
Buffer.byteLength(finalChunk),
usage: {
input: 10,
output: 3,
cacheRead: 50,
cacheWrite: 2,
},
});
expect(completed?.privateData.modelContent?.outputMessages).toEqual([
{ role: "assistant", content: [{ type: "text", text: "working" }] },
{ role: "assistant", content: [{ type: "text", text: "finished" }] },
]);
expect(cancel).not.toHaveBeenCalled();
} finally {
diagnostics.stop();
}
});
it("emits one terminal model-call error for a managed Claude result failure", async () => {
mockClaudeLiveRun(supervisorSpawnMock, {
runId: "live-model-call-error",
pid: 2346,
events: [
{
type: "result",
subtype: "error_during_execution",
is_error: true,
session_id: "live-error",
result: "managed turn failed",
usage: { input_tokens: 8, output_tokens: 2, cache_read_input_tokens: 40 },
},
],
});
const diagnostics = captureModelCallDiagnostics("run-live-model-call-error");
try {
await expect(
executePreparedCliRun(
buildClaudeLiveRunContext({
model: "claude-sonnet-4-6",
runId: "run-live-model-call-error",
}),
),
).rejects.toThrow(/managed turn failed/i);
await waitForDiagnosticEventsDrained();
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.error"]);
expect(diagnostics.events[1]?.event).toMatchObject({
transport: "stdio-live",
usage: { input: 8, output: 2, cacheRead: 40 },
});
} finally {
diagnostics.stop();
}
});
it("extends the live no-output watchdog to the blocked-tool floor while a tool is outstanding", async () => {
const toolErrorEvents: Array<Record<string, unknown>> = [];
const stopDiagnostics = onTrustedToolExecutionEvent((event) => {
if (event.type === "tool.execution.error") {
toolErrorEvents.push(event as unknown as Record<string, unknown>);
}
});
let stdoutListener: ((chunk: string) => void) | undefined;
const cancel = vi.fn();
const stdin = {
write: vi.fn((data: string, callback?: (error?: Error | null) => void) => {
emitClaudeInputStarted(stdoutListener, data);
stdoutListener?.(
[
JSON.stringify({ type: "system", subtype: "init", session_id: "live-quiet-tool" }),
JSON.stringify({
type: "assistant",
message: {
content: [{ type: "tool_use", id: "tool-quiet-1", name: "Bash", input: {} }],
},
}),
].join("\n") + "\n",
);
callback?.();
}),
end: vi.fn(),
};
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = (args[0] ?? {}) as { onStdout?: (chunk: string) => void };
stdoutListener = input.onStdout;
return {
pid: 2345,
startedAtMs: Date.now(),
stdin,
wait: vi.fn(() => new Promise(() => {})),
cancel,
};
});
const run = executePreparedCliRun(
buildClaudeLiveRunContext({
timeoutMs: 3_600_000,
}),
);
const rejection = run.then(
() => undefined,
(error: unknown) => error,
);
await vi.waitFor(() => {
expect(stdin.write).toHaveBeenCalledOnce();
});
// Fake the clock only after the spawn path settled, then emit one more
// stdout line so the watchdog re-arms on the faked setTimeout/Date.
vi.useFakeTimers({ toFake: ["setTimeout", "clearTimeout", "Date"] });
stdoutListener?.(
`${JSON.stringify({
type: "stream_event",
event: { type: "content_block_delta", delta: { type: "text_delta", text: "running" } },
})}\n`,
);
// Base watchdog (600s cap for a 1h budget) must not kill the quiet tool.
vi.advanceTimersByTime(650_000);
expect(cancel).not.toHaveBeenCalled();
// The blocked-tool floor (15min of quiet) still terminates a wedged tool.
try {
vi.advanceTimersByTime(300_000);
expect(cancel).toHaveBeenCalledWith("manual-cancel");
const error = await rejection;
expect(error).toBeInstanceOf(Error);
expect((error as Error).message).toMatch(/produced no output for 900s/);
// Watchdog-killed turns must keep timeout provenance for active tools.
expect(toolErrorEvents).toContainEqual(
expect.objectContaining({
toolCallId: "tool-quiet-1",
terminalReason: "timed_out",
}),
);
} finally {
stopDiagnostics();
}
});
it("keeps non-capture live prepared backend cleanup with the whole-run owner", async () => {
mockClaudeLiveRun(supervisorSpawnMock, {
runId: "live-cleanup-run",
pid: 2346,
events: [
{ type: "system", subtype: "init", session_id: "live-session-cleanup" },
{ type: "result", session_id: "live-session-cleanup", result: "ok" },
],
});
const preparedBackendCleanup = vi.fn(async () => {});
const context = buildClaudeLiveRunContext({
prompt: "first",
backend: {
args: ["-p", "--strict-mcp-config", "--mcp-config", "/tmp/mcp-cleanup.json"],
},
mcpConfigHash: "cleanup-mcp-config",
});
context.preparedBackend.cleanup = preparedBackendCleanup;
const result = await executePreparedCliRun(context);
expect(result.text).toBe("ok");
expect(context.preparedBackend.cleanup).toBe(preparedBackendCleanup);
expect(preparedBackendCleanup).not.toHaveBeenCalled();
resetClaudeLiveSessionsForTest();
expect(preparedBackendCleanup).not.toHaveBeenCalled();
await context.preparedBackend.cleanup?.();
expect(preparedBackendCleanup).toHaveBeenCalledOnce();
});
it("keeps captured live prepared backend cleanup with the whole-run owner", async () => {
const mcpConfigDir = await fs.mkdtemp(
path.join(os.tmpdir(), "openclaw-cli-captured-mcp-config-"),
);
const mcpConfigPath = path.join(mcpConfigDir, "mcp.json");
await fs.writeFile(
mcpConfigPath,
`${JSON.stringify(
{
mcpServers: {
openclaw: {
type: "http",
url: "http://127.0.0.1:23119/mcp",
headers: {},
},
},
},
null,
2,
)}\n`,
"utf-8",
);
try {
mockClaudeLiveRun(supervisorSpawnMock, {
cancelable: true,
pid: 2347,
events: [
{ type: "system", subtype: "init", session_id: "captured-live-cleanup" },
{ type: "result", session_id: "captured-live-cleanup", result: "ok" },
],
});
const preparedBackendCleanup = vi.fn(async () => {});
const context = buildClaudeLiveRunContext({
prompt: "first",
backend: {
args: ["-p", "--strict-mcp-config", "--mcp-config", mcpConfigPath],
},
mcpConfigHash: "captured-cleanup-mcp-config",
mcpDeliveryCapture: true,
});
context.preparedBackend.cleanup = preparedBackendCleanup;
const result = await executePreparedCliRun(context);
expect(result.text).toBe("ok");
expect(context.preparedBackend.cleanup).toBe(preparedBackendCleanup);
expect(preparedBackendCleanup).not.toHaveBeenCalled();
await context.preparedBackend.cleanup?.();
expect(preparedBackendCleanup).toHaveBeenCalledOnce();
} finally {
await fs.rm(mcpConfigDir, { recursive: true, force: true });
}
});
it("preserves completed output when system prompt cleanup fails after delivery", async () => {
const cleanupError = new Error("system prompt cleanup failed");
const logWarnSpy = vi.spyOn(cliBackendLog, "warn").mockImplementation(() => undefined);
setCliRunnerExecuteTestDeps({
writeCliSystemPromptFile: async () => ({
filePath: "/tmp/system-prompt.md",
cleanup: async () => {
throw cleanupError;
},
}),
});
supervisorSpawnMock.mockImplementationOnce(async (...args: unknown[]) => {
const input = args[0] as Parameters<ReturnType<typeof getProcessSupervisor>["spawn"]>[0];
const captureHandle = markMcpLoopbackToolCallStarted({
captureKey: input.env?.OPENCLAW_MCP_CLI_CAPTURE_KEY ?? "",
toolName: "message",
args: { action: "send", target: "chat123", message: "done" },
});
if (!captureHandle) {
throw new Error("Expected message delivery capture");
}
recordMcpLoopbackToolCallResult({
captureHandle,
toolName: "message",
args: { action: "send", target: "chat123", message: "done" },
result: { status: "sent" },
outcome: "completed",
});
markMcpLoopbackToolCallFinished(captureHandle);
input.onStdout?.("done");
return createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
});
});
const context = buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
mcpDeliveryCapture: true,
});
const result = await executePreparedCliRun(context);
setCliRunnerExecuteTestDeps({ writeCliSystemPromptFile });
expect(result.text).toBe("done");
expect(result.didSendViaMessagingTool).toBe(true);
expect(logWarnSpy).toHaveBeenCalledWith(
expect.stringContaining("outer resource cleanup failed after confirmed message delivery"),
);
});
it("emits a model-call error when successful Claude output is followed by cleanup failure", async () => {
const runId = "run-claude-cleanup-failure";
const diagnostics = captureModelCallDiagnostics(runId);
const cleanupError = new Error("system prompt cleanup failed");
setCliRunnerExecuteTestDeps({
writeCliSystemPromptFile: async () => ({
filePath: "/tmp/system-prompt.md",
cleanup: async () => {
throw cleanupError;
},
}),
});
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
try {
await expect(
executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
runId,
}),
),
).rejects.toThrow("system prompt cleanup failed");
await waitForDiagnosticEventsDrained();
expectModelCallTypes(diagnostics, ["model.call.started", "model.call.error"]);
expect(diagnostics.events[1]?.event.callId).toBe(diagnostics.events[0]?.event.callId);
} finally {
diagnostics.stop();
setCliRunnerExecuteTestDeps({ writeCliSystemPromptFile });
}
});
it("wraps primitive and frozen failures to preserve delivery evidence", () => {
const evidence = { didSendViaMessagingTool: true };
const primitive = attachCliMessagingDeliveryEvidence("failed", evidence);
const frozen = attachCliMessagingDeliveryEvidence(Object.freeze(new Error("frozen")), evidence);
expect(primitive).toBeInstanceOf(Error);
expect(frozen).toBeInstanceOf(Error);
expect(getCliMessagingDeliveryEvidence(primitive)?.didSendViaMessagingTool).toBe(true);
expect(getCliMessagingDeliveryEvidence(frozen)?.didSendViaMessagingTool).toBe(true);
});
it("sanitizes dangerous backend env overrides before spawn", async () => {
mockSuccessfulCliRun();
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
backend: {
env: {
NODE_OPTIONS: "--require ./malicious.js",
LD_PRELOAD: "/tmp/pwn.so",
PATH: "/tmp/evil",
HOME: "/tmp/evil-home",
SAFE_KEY: "ok",
},
},
}),
"thread-123",
);
const input = mockCallArg(supervisorSpawnMock) as {
env?: Record<string, string | undefined>;
};
expect(input.env?.SAFE_KEY).toBe("ok");
expect(input.env?.PATH).toBe(process.env.PATH);
expect(input.env?.HOME).toBe(process.env.HOME);
expect(input.env?.NODE_OPTIONS).toBeUndefined();
expect(input.env?.LD_PRELOAD).toBeUndefined();
});
it.each([
{
name: "applies clearEnv after sanitizing backend env overrides",
baseEnv: { SAFE_CLEAR: "from-base" },
backend: { env: { SAFE_KEEP: "keep-me" }, clearEnv: ["SAFE_CLEAR"] },
expected: { SAFE_KEEP: "keep-me", SAFE_CLEAR: undefined },
},
{
name: "can preserve selected clearEnv keys for live CLI backend probes",
baseEnv: { SAFE_CLEAR: "from-base" },
preserve: ["SAFE_CLEAR"],
backend: { clearEnv: ["SAFE_CLEAR", "SAFE_DROP"] },
expected: { SAFE_CLEAR: "from-base", SAFE_DROP: undefined },
},
{
name: "keeps explicit backend env overrides even when clearEnv drops inherited values",
baseEnv: { SAFE_OVERRIDE: "from-base" },
backend: { env: { SAFE_OVERRIDE: "from-override" }, clearEnv: ["SAFE_OVERRIDE"] },
expected: { SAFE_OVERRIDE: "from-override" },
},
])("$name", async (testCase) => {
Object.assign(process.env, testCase.baseEnv);
if (testCase.preserve) {
process.env.OPENCLAW_LIVE_CLI_BACKEND_PRESERVE_ENV = JSON.stringify(testCase.preserve);
}
try {
mockSuccessfulCliRun();
await executePreparedCliRun(
buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
backend: testCase.backend as Partial<PreparedCliRunContext["preparedBackend"]["backend"]>,
}),
"thread-123",
);
const input = mockCallArg(supervisorSpawnMock) as {
env?: Record<string, string | undefined>;
};
for (const [key, value] of Object.entries(testCase.expected)) {
expect(input.env?.[key]).toBe(value);
}
} finally {
delete process.env.OPENCLAW_LIVE_CLI_BACKEND_PRESERVE_ENV;
for (const key of Object.keys(testCase.baseEnv)) {
delete process.env[key];
}
}
});
it("keeps selected Claude auth authoritative over ambient and configured credentials", async () => {
vi.stubEnv("OPENCLAW_LIVE_CLI_BACKEND_PRESERVE_ENV", '["ANTHROPIC_API_KEY"]');
vi.stubEnv("ANTHROPIC_API_KEY", "ambient-api-key");
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
await executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
preparedEnv: {
CLAUDE_CODE_OAUTH_TOKEN: "selected-oauth-token",
CLAUDE_CODE_SUBPROCESS_ENV_SCRUB: "1",
},
backend: {
env: { ANTHROPIC_API_KEY: "configured-api-key" },
clearEnv: ["ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"],
},
}),
);
const input = mockCallArg(supervisorSpawnMock) as {
env?: Record<string, string | undefined>;
};
expect(input.env?.ANTHROPIC_API_KEY).toBeUndefined();
expect(input.env?.CLAUDE_CODE_OAUTH_TOKEN).toBe("selected-oauth-token");
});
it("clears claude-cli provider-routing, auth, telemetry, compaction, and host-managed env", async () => {
vi.stubEnv("ANTHROPIC_BASE_URL", "https://proxy.example.com/v1");
vi.stubEnv("ANTHROPIC_API_TOKEN", "env-api-token");
vi.stubEnv("ANTHROPIC_CUSTOM_HEADERS", "x-test-header: env");
vi.stubEnv("ANTHROPIC_OAUTH_TOKEN", "env-oauth-token");
vi.stubEnv("CLAUDE_CODE_USE_BEDROCK", "1");
vi.stubEnv("ANTHROPIC_AUTH_TOKEN", "env-auth-token");
vi.stubEnv("CLAUDE_CODE_OAUTH_TOKEN", "env-oauth-token");
vi.stubEnv("CLAUDE_CODE_AUTO_COMPACT_WINDOW", "1048576");
vi.stubEnv("CLAUDE_CODE_REMOTE", "1");
vi.stubEnv("ANTHROPIC_UNIX_SOCKET", "/tmp/anthropic.sock");
vi.stubEnv("OTEL_LOGS_EXPORTER", "none");
vi.stubEnv("OTEL_METRICS_EXPORTER", "none");
vi.stubEnv("OTEL_TRACES_EXPORTER", "none");
vi.stubEnv("OTEL_EXPORTER_OTLP_PROTOCOL", "none");
vi.stubEnv("OTEL_SDK_DISABLED", "true");
vi.stubEnv("CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST", "1");
mockSuccessfulCliRun(CLAUDE_OK_JSONL);
await executePreparedCliRun(
buildPreparedCliRunContext({
model: "claude-sonnet-4-6",
preparedEnv: {
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "100000",
},
backend: {
env: {
SAFE_KEEP: "ok",
ANTHROPIC_BASE_URL: "https://override.example.com/v1",
CLAUDE_CODE_OAUTH_TOKEN: "override-oauth-token",
CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST: "1",
},
clearEnv: [
"ANTHROPIC_BASE_URL",
"ANTHROPIC_API_TOKEN",
"ANTHROPIC_CUSTOM_HEADERS",
"ANTHROPIC_OAUTH_TOKEN",
"CLAUDE_CODE_USE_BEDROCK",
"ANTHROPIC_AUTH_TOKEN",
"CLAUDE_CODE_OAUTH_TOKEN",
"CLAUDE_CODE_AUTO_COMPACT_WINDOW",
"CLAUDE_CODE_REMOTE",
"ANTHROPIC_UNIX_SOCKET",
"OTEL_LOGS_EXPORTER",
"OTEL_METRICS_EXPORTER",
"OTEL_TRACES_EXPORTER",
"OTEL_EXPORTER_OTLP_PROTOCOL",
"OTEL_SDK_DISABLED",
],
},
}),
);
const input = mockCallArg(supervisorSpawnMock) as {
env?: Record<string, string | undefined>;
};
expect(input.env?.SAFE_KEEP).toBe("ok");
expect(input.env?.CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST).toBeUndefined();
expect(input.env?.ANTHROPIC_BASE_URL).toBe("https://override.example.com/v1");
expect(input.env?.ANTHROPIC_API_TOKEN).toBeUndefined();
expect(input.env?.ANTHROPIC_CUSTOM_HEADERS).toBeUndefined();
expect(input.env?.ANTHROPIC_OAUTH_TOKEN).toBeUndefined();
expect(input.env?.CLAUDE_CODE_USE_BEDROCK).toBeUndefined();
expect(input.env?.ANTHROPIC_AUTH_TOKEN).toBeUndefined();
expect(input.env?.CLAUDE_CODE_OAUTH_TOKEN).toBe("override-oauth-token");
expect(input.env?.CLAUDE_CODE_AUTO_COMPACT_WINDOW).toBe("100000");
expect(input.env?.CLAUDE_CODE_REMOTE).toBeUndefined();
expect(input.env?.ANTHROPIC_UNIX_SOCKET).toBeUndefined();
expect(input.env?.OTEL_LOGS_EXPORTER).toBeUndefined();
expect(input.env?.OTEL_METRICS_EXPORTER).toBeUndefined();
expect(input.env?.OTEL_TRACES_EXPORTER).toBeUndefined();
expect(input.env?.OTEL_EXPORTER_OTLP_PROTOCOL).toBeUndefined();
expect(input.env?.OTEL_SDK_DISABLED).toBeUndefined();
});
it("formats CLI auth env diagnostics as key names without secret values", () => {
vi.stubEnv("ANTHROPIC_API_KEY", "sk-ant-host");
vi.stubEnv("ANTHROPIC_API_TOKEN", "token-host");
vi.stubEnv("GEMINI_CLI_SYSTEM_SETTINGS_PATH", "/tmp/host-gemini-settings.json");
vi.stubEnv("OPENAI_API_KEY", "sk-openai-host");
const log = buildCliEnvAuthLog({
ANTHROPIC_API_TOKEN: "token-child",
CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST: "1",
GEMINI_CLI_HOME: "/tmp/child-gemini-home",
OPENAI_API_KEY: "sk-openai-child",
});
expect(log).toMatch(/host=.*ANTHROPIC_API_KEY/);
expect(log).toMatch(/host=.*ANTHROPIC_API_TOKEN/);
expect(log).toMatch(/host=.*OPENAI_API_KEY/);
expect(log).toMatch(/child=.*ANTHROPIC_API_TOKEN/);
expect(log).toMatch(/child=.*CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST/);
expect(log).toMatch(/child=.*OPENAI_API_KEY/);
expect(log).toMatch(/cleared=.*ANTHROPIC_API_KEY/);
expect(log).toMatch(/runtimeHost=.*GEMINI_CLI_SYSTEM_SETTINGS_PATH/);
expect(log).toMatch(/runtimeChild=.*GEMINI_CLI_HOME/);
expect(log).toMatch(/runtimeCleared=.*GEMINI_CLI_SYSTEM_SETTINGS_PATH/);
expect(log).not.toContain("sk-ant-host");
expect(log).not.toContain("token-child");
expect(log).not.toContain("/tmp/child-gemini-home");
expect(log).not.toContain("sk-openai-child");
});
it("prepends bootstrap warnings to the CLI prompt body", async () => {
supervisorSpawnMock.mockResolvedValueOnce(
createManagedRun({
reason: "exit",
exitCode: 0,
exitSignal: null,
durationMs: 50,
stdout: "ok",
stderr: "",
timedOut: false,
noOutputTimedOut: false,
}),
);
const context = buildPreparedCliRunContext({
provider: "codex-cli",
model: "gpt-5.4",
});
context.reusableCliSession = { mode: "reuse", sessionId: "thread-123" };
context.bootstrapPromptWarningLines = [
"[Bootstrap truncation warning]",
"- AGENTS.md: 200 raw -> 20 injected",
];
await executePreparedCliRun(context, "thread-123");
const input = mockCallArg(supervisorSpawnMock) as {
argv?: string[];
input?: string;
};
const promptCarrier = [input.input ?? "", ...(input.argv ?? [])].join("\n");
expect(promptCarrier).toContain("[Bootstrap truncation warning]");
expect(promptCarrier).toContain("- AGENTS.md: 200 raw -> 20 injected");
expect(promptCarrier).toContain("hi");
});
});
/* oxlint-disable max-lines -- TODO: split this grandfathered oversized file. */