fix(qa): retain runtime tool evidence after agent completion (#113484)

Wait for durable transcript evidence before runtime parity capture and scope capture to the fixture-owned sessions.

Co-authored-by: Peter Steinberger <steipete@gmail.com>
This commit is contained in:
Vincent Koc
2026-07-25 12:03:49 +08:00
committed by GitHub
parent 2a5ad61ce9
commit c780bfbf0d
6 changed files with 426 additions and 111 deletions
+33 -2
View File
@@ -463,7 +463,7 @@ describe("runtime parity", () => {
expect(cell.toolCalls[0]?.argsHash).toBe(stableHash({ query: "release marker" }));
});
it("captures tool evidence across multiple root sessions", async () => {
it("captures fixture-owned evidence across multiple root sessions", async () => {
const now = Date.now();
const tempRoot = await seedRuntimeParityTranscript({
sessionId: "session-status-happy",
@@ -502,17 +502,48 @@ describe("runtime parity", () => {
],
updatedAt: now,
});
await seedRuntimeParityTranscript({
tempRoot,
sessionId: "unrelated-newer-root",
sessionKey: "agent:qa:unrelated-newer-root",
messages: [{ role: "user", content: "Unrelated setup." }],
updatedAt: now + 1_000,
});
const cell = await captureRuntimeParityCell({
runtime: "codex",
gateway: { tempRoot },
scenarioResult: { status: "pass" },
scenarioResult: {
status: "pass",
steps: [
{
status: "pass",
details: [
"RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:session_status:happy",
"RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:session_status:failure",
].join("\n"),
},
],
},
wallClockMs: 10,
});
expect(cell.transcriptBytes).toContain("target=session_status");
expect(cell.transcriptBytes).toContain("failure target=session_status");
expect(cell.transcriptBytes).not.toContain("Unrelated setup.");
expect(cell.toolCalls).toEqual([expect.objectContaining({ tool: "session_status" })]);
const missingCell = await captureRuntimeParityCell({
runtime: "codex",
gateway: { tempRoot },
scenarioResult: {
status: "fail",
details: "RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:missing:happy",
},
wallClockMs: 10,
});
expect(missingCell.transcriptBytes).toBe("");
expect(missingCell.toolCalls).toEqual([]);
});
it("keeps an explicitly identified orphan result separate", async () => {
+31 -2
View File
@@ -189,6 +189,7 @@ const HEARTBEAT_TRANSCRIPT_PROMPT = "[OpenClaw heartbeat poll]";
const HEARTBEAT_TASK_PROMPT_PREFIX =
"Run the following periodic tasks (only those due based on their intervals):";
const TOOL_RESULT_MISSING_ERROR_CLASS = "tool-result-missing";
const RUNTIME_PARITY_SESSION_KEY_DETAIL_PREFIX = "RUNTIME_PARITY_SESSION_KEY=";
const BOOT_STATE_LINE_RE =
/\b(?:FailoverError|No API key found|Codex app-server|auth profile|runtime policy|restart mode:|plugin|doctor)\b/i;
const TOOL_RESULT_ERROR_RE = /\b(?:error|failed|failure|timeout|denied|enoent|not found)\b/i;
@@ -1204,6 +1205,7 @@ function runtimeParitySessionEnv(stateDir: string): NodeJS.ProcessEnv {
function readRuntimeParitySessionEntries(params: {
stateDir: string;
agentId: string;
preferredSessionKeys?: ReadonlySet<string>;
}): RuntimeParitySessionCandidate[] {
try {
const entries = listSessionEntries({
@@ -1217,8 +1219,11 @@ function readRuntimeParitySessionEntries(params: {
sessionKey,
}))
.filter(({ entry }) => !readNonEmptyString(entry.heartbeatIsolatedBaseSessionKey));
const rootEntries = entries.filter(({ entry }) => isRuntimeParityRootSession(entry));
const candidates = rootEntries.length > 0 ? rootEntries : entries;
const selectedEntries = params.preferredSessionKeys
? entries.filter(({ sessionKey }) => params.preferredSessionKeys?.has(sessionKey))
: entries;
const rootEntries = selectedEntries.filter(({ entry }) => isRuntimeParityRootSession(entry));
const candidates = rootEntries.length > 0 ? rootEntries : selectedEntries;
return candidates.toSorted((left, right) => {
const leftCreatedAt = left.entry.createdAt ?? left.entry.updatedAt ?? 0;
const rightCreatedAt = right.entry.createdAt ?? right.entry.updatedAt ?? 0;
@@ -1232,6 +1237,7 @@ function readRuntimeParitySessionEntries(params: {
async function loadRuntimeParityCaptureSources(params: {
gateway: QaGatewayLike;
agentId: string;
preferredSessionKeys?: readonly string[];
}): Promise<RuntimeParityCaptureSources> {
const stateDir = `${params.gateway.tempRoot}/state`;
const env = runtimeParitySessionEnv(stateDir);
@@ -1239,6 +1245,9 @@ async function loadRuntimeParityCaptureSources(params: {
const sessionEntries = readRuntimeParitySessionEntries({
stateDir,
agentId: params.agentId,
...(params.preferredSessionKeys?.length
? { preferredSessionKeys: new Set(params.preferredSessionKeys) }
: {}),
});
const sessions: RuntimeParityCaptureSources["sessions"] = [];
for (const { entry, sessionKey } of sessionEntries) {
@@ -1289,6 +1298,25 @@ async function loadRuntimeParityCaptureSources(params: {
};
}
function runtimeParitySessionKeysFromScenarioResult(result: QaSuiteScenarioLike) {
const sessionKeys = new Set<string>();
const detailBlocks = [result.details, ...(result.steps ?? []).map((step) => step.details)];
for (const detailBlock of detailBlocks) {
for (const line of detailBlock?.split(/\r?\n/u) ?? []) {
if (!line.startsWith(RUNTIME_PARITY_SESSION_KEY_DETAIL_PREFIX)) {
continue;
}
const sessionKey = readNonEmptyString(
line.slice(RUNTIME_PARITY_SESSION_KEY_DETAIL_PREFIX.length),
);
if (sessionKey) {
sessionKeys.add(sessionKey);
}
}
}
return [...sessionKeys];
}
async function loadRuntimeParityMockToolCalls(
mockBaseUrl: string | undefined,
parentPrompt: string,
@@ -1341,6 +1369,7 @@ export async function captureRuntimeParityCell(
const { sessions, transcriptBytes } = await loadRuntimeParityCaptureSources({
gateway: params.gateway,
agentId,
preferredSessionKeys: runtimeParitySessionKeysFromScenarioResult(params.scenarioResult),
});
const transcriptRecords = buildTranscriptRecords(transcriptBytes);
// Runtime-tool fixtures split happy and failure paths across root sessions.
@@ -5,6 +5,7 @@ import path from "node:path";
import { upsertSessionEntry } from "openclaw/plugin-sdk/session-store-runtime";
import { appendSessionTranscriptMessageByIdentity } from "openclaw/plugin-sdk/session-transcript-runtime";
import { afterEach, describe, expect, it, vi } from "vitest";
import { QaSuiteInfraError } from "./errors.js";
import { runRuntimeToolFixture } from "./runtime-tool-fixture.js";
import { readRawQaSessionStore } from "./suite-runtime-agent-session.js";
import type { QaSuiteRuntimeEnv } from "./suite-runtime-types.js";
@@ -181,12 +182,16 @@ describe("runtime tool fixture", () => {
);
const createdKeys: string[] = [];
const promptKeys: string[] = [];
const promptEvidence: Array<{
requireSuccessfulTranscriptToolResult?: boolean;
transcriptToolName?: string;
}> = [];
const readEffectiveTools = vi.fn(async (_env, sessionKey: string) => {
expect(sessionKey).toBe("agent:qa:runtime-tool:read:happy");
return new Set(["read"]);
});
await runRuntimeToolFixture(
const details = await runRuntimeToolFixture(
env,
{
toolName: "read",
@@ -203,6 +208,10 @@ describe("runtime tool fixture", () => {
readEffectiveTools,
runAgentPrompt: vi.fn(async (_env, params) => {
promptKeys.push(params.sessionKey);
promptEvidence.push({
transcriptToolName: params.transcriptToolName,
requireSuccessfulTranscriptToolResult: params.requireSuccessfulTranscriptToolResult,
});
return {};
}),
fetchJson: vi.fn(),
@@ -218,6 +227,45 @@ describe("runtime tool fixture", () => {
"agent:qa:runtime-tool:read:happy",
"agent:qa:runtime-tool:read:failure",
]);
expect(promptEvidence).toEqual([
{ transcriptToolName: "read", requireSuccessfulTranscriptToolResult: true },
{ transcriptToolName: "read", requireSuccessfulTranscriptToolResult: undefined },
]);
expect(details).toContain("RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:read:happy");
expect(details).toContain("RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:read:failure");
});
it("retains both fixture session keys when the failure prompt throws", async () => {
const env = await makeEnv();
const infraError = new QaSuiteInfraError("agent_wait_failed", "failure prompt did not settle");
const runAgentPrompt = vi.fn().mockResolvedValueOnce({}).mockRejectedValueOnce(infraError);
const result = runRuntimeToolFixture(
env,
{
toolName: "read",
toolCoverage: {
bucket: "openclaw-dynamic-integration",
expectedLayer: "openclaw-dynamic",
},
},
{
createSession: vi.fn(async (_env, _label, key) => key),
readEffectiveTools: vi.fn(async () => new Set(["read"])),
runAgentPrompt,
fetchJson: vi.fn(),
ensureImageGenerationConfigured: vi.fn(),
},
);
await expect(result).rejects.toBeInstanceOf(QaSuiteInfraError);
await expect(result).rejects.toMatchObject({ code: "agent_wait_failed", cause: infraError });
await expect(result).rejects.toThrow(
[
"RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:read:happy",
"RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:read:failure",
"failure prompt did not settle",
].join("\n"),
);
});
it("requires live runtime tool fixtures to produce transcript tool output", async () => {
@@ -536,6 +584,11 @@ describe("runtime tool fixture", () => {
},
]);
const transcriptToolNames: Array<string | undefined> = [];
const runAgentPrompt = vi.fn(async (_env: unknown, params: { transcriptToolName?: string }) => {
transcriptToolNames.push(params.transcriptToolName);
return {};
});
const details = await runRuntimeToolFixture(
env,
{
@@ -551,7 +604,7 @@ describe("runtime tool fixture", () => {
{
createSession: vi.fn(async (_env, _label, key) => key!),
readEffectiveTools: vi.fn(async () => new Set<string>()),
runAgentPrompt: vi.fn(async () => ({})),
runAgentPrompt,
fetchJson,
ensureImageGenerationConfigured: vi.fn(),
},
@@ -560,6 +613,8 @@ describe("runtime tool fixture", () => {
expect(details).toContain("codex-native-workspace read");
expect(details).toContain("OpenClaw dynamic exposure is intentionally omitted");
expect(details).toContain("mock provider happy planned args (diagnostic only)");
expect(runAgentPrompt).toHaveBeenCalledTimes(2);
expect(transcriptToolNames).toEqual([undefined, undefined]);
});
it("reports Codex-native async planned-only happy fixtures without dereferencing missing output", async () => {
@@ -1083,7 +1138,13 @@ describe("runtime tool fixture", () => {
ensureImageGenerationConfigured: vi.fn(),
},
),
).rejects.toThrow("expected mock happy-path successful tool output for read");
).rejects.toThrow(
[
"RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:read:happy",
"RUNTIME_PARITY_SESSION_KEY=agent:qa:runtime-tool:read:failure",
"expected mock happy-path successful tool output for read",
].join("\n"),
);
});
it("requires mock failure fixtures to produce failure-shaped tool output", async () => {
+193 -104
View File
@@ -1,8 +1,10 @@
// Qa Lab plugin module implements runtime tool fixture behavior.
import fs from "node:fs/promises";
import path from "node:path";
import { formatErrorMessage } from "openclaw/plugin-sdk/error-runtime";
import { loadTranscriptEventsSync } from "openclaw/plugin-sdk/session-store-runtime";
import { isRecord } from "openclaw/plugin-sdk/string-coerce-runtime";
import { QaSuiteInfraError } from "./errors.js";
import {
qaMockRequestCursorUrl,
qaMockRequestsAfterUrl,
@@ -54,6 +56,28 @@ type QaRuntimeToolFixtureTranscriptToolResult = {
structuredFailure: boolean;
};
const RUNTIME_PARITY_SESSION_KEY_DETAIL_PREFIX = "RUNTIME_PARITY_SESSION_KEY=";
function runtimeParitySessionKeyDetails(...sessionKeys: string[]) {
return sessionKeys.map(
(sessionKey) => `${RUNTIME_PARITY_SESSION_KEY_DETAIL_PREFIX}${sessionKey}`,
);
}
function runtimeToolFixtureDetails(details: string, ...sessionKeys: string[]) {
return [details, ...runtimeParitySessionKeyDetails(...sessionKeys)].join("\n");
}
function runtimeToolFixtureError(error: unknown, ...sessionKeys: string[]) {
const message = [
...runtimeParitySessionKeyDetails(...sessionKeys),
formatErrorMessage(error),
].join("\n");
return error instanceof QaSuiteInfraError
? new QaSuiteInfraError(error.code, message, { cause: error })
: new Error(message, { cause: error });
}
type QaRuntimeToolFixtureDeps = {
createSession: (
env: Pick<QaSuiteRuntimeEnv, "gateway" | "primaryModel" | "alternateModel" | "providerMode">,
@@ -70,6 +94,8 @@ type QaRuntimeToolFixtureDeps = {
sessionKey: string;
message: string;
timeoutMs?: number;
transcriptToolName?: string;
requireSuccessfulTranscriptToolResult?: boolean;
},
) => Promise<unknown>;
fetchJson: (url: string) => Promise<unknown>;
@@ -631,7 +657,18 @@ export async function runRuntimeToolFixture(
`Runtime tool fixture: ${toolName} failure`,
`agent:qa:runtime-tool:${toolName}:failure`,
);
const tools = await deps.readEffectiveTools(env, happySessionKey);
const sessionKeys = [happySessionKey, failureSessionKey] as const;
const withSessionDetails = (details: string) =>
runtimeToolFixtureDetails(details, ...sessionKeys);
const fixtureError = (error: unknown) => runtimeToolFixtureError(error, ...sessionKeys);
const runFixtureOperation = async <T>(operation: () => Promise<T>): Promise<T> => {
try {
return await operation();
} catch (error) {
throw fixtureError(error);
}
};
const tools = await runFixtureOperation(() => deps.readEffectiveTools(env, happySessionKey));
const metadata = readRuntimeToolCoverageMetadata({
config,
});
@@ -641,16 +678,18 @@ export async function runRuntimeToolFixture(
const expectedAvailable = readBoolean(config.expectedAvailable, true);
if (!tools.has(toolName) && !dynamicExposureIntentionallyExcluded) {
if (!expectedAvailable) {
return formatExpectedUnavailableDetails(toolName, tools);
return withSessionDetails(formatExpectedUnavailableDetails(toolName, tools));
}
if (isKnownBroken(config.knownBroken)) {
return formatKnownBrokenDetails(toolName, tools, config);
return withSessionDetails(formatKnownBrokenDetails(toolName, tools, config));
}
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(
`${toolName} not present in effective tools. Available tools: ${[...tools].toSorted().join(", ")}`,
throw fixtureError(
new Error(
`${toolName} not present in effective tools. Available tools: ${[...tools].toSorted().join(", ")}`,
),
);
}
@@ -668,27 +707,44 @@ export async function runRuntimeToolFixture(
`failure target=${toolName}`,
);
const happyPathOutputRequired = readBoolean(config.happyPathOutputRequired, true);
const requestCursorBefore = env.mock
? readQaMockRequestCursor(await deps.fetchJson(qaMockRequestCursorUrl(env.mock.baseUrl)))
const requireTranscriptEvidence =
metadata.required &&
!dynamicExposureIntentionallyExcluded &&
!isKnownHarnessGap(config.knownHarnessGap);
const mockBaseUrl = env.mock?.baseUrl;
const requestCursorBefore = mockBaseUrl
? await runFixtureOperation(async () =>
readQaMockRequestCursor(await deps.fetchJson(qaMockRequestCursorUrl(mockBaseUrl))),
)
: 0;
await deps.runAgentPrompt(env, {
sessionKey: happySessionKey,
message: happyPrompt,
timeoutMs: liveTurnTimeoutMs(env, 45_000),
});
await deps.runAgentPrompt(env, {
sessionKey: failureSessionKey,
message: failurePrompt,
timeoutMs: liveTurnTimeoutMs(env, 45_000),
});
await runFixtureOperation(() =>
deps.runAgentPrompt(env, {
sessionKey: happySessionKey,
message: happyPrompt,
timeoutMs: liveTurnTimeoutMs(env, 45_000),
...(happyPathOutputRequired && requireTranscriptEvidence
? { transcriptToolName: toolName, requireSuccessfulTranscriptToolResult: true }
: {}),
}),
);
await runFixtureOperation(() =>
deps.runAgentPrompt(env, {
sessionKey: failureSessionKey,
message: failurePrompt,
timeoutMs: liveTurnTimeoutMs(env, 45_000),
...(requireTranscriptEvidence ? { transcriptToolName: toolName } : {}),
}),
);
if (!env.mock) {
const happyRequest = await readLiveToolEvidence({
env,
sessionKey: happySessionKey,
toolName,
});
const happyRequest = await runFixtureOperation(() =>
readLiveToolEvidence({
env,
sessionKey: happySessionKey,
toolName,
}),
);
if (!happyRequest.outputRequest) {
const happyPlannedOnly = happyRequest.plannedRequest && !happyPathOutputRequired;
if (happyPlannedOnly) {
@@ -696,55 +752,70 @@ export async function runRuntimeToolFixture(
// by their task lifecycle scenarios.
} else {
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(
happyRequest.plannedRequest
? `expected live happy-path tool output for ${toolName}`
: `expected live happy-path tool call for ${toolName}`,
throw fixtureError(
new Error(
happyRequest.plannedRequest
? `expected live happy-path tool output for ${toolName}`
: `expected live happy-path tool call for ${toolName}`,
),
);
}
}
if (happyRequest.outputRequest?.structuredFailure) {
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(`expected live happy-path successful tool output for ${toolName}`);
throw fixtureError(
new Error(`expected live happy-path successful tool output for ${toolName}`),
);
}
const failureRequest = await readLiveToolEvidence({
env,
sessionKey: failureSessionKey,
toolName,
});
const failureRequest = await runFixtureOperation(() =>
readLiveToolEvidence({
env,
sessionKey: failureSessionKey,
toolName,
}),
);
if (!failureRequest.outputRequest) {
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(
failureRequest.plannedRequest
? `expected live failure-path tool output for ${toolName}`
: `expected live failure-path tool call for ${toolName}`,
throw fixtureError(
new Error(
failureRequest.plannedRequest
? `expected live failure-path tool output for ${toolName}`
: `expected live failure-path tool call for ${toolName}`,
),
);
}
if (!failureRequest.failureOutputRequest) {
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(`expected live failure-path tool failure output for ${toolName}`);
throw fixtureError(
new Error(`expected live failure-path tool failure output for ${toolName}`),
);
}
return [
`${toolName} live provider happy planned args (diagnostic only): ${JSON.stringify(happyRequest.plannedRequest?.args ?? {})}`,
happyPathOutputRequired
? undefined
: `${toolName} live provider happy direct output not required for this async fixture`,
`${toolName} live provider failure planned args (diagnostic only): ${JSON.stringify(failureRequest.plannedRequest?.args ?? {})}`,
]
.filter(Boolean)
.join("\n");
return withSessionDetails(
[
`${toolName} live provider happy planned args (diagnostic only): ${JSON.stringify(happyRequest.plannedRequest?.args ?? {})}`,
happyPathOutputRequired
? undefined
: `${toolName} live provider happy direct output not required for this async fixture`,
`${toolName} live provider failure planned args (diagnostic only): ${JSON.stringify(failureRequest.plannedRequest?.args ?? {})}`,
]
.filter(Boolean)
.join("\n"),
);
}
const requests = readQaRuntimeToolFixtureRequests(
await deps.fetchJson(qaMockRequestsAfterUrl(env.mock.baseUrl, requestCursorBefore)),
const activeMockBaseUrl = env.mock.baseUrl;
const requests = await runFixtureOperation(async () =>
readQaRuntimeToolFixtureRequests(
await deps.fetchJson(qaMockRequestsAfterUrl(activeMockBaseUrl, requestCursorBefore)),
),
);
const happyPlannedRequest = findPlannedRequest({
requests,
@@ -775,91 +846,109 @@ export async function runRuntimeToolFixture(
!happyRequest
) {
if (!plannedRequestHasPrompt(happyPlannedRequest)) {
throw new Error(`expected mock happy-path prompt args for ${toolName}`);
throw fixtureError(new Error(`expected mock happy-path prompt args for ${toolName}`));
}
if (!plannedRequestHasDeniedInputFailure(failurePlannedRequest)) {
throw new Error(`expected mock failure-path denied-input args for ${toolName}`);
throw fixtureError(new Error(`expected mock failure-path denied-input args for ${toolName}`));
}
if (failureRequest && !requestHasFailureLikeToolOutput(failureRequest.outputRequest)) {
throw new Error(`expected mock failure-path tool failure output for ${toolName}`);
throw fixtureError(
new Error(`expected mock failure-path tool failure output for ${toolName}`),
);
}
return formatReportOnlyMockDetails({
toolName,
happyRequest: happyPlannedRequest,
failureRequest: failurePlannedRequest,
});
return withSessionDetails(
formatReportOnlyMockDetails({
toolName,
happyRequest: happyPlannedRequest,
failureRequest: failurePlannedRequest,
}),
);
}
// Async runtime tools prove the start call here; completion is covered by
// their task lifecycle scenarios.
const happyPlannedOnly = Boolean(happyPlannedRequest && !happyPathOutputRequired);
if (!happyRequest && !happyPlannedOnly) {
if (dynamicExposureIntentionallyExcluded) {
return formatCodexNativeWorkspaceDetails({
toolName,
tools,
reason: metadata.reason,
happyRequest: happyPlannedRequest,
});
return withSessionDetails(
formatCodexNativeWorkspaceDetails({
toolName,
tools,
reason: metadata.reason,
happyRequest: happyPlannedRequest,
}),
);
}
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(
happyPlannedRequest
? `expected mock happy-path tool output for ${toolName}`
: `expected mock happy-path request for ${toolName}`,
throw fixtureError(
new Error(
happyPlannedRequest
? `expected mock happy-path tool output for ${toolName}`
: `expected mock happy-path request for ${toolName}`,
),
);
}
if (happyRequest && requestHasHappyPathFailureToolOutput(happyRequest.outputRequest)) {
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(`expected mock happy-path successful tool output for ${toolName}`);
throw fixtureError(
new Error(`expected mock happy-path successful tool output for ${toolName}`),
);
}
if (!failureRequest) {
if (dynamicExposureIntentionallyExcluded) {
return formatCodexNativeWorkspaceDetails({
toolName,
tools,
reason: metadata.reason,
happyRequest: happyPlannedRequest,
failureRequest: failurePlannedRequest,
});
return withSessionDetails(
formatCodexNativeWorkspaceDetails({
toolName,
tools,
reason: metadata.reason,
happyRequest: happyPlannedRequest,
failureRequest: failurePlannedRequest,
}),
);
}
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(
failurePlannedRequest
? `expected mock failure-path tool output for ${toolName}`
: `expected mock failure-path request for ${toolName}`,
throw fixtureError(
new Error(
failurePlannedRequest
? `expected mock failure-path tool output for ${toolName}`
: `expected mock failure-path request for ${toolName}`,
),
);
}
if (!requestHasFailureLikeToolOutput(failureRequest.outputRequest)) {
if (isKnownHarnessGap(config.knownHarnessGap)) {
return formatKnownHarnessGapDetails(toolName, config);
return withSessionDetails(formatKnownHarnessGapDetails(toolName, config));
}
throw new Error(`expected mock failure-path tool failure output for ${toolName}`);
throw fixtureError(new Error(`expected mock failure-path tool failure output for ${toolName}`));
}
if (dynamicExposureIntentionallyExcluded) {
return formatCodexNativeWorkspaceDetails({
toolName,
tools,
reason: metadata.reason,
happyRequest: happyRequest?.plannedRequest ?? happyPlannedRequest,
failureRequest: failureRequest.plannedRequest,
});
return withSessionDetails(
formatCodexNativeWorkspaceDetails({
toolName,
tools,
reason: metadata.reason,
happyRequest: happyRequest?.plannedRequest ?? happyPlannedRequest,
failureRequest: failureRequest.plannedRequest,
}),
);
}
return [
`${toolName} mock provider happy planned args (diagnostic only): ${formatPlannedToolArgs((happyRequest?.plannedRequest ?? happyPlannedRequest)?.plannedToolArgs)}`,
happyPathOutputRequired
? undefined
: `${toolName} mock provider happy direct output not required for this async fixture`,
`${toolName} mock provider failure planned args (diagnostic only): ${formatPlannedToolArgs(failureRequest.plannedRequest.plannedToolArgs)}`,
]
.filter(Boolean)
.join("\n");
return withSessionDetails(
[
`${toolName} mock provider happy planned args (diagnostic only): ${formatPlannedToolArgs((happyRequest?.plannedRequest ?? happyPlannedRequest)?.plannedToolArgs)}`,
happyPathOutputRequired
? undefined
: `${toolName} mock provider happy direct output not required for this async fixture`,
`${toolName} mock provider failure planned args (diagnostic only): ${formatPlannedToolArgs(failureRequest.plannedRequest.plannedToolArgs)}`,
]
.filter(Boolean)
.join("\n"),
);
}
/* oxlint-disable max-lines -- TODO: split this grandfathered oversized file. */
@@ -9,6 +9,7 @@ const spawnSyncMock = vi.hoisted(() => vi.fn());
const resolveQaNodeExecPathMock = vi.hoisted(() => vi.fn(async () => "/usr/bin/node"));
const waitForGatewayHealthyMock = vi.hoisted(() => vi.fn(async () => undefined));
const waitForTransportReadyMock = vi.hoisted(() => vi.fn(async () => undefined));
const readSessionTranscriptSummaryMock = vi.hoisted(() => vi.fn());
vi.mock("node:child_process", () => ({
spawn: spawnMock,
@@ -24,6 +25,10 @@ vi.mock("./suite-runtime-gateway.js", () => ({
waitForTransportReady: waitForTransportReadyMock,
}));
vi.mock("./suite-runtime-agent-session.js", () => ({
readSessionTranscriptSummary: readSessionTranscriptSummaryMock,
}));
import { QA_CHILD_STDERR_TAIL_BYTES, QA_CHILD_STDOUT_MAX_BYTES } from "./child-output.js";
import {
findManagedDreamingCronJob,
@@ -86,6 +91,7 @@ describe("qa suite runtime agent process helpers", () => {
resolveQaNodeExecPathMock.mockClear();
waitForGatewayHealthyMock.mockClear();
waitForTransportReadyMock.mockClear();
readSessionTranscriptSummaryMock.mockReset();
});
it("runs the qa cli through the resolved node executable", async () => {
@@ -768,6 +774,53 @@ describe("qa suite runtime agent process helpers", () => {
});
});
it("waits for persisted transcript tool evidence after agent completion", async () => {
vi.useFakeTimers();
try {
const gatewayCall = vi
.fn()
.mockResolvedValueOnce({ runId: "run-transcript-evidence" })
.mockResolvedValueOnce({ status: "completed" });
readSessionTranscriptSummaryMock
.mockResolvedValueOnce({
assistantToolCallCounts: {},
successfulToolCallCounts: {},
finalText: "",
})
.mockResolvedValueOnce({
assistantToolCallCounts: { web_fetch: 1 },
successfulToolCallCounts: { web_fetch: 1 },
finalText: "done",
});
const env = {
gateway: { call: gatewayCall },
transport: {
buildAgentDelivery: vi.fn(() => ({
channel: "qa-channel",
replyChannel: "reply-channel",
replyTo: "reply-target",
})),
},
} as never;
const pending = runAgentPrompt(env, {
sessionKey: "session-transcript-evidence",
message: "call web_fetch",
transcriptToolName: "web_fetch",
requireSuccessfulTranscriptToolResult: true,
});
await vi.advanceTimersByTimeAsync(50);
await expect(pending).resolves.toEqual({
started: { runId: "run-transcript-evidence" },
waited: { status: "completed" },
});
expect(readSessionTranscriptSummaryMock).toHaveBeenCalledTimes(2);
} finally {
vi.useRealTimers();
}
});
it("waits for the latest assistant history reply", async () => {
const gatewayCall = vi
.fn()
@@ -20,6 +20,7 @@ import { QaSuiteInfraError } from "./errors.js";
import { extractGatewayMessageText } from "./gateway-log-sentinel.js";
import { resolveQaNodeExecPath } from "./node-exec.js";
import { liveTurnTimeoutMs } from "./suite-runtime-agent-common.js";
import { readSessionTranscriptSummary } from "./suite-runtime-agent-session.js";
import { waitForGatewayHealthy, waitForTransportReady } from "./suite-runtime-gateway.js";
import type { QaDreamingStatus, QaSuiteRuntimeEnv } from "./suite-runtime-types.js";
import { resolveQaGatewayTimeoutWithGraceMs } from "./timer-timeouts.js";
@@ -56,6 +57,8 @@ const MANAGED_DREAMING_PROMPT = "__openclaw_memory_core_short_term_promotion_dre
const QA_HISTORY_RETRY_DEFAULT_MS = 250;
const QA_HISTORY_RETRY_MIN_MS = 100;
const QA_HISTORY_RETRY_MAX_MS = 5_000;
const QA_TRANSCRIPT_EVIDENCE_TIMEOUT_MS = 5_000;
const QA_TRANSCRIPT_EVIDENCE_POLL_MS = 50;
function stripAnsiCodes(text: string) {
return text.replace(ANSI_ESCAPE_PATTERN, "");
@@ -602,6 +605,46 @@ async function forceMemoryIndex(params: {
return result;
}
async function waitForPersistedTranscriptToolEvidence(
env: Pick<QaSuiteRuntimeEnv, "gateway">,
params: {
sessionKey: string;
toolName: string;
requireSuccessfulResult: boolean;
},
) {
const startedAt = Date.now();
let lastError: unknown;
while (Date.now() - startedAt < QA_TRANSCRIPT_EVIDENCE_TIMEOUT_MS) {
try {
const summary = await readSessionTranscriptSummary(env, params.sessionKey, {
allowEmpty: true,
});
const callCount = summary.assistantToolCallCounts[params.toolName] ?? 0;
const successfulCount = summary.successfulToolCallCounts[params.toolName] ?? 0;
if (
summary.finalText &&
callCount > 0 &&
(!params.requireSuccessfulResult || successfulCount > 0)
) {
return;
}
lastError = undefined;
} catch (error) {
lastError = error;
}
const remainingMs = QA_TRANSCRIPT_EVIDENCE_TIMEOUT_MS - (Date.now() - startedAt);
if (remainingMs <= 0) {
break;
}
await sleep(Math.min(QA_TRANSCRIPT_EVIDENCE_POLL_MS, remainingMs));
}
throw new Error(
`timed out after ${QA_TRANSCRIPT_EVIDENCE_TIMEOUT_MS}ms waiting for persisted ${params.toolName} transcript evidence`,
lastError === undefined ? undefined : { cause: lastError },
);
}
async function runAgentPrompt(
env: Pick<QaSuiteRuntimeEnv, "gateway" | "transport">,
params: {
@@ -612,6 +655,8 @@ async function runAgentPrompt(
provider?: string;
model?: string;
timeoutMs?: number;
transcriptToolName?: string;
requireSuccessfulTranscriptToolResult?: boolean;
attachments?: Array<{
mimeType: string;
fileName: string;
@@ -626,6 +671,13 @@ async function runAgentPrompt(
`agent.wait returned ${waited.status ?? "unknown"}: ${waited.error ?? "no error"}`,
);
}
if (params.transcriptToolName) {
await waitForPersistedTranscriptToolEvidence(env, {
sessionKey: params.sessionKey,
toolName: params.transcriptToolName,
requireSuccessfulResult: params.requireSuccessfulTranscriptToolResult === true,
});
}
return {
started,
waited,