diff --git a/docs/nodes/media-understanding.md b/docs/nodes/media-understanding.md index e091d79e92d9..0f8c6616673f 100644 --- a/docs/nodes/media-understanding.md +++ b/docs/nodes/media-understanding.md @@ -269,7 +269,7 @@ When `mode: "all"`, outputs are labeled `[Image 1/2]`, `[Audio 2/2]`, etc. - At most five skip markers render per message; further skipped attachments collapse into one reason-neutral `[ more attachments skipped]` summary so junk attachments cannot grow the prompt without bound. File and image, audio, or video markers share this five-marker budget. - If a PDF falls back to rendered page images, OpenClaw forwards those images to vision-capable reply models and keeps the placeholder `[PDF content rendered to images]` in the file block. - Image, audio, and video decisions record one closed disposition for every attachment candidate: handled, handed to native vision, not selected after the attachment limit, disabled, missing a model, denied by chat scope, or failed. -- Unhandled media gets a bounded model-visible marker. Images handed to native vision and media turns owned by another harness do not add markers. +- Unhandled media gets a bounded model-visible marker. Images handed to native vision do not add markers. When a native harness owns the turn and OpenClaw runs only audio preprocessing, failed or skipped audio still gets a marker; image, video, and document inputs remain owned by the harness. Too-small audio keeps its placeholder transcript without a duplicate marker. ## Config examples diff --git a/src/media-understanding/apply.test.ts b/src/media-understanding/apply.test.ts index 70b8b2b9d78a..ab759d623d8e 100644 --- a/src/media-understanding/apply.test.ts +++ b/src/media-understanding/apply.test.ts @@ -613,81 +613,92 @@ describe("applyMediaUnderstanding", () => { ); }); - it("injects a placeholder transcript when local-path audio is too small", async () => { - const ctx = await createAudioCtx({ - fileName: "tiny.ogg", - mediaType: "audio/ogg", - content: Buffer.alloc(100), - }); - const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" })); - const cfg: OpenClawConfig = { - tools: { - media: { - models: [{ provider: "groq", capabilities: ["audio"] }], - audio: { - enabled: true, - maxBytes: 1024 * 1024, + it.each([undefined, "audio-only"] as const)( + "injects one placeholder for too-small local audio in %s mode", + async (processingMode) => { + const ctx = await createAudioCtx({ + fileName: "tiny.ogg", + mediaType: "audio/ogg", + content: Buffer.alloc(100), + }); + const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" })); + const cfg: OpenClawConfig = { + tools: { + media: { + models: [{ provider: "groq", capabilities: ["audio"] }], + audio: { + enabled: true, + maxBytes: 1024 * 1024, + }, }, }, - }, - }; + }; - const result = await applyMediaUnderstanding({ - ctx, - cfg, - providers: { - groq: { id: "groq", transcribeAudio }, - }, - }); + const result = await applyMediaUnderstanding({ + ctx, + cfg, + processingMode, + providers: { + groq: { id: "groq", transcribeAudio }, + }, + }); - expect(transcribeAudio).not.toHaveBeenCalled(); - expect(result.appliedAudio).toBe(true); - expect(result.outputs).toEqual([ - { - kind: "audio.transcription", - attachmentIndex: 0, - text: "[Voice note could not be transcribed because the audio attachment was too small]", - provider: "openclaw", - model: "synthetic-empty-audio", - }, - ]); - expect(ctx.Transcript).toBe( - "[Voice note could not be transcribed because the audio attachment was too small]", - ); - expect(ctx.Body).toBe( - "[Audio]\nTranscript:\n[Voice note could not be transcribed because the audio attachment was too small]", - ); - }); + expect(transcribeAudio).not.toHaveBeenCalled(); + expect(result.appliedAudio).toBe(true); + expect(result.outputs).toEqual([ + { + kind: "audio.transcription", + attachmentIndex: 0, + text: "[Voice note could not be transcribed because the audio attachment was too small]", + provider: "openclaw", + model: "synthetic-empty-audio", + }, + ]); + expect(ctx.Transcript).toBe( + "[Voice note could not be transcribed because the audio attachment was too small]", + ); + expect(ctx.Body).toBe( + "[Audio]\nTranscript:\n[Voice note could not be transcribed because the audio attachment was too small]", + ); + }, + ); - it("skips audio transcription when attachment exceeds maxBytes", async () => { - const ctx = await createAudioCtx({ - fileName: "large.wav", - mediaType: "audio/wav", - content: Buffer.from([0, 255, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10]), - }); - const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" })); - const cfg: OpenClawConfig = { - tools: { - media: { - models: [{ provider: "groq", capabilities: ["audio"] }], - audio: { - enabled: true, - maxBytes: 4, + it.each([undefined, "audio-only"] as const)( + "marks audio exceeding maxBytes in %s mode", + async (processingMode) => { + const ctx = await createAudioCtx({ + fileName: "large.wav", + mediaType: "audio/wav", + content: Buffer.from([0, 255, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10]), + }); + const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" })); + const cfg: OpenClawConfig = { + tools: { + media: { + models: [{ provider: "groq", capabilities: ["audio"] }], + audio: { + enabled: true, + maxBytes: 4, + }, }, }, - }, - }; + }; - const result = await applyMediaUnderstanding({ - ctx, - cfg, - providers: { groq: { id: "groq", transcribeAudio } }, - }); + const result = await applyMediaUnderstanding({ + ctx, + cfg, + processingMode, + providers: { groq: { id: "groq", transcribeAudio } }, + }); - expect(result.appliedAudio).toBe(false); - expect(transcribeAudio).not.toHaveBeenCalled(); - expect(ctx.Body).toBe("[Audio attachment could not be analyzed]"); - }); + expect(result.appliedAudio).toBe(false); + expect(result.outputs).toEqual([]); + expect(ctx.Transcript).toBeUndefined(); + expect(transcribeAudio).not.toHaveBeenCalled(); + expect(ctx.Body).toBe("[Audio attachment could not be analyzed]"); + expect(ctx.BodyForAgent).toBe(ctx.Body); + }, + ); it("falls back to CLI model when provider fails", async () => { const ctx = await createAudioCtx(); @@ -1676,7 +1687,11 @@ describe("applyMediaUnderstanding", () => { expect(ctx.BodyForCommands).toBe("audio ok"); }); - it("limits native-harness preprocessing to audio", async () => { + it.each([ + { outcome: "success", body: "[Audio]\nTranscript:\naudio ok" }, + { outcome: "failure", body: "[Audio attachment could not be analyzed]" }, + { outcome: "scope-denied", body: "[Audio attachment not analyzed in this chat]" }, + ])("limits native-harness preprocessing to audio on STT $outcome", async ({ outcome, body }) => { const dir = await createTempMediaDir(); const imagePath = path.join(dir, "photo.jpg"); const audioPath = path.join(dir, "note.ogg"); @@ -1686,11 +1701,17 @@ describe("applyMediaUnderstanding", () => { await fs.writeFile(filePath, "file text"); const describeImage = vi.fn(async () => ({ text: "image ok" })); - const transcribeAudio = vi.fn(async () => ({ text: "audio ok" })); + const transcribeAudio = vi.fn(async () => { + if (outcome === "failure") { + throw new Error("transcription provider unavailable"); + } + return { text: "audio ok" }; + }); const ctx: MsgContext = { Body: "", media: [ { path: imagePath, contentType: "image/jpeg" }, + { url: "https://example.test/clip.mp4", contentType: "video/mp4" }, { path: audioPath, contentType: "audio/ogg" }, { path: filePath, contentType: "text/plain" }, ], @@ -1703,7 +1724,10 @@ describe("applyMediaUnderstanding", () => { { provider: "groq", capabilities: ["audio"] }, ], image: { enabled: true }, - audio: { enabled: true }, + audio: { + enabled: true, + scope: { default: outcome === "scope-denied" ? "deny" : "allow" }, + }, }, }, }; @@ -1719,17 +1743,19 @@ describe("applyMediaUnderstanding", () => { }); expect(describeImage).not.toHaveBeenCalled(); - expect(transcribeAudio).toHaveBeenCalledOnce(); + expect(transcribeAudio).toHaveBeenCalledTimes(outcome === "scope-denied" ? 0 : 1); expect(result).toEqual( expect.objectContaining({ appliedImage: false, - appliedAudio: true, + appliedAudio: outcome === "success", appliedVideo: false, appliedFile: false, extractedFileImages: [], }), ); - expect(ctx.Body).toBe("[Audio]\nTranscript:\naudio ok"); + expect(ctx.Body).toBe(body); + expect(ctx.BodyForAgent).toBe(body); + expect(ctx.Transcript).toBe(outcome === "success" ? "audio ok" : undefined); }); it("orders synthetic too-small audio output between image and video", async () => { diff --git a/src/media-understanding/apply.ts b/src/media-understanding/apply.ts index 6939af788bf0..915ba8e79f2e 100644 --- a/src/media-understanding/apply.ts +++ b/src/media-understanding/apply.ts @@ -611,15 +611,14 @@ export async function applyMediaUnderstanding(params: { // placement, suppress — a wrong path is worse than the plain marker (#122411). selfServePathsEnabled: params.selfServeLocalPaths === true, }); - const mediaMarkers = - params.processingMode === "audio-only" - ? [] - : renderMediaAttachmentMarkers({ - attachments, - decisions, - outputs, - deliveredImageIndexes: params.deliveredImageIndexes, - }); + // Only processed capabilities have decisions, so audio-only runs cannot + // add markers for image/video inputs still owned by the native harness. + const mediaMarkers = renderMediaAttachmentMarkers({ + attachments, + decisions, + outputs, + deliveredImageIndexes: params.deliveredImageIndexes, + }); const contextBlocks = applyAttachmentMarkerBudget([...fileContext.blocks, ...mediaMarkers]); if (contextBlocks.length > 0) { ctx.Body = appendFileBlocks(ctx.Body, contextBlocks);