mirror of
https://github.com/openclaw/openclaw.git
synced 2026-08-28 05:16:23 -06:00
fix(media): keep failure markers visible in audio-only processing (#130869)
This commit is contained in:
committed by
GitHub
parent
47882f48cb
commit
2abc76d1ae
@@ -269,7 +269,7 @@ When `mode: "all"`, outputs are labeled `[Image 1/2]`, `[Audio 2/2]`, etc.
|
||||
- At most five skip markers render per message; further skipped attachments collapse into one reason-neutral `[<n> more attachments skipped]` summary so junk attachments cannot grow the prompt without bound. File and image, audio, or video markers share this five-marker budget.
|
||||
- If a PDF falls back to rendered page images, OpenClaw forwards those images to vision-capable reply models and keeps the placeholder `[PDF content rendered to images]` in the file block.
|
||||
- Image, audio, and video decisions record one closed disposition for every attachment candidate: handled, handed to native vision, not selected after the attachment limit, disabled, missing a model, denied by chat scope, or failed.
|
||||
- Unhandled media gets a bounded model-visible marker. Images handed to native vision and media turns owned by another harness do not add markers.
|
||||
- Unhandled media gets a bounded model-visible marker. Images handed to native vision do not add markers. When a native harness owns the turn and OpenClaw runs only audio preprocessing, failed or skipped audio still gets a marker; image, video, and document inputs remain owned by the harness. Too-small audio keeps its placeholder transcript without a duplicate marker.
|
||||
|
||||
## Config examples
|
||||
|
||||
|
||||
@@ -613,81 +613,92 @@ describe("applyMediaUnderstanding", () => {
|
||||
);
|
||||
});
|
||||
|
||||
it("injects a placeholder transcript when local-path audio is too small", async () => {
|
||||
const ctx = await createAudioCtx({
|
||||
fileName: "tiny.ogg",
|
||||
mediaType: "audio/ogg",
|
||||
content: Buffer.alloc(100),
|
||||
});
|
||||
const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" }));
|
||||
const cfg: OpenClawConfig = {
|
||||
tools: {
|
||||
media: {
|
||||
models: [{ provider: "groq", capabilities: ["audio"] }],
|
||||
audio: {
|
||||
enabled: true,
|
||||
maxBytes: 1024 * 1024,
|
||||
it.each([undefined, "audio-only"] as const)(
|
||||
"injects one placeholder for too-small local audio in %s mode",
|
||||
async (processingMode) => {
|
||||
const ctx = await createAudioCtx({
|
||||
fileName: "tiny.ogg",
|
||||
mediaType: "audio/ogg",
|
||||
content: Buffer.alloc(100),
|
||||
});
|
||||
const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" }));
|
||||
const cfg: OpenClawConfig = {
|
||||
tools: {
|
||||
media: {
|
||||
models: [{ provider: "groq", capabilities: ["audio"] }],
|
||||
audio: {
|
||||
enabled: true,
|
||||
maxBytes: 1024 * 1024,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
};
|
||||
|
||||
const result = await applyMediaUnderstanding({
|
||||
ctx,
|
||||
cfg,
|
||||
providers: {
|
||||
groq: { id: "groq", transcribeAudio },
|
||||
},
|
||||
});
|
||||
const result = await applyMediaUnderstanding({
|
||||
ctx,
|
||||
cfg,
|
||||
processingMode,
|
||||
providers: {
|
||||
groq: { id: "groq", transcribeAudio },
|
||||
},
|
||||
});
|
||||
|
||||
expect(transcribeAudio).not.toHaveBeenCalled();
|
||||
expect(result.appliedAudio).toBe(true);
|
||||
expect(result.outputs).toEqual([
|
||||
{
|
||||
kind: "audio.transcription",
|
||||
attachmentIndex: 0,
|
||||
text: "[Voice note could not be transcribed because the audio attachment was too small]",
|
||||
provider: "openclaw",
|
||||
model: "synthetic-empty-audio",
|
||||
},
|
||||
]);
|
||||
expect(ctx.Transcript).toBe(
|
||||
"[Voice note could not be transcribed because the audio attachment was too small]",
|
||||
);
|
||||
expect(ctx.Body).toBe(
|
||||
"[Audio]\nTranscript:\n[Voice note could not be transcribed because the audio attachment was too small]",
|
||||
);
|
||||
});
|
||||
expect(transcribeAudio).not.toHaveBeenCalled();
|
||||
expect(result.appliedAudio).toBe(true);
|
||||
expect(result.outputs).toEqual([
|
||||
{
|
||||
kind: "audio.transcription",
|
||||
attachmentIndex: 0,
|
||||
text: "[Voice note could not be transcribed because the audio attachment was too small]",
|
||||
provider: "openclaw",
|
||||
model: "synthetic-empty-audio",
|
||||
},
|
||||
]);
|
||||
expect(ctx.Transcript).toBe(
|
||||
"[Voice note could not be transcribed because the audio attachment was too small]",
|
||||
);
|
||||
expect(ctx.Body).toBe(
|
||||
"[Audio]\nTranscript:\n[Voice note could not be transcribed because the audio attachment was too small]",
|
||||
);
|
||||
},
|
||||
);
|
||||
|
||||
it("skips audio transcription when attachment exceeds maxBytes", async () => {
|
||||
const ctx = await createAudioCtx({
|
||||
fileName: "large.wav",
|
||||
mediaType: "audio/wav",
|
||||
content: Buffer.from([0, 255, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10]),
|
||||
});
|
||||
const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" }));
|
||||
const cfg: OpenClawConfig = {
|
||||
tools: {
|
||||
media: {
|
||||
models: [{ provider: "groq", capabilities: ["audio"] }],
|
||||
audio: {
|
||||
enabled: true,
|
||||
maxBytes: 4,
|
||||
it.each([undefined, "audio-only"] as const)(
|
||||
"marks audio exceeding maxBytes in %s mode",
|
||||
async (processingMode) => {
|
||||
const ctx = await createAudioCtx({
|
||||
fileName: "large.wav",
|
||||
mediaType: "audio/wav",
|
||||
content: Buffer.from([0, 255, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10]),
|
||||
});
|
||||
const transcribeAudio = vi.fn(async () => ({ text: "should-not-run" }));
|
||||
const cfg: OpenClawConfig = {
|
||||
tools: {
|
||||
media: {
|
||||
models: [{ provider: "groq", capabilities: ["audio"] }],
|
||||
audio: {
|
||||
enabled: true,
|
||||
maxBytes: 4,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
};
|
||||
|
||||
const result = await applyMediaUnderstanding({
|
||||
ctx,
|
||||
cfg,
|
||||
providers: { groq: { id: "groq", transcribeAudio } },
|
||||
});
|
||||
const result = await applyMediaUnderstanding({
|
||||
ctx,
|
||||
cfg,
|
||||
processingMode,
|
||||
providers: { groq: { id: "groq", transcribeAudio } },
|
||||
});
|
||||
|
||||
expect(result.appliedAudio).toBe(false);
|
||||
expect(transcribeAudio).not.toHaveBeenCalled();
|
||||
expect(ctx.Body).toBe("[Audio attachment could not be analyzed]");
|
||||
});
|
||||
expect(result.appliedAudio).toBe(false);
|
||||
expect(result.outputs).toEqual([]);
|
||||
expect(ctx.Transcript).toBeUndefined();
|
||||
expect(transcribeAudio).not.toHaveBeenCalled();
|
||||
expect(ctx.Body).toBe("[Audio attachment could not be analyzed]");
|
||||
expect(ctx.BodyForAgent).toBe(ctx.Body);
|
||||
},
|
||||
);
|
||||
|
||||
it("falls back to CLI model when provider fails", async () => {
|
||||
const ctx = await createAudioCtx();
|
||||
@@ -1676,7 +1687,11 @@ describe("applyMediaUnderstanding", () => {
|
||||
expect(ctx.BodyForCommands).toBe("audio ok");
|
||||
});
|
||||
|
||||
it("limits native-harness preprocessing to audio", async () => {
|
||||
it.each([
|
||||
{ outcome: "success", body: "[Audio]\nTranscript:\naudio ok" },
|
||||
{ outcome: "failure", body: "[Audio attachment could not be analyzed]" },
|
||||
{ outcome: "scope-denied", body: "[Audio attachment not analyzed in this chat]" },
|
||||
])("limits native-harness preprocessing to audio on STT $outcome", async ({ outcome, body }) => {
|
||||
const dir = await createTempMediaDir();
|
||||
const imagePath = path.join(dir, "photo.jpg");
|
||||
const audioPath = path.join(dir, "note.ogg");
|
||||
@@ -1686,11 +1701,17 @@ describe("applyMediaUnderstanding", () => {
|
||||
await fs.writeFile(filePath, "file text");
|
||||
|
||||
const describeImage = vi.fn(async () => ({ text: "image ok" }));
|
||||
const transcribeAudio = vi.fn(async () => ({ text: "audio ok" }));
|
||||
const transcribeAudio = vi.fn(async () => {
|
||||
if (outcome === "failure") {
|
||||
throw new Error("transcription provider unavailable");
|
||||
}
|
||||
return { text: "audio ok" };
|
||||
});
|
||||
const ctx: MsgContext = {
|
||||
Body: "",
|
||||
media: [
|
||||
{ path: imagePath, contentType: "image/jpeg" },
|
||||
{ url: "https://example.test/clip.mp4", contentType: "video/mp4" },
|
||||
{ path: audioPath, contentType: "audio/ogg" },
|
||||
{ path: filePath, contentType: "text/plain" },
|
||||
],
|
||||
@@ -1703,7 +1724,10 @@ describe("applyMediaUnderstanding", () => {
|
||||
{ provider: "groq", capabilities: ["audio"] },
|
||||
],
|
||||
image: { enabled: true },
|
||||
audio: { enabled: true },
|
||||
audio: {
|
||||
enabled: true,
|
||||
scope: { default: outcome === "scope-denied" ? "deny" : "allow" },
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
@@ -1719,17 +1743,19 @@ describe("applyMediaUnderstanding", () => {
|
||||
});
|
||||
|
||||
expect(describeImage).not.toHaveBeenCalled();
|
||||
expect(transcribeAudio).toHaveBeenCalledOnce();
|
||||
expect(transcribeAudio).toHaveBeenCalledTimes(outcome === "scope-denied" ? 0 : 1);
|
||||
expect(result).toEqual(
|
||||
expect.objectContaining({
|
||||
appliedImage: false,
|
||||
appliedAudio: true,
|
||||
appliedAudio: outcome === "success",
|
||||
appliedVideo: false,
|
||||
appliedFile: false,
|
||||
extractedFileImages: [],
|
||||
}),
|
||||
);
|
||||
expect(ctx.Body).toBe("[Audio]\nTranscript:\naudio ok");
|
||||
expect(ctx.Body).toBe(body);
|
||||
expect(ctx.BodyForAgent).toBe(body);
|
||||
expect(ctx.Transcript).toBe(outcome === "success" ? "audio ok" : undefined);
|
||||
});
|
||||
|
||||
it("orders synthetic too-small audio output between image and video", async () => {
|
||||
|
||||
@@ -611,15 +611,14 @@ export async function applyMediaUnderstanding(params: {
|
||||
// placement, suppress — a wrong path is worse than the plain marker (#122411).
|
||||
selfServePathsEnabled: params.selfServeLocalPaths === true,
|
||||
});
|
||||
const mediaMarkers =
|
||||
params.processingMode === "audio-only"
|
||||
? []
|
||||
: renderMediaAttachmentMarkers({
|
||||
attachments,
|
||||
decisions,
|
||||
outputs,
|
||||
deliveredImageIndexes: params.deliveredImageIndexes,
|
||||
});
|
||||
// Only processed capabilities have decisions, so audio-only runs cannot
|
||||
// add markers for image/video inputs still owned by the native harness.
|
||||
const mediaMarkers = renderMediaAttachmentMarkers({
|
||||
attachments,
|
||||
decisions,
|
||||
outputs,
|
||||
deliveredImageIndexes: params.deliveredImageIndexes,
|
||||
});
|
||||
const contextBlocks = applyAttachmentMarkerBudget([...fileContext.blocks, ...mediaMarkers]);
|
||||
if (contextBlocks.length > 0) {
|
||||
ctx.Body = appendFileBlocks(ctx.Body, contextBlocks);
|
||||
|
||||
Reference in New Issue
Block a user