diff --git a/docs/providers/alibaba.md b/docs/providers/alibaba.md index 3c322fddcafa..0a5d2dbf2021 100644 --- a/docs/providers/alibaba.md +++ b/docs/providers/alibaba.md @@ -82,15 +82,22 @@ The bundled `alibaba` plugin registers a video-generation provider for Wan model ## Capabilities and limits -All three modes share the same per-request video count and duration cap; only the input shape differs. +Each model advertises only its matching runtime mode. Geometry also follows the +vendor protocol for that model family instead of sending one generic parameter shape. -| Mode | Max output videos | Max input images | Max input videos | Max duration | Supported controls | -| ------------------ | ----------------- | ---------------- | ---------------- | ------------ | --------------------------------------------------------- | -| Text-to-video | 1 | n/a | n/a | 10 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | -| Image-to-video | 1 | 1 | n/a | 10 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | -| Reference-to-video | 1 | n/a | 4 | 10 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | +| Mode | Max output videos | Reference limits | Max duration | Supported controls | +| ---------------------------- | ----------------- | ------------------------------------- | ------------ | -------------------------------------------------------------------- | +| Text-to-video | 1 | n/a | 15 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | +| Image-to-video | 1 | 1 image | 15 s | `resolution`, `audio`, `watermark` | +| Reference-to-video (Wan 2.6) | 1 | 5 total images/videos; up to 3 videos | 10 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | +| Reference-to-video (Wan 2.7) | 1 | 5 total images/videos; up to 3 videos | 10 s | `size`, `aspectRatio`, `resolution`, `watermark`; audio is always on | -A request that omits `durationSeconds` gets DashScope's accepted default of **5 seconds**. Set `durationSeconds` explicitly on the [video generation tool](/tools/video-generation) to extend up to 10 s. +Wan 2.6 text/reference models translate `resolution` plus `aspectRatio` to the +documented exact `size`. Wan 2.6 image-to-video sends the `resolution` tier and +uses the input image's aspect ratio. Wan 2.7 reference-to-video sends the newer +`media`, `resolution`, and `ratio` fields and always generates audio. + +A request that omits `durationSeconds` gets DashScope's accepted default of **5 seconds**. Reference image and video inputs must be remote `http(s)` URLs; DashScope's reference modes reject local file paths. Upload to object storage first, or use the [media tool](/tools/media-overview) flow that already produces a public URL. diff --git a/docs/providers/qwen.md b/docs/providers/qwen.md index 4c17b0fb0037..0869dd9b6f87 100644 --- a/docs/providers/qwen.md +++ b/docs/providers/qwen.md @@ -284,12 +284,23 @@ To make Qwen the default video provider: } ``` -Video-generation limits: 1 output video per request, up to 1 input image -(image-to-video), up to 4 input videos (video-to-video), max 10 seconds -duration. Supports `size`, `aspectRatio`, `resolution`, `audio`, and -`watermark`. Reference image/video inputs require remote http(s) URLs; local -file paths are rejected up front because the DashScope video endpoint does not -accept uploaded local buffers for those references. +Each Wan model advertises only its matching runtime mode: + +| Mode | Models | Reference limits | Max duration | Supported controls | +| ---------------------------- | -------------------------------- | ------------------------------------- | ------------ | -------------------------------------------------------------------- | +| Text-to-video | `wan2.6-t2v` | n/a | 15 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | +| Image-to-video | `wan2.6-i2v` | 1 image | 15 s | `resolution`, `audio`, `watermark` | +| Reference-to-video (Wan 2.6) | `wan2.6-r2v`, `wan2.6-r2v-flash` | 5 total images/videos; up to 3 videos | 10 s | `size`, `aspectRatio`, `resolution`, `audio`, `watermark` | +| Reference-to-video (Wan 2.7) | `wan2.7-r2v` | 5 total images/videos; up to 3 videos | 10 s | `size`, `aspectRatio`, `resolution`, `watermark`; audio is always on | + +Wan 2.6 text/reference models translate `resolution` plus `aspectRatio` to the +documented exact `size`. Wan 2.6 image-to-video sends the `resolution` tier and +uses the input image's aspect ratio. Wan 2.7 reference-to-video sends +`media`, `resolution`, and `ratio` and always generates audio. + +Reference image/video inputs require remote http(s) URLs; local file paths are +rejected up front because the DashScope video endpoint does not accept uploaded +local buffers for those references. See [Video generation](/tools/video-generation) for shared tool parameters, provider selection, and failover behavior. diff --git a/extensions/alibaba/video-generation-provider.test.ts b/extensions/alibaba/video-generation-provider.test.ts index 86f9aeb9ea91..590b19de6fe4 100644 --- a/extensions/alibaba/video-generation-provider.test.ts +++ b/extensions/alibaba/video-generation-provider.test.ts @@ -282,10 +282,10 @@ describe("alibaba video generation provider", () => { expect(body.model).toBe("wan2.6-r2v-flash"); const input = requireRecord(body.input, "DashScope request input"); expect(input.prompt).toBe("animate this shot"); - expect(input.img_url).toBe("https://example.com/ref.png"); + expect(input.reference_urls).toEqual(["https://example.com/ref.png"]); const parameters = requireRecord(body.parameters, "DashScope request parameters"); expect(parameters.duration).toBe(6); - expect(parameters.enable_audio).toBe(true); + expect(parameters.audio).toBe(true); expect(parameters.watermark).toBe(false); expectDashscopeVideoTaskPoll(fetchWithTimeoutMock); expectSuccessfulDashscopeVideoResult(result); diff --git a/extensions/qwen/video-generation-provider.test.ts b/extensions/qwen/video-generation-provider.test.ts index 4910f9a83d69..12081dd913c2 100644 --- a/extensions/qwen/video-generation-provider.test.ts +++ b/extensions/qwen/video-generation-provider.test.ts @@ -255,11 +255,11 @@ describe("qwen video generation provider", () => { model: "wan2.6-r2v-flash", input: { prompt: "animate this shot", - img_url: "https://example.com/ref.png", + reference_urls: ["https://example.com/ref.png"], }, parameters: { duration: 6, - enable_audio: true, + audio: true, }, }, }); @@ -378,7 +378,7 @@ describe("qwen video generation provider", () => { await expect( provider.generateVideo({ provider: "qwen", - model: "wan2.6-r2v-flash", + model: "wan2.6-t2v", prompt: "short video", cfg: { agents: { defaults: { mediaMaxMb: 0.000001 } } }, }), diff --git a/src/agents/tools/video-generate-tool.test.ts b/src/agents/tools/video-generate-tool.test.ts index 8a9c3f0e51ec..4c131059592f 100644 --- a/src/agents/tools/video-generate-tool.test.ts +++ b/src/agents/tools/video-generate-tool.test.ts @@ -1641,7 +1641,7 @@ describe("createVideoGenerateTool", () => { }); }); - it("rejects image-to-video when the provider disables that mode", async () => { + it("defers disabled primary modes to the fallback-aware runtime", async () => { vi.spyOn(videoGenerationRuntime, "listRuntimeVideoGenerationProviders").mockReturnValue([ { id: "video-plugin", @@ -1657,7 +1657,7 @@ describe("createVideoGenerateTool", () => { }), }, ]); - const generateSpy = vi.spyOn(videoGenerationRuntime, "generateVideo"); + const generateSpy = mockSavedVideoResult(); const tool = createVideoGenerateTool({ config: asConfig({ @@ -1672,13 +1672,64 @@ describe("createVideoGenerateTool", () => { throw new Error("expected video_generate tool"); } - await expect( - tool.execute("call-1", { - prompt: "lobster timelapse", - image: "data:image/png;base64,cG5n", + await tool.execute("call-1", { + prompt: "lobster timelapse", + image: "data:image/png;base64,cG5n", + }); + + const request = firstMockCallArg(generateSpy) as { inputImages?: unknown[] }; + expect(request.inputImages).toHaveLength(1); + }); + + it("defers model-specific reference limits to runtime overlays", async () => { + vi.spyOn(videoGenerationRuntime, "listRuntimeVideoGenerationProviders").mockReturnValue([ + { + id: "video-plugin", + defaultModel: "r2v", + models: ["r2v"], + capabilities: { + imageToVideo: { + enabled: true, + maxInputImages: 1, + }, + }, + catalogByModel: { + r2v: { + modes: ["imageToVideo"], + capabilities: { + imageToVideo: { + enabled: true, + maxInputImages: 5, + }, + }, + }, + }, + generateVideo: vi.fn(async () => { + throw new Error("not used"); + }), + }, + ]); + const generateSpy = mockSavedVideoResult(); + const tool = createVideoGenerateTool({ + config: asConfig({ + agents: { + defaults: { + videoGenerationModel: { primary: "video-plugin/r2v" }, + }, + }, }), - ).rejects.toThrow("video-plugin does not support image-to-video reference inputs."); - expect(generateSpy).not.toHaveBeenCalled(); + }); + if (!tool) { + throw new Error("expected video_generate tool"); + } + + await tool.execute("call-r2v", { + prompt: "animate both references", + images: ["data:image/png;base64,cG5n", "data:image/png;base64,cG5nMg=="], + }); + + const request = firstMockCallArg(generateSpy) as { inputImages?: unknown[] }; + expect(request.inputImages).toHaveLength(2); }); it("warns when optional provider overrides are ignored", async () => { diff --git a/src/agents/tools/video-generate-tool.ts b/src/agents/tools/video-generate-tool.ts index 177616ab20eb..eab94ec0fe0f 100644 --- a/src/agents/tools/video-generate-tool.ts +++ b/src/agents/tools/video-generate-tool.ts @@ -17,10 +17,6 @@ import { isManifestPluginAvailableForControlPlane } from "../../plugins/manifest import type { PluginMetadataSnapshot } from "../../plugins/plugin-metadata-snapshot.types.js"; import { resolveUserPath } from "../../utils.js"; import type { DeliveryContext } from "../../utils/delivery-context.js"; -import { - resolveVideoGenerationMode, - resolveVideoGenerationModeCapabilities, -} from "../../video-generation/capabilities.js"; import { parseVideoGenerationModelRef } from "../../video-generation/model-ref.js"; import { generateVideo, @@ -459,85 +455,6 @@ function resolveSelectedVideoGenerationProvider(params: { }); } -function validateVideoGenerationCapabilities(params: { - provider: VideoGenerationProvider | undefined; - model?: string; - inputImageCount: number; - inputVideoCount: number; - inputAudioCount: number; - size?: string; - aspectRatio?: string; - resolution?: VideoGenerationResolution; - durationSeconds?: number; - audio?: boolean; - watermark?: boolean; -}) { - const provider = params.provider; - if (!provider) { - return; - } - const mode = resolveVideoGenerationMode({ - inputImageCount: params.inputImageCount, - inputVideoCount: params.inputVideoCount, - }); - const { capabilities: caps } = resolveVideoGenerationModeCapabilities({ - provider, - model: params.model, - inputImageCount: params.inputImageCount, - inputVideoCount: params.inputVideoCount, - }); - if (!caps && mode === "imageToVideo" && params.inputVideoCount === 0) { - throw new ToolInputError(`${provider.id} does not support image-to-video reference inputs.`); - } - if (!caps && mode === "videoToVideo" && params.inputImageCount === 0) { - throw new ToolInputError(`${provider.id} does not support video-to-video reference inputs.`); - } - if (!caps) { - return; - } - if ( - mode === "imageToVideo" && - "enabled" in caps && - !caps.enabled && - params.inputVideoCount === 0 - ) { - throw new ToolInputError(`${provider.id} does not support image-to-video reference inputs.`); - } - if ( - mode === "videoToVideo" && - "enabled" in caps && - !caps.enabled && - params.inputImageCount === 0 - ) { - throw new ToolInputError(`${provider.id} does not support video-to-video reference inputs.`); - } - if (params.inputImageCount > 0) { - const maxInputImages = caps.maxInputImages ?? MAX_INPUT_IMAGES; - if (params.inputImageCount > maxInputImages) { - throw new ToolInputError( - `${provider.id} supports at most ${maxInputImages} reference image${maxInputImages === 1 ? "" : "s"}.`, - ); - } - } - if (params.inputVideoCount > 0) { - const maxInputVideos = caps.maxInputVideos ?? MAX_INPUT_VIDEOS; - if (params.inputVideoCount > maxInputVideos) { - throw new ToolInputError( - `${provider.id} supports at most ${maxInputVideos} reference video${maxInputVideos === 1 ? "" : "s"}.`, - ); - } - } - // Audio-count validation is intentionally deferred to runtime.ts (generateVideo). - // The runtime guard skips per-candidate providers that lack audio support, allowing - // fallback candidates that do support audio to run. A ToolInputError here would fire - // against only the primary provider and prevent valid fallback-based audio requests. - // maxDurationSeconds validation is intentionally deferred to runtime.ts (generateVideo). - // The runtime guard skips per-candidate providers whose hard cap is below the requested - // duration, allowing a fallback with a higher cap to run — same rationale as the audio - // check above. When providers declare an explicit supportedDurationSeconds list, runtime - // normalization snaps to the nearest valid value instead of skipping. -} - function formatIgnoredVideoGenerationOverride(override: VideoGenerationIgnoredOverride): string { return `${sanitizeGeneratedMediaDisplayText(override.key)}=${sanitizeGeneratedMediaDisplayText(String(override.value))}`; } @@ -1223,20 +1140,6 @@ export function createVideoGenerateTool(options?: { asset.sourceAsset.role = role; } } - validateVideoGenerationCapabilities({ - provider: selectedProvider, - model: - parseVideoGenerationModelRef(model)?.model ?? model ?? selectedProvider?.defaultModel, - inputImageCount: loadedReferenceImages.length, - inputVideoCount: loadedReferenceVideos.length, - inputAudioCount: loadedReferenceAudios.length, - size, - aspectRatio, - resolution, - durationSeconds, - audio, - watermark, - }); // Accepted tasks own their paid work independently; cancellation applies only before admission. signal?.throwIfAborted(); const taskHandle = createVideoGenerationTaskRun({ diff --git a/src/plugin-sdk/video-generation.ts b/src/plugin-sdk/video-generation.ts index bb2015befa82..8dcae467ad4d 100644 --- a/src/plugin-sdk/video-generation.ts +++ b/src/plugin-sdk/video-generation.ts @@ -6,6 +6,7 @@ import type { AuthProfileStore } from "../agents/auth-profiles/types.js"; import type { OpenClawConfig } from "../config/types.openclaw.js"; import { + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, DASHSCOPE_WAN_VIDEO_CAPABILITIES, DASHSCOPE_WAN_VIDEO_MODELS, DEFAULT_DASHSCOPE_WAN_VIDEO_MODEL, @@ -288,6 +289,9 @@ export function buildDashscopeVideoGenerationProvider( label: options.label, defaultModel: DEFAULT_DASHSCOPE_WAN_VIDEO_MODEL, models: [...DASHSCOPE_WAN_VIDEO_MODELS], + catalogByModel: DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + resolveModelCapabilities: ({ model }) => + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL[model]?.capabilities, isConfigured: (ctx) => { const baseUrl = ctx.cfg?.models?.providers?.[options.providerId]?.baseUrl; if (options.credentialPolicy?.acceptsBaseUrl?.(baseUrl) === false) { diff --git a/src/video-generation/capabilities.test.ts b/src/video-generation/capabilities.test.ts index 3b782a701ff5..29339cf1ae33 100644 --- a/src/video-generation/capabilities.test.ts +++ b/src/video-generation/capabilities.test.ts @@ -2,7 +2,6 @@ import { describe, expect, it } from "vitest"; import { listSupportedVideoGenerationModes, - resolveVideoGenerationMode, resolveVideoGenerationModeCapabilities, } from "./capabilities.js"; import type { VideoGenerationProvider } from "./types.js"; @@ -64,7 +63,6 @@ describe("video-generation capabilities", () => { supportsAudio: true, }); - expect(resolveVideoGenerationMode({ inputImageCount: 1, inputVideoCount: 1 })).toBeNull(); expect( resolveVideoGenerationModeCapabilities({ provider, @@ -91,7 +89,6 @@ describe("video-generation capabilities", () => { }, }); - expect(resolveVideoGenerationMode({ inputImageCount: 1, inputVideoCount: 1 })).toBeNull(); expect( resolveVideoGenerationModeCapabilities({ provider, diff --git a/src/video-generation/capabilities.ts b/src/video-generation/capabilities.ts index 9b339a520b5a..c650f866ad0a 100644 --- a/src/video-generation/capabilities.ts +++ b/src/video-generation/capabilities.ts @@ -8,7 +8,7 @@ import type { // Video generation mode helpers derive the active mode from reference inputs // and expose the provider capability block that applies to that mode/model. -export function resolveVideoGenerationMode(params: { +function resolveVideoGenerationMode(params: { inputImageCount?: number; inputVideoCount?: number; }): VideoGenerationMode | null { diff --git a/src/video-generation/capability-overlays.test.ts b/src/video-generation/capability-overlays.test.ts index 6264d8e996d0..c23fc2dd567c 100644 --- a/src/video-generation/capability-overlays.test.ts +++ b/src/video-generation/capability-overlays.test.ts @@ -2,9 +2,14 @@ import { describe, expect, it, vi } from "vitest"; import type { OpenClawConfig } from "../config/types.js"; import { - buildReferenceInputCapabilityFailure, + buildVideoGenerationCapabilityFailure, resolveProviderWithModelCapabilities, } from "./capability-overlays.js"; +import { + DASHSCOPE_WAN_VIDEO_CAPABILITIES, + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + DASHSCOPE_WAN_VIDEO_MODELS, +} from "./dashscope-compatible.js"; import type { VideoGenerationProvider, VideoGenerationProviderCapabilities } from "./types.js"; async function resolveCapabilitiesWithOverlay( @@ -156,7 +161,7 @@ describe("video-generation capability overlays", () => { }); expect( - buildReferenceInputCapabilityFailure({ + buildVideoGenerationCapabilityFailure({ providerId: "openrouter", model: "minimax/hailuo-2.3", provider: activeProvider, @@ -166,4 +171,50 @@ describe("video-generation capability overlays", () => { }), ).toMatch(/supports at most 1 reference image\(s\), 2 requested/); }); + + it.each(DASHSCOPE_WAN_VIDEO_MODELS)( + "enforces bundled Wan catalog modes before provider I/O for %s", + async (model) => { + const provider: VideoGenerationProvider = { + id: "qwen", + capabilities: DASHSCOPE_WAN_VIDEO_CAPABILITIES, + catalogByModel: DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + resolveModelCapabilities: ({ model: selectedModel }) => + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL[selectedModel]?.capabilities, + async generateVideo() { + throw new Error("should not be called"); + }, + }; + const activeProvider = await resolveProviderWithModelCapabilities({ + provider, + providerId: "qwen", + model, + cfg: {} as OpenClawConfig, + log: { debug: vi.fn() }, + }); + const declaredModes = DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL[model]?.modes ?? []; + const requests = [ + { mode: "generate", inputImageCount: 0, inputVideoCount: 0 }, + { mode: "imageToVideo", inputImageCount: 1, inputVideoCount: 0 }, + { mode: "videoToVideo", inputImageCount: 0, inputVideoCount: 1 }, + ] as const; + + for (const request of requests) { + const failure = buildVideoGenerationCapabilityFailure({ + providerId: "qwen", + model, + provider: activeProvider, + inputImageCount: request.inputImageCount, + inputVideoCount: request.inputVideoCount, + inputAudioCount: 0, + }); + + if (declaredModes.includes(request.mode)) { + expect(failure, `${model}:${request.mode}`).toBeUndefined(); + } else { + expect(failure, `${model}:${request.mode}`).toMatch(/does not support/u); + } + } + }, + ); }); diff --git a/src/video-generation/capability-overlays.ts b/src/video-generation/capability-overlays.ts index a801bc3da671..85df91467e40 100644 --- a/src/video-generation/capability-overlays.ts +++ b/src/video-generation/capability-overlays.ts @@ -17,7 +17,7 @@ function isVideoGenerationTransformCapabilities( return Boolean(capabilities && "enabled" in capabilities); } -export function buildReferenceInputCapabilityFailure(params: { +export function buildVideoGenerationCapabilityFailure(params: { providerId: string; model: string; provider: VideoGenerationProvider; @@ -27,12 +27,22 @@ export function buildReferenceInputCapabilityFailure(params: { }): string | undefined { const { providerId, model, provider, inputImageCount, inputVideoCount, inputAudioCount } = params; const label = `${providerId}/${model}`; - const { capabilities } = resolveVideoGenerationModeCapabilities({ + const { mode, capabilities } = resolveVideoGenerationModeCapabilities({ provider, model, inputImageCount, inputVideoCount, }); + const catalogModes = provider.catalogByModel?.[model]?.modes; + if (mode && catalogModes && !catalogModes.includes(mode)) { + const modeLabel = + mode === "generate" + ? "text-to-video generation" + : mode === "imageToVideo" + ? "image-to-video generation" + : "video-to-video generation"; + return `${label} does not support ${modeLabel}; skipping`; + } if (inputImageCount > 0 || inputVideoCount > 0) { // Reference inputs must be explicitly supported. Falling back to a provider diff --git a/src/video-generation/dashscope-compatible.test.ts b/src/video-generation/dashscope-compatible.test.ts index aea27362f0d6..1d421dd35ec1 100644 --- a/src/video-generation/dashscope-compatible.test.ts +++ b/src/video-generation/dashscope-compatible.test.ts @@ -1,6 +1,27 @@ -// DashScope-compatible download regressions: body idle after headers. +// DashScope-compatible lifecycle, task status, and generated-video regressions. import { describe, expect, it, vi } from "vitest"; -import { downloadDashscopeGeneratedVideos } from "./dashscope-compatible.js"; +import { + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + buildDashscopeVideoGenerationInput, + buildDashscopeVideoGenerationParameters, + downloadDashscopeGeneratedVideos, + pollDashscopeVideoTaskUntilComplete, + runDashscopeVideoGenerationTask, +} from "./dashscope-compatible.js"; + +const providerLabels = ["Qwen", "Alibaba Wan"] as const; + +const invalidGeneratedVideos = [ + { name: "JSON error", contentType: "application/json", body: '{"error":"not a video"}' }, + { + name: "problem JSON error", + contentType: "application/problem+json", + body: '{"title":"not a video"}', + }, + { name: "HTML error", contentType: "text/html; charset=utf-8", body: "error" }, + { name: "image", contentType: "image/png", body: "image-bytes" }, + { name: "audio", contentType: "audio/mp4", body: "audio-bytes" }, +] as const; function neverChunkingVideoResponse(): Response { return new Response( @@ -16,7 +37,244 @@ function neverChunkingVideoResponse(): Response { ); } +describe("DashScope Wan request contracts", () => { + it("advertises only the modes supported by each bundled Wan model", () => { + expect(DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL["wan2.6-t2v"]?.modes).toEqual(["generate"]); + expect( + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL["wan2.6-t2v"]?.capabilities?.generate + ?.supportsAspectRatio, + ).toBe(true); + expect(DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL["wan2.6-i2v"]?.modes).toEqual(["imageToVideo"]); + expect(DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL["wan2.6-r2v"]?.modes).toEqual([ + "imageToVideo", + "videoToVideo", + ]); + expect( + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL["wan2.7-r2v"]?.capabilities?.videoToVideo?.supportsAudio, + ).toBe(false); + expect( + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL["wan2.7-r2v"]?.capabilities?.videoToVideo + ?.supportsAspectRatio, + ).toBe(true); + }); + + it("builds mode-specific image and reference inputs", () => { + expect( + buildDashscopeVideoGenerationInput({ + providerLabel: "Qwen", + req: { + provider: "qwen", + model: "wan2.6-i2v", + prompt: "animate", + cfg: {}, + inputImages: [{ url: "https://example.com/frame.png" }], + }, + }), + ).toEqual({ prompt: "animate", img_url: "https://example.com/frame.png" }); + + expect( + buildDashscopeVideoGenerationInput({ + providerLabel: "Qwen", + req: { + provider: "qwen", + model: "wan2.6-r2v", + prompt: "character1 waves", + cfg: {}, + inputImages: [{ url: "https://example.com/character.png" }], + }, + }), + ).toEqual({ + prompt: "character1 waves", + reference_urls: ["https://example.com/character.png"], + }); + + expect( + buildDashscopeVideoGenerationInput({ + providerLabel: "Alibaba Wan", + req: { + provider: "alibaba", + model: "wan2.7-r2v", + prompt: "Image 1 greets Video 1", + cfg: {}, + inputImages: [{ url: "https://example.com/character.png" }], + inputVideos: [{ url: "https://example.com/action.mp4", role: "reference_video" }], + }, + }), + ).toEqual({ + prompt: "Image 1 greets Video 1", + media: [ + { type: "reference_image", url: "https://example.com/character.png" }, + { type: "reference_video", url: "https://example.com/action.mp4" }, + ], + }); + }); + + it("rejects model and reference mode mismatches before submission", () => { + expect(() => + buildDashscopeVideoGenerationInput({ + providerLabel: "Qwen", + req: { + provider: "qwen", + model: "wan2.6-t2v", + prompt: "animate", + cfg: {}, + inputImages: [{ url: "https://example.com/frame.png" }], + }, + }), + ).toThrow(/text-to-video.*does not accept reference media/u); + }); + + it.each([ + { + name: "Wan 2.6 text-to-video", + req: { + provider: "qwen", + model: "wan2.6-t2v", + prompt: "video", + cfg: {}, + resolution: "720P", + aspectRatio: "9:16", + audio: false, + }, + expected: { size: "720*1280", audio: false }, + }, + { + name: "Wan 2.6 image-to-video", + req: { + provider: "qwen", + model: "wan2.6-i2v", + prompt: "video", + cfg: {}, + resolution: "1080P", + inputImages: [{ url: "https://example.com/frame.png" }], + audio: true, + }, + expected: { resolution: "1080P", audio: true }, + }, + { + name: "Wan 2.7 reference-to-video", + req: { + provider: "alibaba", + model: "wan2.7-r2v", + prompt: "video", + cfg: {}, + size: "1920x1080", + inputVideos: [{ url: "https://example.com/reference.mp4" }], + audio: false, + }, + expected: { resolution: "1080P", ratio: "16:9" }, + }, + ])("builds documented $name parameters", ({ req, expected }) => { + expect(buildDashscopeVideoGenerationParameters(req)).toEqual(expected); + }); +}); + describe("downloadDashscopeGeneratedVideos", () => { + it.each( + providerLabels.flatMap((providerLabel) => + invalidGeneratedVideos.map(({ name, contentType, body }) => ({ + providerLabel, + name, + contentType, + body, + })), + ), + )("rejects $providerLabel $name responses instead of returning a video", async (invalid) => { + const fetchFn = vi.fn( + async () => + new Response(invalid.body, { + status: 200, + headers: { "content-type": invalid.contentType }, + }), + ); + + await expect( + downloadDashscopeGeneratedVideos({ + providerLabel: invalid.providerLabel, + urls: ["https://example.com/not-video.mp4"], + timeoutMs: 5_000, + fetchFn: fetchFn as typeof fetch, + maxBytes: 10 * 1024 * 1024, + }), + ).rejects.toThrow( + `${invalid.providerLabel} generated video download: malformed video response`, + ); + + expect(fetchFn).toHaveBeenCalledOnce(); + }); + + it.each(providerLabels)( + "cancels unread invalid %s video bodies before releasing them", + async (providerLabel) => { + const cancellationOrder: string[] = []; + const cancelBody = vi.fn(async () => { + cancellationOrder.push("cancel-started"); + await Promise.resolve(); + cancellationOrder.push("cancel-completed"); + }); + const fetchFn = vi.fn( + async () => + new Response( + new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode('{"error":"still streaming"}')); + }, + cancel: cancelBody, + }), + { + status: 200, + headers: { "content-type": "application/json" }, + }, + ), + ); + + await expect( + downloadDashscopeGeneratedVideos({ + providerLabel, + urls: ["https://example.com/still-streaming.mp4"], + timeoutMs: 80, + fetchFn: fetchFn as typeof fetch, + maxBytes: 10 * 1024 * 1024, + }), + ).rejects.toThrow(`${providerLabel} generated video download: malformed video response`); + + expect(cancelBody).toHaveBeenCalledOnce(); + expect(cancellationOrder).toEqual(["cancel-started", "cancel-completed"]); + }, + ); + + it.each([ + { contentType: "video/mp4", expectedMimeType: "video/mp4" }, + { contentType: "VIDEO/MP4; codecs=avc1", expectedMimeType: "VIDEO/MP4; codecs=avc1" }, + { contentType: "application/octet-stream", expectedMimeType: "application/octet-stream" }, + { contentType: undefined, expectedMimeType: "video/mp4" }, + ])( + "preserves valid generated video content type $contentType", + async ({ contentType, expectedMimeType }) => { + const fetchFn = vi.fn( + async () => + new Response(new TextEncoder().encode("mp4-bytes"), { + status: 200, + ...(contentType ? { headers: { "content-type": contentType } } : {}), + }), + ); + + const videos = await downloadDashscopeGeneratedVideos({ + providerLabel: "Alibaba Wan", + urls: ["https://example.com/video.mp4"], + timeoutMs: 5_000, + fetchFn: fetchFn as typeof fetch, + maxBytes: 10 * 1024 * 1024, + }); + + expect(videos[0]).toMatchObject({ + buffer: Buffer.from("mp4-bytes"), + fileName: "video-1.mp4", + mimeType: expectedMimeType, + }); + }, + ); + it("aborts a stalled generated video body via chunk idle timeout", async () => { const fetchFn = vi.fn(async () => neverChunkingVideoResponse()); const timeoutMs = 80; @@ -141,3 +399,116 @@ describe("downloadDashscopeGeneratedVideos", () => { } }); }); + +describe("pollDashscopeVideoTaskUntilComplete", () => { + it.each(providerLabels)( + "immediately rejects documented UNKNOWN %s tasks", + async (providerLabel) => { + const fetchFn = vi.fn( + async () => + new Response(JSON.stringify({ output: { task_status: " UNKNOWN " } }), { + status: 200, + headers: { "content-type": "application/json" }, + }), + ); + + await expect( + pollDashscopeVideoTaskUntilComplete({ + providerLabel, + taskId: "expired-task", + headers: new Headers(), + timeoutMs: 80, + fetchFn: fetchFn as typeof fetch, + baseUrl: "https://example.com", + }), + ).rejects.toThrow( + `${providerLabel} video generation task expired-task is unknown or expired`, + ); + + expect(fetchFn).toHaveBeenCalledOnce(); + }, + ); + + it.each(providerLabels)( + "includes the provider reason when an UNKNOWN %s task expires", + async (providerLabel) => { + const fetchFn = vi.fn( + async () => + new Response( + JSON.stringify({ output: { task_status: "UNKNOWN", message: "task was deleted" } }), + { + status: 200, + headers: { "content-type": "application/json" }, + }, + ), + ); + + await expect( + pollDashscopeVideoTaskUntilComplete({ + providerLabel, + taskId: "deleted-task", + headers: new Headers(), + timeoutMs: 80, + fetchFn: fetchFn as typeof fetch, + baseUrl: "https://example.com", + }), + ).rejects.toThrow( + `${providerLabel} video generation task deleted-task is unknown or expired: task was deleted`, + ); + + expect(fetchFn).toHaveBeenCalledOnce(); + }, + ); +}); + +describe("runDashscopeVideoGenerationTask", () => { + it("releases the submission request timeout before polling the task", async () => { + vi.useFakeTimers(); + try { + let submissionTimerCount: number | undefined; + let pollTimerCount: number | undefined; + const fetchFn = vi.fn(async (url: string | URL | Request) => { + const requestUrl = url instanceof Request ? url.url : String(url); + if (requestUrl.includes("/video-synthesis")) { + submissionTimerCount = vi.getTimerCount(); + return new Response(JSON.stringify({ output: { task_id: "task-123" } }), { + status: 200, + headers: { "content-type": "application/json" }, + }); + } + if (requestUrl.includes("/tasks/task-123")) { + pollTimerCount = vi.getTimerCount(); + return new Response( + JSON.stringify({ + output: { task_status: "SUCCEEDED", video_url: "https://example.com/result.mp4" }, + }), + { + status: 200, + headers: { "content-type": "application/json" }, + }, + ); + } + return new Response(new TextEncoder().encode("mp4-bytes"), { + status: 200, + headers: { "content-type": "video/mp4" }, + }); + }); + + await runDashscopeVideoGenerationTask({ + providerLabel: "Qwen", + model: "wan2.6-t2v", + req: { provider: "qwen", model: "wan2.6-t2v", prompt: "video", cfg: {} }, + url: "https://example.com/video-synthesis", + headers: new Headers(), + baseUrl: "https://example.com", + timeoutMs: 5_000, + fetchFn: fetchFn as typeof fetch, + }); + + expect(submissionTimerCount).toBeGreaterThan(0); + expect(pollTimerCount).toBe(submissionTimerCount); + } finally { + vi.useRealTimers(); + } + }); +}); diff --git a/src/video-generation/dashscope-compatible.ts b/src/video-generation/dashscope-compatible.ts index 0565818a9850..eda8351e3c75 100644 --- a/src/video-generation/dashscope-compatible.ts +++ b/src/video-generation/dashscope-compatible.ts @@ -1,3 +1,4 @@ +import { kindFromMime, normalizeMimeType } from "@openclaw/media-core/mime"; import { normalizeLowercaseStringOrEmpty } from "@openclaw/normalization-core/string-coerce"; import { uniqueStrings } from "@openclaw/normalization-core/string-normalization"; // DashScope-compatible video provider adapts DashScope-style generation APIs. @@ -17,6 +18,7 @@ import { } from "../plugin-sdk/provider-http.js"; import type { GeneratedVideoAsset, + VideoGenerationCatalogModelEntry, VideoGenerationProviderCapabilities, VideoGenerationRequest, VideoGenerationResult, @@ -33,10 +35,48 @@ export const DASHSCOPE_WAN_VIDEO_MODELS = [ "wan2.6-r2v-flash", "wan2.7-r2v", ]; + +const DASHSCOPE_WAN_VIDEO_RESOLUTIONS = ["720P", "1080P"] as const; +const DASHSCOPE_WAN_VIDEO_ASPECT_RATIOS = ["16:9", "9:16", "1:1", "4:3", "3:4"] as const; +const DASHSCOPE_WAN_LONG_VIDEO_DURATIONS = [ + 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, +] as const; +const DASHSCOPE_WAN_SHORT_VIDEO_DURATIONS = [2, 3, 4, 5, 6, 7, 8, 9, 10] as const; +const DASHSCOPE_WAN_VIDEO_SIZE_BY_GEOMETRY: Readonly< + Record>> +> = { + "480P": { + "16:9": "832*480", + "9:16": "480*832", + "1:1": "624*624", + }, + "720P": { + "16:9": "1280*720", + "9:16": "720*1280", + "1:1": "960*960", + "4:3": "1088*832", + "3:4": "832*1088", + }, + "1080P": { + "16:9": "1920*1080", + "9:16": "1080*1920", + "1:1": "1440*1440", + "4:3": "1632*1248", + "3:4": "1248*1632", + }, +}; +const DASHSCOPE_WAN_VIDEO_SIZES = DASHSCOPE_WAN_VIDEO_RESOLUTIONS.flatMap((resolution) => + Object.values(DASHSCOPE_WAN_VIDEO_SIZE_BY_GEOMETRY[resolution] ?? {}), +); + export const DASHSCOPE_WAN_VIDEO_CAPABILITIES = { generate: { maxVideos: 1, - maxDurationSeconds: 10, + maxDurationSeconds: 15, + supportedDurationSeconds: DASHSCOPE_WAN_LONG_VIDEO_DURATIONS, + sizes: DASHSCOPE_WAN_VIDEO_SIZES, + aspectRatios: DASHSCOPE_WAN_VIDEO_ASPECT_RATIOS, + resolutions: DASHSCOPE_WAN_VIDEO_RESOLUTIONS, supportsSize: true, supportsAspectRatio: true, supportsResolution: true, @@ -47,9 +87,11 @@ export const DASHSCOPE_WAN_VIDEO_CAPABILITIES = { enabled: true, maxVideos: 1, maxInputImages: 1, - maxDurationSeconds: 10, - supportsSize: true, - supportsAspectRatio: true, + maxDurationSeconds: 15, + supportedDurationSeconds: DASHSCOPE_WAN_LONG_VIDEO_DURATIONS, + resolutions: DASHSCOPE_WAN_VIDEO_RESOLUTIONS, + supportsSize: false, + supportsAspectRatio: false, supportsResolution: true, supportsAudio: true, supportsWatermark: true, @@ -57,8 +99,13 @@ export const DASHSCOPE_WAN_VIDEO_CAPABILITIES = { videoToVideo: { enabled: true, maxVideos: 1, - maxInputVideos: 4, + maxInputImages: 5, + maxInputVideos: 3, maxDurationSeconds: 10, + supportedDurationSeconds: DASHSCOPE_WAN_SHORT_VIDEO_DURATIONS, + sizes: DASHSCOPE_WAN_VIDEO_SIZES, + aspectRatios: DASHSCOPE_WAN_VIDEO_ASPECT_RATIOS, + resolutions: DASHSCOPE_WAN_VIDEO_RESOLUTIONS, supportsSize: true, supportsAspectRatio: true, supportsResolution: true, @@ -67,6 +114,61 @@ export const DASHSCOPE_WAN_VIDEO_CAPABILITIES = { }, } satisfies VideoGenerationProviderCapabilities; +const disabledVideoTransform = { enabled: false } as const; +const dashscopeWanR2vCapabilities = { + ...DASHSCOPE_WAN_VIDEO_CAPABILITIES, + imageToVideo: { + ...DASHSCOPE_WAN_VIDEO_CAPABILITIES.videoToVideo, + enabled: true, + }, +}; + +// One model catalog drives both agent-visible modes and request-local runtime +// capability overlays, so the tool cannot advertise a mode the model rejects. +export const DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL: Readonly< + Record +> = { + "wan2.6-t2v": { + modes: ["generate"], + capabilities: { + generate: DASHSCOPE_WAN_VIDEO_CAPABILITIES.generate, + imageToVideo: disabledVideoTransform, + videoToVideo: disabledVideoTransform, + }, + }, + "wan2.6-i2v": { + modes: ["imageToVideo"], + capabilities: { + imageToVideo: DASHSCOPE_WAN_VIDEO_CAPABILITIES.imageToVideo, + videoToVideo: disabledVideoTransform, + }, + }, + "wan2.6-r2v": { + modes: ["imageToVideo", "videoToVideo"], + capabilities: dashscopeWanR2vCapabilities, + }, + "wan2.6-r2v-flash": { + modes: ["imageToVideo", "videoToVideo"], + capabilities: dashscopeWanR2vCapabilities, + }, + "wan2.7-r2v": { + modes: ["imageToVideo", "videoToVideo"], + capabilities: { + ...dashscopeWanR2vCapabilities, + imageToVideo: { + ...dashscopeWanR2vCapabilities.imageToVideo, + supportsAspectRatio: true, + supportsAudio: false, + }, + videoToVideo: { + ...dashscopeWanR2vCapabilities.videoToVideo, + supportsAspectRatio: true, + supportsAudio: false, + }, + }, + }, +}; + export const DEFAULT_VIDEO_GENERATION_DURATION_SECONDS = 5; export const DEFAULT_VIDEO_GENERATION_TIMEOUT_MS = 120_000; export const DEFAULT_VIDEO_RESOLUTION_TO_SIZE: Record = { @@ -97,12 +199,62 @@ export type DashscopeVideoGenerationResponse = { message?: string; }; +type DashscopeWanVideoMode = "t2v" | "i2v" | "r2v"; + +function resolveDashscopeWanVideoMode(req: VideoGenerationRequest): DashscopeWanVideoMode { + const model = req.model.trim().toLowerCase(); + if (model.includes("-i2v")) { + return "i2v"; + } + if (model.includes("-r2v")) { + return "r2v"; + } + if (model.includes("-t2v")) { + return "t2v"; + } + if ((req.inputVideos?.length ?? 0) > 0 || (req.inputImages?.length ?? 0) > 1) { + return "r2v"; + } + return (req.inputImages?.length ?? 0) === 1 ? "i2v" : "t2v"; +} + +function isDashscopeWan27Model(model: string): boolean { + return model.trim().toLowerCase().startsWith("wan2.7"); +} + +function assertDashscopeWanVideoInputs(params: { + providerLabel: string; + req: VideoGenerationRequest; + mode: DashscopeWanVideoMode; +}): void { + const imageCount = params.req.inputImages?.length ?? 0; + const videoCount = params.req.inputVideos?.length ?? 0; + if (params.mode === "t2v" && imageCount + videoCount > 0) { + throw new Error( + `${params.providerLabel} model ${params.req.model} is text-to-video and does not accept reference media; use an i2v or r2v Wan model.`, + ); + } + if (params.mode === "i2v" && (imageCount !== 1 || videoCount > 0)) { + throw new Error( + `${params.providerLabel} model ${params.req.model} requires exactly one reference image and no reference videos.`, + ); + } + if (params.mode === "r2v") { + const total = imageCount + videoCount; + if (total === 0 || total > 5 || videoCount > 3) { + throw new Error( + `${params.providerLabel} model ${params.req.model} requires 1-5 reference images/videos, with at most 3 videos.`, + ); + } + } +} + export function buildDashscopeVideoGenerationInput(params: { providerLabel: string; req: VideoGenerationRequest; }): Record { const unsupported = [...(params.req.inputImages ?? []), ...(params.req.inputVideos ?? [])].some( - (asset) => !asset.url?.trim() && asset.buffer, + (asset) => !asset.url?.trim(), ); // DashScope accepts remote references in this path; buffer uploads require a // different provider-specific flow, so fail before silently dropping refs. @@ -114,17 +266,26 @@ export function buildDashscopeVideoGenerationInput(params: { const input: Record = { prompt: params.req.prompt, }; + const mode = resolveDashscopeWanVideoMode(params.req); + assertDashscopeWanVideoInputs({ ...params, mode }); const referenceUrls = resolveVideoGenerationReferenceUrls( params.req.inputImages, params.req.inputVideos, ); - if ( - referenceUrls.length === 1 && - (params.req.inputImages?.length ?? 0) === 1 && - !params.req.inputVideos?.length - ) { + if (mode === "i2v") { input.img_url = referenceUrls[0]; - } else if (referenceUrls.length > 0) { + } else if (mode === "r2v" && isDashscopeWan27Model(params.req.model)) { + input.media = [ + ...(params.req.inputImages ?? []).map((asset) => ({ + type: asset.role?.trim() || "reference_image", + url: asset.url?.trim() ?? "", + })), + ...(params.req.inputVideos ?? []).map((asset) => ({ + type: asset.role?.trim() || "reference_video", + url: asset.url?.trim() ?? "", + })), + ]; + } else if (mode === "r2v") { input.reference_urls = referenceUrls; } return input; @@ -144,18 +305,42 @@ export function buildDashscopeVideoGenerationParameters( resolutionToSize: Record = DEFAULT_VIDEO_RESOLUTION_TO_SIZE, ): Record | undefined { const parameters: Record = {}; - const size = req.size?.trim() || (req.resolution ? resolutionToSize[req.resolution] : undefined); - if (size) { - parameters.size = size; - } - if (req.aspectRatio?.trim()) { - parameters.aspect_ratio = req.aspectRatio.trim(); + const mode = resolveDashscopeWanVideoMode(req); + const wan27 = isDashscopeWan27Model(req.model); + const requestedSize = req.size?.trim(); + const sizeGeometry = requestedSize + ? resolveDashscopeWanVideoSizeGeometry(requestedSize) + : undefined; + // Wan 2.6 I2V and all Wan 2.7 models use resolution tiers. Wan 2.6 T2V/R2V + // use exact dimensions in `size`; folding these together causes API rejection. + if (wan27 || mode === "i2v") { + const resolution = req.resolution?.trim() || sizeGeometry?.resolution; + if (resolution) { + parameters.resolution = resolution; + } + if (wan27 && mode !== "i2v") { + const ratio = req.aspectRatio?.trim() || sizeGeometry?.aspectRatio; + if (ratio) { + parameters.ratio = ratio; + } + } + } else { + const ratio = req.aspectRatio?.trim() || "16:9"; + const size = + requestedSize || + (req.resolution + ? (DASHSCOPE_WAN_VIDEO_SIZE_BY_GEOMETRY[req.resolution]?.[ratio] ?? + resolutionToSize[req.resolution]) + : undefined); + if (size) { + parameters.size = size; + } } if (typeof req.durationSeconds === "number" && Number.isFinite(req.durationSeconds)) { parameters.duration = Math.max(1, Math.round(req.durationSeconds)); } - if (typeof req.audio === "boolean") { - parameters.enable_audio = req.audio; + if (typeof req.audio === "boolean" && !wan27) { + parameters.audio = req.audio; } if (typeof req.watermark === "boolean") { parameters.watermark = req.watermark; @@ -163,6 +348,20 @@ export function buildDashscopeVideoGenerationParameters( return Object.keys(parameters).length > 0 ? parameters : undefined; } +function resolveDashscopeWanVideoSizeGeometry( + size: string, +): { resolution: string; aspectRatio: string } | undefined { + const normalizedSize = size.trim().toLowerCase().replace("x", "*"); + for (const [resolution, sizes] of Object.entries(DASHSCOPE_WAN_VIDEO_SIZE_BY_GEOMETRY)) { + for (const [aspectRatio, candidate] of Object.entries(sizes)) { + if (candidate.toLowerCase() === normalizedSize) { + return { resolution, aspectRatio }; + } + } + } + return undefined; +} + // DashScope may return videos in results[] or a top-level output.video_url. // De-dupe so downstream downloads produce one asset per unique URL. export function extractDashscopeVideoUrls(payload: DashscopeVideoGenerationResponse): string[] { @@ -232,6 +431,16 @@ export async function pollDashscopeVideoTaskUntilComplete(params: { if (status === "SUCCEEDED") { return payload; } + // DashScope reports missing or expired task IDs as UNKNOWN, not PENDING; + // waiting cannot recover them and hides the actionable provider outcome. + if (status === "UNKNOWN") { + const reason = payload.output?.message?.trim() || payload.message?.trim(); + throw new Error( + `${params.providerLabel} video generation task ${params.taskId} is unknown or expired${ + reason ? `: ${reason}` : "" + }`, + ); + } // Terminal failure statuses carry provider messages; nonterminal statuses // continue until the shared operation deadline or max poll attempts wins. if (status === "FAILED" || status === "CANCELED") { @@ -292,55 +501,55 @@ export async function runDashscopeVideoGenerationTask(params: { dispatcherPolicy: params.dispatcherPolicy, }); + let submitted: DashscopeVideoGenerationResponse; try { await assertOkOrThrowHttpError(response, `${params.providerLabel} video generation failed`); - const submitted = await readProviderJsonResponse( + submitted = await readProviderJsonResponse( response, `${params.providerLabel} video generation`, ); - const taskId = submitted.output?.task_id?.trim(); - if (!taskId) { - throw new Error(`${params.providerLabel} video generation response missing task_id`); - } - const completed = await pollDashscopeVideoTaskUntilComplete({ - providerLabel: params.providerLabel, - taskId, - headers: params.headers, - timeoutMs: resolveProviderOperationTimeoutMs({ deadline, defaultTimeoutMs }), - fetchFn: params.fetchFn, - baseUrl: params.baseUrl, - allowPrivateNetwork: params.allowPrivateNetwork, - dispatcherPolicy: params.dispatcherPolicy, - defaultTimeoutMs, - }); - const urls = extractDashscopeVideoUrls(completed); - if (urls.length === 0) { - throw new Error( - `${params.providerLabel} video generation completed without output video URLs`, - ); - } - const videos = await downloadDashscopeGeneratedVideos({ - providerLabel: params.providerLabel, - urls, - timeoutMs: createProviderOperationTimeoutResolver({ deadline, defaultTimeoutMs }), - fetchFn: params.fetchFn, - allowPrivateNetwork: params.allowPrivateNetwork, - dispatcherPolicy: params.dispatcherPolicy, - defaultTimeoutMs, - maxBytes: resolveGeneratedMediaMaxBytes(params.req.cfg, "video"), - }); - return { - videos, - model: params.model, - metadata: { - requestId: submitted.request_id, - taskId, - taskStatus: completed.output?.task_status, - }, - }; } finally { await release(); } + + const taskId = submitted.output?.task_id?.trim(); + if (!taskId) { + throw new Error(`${params.providerLabel} video generation response missing task_id`); + } + const completed = await pollDashscopeVideoTaskUntilComplete({ + providerLabel: params.providerLabel, + taskId, + headers: params.headers, + timeoutMs: resolveProviderOperationTimeoutMs({ deadline, defaultTimeoutMs }), + fetchFn: params.fetchFn, + baseUrl: params.baseUrl, + allowPrivateNetwork: params.allowPrivateNetwork, + dispatcherPolicy: params.dispatcherPolicy, + defaultTimeoutMs, + }); + const urls = extractDashscopeVideoUrls(completed); + if (urls.length === 0) { + throw new Error(`${params.providerLabel} video generation completed without output video URLs`); + } + const videos = await downloadDashscopeGeneratedVideos({ + providerLabel: params.providerLabel, + urls, + timeoutMs: createProviderOperationTimeoutResolver({ deadline, defaultTimeoutMs }), + fetchFn: params.fetchFn, + allowPrivateNetwork: params.allowPrivateNetwork, + dispatcherPolicy: params.dispatcherPolicy, + defaultTimeoutMs, + maxBytes: resolveGeneratedMediaMaxBytes(params.req.cfg, "video"), + }); + return { + videos, + model: params.model, + metadata: { + requestId: submitted.request_id, + taskId, + taskStatus: completed.output?.task_status, + }, + }; } function resolveDashscopeVideoDownloadTimeoutMs( @@ -374,6 +583,7 @@ export async function downloadDashscopeGeneratedVideos(params: { maxBytes: number; }): Promise { const videos: GeneratedVideoAsset[] = []; + const downloadLabel = `${params.providerLabel} generated video download`; for (const [index, url] of params.urls.entries()) { const result = await executeProviderOperationWithRetry({ provider: params.providerLabel, @@ -409,6 +619,22 @@ export async function downloadDashscopeGeneratedVideos(params: { let buffer: Buffer; let mimeType: string; try { + try { + const contentType = normalizeMimeType(result.response.headers.get("content-type")); + if ( + contentType && + contentType !== "application/octet-stream" && + kindFromMime(contentType) !== "video" + ) { + throw new Error(`${downloadLabel}: malformed video response`); + } + } catch (error) { + // Header rejection happens before the body reader, so explicitly cancel + // unread streams before their guarded dispatcher and timeout release. + await result.response.body?.cancel(error).catch(() => undefined); + throw error; + } + // Re-resolve after headers so the body uses the remaining operation budget. let downloadTimeoutMs: number; try { diff --git a/src/video-generation/live-test-helpers.test.ts b/src/video-generation/live-test-helpers.test.ts index 96b68aeddea5..27dcab9887fc 100644 --- a/src/video-generation/live-test-helpers.test.ts +++ b/src/video-generation/live-test-helpers.test.ts @@ -9,9 +9,20 @@ import { redactLiveApiKey, resolveConfiguredLiveVideoModels, resolveLiveVideoAuthStore, + resolveLiveVideoResolution, } from "./live-test-helpers.js"; describe("video-generation live-test helpers", () => { + it.each([ + ["alibaba", "alibaba/wan2.6-t2v", "720P"], + ["qwen", "qwen/wan2.6-t2v", "720P"], + ["minimax", "minimax/MiniMax-Hailuo-2.3", "768P"], + ["pixverse", "pixverse/v6", "540P"], + ["google", "google/veo-3.1-fast-generate-preview", "480P"], + ] as const)("uses a supported %s live resolution", (providerId, modelRef, expected) => { + expect(resolveLiveVideoResolution({ providerId, modelRef })).toBe(expected); + }); + it("parses provider filters and treats empty/all as unfiltered", () => { expect(parseCsvFilter()).toBeNull(); expect(parseCsvFilter("all")).toBeNull(); diff --git a/src/video-generation/live-test-helpers.ts b/src/video-generation/live-test-helpers.ts index 6269c494464e..7aeddf66fb12 100644 --- a/src/video-generation/live-test-helpers.ts +++ b/src/video-generation/live-test-helpers.ts @@ -50,6 +50,9 @@ export function resolveLiveVideoResolution(params: { if (providerId === "pixverse") { return "540P"; } + if (providerId === "alibaba" || providerId === "qwen") { + return "720P"; + } return "480P"; } diff --git a/src/video-generation/runtime.test.ts b/src/video-generation/runtime.test.ts index 1af8ae464832..3533c13a58f7 100644 --- a/src/video-generation/runtime.test.ts +++ b/src/video-generation/runtime.test.ts @@ -1,6 +1,12 @@ // Video generation runtime tests cover provider execution and fallback behavior. import { beforeEach, describe, expect, it } from "vitest"; import type { OpenClawConfig } from "../config/types.js"; +import { + DASHSCOPE_WAN_VIDEO_CAPABILITIES, + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + DASHSCOPE_WAN_VIDEO_MODELS, + buildDashscopeVideoGenerationParameters, +} from "./dashscope-compatible.js"; import { generateVideo, listRuntimeVideoGenerationProviders, @@ -782,6 +788,138 @@ describe("video-generation runtime", () => { expect(attempt.error).toMatch(/supports at most 1 reference image\(s\), 2 requested/); }); + it("falls back when the primary model catalog rejects the requested mode", async () => { + const seenModels: string[] = []; + providers = [ + { + id: "qwen", + defaultModel: "wan2.6-t2v", + models: [...DASHSCOPE_WAN_VIDEO_MODELS], + capabilities: DASHSCOPE_WAN_VIDEO_CAPABILITIES, + catalogByModel: DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + resolveModelCapabilities: ({ model }) => + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL[model]?.capabilities, + isConfigured: () => true, + async generateVideo(req) { + seenModels.push(req.model); + return { + videos: [{ buffer: Buffer.from("mp4-bytes"), mimeType: "video/mp4" }], + model: req.model, + }; + }, + }, + ]; + + const result = await runGenerateVideo({ + cfg: { + agents: { + defaults: { + videoGenerationModel: { + primary: "qwen/wan2.6-t2v", + fallbacks: ["qwen/wan2.6-i2v"], + }, + }, + }, + } as OpenClawConfig, + prompt: "animate the reference", + inputImages: [{ url: "https://example.com/reference.png" }], + }); + + expect(seenModels).toEqual(["wan2.6-i2v"]); + expect(result.model).toBe("wan2.6-i2v"); + expect(result.attempts).toHaveLength(1); + expect(requireAttempt(result, 0).error).toMatch(/does not support image-to-video generation/u); + }); + + it("applies model-specific R2V reference limits during fallback-aware selection", async () => { + let seenImageCount = 0; + providers = [ + { + id: "qwen", + defaultModel: "wan2.6-t2v", + models: [...DASHSCOPE_WAN_VIDEO_MODELS], + capabilities: DASHSCOPE_WAN_VIDEO_CAPABILITIES, + catalogByModel: DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + resolveModelCapabilities: ({ model }) => + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL[model]?.capabilities, + async generateVideo(req) { + seenImageCount = req.inputImages?.length ?? 0; + return { + videos: [{ buffer: Buffer.from("mp4-bytes"), mimeType: "video/mp4" }], + model: req.model, + }; + }, + }, + ]; + + const result = await runGenerateVideo({ + cfg: { + agents: { + defaults: { + videoGenerationModel: { primary: "qwen/wan2.6-r2v" }, + }, + }, + } as OpenClawConfig, + prompt: "animate all references", + inputImages: Array.from({ length: 5 }, (_, index) => ({ + url: `https://example.com/reference-${index}.png`, + })), + }); + + expect(seenImageCount).toBe(5); + expect(result.model).toBe("wan2.6-r2v"); + expect(result.attempts).toEqual([]); + }); + + it("preserves Wan 2.6 resolution and aspect ratio until adapter mapping", async () => { + let seenRequest: + | { size?: string; resolution?: string; aspectRatio?: string; parameters?: unknown } + | undefined; + providers = [ + { + id: "qwen", + defaultModel: "wan2.6-t2v", + models: [...DASHSCOPE_WAN_VIDEO_MODELS], + capabilities: DASHSCOPE_WAN_VIDEO_CAPABILITIES, + catalogByModel: DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL, + resolveModelCapabilities: ({ model }) => + DASHSCOPE_WAN_VIDEO_CATALOG_BY_MODEL[model]?.capabilities, + async generateVideo(req) { + seenRequest = { + size: req.size, + resolution: req.resolution, + aspectRatio: req.aspectRatio, + parameters: buildDashscopeVideoGenerationParameters(req), + }; + return { + videos: [{ buffer: Buffer.from("mp4-bytes"), mimeType: "video/mp4" }], + model: req.model, + }; + }, + }, + ]; + + await runGenerateVideo({ + cfg: { + agents: { + defaults: { + videoGenerationModel: { primary: "qwen/wan2.6-t2v" }, + }, + }, + } as OpenClawConfig, + prompt: "portrait video", + resolution: "1080P", + aspectRatio: "9:16", + }); + + expect(seenRequest).toEqual({ + size: undefined, + resolution: "1080P", + aspectRatio: "9:16", + parameters: { size: "1080*1920" }, + }); + }); + it("skips providers whose live model capabilities disable video inputs", async () => { providers = [ { diff --git a/src/video-generation/runtime.ts b/src/video-generation/runtime.ts index 898dbaf99a7b..44d810319220 100644 --- a/src/video-generation/runtime.ts +++ b/src/video-generation/runtime.ts @@ -14,7 +14,7 @@ import { import { getProviderEnvVars } from "../secrets/provider-env-vars.js"; import { resolveVideoGenerationModeCapabilities } from "./capabilities.js"; import { - buildReferenceInputCapabilityFailure, + buildVideoGenerationCapabilityFailure, resolveProviderWithModelCapabilities, } from "./capability-overlays.js"; import { resolveVideoGenerationSupportedDurations } from "./duration-support.js"; @@ -176,13 +176,12 @@ export async function generateVideo( log: logger, }); - // Guard: skip candidates that cannot satisfy reference-input counts so - // we never silently drop audio/image/video refs by falling over to a - // provider that ignores them and "succeeds" without the caller's assets. + // Guard: catalog modes and reference counts are authoritative before I/O, + // so fallback cannot select a model that will reject or drop the request. const inputImageCount = params.inputImages?.length ?? 0; const inputVideoCount = params.inputVideos?.length ?? 0; const inputAudioCount = params.inputAudios?.length ?? 0; - const referenceInputMismatch = buildReferenceInputCapabilityFailure({ + const capabilityMismatch = buildVideoGenerationCapabilityFailure({ providerId: candidate.provider, model: candidate.model, provider: activeProvider, @@ -190,16 +189,16 @@ export async function generateVideo( inputVideoCount, inputAudioCount, }); - if (referenceInputMismatch) { + if (capabilityMismatch) { attempts.push({ provider: candidate.provider, model: candidate.model, - error: referenceInputMismatch, + error: capabilityMismatch, }); - lastError = new Error(referenceInputMismatch); - warnOnFirstSkip(referenceInputMismatch); + lastError = new Error(capabilityMismatch); + warnOnFirstSkip(capabilityMismatch); logger.debug( - `video-generation candidate skipped (reference input capability): ${candidate.provider}/${candidate.model}`, + `video-generation candidate skipped (mode or reference capability): ${candidate.provider}/${candidate.model}`, ); continue; } diff --git a/test/e2e/qa-lab/media/video-generation-invocation.e2e.test.ts b/test/e2e/qa-lab/media/video-generation-invocation.e2e.test.ts index e47829665a45..728b293ae5b8 100644 --- a/test/e2e/qa-lab/media/video-generation-invocation.e2e.test.ts +++ b/test/e2e/qa-lab/media/video-generation-invocation.e2e.test.ts @@ -5,7 +5,7 @@ import type { PreparedModelRuntimeSnapshot } from "../../../../src/agents/prepar import { createVideoGenerateTool } from "../../../../src/agents/tools/video-generate-tool.js"; import type { OpenClawConfig } from "../../../../src/config/types.js"; import { withEnvAsync } from "../../../../src/test-utils/env.js"; -import { resolveVideoGenerationMode } from "../../../../src/video-generation/capabilities.js"; +import { resolveVideoGenerationModeCapabilities } from "../../../../src/video-generation/capabilities.js"; import type { VideoGenerationProvider, VideoGenerationRequest, @@ -143,10 +143,12 @@ describe("video generation invocation QA", () => { providerOptions, }); expect( - resolveVideoGenerationMode({ + resolveVideoGenerationModeCapabilities({ + provider: fallbackProvider, + model: fallbackRequest?.model, inputImageCount: fallbackRequest?.inputImages?.length, inputVideoCount: fallbackRequest?.inputVideos?.length, - }), + }).mode, ).toBe("imageToVideo"); expect(fallbackRequest?.inputImages).toEqual([ {