feat(canvas): remove Volcengine Ark protocol support and streamline video generation references

This commit is contained in:
HouYunFei
2026-08-18 15:35:28 +08:00
parent b8053ee9f8
commit 0bdc820247
22 changed files with 68 additions and 667 deletions
@@ -1,7 +1,6 @@
import type { AiTextMessage } from "@/services/api/image";
import i18n from "@/i18n";
import { imageReferenceLabel } from "@/lib/image-reference-prompt";
import { seedanceReferenceLabel } from "@/lib/seedance-video";
import type { ReferenceImage } from "@/types/image";
import type { ReferenceAudio, ReferenceVideo } from "@/types/media";
import { CanvasNodeType, type CanvasConnection, type CanvasNodeData } from "@/types/canvas";
@@ -153,8 +152,8 @@ function readNodeTextInput(node: CanvasNodeData) {
function generationLabel(type: NodeGenerationInput["type"], index: number) {
if (type === "image") return imageReferenceLabel(index);
if (type === "video") return seedanceReferenceLabel("video", index);
if (type === "audio") return seedanceReferenceLabel("audio", index);
if (type === "video") return i18n.t("canvas.configNode.videoReferences") + ` ${index + 1}`;
if (type === "audio") return i18n.t("canvas.configNode.audioReferences") + ` ${index + 1}`;
return i18n.t("canvas.composer.resources.text", { index: index + 1 });
}
@@ -196,7 +196,7 @@ export function AppConfigPanel({ showDoneButton = false, initialTab = "channels"
<div className="min-w-0">
<div className="truncate text-sm font-semibold">{channel.name || t("config.channels.unnamed")}</div>
<div className="mt-1 truncate text-xs text-stone-500">
{apiFormatLabel(channel.apiFormat, t)} · {t("config.channels.modelCount", { count: channel.models.length })} · {channel.baseUrl || t("config.channels.missingUrl")}
{apiFormatLabel(channel.apiFormat)} · {t("config.channels.modelCount", { count: channel.models.length })} · {channel.baseUrl || t("config.channels.missingUrl")}
</div>
</div>
<div className="flex shrink-0 gap-2">
@@ -384,9 +384,8 @@ function normalizeImageCount(value: string) {
return String(Math.max(1, Math.min(15, Math.floor(Math.abs(Number(value)) || 3))));
}
function apiFormatLabel(apiFormat: ApiCallFormat, t: TFunction) {
function apiFormatLabel(apiFormat: ApiCallFormat) {
if (apiFormat === "gemini") return "Gemini";
if (apiFormat === "ark") return t("config.protocols.ark");
return "OpenAI";
}
@@ -17,7 +17,6 @@ export function ChannelEditorDrawer({ open, channel, onSave, onClose }: { open:
const apiFormatOptions: Array<{ label: string; value: ApiCallFormat }> = [
{ label: "OpenAI", value: "openai" },
{ label: "Gemini", value: "gemini" },
{ label: t("config.protocols.ark"), value: "ark" },
];
const capabilityOptions: Array<{ label: string; value: ModelCapability }> = ["image", "video", "text", "audio"].map((value) => ({ label: t(`config.channelEditor.capabilities.${value}`), value: value as ModelCapability }));
@@ -1,10 +1,8 @@
import { type ReactNode } from "react";
import { Switch } from "antd";
import { useTranslation } from "react-i18next";
import i18n from "@/i18n";
import { ImageSettingsTheme } from "@/components/image-settings-panel";
import { boolConfig, isSeedanceVideoConfig, normalizeSeedanceDuration, normalizeSeedanceRatio, normalizeSeedanceResolution, seedanceDurationOptions, seedancePixelLabel, seedanceRatioOptions, seedanceResolutionOptions } from "@/lib/seedance-video";
import { type CanvasTheme } from "@/lib/canvas-theme";
import { type AiConfig } from "@/stores/use-config-store";
@@ -23,7 +21,6 @@ const sizeOptions = [
];
const secondOptions = [6, 10, 12, 16, 20];
const seedanceRatioLabelKeys: Record<string, string> = { "16:9": "landscape", "9:16": "portrait", "1:1": "square", "4:3": "standardLandscape", "3:4": "standardPortrait", "21:9": "cinematic", adaptive: "adaptive" };
export const videoResolutionOptions = resolutionOptions.map((item) => ({ value: item.value, label: item.label }));
export const videoSizeOptions = sizeOptions.map((item) => ({ value: item.value, get label() { return i18n.t(`settingsPanels.video.sizes.${item.labelKey}`); } }));
@@ -39,10 +36,6 @@ type VideoSettingsPanelProps = {
export function VideoSettingsPanel({ config, onConfigChange, theme, showTitle = true, className = "w-[320px] space-y-4 rounded-2xl px-1 py-0.5" }: VideoSettingsPanelProps) {
const { t } = useTranslation();
if (isSeedanceVideoConfig(config)) {
return <SeedanceVideoSettingsPanel config={config} onConfigChange={onConfigChange} theme={theme} showTitle={showTitle} className={className} />;
}
const seconds = config.videoSeconds || "6";
const size = normalizeVideoSizeValue(config.size);
const dimensions = readSizeDimensions(size);
@@ -108,74 +101,12 @@ export function VideoSettingsPanel({ config, onConfigChange, theme, showTitle =
);
}
function SeedanceVideoSettingsPanel({ config, onConfigChange, theme, showTitle, className }: VideoSettingsPanelProps) {
const { t } = useTranslation();
const resolution = normalizeSeedanceResolution(config.vquality);
const ratio = normalizeSeedanceRatio(config.size);
const duration = normalizeSeedanceDuration(config.videoSeconds);
const generateAudio = boolConfig(config.videoGenerateAudio, true);
const watermark = boolConfig(config.videoWatermark, false);
return (
<ImageSettingsTheme theme={theme}>
<div className={className} style={{ color: theme.node.text }} onMouseDown={(event) => event.stopPropagation()}>
{showTitle ? <div className="text-lg font-semibold">{t("settingsPanels.video.title")}</div> : null}
<SettingGroup title={t("settingsPanels.video.resolution")} color={theme.node.muted}>
<div className="grid grid-cols-3 gap-2.5">
{seedanceResolutionOptions.map((item) => (
<OptionPill key={item.value} selected={resolution === item.value} theme={theme} onClick={() => onConfigChange("vquality", item.value)}>
{item.label}
</OptionPill>
))}
</div>
</SettingGroup>
<SettingGroup title={t("settingsPanels.video.ratio")} color={theme.node.muted}>
<div className="grid grid-cols-3 gap-2.5">
{seedanceRatioOptions.map((item) => (
<button
key={item.value}
type="button"
className="flex h-[68px] cursor-pointer flex-col items-center justify-center gap-1 rounded-xl border bg-transparent px-1 text-sm transition hover:opacity-80"
style={{ borderColor: ratio === item.value ? theme.node.text : theme.node.stroke, color: theme.node.text }}
onMouseDown={(event) => event.stopPropagation()}
onClick={() => onConfigChange("size", item.value)}
>
<SizePreview width={ratioPreview(item.value).width} height={ratioPreview(item.value).height} color={theme.node.text} />
<span>{i18n.t(`settingsPanels.video.ratios.${seedanceRatioLabelKeys[item.value]}`)}</span>
<span className="text-[10px] leading-none opacity-55">{item.value === "adaptive" ? "adaptive" : seedancePixelLabel(resolution, item.value)}</span>
</button>
))}
</div>
</SettingGroup>
<SettingGroup title={t("settingsPanels.video.duration")} color={theme.node.muted}>
<div className="grid grid-cols-4 gap-2.5">
{seedanceDurationOptions.map((value) => (
<OptionPill key={value} selected={duration === value} theme={theme} onClick={() => onConfigChange("videoSeconds", String(value))}>
{value === -1 ? t("settingsPanels.video.smart") : `${value}s`}
</OptionPill>
))}
</div>
<NumberInput value={String(duration)} min={-1} max={15} theme={theme} onChange={(value) => onConfigChange("videoSeconds", value)} />
</SettingGroup>
<SettingGroup title={t("settingsPanels.video.output")} color={theme.node.muted}>
<div className="grid gap-2 rounded-xl border p-2.5" style={{ borderColor: theme.node.stroke }}>
<SwitchRow label={t("settingsPanels.video.generateAudio")} checked={generateAudio} theme={theme} onChange={(checked) => onConfigChange("videoGenerateAudio", String(checked))} />
<SwitchRow label={t("settingsPanels.video.watermark")} checked={watermark} theme={theme} onChange={(checked) => onConfigChange("videoWatermark", String(checked))} />
</div>
</SettingGroup>
</div>
</ImageSettingsTheme>
);
}
export function videoResolutionLabel(value: string) {
return `${normalizeVideoResolutionValue(value)}p`;
}
export function videoSizeLabel(value: string) {
const ratio = normalizeSeedanceRatio(value);
if (value === "adaptive" || value === "auto") return i18n.t("settingsPanels.video.adaptive");
if (ratio === value) return i18n.t(`settingsPanels.video.ratios.${seedanceRatioLabelKeys[ratio]}`);
const size = normalizeVideoSizeValue(value);
const option = sizeOptions.find((item) => item.value === size);
return option ? i18n.t(`settingsPanels.video.sizes.${option.labelKey}`) : size;
@@ -251,29 +182,6 @@ function SizePreview({ width, height, color }: { width: number; height: number;
return <span className="rounded-[3px] border-2" style={{ width: previewWidth, height: previewHeight, borderColor: color }} />;
}
function ratioPreview(ratio: string) {
if (ratio === "9:16") return { width: 9, height: 16 };
if (ratio === "1:1") return { width: 1, height: 1 };
if (ratio === "4:3") return { width: 4, height: 3 };
if (ratio === "3:4") return { width: 3, height: 4 };
if (ratio === "21:9") return { width: 21, height: 9 };
if (ratio === "adaptive") return { width: 0, height: 0 };
return { width: 16, height: 9 };
}
function SwitchRow({ label, checked, theme, onChange }: { label: string; checked: boolean; theme: CanvasTheme; onChange: (checked: boolean) => void }) {
return (
<div className="flex h-8 items-center justify-between gap-3">
<span className="text-sm" style={{ color: theme.node.text }}>
{label}
</span>
<span onMouseDown={(event) => event.stopPropagation()}>
<Switch size="small" checked={checked} onChange={onChange} />
</span>
</div>
);
}
function readSizeDimensions(size: string) {
if (size === "auto") return { width: 0, height: 0 };
const match = size.match(/^(\d+)x(\d+)$/);
+1 -3
View File
@@ -38,14 +38,13 @@ export default {
},
generation: { pending: ["Creating image", "Almost there", "Just a little longer", "Refining details"] },
imageReferences: { label: "Image {{index}}", separator: ", ", promptPrefix: "Reference image labels: {{labels}}. Use these labels to interpret image references in the prompt.\n\n{{prompt}}" },
seedance: { autoMatch: "Automatic", separator: ", ", references: { image: "Image {{index}}", video: "Video {{index}}", audio: "Audio {{index}}" }, promptPrefix: "Reference asset labels: {{labels}}. Use these labels to interpret image, video, and audio references in the prompt.\n\n{{prompt}}", referenceHint: "Reference videos must be mp4/mov, H.264/H.265, and 24–60 FPS. Use authorized Volcengine asset:// assets when real faces are included.", errors: { format: "{{label}} supports mp4/mov only", size: "{{label}} exceeds 200 MB. Compress it before uploading.", duration: "{{label}} must be 2–15 seconds", dimensions: "{{label}} dimensions must be between 300 and 6000px", ratio: "{{label}} aspect ratio must be between 0.4 and 2.5", pixels: "{{label}} total pixel count must be between 409600 and 8295044", totalDuration: "Seedance reference videos cannot exceed 15 seconds in total" } },
modelPlugin: {
pollTimeout: "Plugin polling timed out. Check the request script or try again later.", executionFailed: "Model request script failed: {{message}}", noImages: "The model request script did not return any images",
variables: { prompt: "User prompt with the system prompt already included", images: "Reference images as a data URL array, available for image editing and image-to-video", messages: "Conversation message array including the system message", params: "Generation parameters: image {size,quality,count}, video {seconds,size,resolution,ratio,generateAudio,watermark}, audio {voice,format,speed,instructions}", model: "Model name without the provider prefix", baseUrl: "Provider endpoint as entered, without appending /v1", apiKey: "Provider API key; add it to request headers yourself", systemPrompt: "Original system prompt", reasoningEffort: "Text reasoning effort; auto lets the script decide whether to send it", http: "Convenience client: http.post(path, body, {headers,params,responseType}), http.get(path, opts), and http.url(path). Authorization: Bearer apiKey is included by default and can be overridden. Relative paths append /v1 to baseUrl.", request: "Raw request({ method, url, headers, params, data, responseType }) with no default headers. Add authentication yourself. Relative URLs are joined to baseUrl without /v1.", poll: "poll(request, extract, {intervalMs,timeoutMs}) continues until extract returns a truthy value", sleep: "Delay with sleep(ms)", signal: "Cancellation signal that can be passed to http/request", onDelta: "Push streaming text with onDelta(text) for text models" },
returns: { image: "Text-to-image and image editing use different APIs; distinguish them by whether images is empty. Return an image URL or data URL, an array of them, or [{ dataUrl }] / [{ url }] / [{ b64_json }].", video: "Poll inside the script and return { url }, { blob }, or a video URL string.", audio: "Return a Blob, base64/data URL string, or { b64_json } / { data } / { url }.", text: "Push streaming output with onDelta(text), then return the complete text string." },
templates: { openai: "OpenAI format", gemini: "Gemini format", imageOpenai: "Image generation and editing use different endpoints; distinguish them by whether images is empty.", availableImage: "Available: prompt, images(dataURL[]), params{size,quality,count}, model, baseUrl, apiKey", textToImage: "Text to image: /images/generations (JSON)", imageToImage: "Image editing: /images/edits (multipart/form-data with reference images uploaded as files)", formDataHeader: "Do not set Content-Type manually; let the browser add the boundary", imageGemini: "Gemini image generation and editing both use generateContent, with references in parts.inline_data.", availableImageGemini: "Available: prompt, images(dataURL[]), model, baseUrl, apiKey", videoOpenai: "Video with polling handled inside the script. Available: prompt, images(dataURL[]), params{seconds,size,resolution,ratio}", videoGemini: "Gemini (Veo) video: submit with predictLongRunning and poll the operation for the video URI.", availableVideoGemini: "Available: prompt, images(dataURL[]), params, model, baseUrl, apiKey", geminiNoVideoUri: "Gemini did not return a video URI", audioOpenai: "Audio TTS. Available: prompt, params{voice,format,speed,instructions}, model", audioGemini: "Gemini TTS: generateContent with AUDIO modality, returning base64 PCM in inlineData.data.", availableAudioGemini: "Available: prompt, params{voice}, model, baseUrl, apiKey", geminiNoAudio: "Gemini did not return audio", textOpenai: "Text chat using the OpenAI Responses API. Available: messages([{role,content}]), systemPrompt, model, reasoningEffort", textGemini: "Gemini text: generateContent with the system message in systemInstruction.", availableTextGemini: "Available: messages([{role,content}]), systemPrompt, model, baseUrl, apiKey" },
},
apiErrors: { requestFailed: "Request failed", requestCanceled: "Request canceled", baseUrlRequired: "Configure the Base URL first", apiKeyRequired: "Configure the API key first", authenticationFailed: "Authentication failed. Check the API key, plan permissions, and model permissions.", rateLimited: "The request was rate-limited or the quota is insufficient. Try again later.", notFound: "The endpoint was not found (404). Check the Base URL and selected model.", badGateway: "Gateway error (502). The API service is temporarily unavailable.", serviceBusy: "Service unavailable (503). Try again later.", httpFailed: "Request failed (HTTP {{status}}). Check the Base URL and API key.", htmlError: "The service returned an HTML error page ({{preview}})", audioModelRequired: "Configure an audio model first", audioGenerationFailed: "Audio generation failed", scriptNoAudio: "The model request script did not return audio", geminiAudioUnsupported: "The Gemini API format does not support audio generation. Use an OpenAI-format provider.", invalidImageSizeFormat: "Unsupported image size. Use auto, 9:16, or 1024x1024.", positiveImageRatio: "The image ratio must use positive numbers, such as 9:16.", imageRatioLimit: "The image aspect ratio cannot exceed 3:1. Adjust the dimensions.", positiveImageDimensions: "Image dimensions must be positive integers, such as 1024x1024.", imageDimensionStep: "Image width and height must be multiples of 16.", imageEdgeLimit: "The longest image edge cannot exceed 3840px.", imagePixelLimit: "The total image pixel count must be between 655360 and 8294400.", unknownImageResponse: "The API returned data in an unknown format (fields: {{fields}}). Check model or API compatibility.", noImageReturned: "The API did not return an image. The prompt may have triggered safety review or the model may not support this operation.", geminiRejected: "Gemini rejected the request: {{reason}}", geminiNoImage: "The Gemini API did not return an image", geminiMaskUnsupported: "The Gemini API format does not support mask editing", maskModelUnsupported: "This model does not support mask editing. Use another provider.", noContent: "No content returned", modelReadFailed: "Failed to load models", videoTimeout: "{{provider}}video generation timed out. Try again later.", videoReferencesUnsupported: "This video API does not support reference video or audio. Switch to Seedance 2.0 / Volcengine Agent Plan, or remove the reference assets.", pluginVideoExpired: "The plugin video task has expired. Generate it again.", scriptNoVideo: "The model request script did not return a video", noPlayableVideo: "The video API did not return a playable video", noVideoTaskId: "The video API did not return a task ID", videoTaskCreateFailed: "Failed to create video task", videoGenerationFailed: "Video generation failed", videoTaskQueryFailed: "Failed to query video task", seedanceAudioRequiresVisual: "Seedance reference audio cannot be used alone. Add a reference image or video.", videoPromptRequired: "Enter a video prompt or connect a reference image, video, or audio asset", seedanceNoTaskId: "The Seedance API did not return a task ID", seedanceTaskCreateFailed: "Failed to create Seedance task", seedanceNoVideoUrl: "The Seedance task succeeded but did not return a video URL", seedanceVideoTimeout: "Seedance video generation timed out", seedanceVideoFailed: "Seedance video generation failed", seedanceTaskQueryFailed: "Failed to query Seedance task", seedanceVideoDuration: "Each Seedance reference video must be 2–15 seconds", seedanceVideoTotalDuration: "Seedance reference videos cannot exceed 15 seconds in total", seedanceAudioDuration: "Each Seedance reference audio file must be 2–15 seconds", seedanceAudioTotalDuration: "Seedance reference audio cannot exceed 15 seconds in total", referenceImageReadFailed: "Failed to read the reference image. Choose another image or upload it again.", invalidReferenceVideo: "A reference video must be a public URL, asset ID, or locally saved video", invalidReferenceAudio: "Reference audio must be a public URL, asset ID, or locally saved audio file", videoModelRequired: "Configure a video model first", geminiVideoUnsupported: "The Gemini API format does not support video generation. Use an OpenAI-format provider.", noVideoTask: "The API did not return a video task", seedanceNoTask: "The Seedance API did not return a task", videoDownloadFailed: "Failed to download video", localAssetReadFailed: "Failed to read local asset" },
apiErrors: { requestFailed: "Request failed", requestCanceled: "Request canceled", baseUrlRequired: "Configure the Base URL first", apiKeyRequired: "Configure the API key first", authenticationFailed: "Authentication failed. Check the API key, plan permissions, and model permissions.", rateLimited: "The request was rate-limited or the quota is insufficient. Try again later.", notFound: "The endpoint was not found (404). Check the Base URL and selected model.", badGateway: "Gateway error (502). The API service is temporarily unavailable.", serviceBusy: "Service unavailable (503). Try again later.", httpFailed: "Request failed (HTTP {{status}}). Check the Base URL and API key.", htmlError: "The service returned an HTML error page ({{preview}})", audioModelRequired: "Configure an audio model first", audioGenerationFailed: "Audio generation failed", scriptNoAudio: "The model request script did not return audio", geminiAudioUnsupported: "The Gemini API format does not support audio generation. Use an OpenAI-format provider.", invalidImageSizeFormat: "Unsupported image size. Use auto, 9:16, or 1024x1024.", positiveImageRatio: "The image ratio must use positive numbers, such as 9:16.", imageRatioLimit: "The image aspect ratio cannot exceed 3:1. Adjust the dimensions.", positiveImageDimensions: "Image dimensions must be positive integers, such as 1024x1024.", imageDimensionStep: "Image width and height must be multiples of 16.", imageEdgeLimit: "The longest image edge cannot exceed 3840px.", imagePixelLimit: "The total image pixel count must be between 655360 and 8294400.", unknownImageResponse: "The API returned data in an unknown format (fields: {{fields}}). Check model or API compatibility.", noImageReturned: "The API did not return an image. The prompt may have triggered safety review or the model may not support this operation.", geminiRejected: "Gemini rejected the request: {{reason}}", geminiNoImage: "The Gemini API did not return an image", geminiMaskUnsupported: "The Gemini API format does not support mask editing", maskModelUnsupported: "This model does not support mask editing. Use another provider.", noContent: "No content returned", modelReadFailed: "Failed to load models", videoTimeout: "{{provider}}video generation timed out. Try again later.", pluginVideoExpired: "The plugin video task has expired. Generate it again.", scriptNoVideo: "The model request script did not return a video", noPlayableVideo: "The video API did not return a playable video", noVideoTaskId: "The video API did not return a task ID", videoTaskCreateFailed: "Failed to create video task", videoGenerationFailed: "Video generation failed", videoTaskQueryFailed: "Failed to query video task", videoPromptRequired: "Enter a video prompt or connect a reference image, video, or audio asset", referenceImageReadFailed: "Failed to read the reference image. Choose another image or upload it again.", invalidReferenceVideo: "A reference video must be a public URL, asset ID, or locally saved video", invalidReferenceAudio: "Reference audio must be a public URL, asset ID, or locally saved audio file", videoModelRequired: "Configure a video model first", geminiVideoUnsupported: "The Gemini API format does not support video generation. Use an OpenAI-format provider.", noVideoTask: "The API did not return a video task", videoDownloadFailed: "Failed to download video", localAssetReadFailed: "Failed to read local asset" },
prompts: {
title: "Prompt Center",
library: "Prompt Library",
@@ -571,7 +570,6 @@ export default {
errors: { testFailed: "WebDAV connection test failed", downloadFailed: "Failed to read the WebDAV sync file", downloadTimeout: "Timed out while reading the WebDAV sync file", emptyUpload: "The upload file is empty; upload canceled", uploadFailed: "Failed to upload the WebDAV sync file", directoryFailed: "Failed to create the remote WebDAV directory", requestTimeout: "The WebDAV request timed out. Check the network or remote service.", connectionFailed: "Could not connect to WebDAV. Check the address, HTTPS certificate, CORS, and network.", urlRequired: "Enter a WebDAV URL first", authenticationFailed: "WebDAV authentication failed. Check the username, password, or app password.", pathMissing: "The WebDAV path does not exist. Check the address and remote directory.", responseFailed: "{{fallback}}: {{status}}{{detail}}", syncFailed: "Sync failed", invalidManifest: "The {{domain}} sync manifest does not belong to this app" },
},
protocols: {
ark: "Volcengine Ark",
},
},
agent: {
+1 -3
View File
@@ -38,14 +38,13 @@ export default {
},
generation: { pending: ["正在创建图片", "马上就好了", "再等等", "正在整理细节"] },
imageReferences: { label: "图片{{index}}", separator: "、", promptPrefix: "参考图片编号:{{labels}}。请按这些编号理解提示词中的图片引用。\n\n{{prompt}}" },
seedance: { autoMatch: "自动匹配", separator: "、", references: { image: "图片{{index}}", video: "视频{{index}}", audio: "音频{{index}}" }, promptPrefix: "参考资产编号:{{labels}}。请按这些编号理解提示词中的图片、视频和音频引用。\n\n{{prompt}}", referenceHint: "参考视频需为 mp4/mov,H.264/H.265,FPS 24-60;含真人人脸资产请使用火山授权 asset:// 资产。", errors: { format: "{{label}} 仅支持 mp4/mov 格式", size: "{{label}} 超过 200MB,请压缩后再上传", duration: "{{label}} 时长需要在 2-15 秒之间", dimensions: "{{label}} 宽高需要在 300-6000px 之间", ratio: "{{label}} 宽高比需要在 0.4-2.5 之间", pixels: "{{label}} 总像素需要在 409600-8295044 之间", totalDuration: "Seedance 参考视频总时长不能超过 15 秒" } },
modelPlugin: {
pollTimeout: "插件轮询超时,请检查调用脚本或稍后重试", executionFailed: "模型调用脚本执行失败:{{message}}", noImages: "模型调用脚本没有返回图片",
variables: { prompt: "用户输入的提示词(已拼接系统提示词)", images: "参考图,dataURL 数组(改图 / 图生视频时有值)", messages: "对话消息数组,含系统消息", params: "生成参数:生图 {size,quality,count}、视频 {seconds,size,resolution,ratio,generateAudio,watermark}、音频 {voice,format,speed,instructions}", model: "模型名称(不含渠道前缀)", baseUrl: "渠道接口地址(原样,未拼 /v1)", apiKey: "渠道 API Key,请求头里自己带上", systemPrompt: "系统提示词原文", reasoningEffort: "文本推理强度;auto 表示由脚本决定是否传递", http: "便捷请求:http.post(path, body, {headers,params,responseType})、http.get(path, opts)、http.url(path);默认带 Authorization: Bearer apiKey,可用 headers 覆盖;path 相对时按 baseUrl 拼 /v1", request: "原始请求 request({ method, url, headers, params, data, responseType }),不加任何默认头,鉴权头自己写;url 相对时按 baseUrl 拼接(不加 /v1)", poll: "轮询 poll(request, extract, {intervalMs,timeoutMs}),extract 返回真值即结束", sleep: "sleep(ms) 延时", signal: "取消信号,可透传给 http/request", onDelta: "onDelta(text) 推送流式文本(文本模型)" },
returns: { image: "文生图(images 为空)和图生图(images 有参考图)接口不同,脚本需自行区分;返回图片 URL 或 dataURL 字符串,也可返回它们的数组,或 [{ dataUrl }] / [{ url }] / [{ b64_json }]", video: "脚本内部完成轮询,返回 { url } 或 { blob } 或视频 URL 字符串", audio: "返回 Blob,或 base64 / dataURL 字符串,或 { b64_json } / { data } / { url }", text: "用 onDelta(text) 推送流式,最终 return 完整文本字符串" },
templates: { openai: "OpenAI 规范", gemini: "Gemini 规范", imageOpenai: "生图 / 改图:两者接口不同,用 images 是否为空来区分。", availableImage: "可用:prompt、images(dataURL[])、params{size,quality,count}、model、baseUrl、apiKey", textToImage: "文生图:/images/generations(JSON)", imageToImage: "图生图:/images/edits(multipart/form-data,参考图作为文件上传)", formDataHeader: "不要手动设 Content-Type,交给浏览器带 boundary", imageGemini: "Gemini 文生图 / 图生图:都走 generateContent,参考图放进 parts 的 inline_data。", availableImageGemini: "可用:prompt、images(dataURL[])、model、baseUrl、apiKey", videoOpenai: "视频(脚本内部自行轮询)。可用:prompt、images(dataURL[])、params{seconds,size,resolution,ratio}", videoGemini: "Gemini(Veo) 视频:predictLongRunning 提交,轮询 operation 拿视频 URI。", availableVideoGemini: "可用:prompt、images(dataURL[])、params、model、baseUrl、apiKey", geminiNoVideoUri: "Gemini 未返回视频 URI", audioOpenai: "音频 TTS。可用:prompt、params{voice,format,speed,instructions}、model", audioGemini: "Gemini TTS:generateContent + AUDIO 模态,返回 base64 PCM(音频数据在 inlineData.data)。", availableAudioGemini: "可用:prompt、params{voice}、model、baseUrl、apiKey", geminiNoAudio: "Gemini 未返回音频", textOpenai: "文本对话(OpenAI Responses 接口)。可用:messages([{role,content}])、systemPrompt、model、reasoningEffort", textGemini: "Gemini 文本:generateContent,system 消息放 systemInstruction。", availableTextGemini: "可用:messages([{role,content}])、systemPrompt、model、baseUrl、apiKey" },
},
apiErrors: { requestFailed: "请求失败", requestCanceled: "请求已取消", baseUrlRequired: "请先配置 Base URL", apiKeyRequired: "请先配置 API Key", authenticationFailed: "鉴权失败,请检查 API Key、套餐权限或模型权限", rateLimited: "请求被限流或额度不足,请稍后重试", notFound: "接口地址不存在(404),请检查 Base URL 和模型选择", badGateway: "网关错误(502),接口服务暂时不可用,请稍后重试", serviceBusy: "服务繁忙(503),请稍后重试", httpFailed: "请求失败(HTTP {{status}}),请检查 Base URL 和 API Key 是否正确", htmlError: "服务返回了 HTML 错误页面({{preview}})", audioModelRequired: "请先配置音频模型", audioGenerationFailed: "音频生成失败", scriptNoAudio: "模型调用脚本没有返回音频", geminiAudioUnsupported: "Gemini 调用格式暂不支持音频生成,请使用 OpenAI 格式渠道", invalidImageSizeFormat: "图像尺寸格式不支持,请使用 auto、9:16 或 1024x1024", positiveImageRatio: "图像比例必须是正数,例如 9:16", imageRatioLimit: "图像宽高比不能超过 3:1,请调整尺寸", positiveImageDimensions: "图像尺寸必须是正整数,例如 1024x1024", imageDimensionStep: "图像尺寸的宽高必须是 16 的倍数,请调整尺寸", imageEdgeLimit: "图像尺寸最长边不能超过 3840px,请调整尺寸", imagePixelLimit: "图像总像素需在 655360 到 8294400 之间,请调整尺寸", unknownImageResponse: "接口返回了未知格式的数据(字段:{{fields}}),请检查模型或接口兼容性", noImageReturned: "接口没有返回图片,请检查提示词是否触发安全审核或模型是否支持该操作", geminiRejected: "Gemini 拒绝了本次请求:{{reason}}", geminiNoImage: "Gemini 接口没有返回图片", geminiMaskUnsupported: "Gemini 调用格式暂不支持蒙版编辑", maskModelUnsupported: "蒙版编辑暂不支持该模型,请使用其他渠道", noContent: "没有返回内容", modelReadFailed: "读取模型失败", videoTimeout: "{{provider}}视频生成超时,请稍后重试", videoReferencesUnsupported: "当前视频接口不支持参考视频或参考音频,请切换到 Seedance 2.0 / 火山 Agent Plan 模型,或移除参考资产", pluginVideoExpired: "插件视频任务已失效,请重新生成", scriptNoVideo: "模型调用脚本没有返回视频", noPlayableVideo: "视频接口没有返回可播放的视频", noVideoTaskId: "视频接口没有返回任务 ID", videoTaskCreateFailed: "视频任务创建失败", videoGenerationFailed: "视频生成失败", videoTaskQueryFailed: "视频任务查询失败", seedanceAudioRequiresVisual: "Seedance 参考音频不能单独使用,请同时添加参考图或参考视频", videoPromptRequired: "请输入视频提示词,或连接参考图片/视频/音频", seedanceNoTaskId: "Seedance 接口没有返回任务 ID", seedanceTaskCreateFailed: "Seedance 任务创建失败", seedanceNoVideoUrl: "Seedance 任务成功但没有返回视频 URL", seedanceVideoTimeout: "Seedance 视频生成超时", seedanceVideoFailed: "Seedance 视频生成失败", seedanceTaskQueryFailed: "Seedance 任务查询失败", seedanceVideoDuration: "Seedance 参考视频单个时长需要在 2-15 秒之间", seedanceVideoTotalDuration: "Seedance 参考视频总时长不能超过 15 秒", seedanceAudioDuration: "Seedance 参考音频单个时长需要在 2-15 秒之间", seedanceAudioTotalDuration: "Seedance 参考音频总时长不能超过 15 秒", referenceImageReadFailed: "参考图读取失败,请换一张图片或重新上传", invalidReferenceVideo: "参考视频必须是公网 URL、资产 ID,或本地已保存的视频", invalidReferenceAudio: "参考音频必须是公网 URL、资产 ID,或本地已保存的音频", videoModelRequired: "请先配置视频模型", geminiVideoUnsupported: "Gemini 调用格式暂不支持视频生成,请使用 OpenAI 格式渠道", noVideoTask: "接口没有返回视频任务", seedanceNoTask: "Seedance 接口没有返回任务", videoDownloadFailed: "视频下载失败", localAssetReadFailed: "读取本地资产失败" },
apiErrors: { requestFailed: "请求失败", requestCanceled: "请求已取消", baseUrlRequired: "请先配置 Base URL", apiKeyRequired: "请先配置 API Key", authenticationFailed: "鉴权失败,请检查 API Key、套餐权限或模型权限", rateLimited: "请求被限流或额度不足,请稍后重试", notFound: "接口地址不存在(404),请检查 Base URL 和模型选择", badGateway: "网关错误(502),接口服务暂时不可用,请稍后重试", serviceBusy: "服务繁忙(503),请稍后重试", httpFailed: "请求失败(HTTP {{status}}),请检查 Base URL 和 API Key 是否正确", htmlError: "服务返回了 HTML 错误页面({{preview}})", audioModelRequired: "请先配置音频模型", audioGenerationFailed: "音频生成失败", scriptNoAudio: "模型调用脚本没有返回音频", geminiAudioUnsupported: "Gemini 调用格式暂不支持音频生成,请使用 OpenAI 格式渠道", invalidImageSizeFormat: "图像尺寸格式不支持,请使用 auto、9:16 或 1024x1024", positiveImageRatio: "图像比例必须是正数,例如 9:16", imageRatioLimit: "图像宽高比不能超过 3:1,请调整尺寸", positiveImageDimensions: "图像尺寸必须是正整数,例如 1024x1024", imageDimensionStep: "图像尺寸的宽高必须是 16 的倍数,请调整尺寸", imageEdgeLimit: "图像尺寸最长边不能超过 3840px,请调整尺寸", imagePixelLimit: "图像总像素需在 655360 到 8294400 之间,请调整尺寸", unknownImageResponse: "接口返回了未知格式的数据(字段:{{fields}}),请检查模型或接口兼容性", noImageReturned: "接口没有返回图片,请检查提示词是否触发安全审核或模型是否支持该操作", geminiRejected: "Gemini 拒绝了本次请求:{{reason}}", geminiNoImage: "Gemini 接口没有返回图片", geminiMaskUnsupported: "Gemini 调用格式暂不支持蒙版编辑", maskModelUnsupported: "蒙版编辑暂不支持该模型,请使用其他渠道", noContent: "没有返回内容", modelReadFailed: "读取模型失败", videoTimeout: "{{provider}}视频生成超时,请稍后重试", pluginVideoExpired: "插件视频任务已失效,请重新生成", scriptNoVideo: "模型调用脚本没有返回视频", noPlayableVideo: "视频接口没有返回可播放的视频", noVideoTaskId: "视频接口没有返回任务 ID", videoTaskCreateFailed: "视频任务创建失败", videoGenerationFailed: "视频生成失败", videoTaskQueryFailed: "视频任务查询失败", videoPromptRequired: "请输入视频提示词,或连接参考图片/视频/音频", referenceImageReadFailed: "参考图读取失败,请换一张图片或重新上传", invalidReferenceVideo: "参考视频必须是公网 URL、资产 ID,或本地已保存的视频", invalidReferenceAudio: "参考音频必须是公网 URL、资产 ID,或本地已保存的音频", videoModelRequired: "请先配置视频模型", geminiVideoUnsupported: "Gemini 调用格式暂不支持视频生成,请使用 OpenAI 格式渠道", noVideoTask: "接口没有返回视频任务", videoDownloadFailed: "视频下载失败", localAssetReadFailed: "读取本地资产失败" },
prompts: {
title: "提示词中心",
library: "提示词库",
@@ -571,7 +570,6 @@ export default {
errors: { testFailed: "WebDAV 连接测试失败", downloadFailed: "读取 WebDAV 同步文件失败", downloadTimeout: "读取 WebDAV 同步文件超时", emptyUpload: "上传文件为空,已取消上传", uploadFailed: "上传 WebDAV 同步文件失败", directoryFailed: "创建 WebDAV 远程目录失败", requestTimeout: "WebDAV 请求超时,请检查网络或远端服务状态", connectionFailed: "无法连接 WebDAV,请检查地址、HTTPS 证书、CORS 或网络状态", urlRequired: "请先填写 WebDAV 地址", authenticationFailed: "WebDAV 认证失败,请检查用户名、密码或应用密码", pathMissing: "WebDAV 路径不存在,请检查地址和远程目录", responseFailed: "{{fallback}}:{{status}}{{detail}}", syncFailed: "同步失败", invalidManifest: "{{domain}} 同步清单不是当前应用的数据" },
},
protocols: {
ark: "火山方舟",
},
},
agent: {
@@ -1,6 +1,5 @@
import { imageReferenceLabel } from "@/lib/image-reference-prompt";
import i18n from "@/i18n";
import { seedanceReferenceLabel } from "@/lib/seedance-video";
import { getNodeDefinition } from "@/lib/canvas/node-registry";
import { getDataUrlByteSize, readImageMeta } from "@/lib/image-utils";
import { imageToDataUrl } from "@/services/image-storage";
@@ -106,8 +105,8 @@ function labelResourceNodes(nodes: CanvasNodeData[], active: boolean) {
function labelForKind(kind: CanvasResourceKind, index: number) {
if (kind === "image") return imageReferenceLabel(index);
if (kind === "video") return seedanceReferenceLabel("video", index);
if (kind === "audio") return seedanceReferenceLabel("audio", index);
if (kind === "video") return i18n.t("canvas.configNode.videoReferences") + ` ${index + 1}`;
if (kind === "audio") return i18n.t("canvas.configNode.audioReferences") + ` ${index + 1}`;
return i18n.t("canvas.composer.resources.text", { index: index + 1 });
}
-157
View File
@@ -1,157 +0,0 @@
import i18n from "@/i18n";
import { resolveModelRequestConfig, type AiConfig } from "@/stores/use-config-store";
import type { ReferenceImage } from "@/types/image";
import type { ReferenceAudio, ReferenceVideo } from "@/types/media";
export const SEEDANCE_REFERENCE_LIMITS = {
images: 9,
videos: 3,
audios: 3,
imageMaxBytes: 30 * 1024 * 1024,
videoMaxBytes: 200 * 1024 * 1024,
audioMaxBytes: 15 * 1024 * 1024,
};
export const SEEDANCE_VIDEO_MIME_TYPES = ["video/mp4", "video/quicktime"];
export const seedanceResolutionOptions = [
{ value: "480p", label: "480p" },
{ value: "720p", label: "720p" },
{ value: "1080p", label: "1080p" },
] as const;
export const seedanceRatioOptions = [
{ value: "16:9" },
{ value: "9:16" },
{ value: "1:1" },
{ value: "4:3" },
{ value: "3:4" },
{ value: "21:9" },
{ value: "adaptive" },
] as const;
export const seedanceDurationOptions = [-1, 4, 5, 6, 8, 10, 12, 15] as const;
const seedancePixels = {
"480p": {
"16:9": "864x496",
"4:3": "752x560",
"1:1": "640x640",
"3:4": "560x752",
"9:16": "496x864",
"21:9": "992x432",
},
"720p": {
"16:9": "1280x720",
"4:3": "1112x834",
"1:1": "960x960",
"3:4": "834x1112",
"9:16": "720x1280",
"21:9": "1470x630",
},
"1080p": {
"16:9": "1920x1080",
"4:3": "1664x1248",
"1:1": "1440x1440",
"3:4": "1248x1664",
"9:16": "1080x1920",
"21:9": "2206x946",
},
} as const;
export function isSeedanceVideoConfig(config: AiConfig | Pick<AiConfig, "model" | "videoModel" | "apiFormat">) {
const requestConfig = "channels" in config ? resolveModelRequestConfig(config, config.model || config.videoModel) : config;
return requestConfig.apiFormat === "ark";
}
export function normalizeSeedanceResolution(value: string) {
const normalized = normalizeResolutionToken(value);
return seedanceResolutionOptions.some((item) => item.value === normalized) ? normalized : "720p";
}
export function normalizeResolutionToken(value: string) {
if (value === "low") return "480p";
if (value === "auto" || value === "high" || value === "medium") return "720p";
const resolution = String(value || "").replace(/p$/i, "") || "720";
return `${resolution}p`;
}
export function normalizeSeedanceDuration(value: string) {
if (String(value).trim() === "-1") return -1;
const seconds = Math.floor(Number(value) || 5);
return Math.max(4, Math.min(15, seconds));
}
export function normalizeSeedanceRatio(value: string) {
if (!value || value === "auto" || value === "adaptive") return "adaptive";
if (seedanceRatioOptions.some((item) => item.value === value)) return value;
const match = value.match(/^(\d+)x(\d+)$/);
if (!match) return "adaptive";
const width = Number(match[1]);
const height = Number(match[2]);
if (!width || !height) return "adaptive";
const ratio = width / height;
const options = [
["16:9", 16 / 9],
["4:3", 4 / 3],
["1:1", 1],
["3:4", 3 / 4],
["9:16", 9 / 16],
["21:9", 21 / 9],
] as const;
return options.reduce((best, item) => (Math.abs(item[1] - ratio) < Math.abs(best[1] - ratio) ? item : best), options[0])[0];
}
export function seedancePixelLabel(resolution: string, ratio: string) {
const normalizedResolution = normalizeSeedanceResolution(resolution) as keyof typeof seedancePixels;
const normalizedRatio = normalizeSeedanceRatio(ratio) as keyof (typeof seedancePixels)[typeof normalizedResolution] | "adaptive";
if (normalizedRatio === "adaptive") return i18n.t("seedance.autoMatch");
return seedancePixels[normalizedResolution][normalizedRatio] || "";
}
export function boolConfig(value: string | undefined, fallback: boolean) {
if (value === "true") return true;
if (value === "false") return false;
return fallback;
}
export function seedanceReferenceLabel(kind: "image" | "video" | "audio", index: number) {
return i18n.t(`seedance.references.${kind}`, { index: index + 1 });
}
export function buildSeedancePromptText(prompt: string, images: ReferenceImage[], videos: ReferenceVideo[], audios: ReferenceAudio[]) {
const labels = [
...images.map((_, index) => seedanceReferenceLabel("image", index)),
...videos.map((_, index) => seedanceReferenceLabel("video", index)),
...audios.map((_, index) => seedanceReferenceLabel("audio", index)),
];
const text = prompt.trim();
if (!labels.length) return text;
return i18n.t("seedance.promptPrefix", { labels: labels.join(i18n.t("seedance.separator")), prompt: text });
}
export function seedanceVideoReferenceError(videos: ReferenceVideo[]) {
let totalDurationMs = 0;
for (let index = 0; index < videos.length; index += 1) {
const video = videos[index];
const label = seedanceReferenceLabel("video", index);
if (!SEEDANCE_VIDEO_MIME_TYPES.includes(video.type)) return i18n.t("seedance.errors.format", { label });
if (video.bytes && video.bytes > SEEDANCE_REFERENCE_LIMITS.videoMaxBytes) return i18n.t("seedance.errors.size", { label });
if (video.durationMs) {
if (video.durationMs < 2000 || video.durationMs > 15000) return i18n.t("seedance.errors.duration", { label });
totalDurationMs += video.durationMs;
}
if (video.width && video.height) {
if (video.width < 300 || video.width > 6000 || video.height < 300 || video.height > 6000) return i18n.t("seedance.errors.dimensions", { label });
const ratio = video.width / video.height;
if (ratio < 0.4 || ratio > 2.5) return i18n.t("seedance.errors.ratio", { label });
const pixels = video.width * video.height;
if (pixels < 640 * 640 || pixels > 3326 * 2494) return i18n.t("seedance.errors.pixels", { label });
}
}
if (totalDurationMs > 15000) return i18n.t("seedance.errors.totalDuration");
return "";
}
export function seedanceVideoReferenceHint() {
return i18n.t("seedance.referenceHint");
}
@@ -64,7 +64,7 @@ export function usePluginHost(params: PluginHostParams) {
...(options?.seconds ? { videoSeconds: options.seconds } : {}),
};
ensureReady(config);
const file = await storeGeneratedVideo(await requestVideoGeneration(config, prompt, toReferences(options?.references), [], [], { signal: options?.signal }));
const file = await storeGeneratedVideo(await requestVideoGeneration(config, prompt, toReferences(options?.references), { signal: options?.signal }));
return { url: file.url, mimeType: file.mimeType, width: file.width, height: file.height, durationMs: file.durationMs };
},
generateText: async (prompt, options) => {
+2 -2
View File
@@ -2248,7 +2248,7 @@ function InfiniteCanvasPage() {
const controller = startGenerationRequest(videoId, nodeId, nodeId, runController);
try {
const video = await storeGeneratedVideo(
await requestVideoGeneration(generationConfig, effectivePrompt, generationContext.referenceImages, generationContext.referenceVideos, generationContext.referenceAudios, { signal: controller.signal }),
await requestVideoGeneration(generationConfig, effectivePrompt, generationContext.referenceImages, { signal: controller.signal }),
);
const videoSize = fitNodeSize(video.width || spec.width, video.height || spec.height, VIDEO_NODE_MAX_WIDTH, VIDEO_NODE_MAX_HEIGHT);
setNodes((prev) =>
@@ -2453,7 +2453,7 @@ function InfiniteCanvasPage() {
return;
}
if (node.type === CanvasNodeType.Video) {
const video = await storeGeneratedVideo(await requestVideoGeneration(generationConfig, prompt, retryImages, context?.referenceVideos || [], context?.referenceAudios || [], { signal: controller.signal }));
const video = await storeGeneratedVideo(await requestVideoGeneration(generationConfig, prompt, retryImages, { signal: controller.signal }));
const videoSize = fitNodeSize(video.width || node.width, video.height || node.height, VIDEO_NODE_MAX_WIDTH, VIDEO_NODE_MAX_HEIGHT);
setNodes((prev) =>
prev.map((item) =>
+26 -175
View File
@@ -1,4 +1,4 @@
import { ArrowLeft, ArrowRight, BookOpen, CheckSquare, ClipboardPaste, Download, FolderPlus, History, LoaderCircle, Music2, Plus, SlidersHorizontal, Sparkles, Trash2, Upload, VideoIcon } from "lucide-react";
import { ArrowLeft, ArrowRight, BookOpen, CheckSquare, ClipboardPaste, Download, FolderPlus, History, LoaderCircle, Plus, SlidersHorizontal, Sparkles, Trash2, Upload, VideoIcon } from "lucide-react";
import { useEffect, useRef, useState, type DragEvent } from "react";
import { App, Button, Checkbox, Drawer, Empty, Input, Modal, Tag, Typography } from "antd";
import localforage from "localforage";
@@ -12,16 +12,14 @@ import { PromptSelectDialog } from "@/components/prompts/prompt-select-dialog";
import { VideoSettingsPanel, normalizeVideoResolutionValue, normalizeVideoSizeValue, videoSizeLabel } from "@/components/video-settings-panel";
import { canvasThemes } from "@/lib/canvas-theme";
import { formatBytes, formatDuration } from "@/lib/image-utils";
import { boolConfig, isSeedanceVideoConfig, normalizeSeedanceRatio, seedanceReferenceLabel, seedanceVideoReferenceError, seedanceVideoReferenceHint, SEEDANCE_REFERENCE_LIMITS, SEEDANCE_VIDEO_MIME_TYPES } from "@/lib/seedance-video";
import { deleteStoredMedia, resolveMediaUrl, uploadMediaFile } from "@/services/file-storage";
import { deleteStoredMedia, resolveMediaUrl } from "@/services/file-storage";
import { resolveImageUrl, uploadImage } from "@/services/image-storage";
import { createVideoGenerationTask, pollVideoGenerationTask, storeGeneratedVideo, type VideoGenerationTask } from "@/services/api/video";
import { useAssetStore } from "@/stores/use-asset-store";
import { useWorkbenchAgentStore } from "@/stores/use-workbench-agent-store";
import { modelOptionLabel, useConfigStore, useEffectiveConfig, type AiConfig } from "@/stores/use-config-store";
import { boolConfig, modelOptionLabel, useConfigStore, useEffectiveConfig, type AiConfig } from "@/stores/use-config-store";
import { useThemeStore } from "@/stores/use-theme-store";
import type { ReferenceImage } from "@/types/image";
import type { ReferenceAudio, ReferenceVideo } from "@/types/media";
import i18n from "@/i18n";
type GeneratedVideo = {
@@ -51,8 +49,6 @@ type GenerationLog = {
model: string;
config: GenerationLogConfig;
references: ReferenceImage[];
videoReferences: ReferenceVideo[];
audioReferences: ReferenceAudio[];
durationMs: number;
size: string;
resolution: string;
@@ -84,8 +80,6 @@ export default function VideoPage() {
const addAsset = useAssetStore((state) => state.addAsset);
const [prompt, setPrompt] = useState("");
const [references, setReferences] = useState<ReferenceImage[]>([]);
const [videoReferences, setVideoReferences] = useState<ReferenceVideo[]>([]);
const [audioReferences, setAudioReferences] = useState<ReferenceAudio[]>([]);
const [results, setResults] = useState<GenerationResult[]>([]);
const [logs, setLogs] = useState<GenerationLog[]>([]);
const [running, setRunning] = useState(false);
@@ -98,7 +92,7 @@ export default function VideoPage() {
const [selectedLogIds, setSelectedLogIds] = useState<string[]>([]);
const [previewLog, setPreviewLog] = useState<GenerationLog | null>(null);
const [deleteConfirmOpen, setDeleteConfirmOpen] = useState(false);
const [referenceDragTarget, setReferenceDragTarget] = useState<"image" | "video" | "audio" | null>(null);
const [referenceDragTarget, setReferenceDragTarget] = useState(false);
const [autoRunToken, setAutoRunToken] = useState(0);
const videoCommand = useWorkbenchAgentStore((state) => state.videoCommand);
const clearVideoCommand = useWorkbenchAgentStore((state) => state.clearVideoCommand);
@@ -121,57 +115,34 @@ export default function VideoPage() {
const addReferences = async (files?: FileList | null) => {
const selectedFiles = Array.from(files || []);
const unsupported = selectedFiles.filter((file) => !file.type.startsWith("image/") && !SEEDANCE_VIDEO_MIME_TYPES.includes(file.type) && !isSupportedAudioFile(file));
const unsupported = selectedFiles.filter((file) => !file.type.startsWith("image/"));
if (unsupported.length) message.warning(t("videoWorkbench.unsupportedFiles"));
const imageFiles = selectedFiles.filter((file) => file.type.startsWith("image/") && file.size <= SEEDANCE_REFERENCE_LIMITS.imageMaxBytes).slice(0, SEEDANCE_REFERENCE_LIMITS.images - references.length);
const videoFiles = selectedFiles.filter((file) => SEEDANCE_VIDEO_MIME_TYPES.includes(file.type) && file.size <= SEEDANCE_REFERENCE_LIMITS.videoMaxBytes).slice(0, SEEDANCE_REFERENCE_LIMITS.videos - videoReferences.length);
const audioFiles = selectedFiles.filter((file) => isSupportedAudioFile(file) && file.size <= SEEDANCE_REFERENCE_LIMITS.audioMaxBytes).slice(0, SEEDANCE_REFERENCE_LIMITS.audios - audioReferences.length);
if (selectedFiles.some((file) => file.type.startsWith("image/") && file.size > SEEDANCE_REFERENCE_LIMITS.imageMaxBytes)) message.warning(t("videoWorkbench.imageTooLarge"));
if (selectedFiles.some((file) => SEEDANCE_VIDEO_MIME_TYPES.includes(file.type) && file.size > SEEDANCE_REFERENCE_LIMITS.videoMaxBytes)) message.warning(t("videoWorkbench.videoTooLarge"));
if (selectedFiles.some((file) => isSupportedAudioFile(file) && file.size > SEEDANCE_REFERENCE_LIMITS.audioMaxBytes)) message.warning(t("videoWorkbench.audioTooLarge"));
const imageFiles = selectedFiles.filter((file) => file.type.startsWith("image/")).slice(0, 7 - references.length);
const nextReferences = await Promise.all(
imageFiles.map(async (file) => {
const image = await uploadImage(file);
return { id: nanoid(), name: file.name, type: image.mimeType, dataUrl: image.url, storageKey: image.storageKey };
}),
);
const nextVideoReferences = await Promise.all(
videoFiles.map(async (file) => {
const video = await uploadMediaFile(file, "video-reference");
return { id: nanoid(), name: file.name, type: video.mimeType, url: video.url, storageKey: video.storageKey, bytes: video.bytes, width: video.width, height: video.height, durationMs: video.durationMs };
}),
);
const nextAudioReferences = filterAudioReferencesByDuration(
audioReferences,
await Promise.all(
audioFiles.map(async (file) => {
const audio = await uploadMediaFile(file, "audio-reference");
return { id: nanoid(), name: file.name, type: audio.mimeType, url: audio.url, storageKey: audio.storageKey, durationMs: audio.durationMs };
}),
),
message.warning,
);
setReferences((value) => [...value, ...nextReferences].slice(0, SEEDANCE_REFERENCE_LIMITS.images));
setVideoReferences((value) => [...value, ...nextVideoReferences].slice(0, SEEDANCE_REFERENCE_LIMITS.videos));
setAudioReferences((value) => [...value, ...nextAudioReferences].slice(0, SEEDANCE_REFERENCE_LIMITS.audios));
setReferences((value) => [...value, ...nextReferences].slice(0, 7));
};
const handleReferenceDragEnter = (event: DragEvent<HTMLDivElement>, target: "image" | "video" | "audio") => {
const handleReferenceDragEnter = (event: DragEvent<HTMLDivElement>) => {
event.preventDefault();
dragDepthRef.current += 1;
if (event.dataTransfer.types.includes("Files")) setReferenceDragTarget(target);
if (event.dataTransfer.types.includes("Files")) setReferenceDragTarget(true);
};
const handleReferenceDragLeave = (event: DragEvent<HTMLDivElement>) => {
event.preventDefault();
dragDepthRef.current = Math.max(0, dragDepthRef.current - 1);
if (!dragDepthRef.current) setReferenceDragTarget(null);
if (!dragDepthRef.current) setReferenceDragTarget(false);
};
const handleReferenceDrop = (event: DragEvent<HTMLDivElement>) => {
event.preventDefault();
dragDepthRef.current = 0;
setReferenceDragTarget(null);
setReferenceDragTarget(false);
void addReferences(event.dataTransfer.files);
};
@@ -184,12 +155,12 @@ export default function VideoPage() {
return;
}
const nextReferences = await Promise.all(
blobs.slice(0, SEEDANCE_REFERENCE_LIMITS.images - references.length).map(async (blob, index) => {
blobs.slice(0, 7 - references.length).map(async (blob, index) => {
const image = await uploadImage(blob);
return { id: nanoid(), name: `clipboard-${index + 1}.png`, type: image.mimeType, dataUrl: image.url, storageKey: image.storageKey };
}),
);
setReferences((value) => [...value, ...nextReferences].slice(0, SEEDANCE_REFERENCE_LIMITS.images));
setReferences((value) => [...value, ...nextReferences].slice(0, 7));
message.success(t("videoWorkbench.clipboardAdded", { count: nextReferences.length }));
} catch {
message.error(t("videoWorkbench.clipboardEmpty"));
@@ -211,15 +182,15 @@ export default function VideoPage() {
const batchStartedAt = performance.now();
setStartedAt(batchStartedAt);
try {
const task = await createVideoGenerationTask(snapshot.config, snapshot.text, snapshot.references, snapshot.videoReferences, snapshot.audioReferences);
const log = buildLog({ prompt: snapshot.text, model, config: snapshot.config, references: snapshot.references, videoReferences: snapshot.videoReferences, audioReferences: snapshot.audioReferences, durationMs: 0, status: "pending", task });
const task = await createVideoGenerationTask(snapshot.config, snapshot.text, snapshot.references);
const log = buildLog({ prompt: snapshot.text, model, config: snapshot.config, references: snapshot.references, durationMs: 0, status: "pending", task });
await saveLog(log, false);
void pollGenerationLog(log, snapshot.config, agentTaskId);
} catch (error) {
const errorMessage = error instanceof Error ? error.message : t("workbench.generationFailed");
setResults([{ id: nanoid(), status: "failed", error: errorMessage }]);
if (agentTaskId) updateAgentTask(agentTaskId, { status: "failed", successCount: 0, failCount: 1, error: errorMessage });
await saveLog(buildLog({ prompt: snapshot.text, model, config: snapshot.config, references: snapshot.references, videoReferences: snapshot.videoReferences, audioReferences: snapshot.audioReferences, durationMs: performance.now() - batchStartedAt, status: "failed", error: errorMessage }));
await saveLog(buildLog({ prompt: snapshot.text, model, config: snapshot.config, references: snapshot.references, durationMs: performance.now() - batchStartedAt, status: "failed", error: errorMessage }));
message.error(errorMessage);
setRunning(false);
}
@@ -258,12 +229,7 @@ export default function VideoPage() {
openConfigDialog(true);
return null;
}
const videoReferenceError = seedanceVideoReferenceError(videoReferences);
if (videoReferenceError) {
message.error(t("videoWorkbench.referenceError", { error: videoReferenceError, hint: seedanceVideoReferenceHint() }));
return null;
}
return { text, config: buildVideoConfig(effectiveConfig, model), references: [...references], videoReferences: [...videoReferences], audioReferences: [...audioReferences] };
return { text, config: buildVideoConfig(effectiveConfig, model), references: [...references] };
};
const retryResult = () => {
@@ -292,9 +258,7 @@ export default function VideoPage() {
setPrompt(payload.content);
} else if (payload.kind === "image") {
const stored = await uploadImage(payload.dataUrl);
setReferences((value) => [...value, { id: nanoid(), name: payload.title, type: stored.mimeType, dataUrl: stored.url, storageKey: stored.storageKey }].slice(0, SEEDANCE_REFERENCE_LIMITS.images));
} else if (payload.kind === "video") {
setVideoReferences((value) => [...value, { id: nanoid(), name: payload.title, type: "video/mp4", url: payload.url, storageKey: payload.storageKey, width: payload.width, height: payload.height }].slice(0, SEEDANCE_REFERENCE_LIMITS.videos));
setReferences((value) => [...value, { id: nanoid(), name: payload.title, type: stored.mimeType, dataUrl: stored.url, storageKey: stored.storageKey }].slice(0, 7));
}
setAssetPickerOpen(false);
};
@@ -302,8 +266,6 @@ export default function VideoPage() {
const createSession = () => {
setPrompt("");
setReferences([]);
setVideoReferences([]);
setAudioReferences([]);
setResults([]);
setElapsedMs(0);
setStartedAt(0);
@@ -373,7 +335,7 @@ export default function VideoPage() {
}
if (state.status === "failed") throw new Error(state.error);
if (attempt === 119) throw new Error(t("videoWorkbench.timeout"));
await delay(log.task.provider === "seedance" ? 5000 : 2500);
await delay(2500);
}
} catch (error) {
const errorMessage = error instanceof Error ? error.message : t("workbench.generationFailed");
@@ -395,8 +357,6 @@ export default function VideoPage() {
setLogsOpen(false);
setPrompt(log.prompt);
setReferences(log.references || []);
setVideoReferences(log.videoReferences || []);
setAudioReferences(log.audioReferences || []);
if (log.config.videoModel || log.model) updateConfig("videoModel", log.config.videoModel || log.model);
if (log.config.size) updateConfig("size", log.config.size);
if (log.config.vquality) updateConfig("vquality", log.config.vquality);
@@ -456,8 +416,8 @@ export default function VideoPage() {
</div>
</div>
<div
className={`hover-scrollbar hover-scrollbar-hint flex min-h-24 w-full min-w-0 max-w-full gap-2 overflow-x-scroll overflow-y-hidden rounded-lg border border-dashed p-2 pb-3 overscroll-x-contain transition-colors ${referenceDragTarget === "image" ? "border-stone-900 bg-stone-100/80 dark:border-stone-100 dark:bg-stone-900/80" : "border-stone-300 dark:border-stone-700"}`}
onDragEnter={(event) => handleReferenceDragEnter(event, "image")}
className={`hover-scrollbar hover-scrollbar-hint flex min-h-24 w-full min-w-0 max-w-full gap-2 overflow-x-scroll overflow-y-hidden rounded-lg border border-dashed p-2 pb-3 overscroll-x-contain transition-colors ${referenceDragTarget ? "border-stone-900 bg-stone-100/80 dark:border-stone-100 dark:bg-stone-900/80" : "border-stone-300 dark:border-stone-700"}`}
onDragEnter={handleReferenceDragEnter}
onDragOver={(event) => {
event.preventDefault();
event.dataTransfer.dropEffect = "copy";
@@ -468,80 +428,14 @@ export default function VideoPage() {
{references.map((item, index) => (
<div key={item.id} className="group relative size-20 shrink-0 overflow-hidden rounded-md border border-stone-200 dark:border-stone-800">
<img src={item.dataUrl} alt={item.name} className="size-full object-cover" />
<span className="absolute left-1 top-1 rounded bg-black/60 px-1.5 py-0.5 text-[10px] font-medium text-white">{seedanceReferenceLabel("image", index)}</span>
<span className="absolute left-1 top-1 rounded bg-black/60 px-1.5 py-0.5 text-[10px] font-medium text-white">{index + 1}</span>
<ReferenceOrderButtons index={index} total={references.length} onMove={(offset) => setReferences((value) => moveListItem(value, index, offset))} />
<button type="button" className="absolute right-1 top-1 hidden size-6 items-center justify-center rounded bg-black/60 text-white group-hover:flex" onClick={() => setReferences((value) => value.filter((ref) => ref.id !== item.id))} aria-label={t("videoWorkbench.removeImage")}>
<Trash2 className="size-3.5" />
</button>
</div>
))}
{!references.length ? <div className="flex min-w-full items-center justify-center text-sm text-stone-500">{referenceDragTarget === "image" ? t("videoWorkbench.dropReferences") : t("videoWorkbench.noImages")}</div> : null}
</div>
</div>
<div className="min-w-0">
<div className="mb-2 flex items-center justify-between gap-3">
<span className="text-base font-semibold">{t("videoWorkbench.videoReferences")}</span>
<Button size="small" icon={<Upload className="size-3.5" />} onClick={() => fileInputRef.current?.click()}>
{t("workbench.upload")}
</Button>
</div>
<div
className={`hover-scrollbar hover-scrollbar-hint flex min-h-24 w-full min-w-0 max-w-full gap-2 overflow-x-scroll overflow-y-hidden rounded-lg border border-dashed p-2 pb-3 overscroll-x-contain transition-colors ${referenceDragTarget === "video" ? "border-stone-900 bg-stone-100/80 dark:border-stone-100 dark:bg-stone-900/80" : "border-stone-300 dark:border-stone-700"}`}
onDragEnter={(event) => handleReferenceDragEnter(event, "video")}
onDragOver={(event) => {
event.preventDefault();
event.dataTransfer.dropEffect = "copy";
}}
onDragLeave={handleReferenceDragLeave}
onDrop={handleReferenceDrop}
>
{videoReferences.map((item, index) => (
<div key={item.id} className="group relative h-20 w-32 shrink-0 overflow-hidden rounded-md border border-stone-200 bg-black dark:border-stone-800">
<video src={item.url} className="size-full object-cover" muted preload="metadata" />
<span className="absolute left-1 top-1 rounded bg-black/60 px-1.5 py-0.5 text-[10px] font-medium text-white">{seedanceReferenceLabel("video", index)}</span>
<ReferenceOrderButtons index={index} total={videoReferences.length} onMove={(offset) => setVideoReferences((value) => moveListItem(value, index, offset))} />
<button type="button" className="absolute right-1 top-1 hidden size-6 items-center justify-center rounded bg-black/60 text-white group-hover:flex" onClick={() => setVideoReferences((value) => value.filter((ref) => ref.id !== item.id))} aria-label={t("videoWorkbench.removeVideo")}>
<Trash2 className="size-3.5" />
</button>
</div>
))}
{!videoReferences.length ? <div className="flex min-w-full items-center justify-center text-sm text-stone-500">{referenceDragTarget === "video" ? t("videoWorkbench.dropReferences") : t("videoWorkbench.noVideos")}</div> : null}
</div>
</div>
<div className="min-w-0">
<div className="mb-2 flex items-center justify-between gap-3">
<span className="text-base font-semibold">{t("videoWorkbench.audioReferences")}</span>
<Button size="small" icon={<Upload className="size-3.5" />} onClick={() => fileInputRef.current?.click()}>
{t("workbench.upload")}
</Button>
</div>
<div
className={`hover-scrollbar hover-scrollbar-hint flex min-h-24 w-full min-w-0 max-w-full gap-2 overflow-x-scroll overflow-y-hidden rounded-lg border border-dashed p-2 pb-3 overscroll-x-contain transition-colors ${referenceDragTarget === "audio" ? "border-stone-900 bg-stone-100/80 dark:border-stone-100 dark:bg-stone-900/80" : "border-stone-300 dark:border-stone-700"}`}
onDragEnter={(event) => handleReferenceDragEnter(event, "audio")}
onDragOver={(event) => {
event.preventDefault();
event.dataTransfer.dropEffect = "copy";
}}
onDragLeave={handleReferenceDragLeave}
onDrop={handleReferenceDrop}
>
{audioReferences.map((item, index) => (
<div key={item.id} className="group relative flex h-20 w-48 shrink-0 flex-col justify-center gap-2 rounded-md border border-stone-200 bg-stone-50 px-2 dark:border-stone-800 dark:bg-stone-900">
<div className="flex min-w-0 items-center gap-2 text-xs text-stone-500 dark:text-stone-400">
<Music2 className="size-4 shrink-0" />
<span className="shrink-0 rounded bg-stone-200 px-1 text-[10px] text-stone-700 dark:bg-stone-800 dark:text-stone-200">{seedanceReferenceLabel("audio", index)}</span>
<span className="truncate">{item.name}</span>
</div>
<audio src={item.url} controls className="h-8 w-full" preload="metadata" />
<ReferenceOrderButtons index={index} total={audioReferences.length} onMove={(offset) => setAudioReferences((value) => moveListItem(value, index, offset))} />
<button type="button" className="absolute right-1 top-1 hidden size-6 items-center justify-center rounded bg-black/60 text-white group-hover:flex" onClick={() => setAudioReferences((value) => value.filter((ref) => ref.id !== item.id))} aria-label={t("videoWorkbench.removeAudio")}>
<Trash2 className="size-3.5" />
</button>
</div>
))}
{!audioReferences.length ? <div className="flex min-w-full items-center justify-center text-center text-sm text-stone-500">{referenceDragTarget === "audio" ? t("videoWorkbench.dropReferences") : t("videoWorkbench.noAudio")}</div> : null}
{!references.length ? <div className="flex min-w-full items-center justify-center text-sm text-stone-500">{referenceDragTarget ? t("videoWorkbench.dropReferences") : t("videoWorkbench.noImages")}</div> : null}
</div>
</div>
@@ -587,7 +481,7 @@ export default function VideoPage() {
<input
ref={fileInputRef}
type="file"
accept="image/*,video/mp4,video/quicktime,audio/mpeg,audio/wav,audio/x-wav,.mp3,.wav"
accept="image/*"
multiple
className="hidden"
onChange={(event) => {
@@ -776,18 +670,6 @@ async function readStoredLogs() {
async function normalizeLog(log: Partial<GenerationLog>): Promise<GenerationLog> {
const video = log.video?.storageKey ? { ...log.video, url: await resolveMediaUrl(log.video.storageKey, log.video.url) } : log.video;
const videoReferences = await Promise.all(
(log.videoReferences || []).map(async (item) => ({
...item,
url: item.storageKey ? await resolveMediaUrl(item.storageKey, item.url) : item.url,
})),
);
const audioReferences = await Promise.all(
(log.audioReferences || []).map(async (item) => ({
...item,
url: item.storageKey ? await resolveMediaUrl(item.storageKey, item.url) : item.url,
})),
);
const references = await Promise.all(
(log.references || []).map(async (item) => ({
...item,
@@ -804,8 +686,6 @@ async function normalizeLog(log: Partial<GenerationLog>): Promise<GenerationLog>
model: log.model || config.videoModel || "",
config,
references,
videoReferences,
audioReferences,
durationMs: log.durationMs || 0,
size: log.size || config.size || "",
resolution: normalizeResolution(log.resolution || config.vquality || ""),
@@ -821,36 +701,10 @@ function serializeLog(log: GenerationLog): GenerationLog {
return {
...log,
references: log.references.map((item) => ({ ...item, dataUrl: item.storageKey ? "" : item.dataUrl })),
videoReferences: log.videoReferences.map((item) => (item.storageKey ? { ...item, url: "" } : item)),
audioReferences: log.audioReferences.map((item) => (item.storageKey ? { ...item, url: "" } : item)),
video: log.video?.storageKey ? { ...log.video, url: "" } : log.video,
};
}
function isSupportedAudioFile(file: File) {
return file.type === "audio/mpeg" || file.type === "audio/mp3" || file.type === "audio/wav" || file.type === "audio/x-wav" || /\.(mp3|wav)$/i.test(file.name);
}
function filterAudioReferencesByDuration(existing: ReferenceAudio[], next: ReferenceAudio[], warn: (content: string) => void) {
let total = existing.reduce((sum, item) => sum + (item.durationMs || 0), 0);
const accepted: ReferenceAudio[] = [];
let skipped = false;
for (const item of next) {
if (item.durationMs && (item.durationMs < 2000 || item.durationMs > 15000)) {
skipped = true;
continue;
}
if (item.durationMs && total + item.durationMs > 15000) {
skipped = true;
continue;
}
total += item.durationMs || 0;
accepted.push(item);
}
if (skipped) warn(i18n.t("videoWorkbench.audioDurationInvalid"));
return accepted;
}
function moveListItem<T>(items: T[], index: number, offset: number) {
const targetIndex = index + offset;
if (targetIndex < 0 || targetIndex >= items.length) return items;
@@ -881,7 +735,7 @@ function normalizeLogConfig(log: Partial<GenerationLog>): GenerationLogConfig {
};
}
function buildLog({ prompt, model, config, references, videoReferences, audioReferences, durationMs, status, task, video, error }: { prompt: string; model: string; config: AiConfig; references: ReferenceImage[]; videoReferences: ReferenceVideo[]; audioReferences: ReferenceAudio[]; durationMs: number; status: GenerationLog["status"]; task?: VideoGenerationTask; video?: GeneratedVideo; error?: string }): GenerationLog {
function buildLog({ prompt, model, config, references, durationMs, status, task, video, error }: { prompt: string; model: string; config: AiConfig; references: ReferenceImage[]; durationMs: number; status: GenerationLog["status"]; task?: VideoGenerationTask; video?: GeneratedVideo; error?: string }): GenerationLog {
const logConfig = {
model: config.model,
videoModel: config.videoModel,
@@ -900,8 +754,6 @@ function buildLog({ prompt, model, config, references, videoReferences, audioRef
model,
config: logConfig,
references,
videoReferences,
audioReferences,
durationMs,
size: logConfig.size,
resolution: logConfig.vquality,
@@ -914,12 +766,11 @@ function buildLog({ prompt, model, config, references, videoReferences, audioRef
}
function buildVideoConfig(config: AiConfig, model: string): AiConfig {
const seedance = isSeedanceVideoConfig({ ...config, model });
return {
...config,
model,
videoModel: model,
size: seedance ? normalizeSeedanceRatio(config.size) : normalizeVideoSize(config.size),
size: normalizeVideoSize(config.size),
videoSeconds: normalizeVideoSeconds(config.videoSeconds),
vquality: normalizeResolution(config.vquality),
videoGenerateAudio: String(boolConfig(config.videoGenerateAudio, true)),
-31
View File
@@ -805,37 +805,6 @@ export async function requestEdit(config: AiConfig, prompt: string, references:
}
}
if (requestConfig.apiFormat === "ark") {
if (mask) throw new Error(apiText("maskModelUnsupported"));
const quality = normalizeQuality(config.quality);
const requestSize = resolveRequestSize(quality, config.size);
const background = normalizeBackground(config.background);
const refs = await Promise.all(references.map((image) => imageToDataUrl(image)));
try {
const response = await axios.post<ImageApiResponse>(
aiApiUrl(requestConfig, "/images/generations"),
{
model: requestConfig.model,
prompt: withSystemPrompt(requestConfig, requestPrompt),
n,
response_format: "b64_json",
output_format: IMAGE_OUTPUT_FORMAT,
image: refs,
...(quality ? { quality } : {}),
...(requestSize ? { size: requestSize } : {}),
...(background ? { background } : {}),
},
{
headers: aiHeaders(requestConfig, "application/json"),
signal: options?.signal,
},
);
return parseImagePayload(response.data);
} catch (error) {
throw new Error(readAxiosError(error, apiText("requestFailed")));
}
}
const quality = normalizeQuality(config.quality);
const requestSize = resolveRequestSize(quality, config.size);
const background = normalizeBackground(config.background);
+10 -149
View File
@@ -3,31 +3,20 @@ import { nanoid } from "nanoid";
import i18n from "@/i18n";
import { dataUrlToFile } from "@/lib/image-utils";
import { getMediaBlob, uploadMediaFile, type UploadedFile } from "@/services/file-storage";
import { uploadMediaFile, type UploadedFile } from "@/services/file-storage";
import { imageToDataUrl } from "@/services/image-storage";
import { boolConfig, buildSeedancePromptText, isSeedanceVideoConfig, normalizeSeedanceDuration, normalizeSeedanceRatio, normalizeSeedanceResolution, seedanceVideoReferenceError, SEEDANCE_REFERENCE_LIMITS } from "@/lib/seedance-video";
import { buildApiUrl, modelOptionName, resolveModelRequestConfig, resolveModelScript, type AiConfig } from "@/stores/use-config-store";
import { boolConfig, buildApiUrl, modelOptionName, resolveModelRequestConfig, resolveModelScript, type AiConfig } from "@/stores/use-config-store";
import { runModelPlugin } from "./model-plugin";
import type { ReferenceImage } from "@/types/image";
import type { ReferenceAudio, ReferenceVideo } from "@/types/media";
type VideoResponse = { id: string; status?: string; error?: { message?: string }; url?: string; result_url?: string; video_url?: string; content?: { video_url?: string; url?: string } | null };
type ApiVideoResponse = VideoResponse | { code?: number | string; data?: VideoResponse | null; msg?: string; message?: string; error?: { message?: string } };
type SeedanceTask = {
id: string;
status?: "queued" | "running" | "succeeded" | "completed" | "failed" | "cancelled" | "expired";
error?: { code?: string; message?: string } | null;
content?: { video_url?: string; url?: string; last_frame_url?: string } | null;
url?: string;
result_url?: string;
video_url?: string;
};
type ApiEnvelope<T> = T | { code?: number | string; data?: T | null; msg?: string; message?: string; error?: { message?: string } };
type RequestOptions = { signal?: AbortSignal };
const apiText = (key: string, options?: Record<string, unknown>) => i18n.t(`apiErrors.${key}`, options);
export type VideoGenerationResult = { blob?: Blob; url?: string; mimeType?: string };
export type VideoGenerationTask = { id: string; provider: "openai" | "seedance" | "plugin"; model: string };
export type VideoGenerationTask = { id: string; provider: "openai" | "plugin"; model: string };
export type VideoGenerationTaskState = { status: "pending" } | { status: "completed"; result: VideoGenerationResult } | { status: "failed"; error: string };
/** Results for scripted (plugin) video models, which run their own create+poll in one shot at task creation. */
@@ -44,32 +33,25 @@ function aiHeaders(config: AiConfig, contentType?: string) {
};
}
export async function requestVideoGeneration(config: AiConfig, prompt: string, references: ReferenceImage[] = [], videoReferences: ReferenceVideo[] = [], audioReferences: ReferenceAudio[] = [], options?: RequestOptions): Promise<VideoGenerationResult> {
const task = await createVideoGenerationTask(config, prompt, references, videoReferences, audioReferences, options);
const delayMs = task.provider === "seedance" ? 5000 : 2500;
export async function requestVideoGeneration(config: AiConfig, prompt: string, references: ReferenceImage[] = [], options?: RequestOptions): Promise<VideoGenerationResult> {
const task = await createVideoGenerationTask(config, prompt, references, options);
for (let attempt = 0; attempt < 120; attempt += 1) {
if (options?.signal?.aborted) throw new DOMException("Aborted", "AbortError");
const state = await pollVideoGenerationTask(config, task, options);
if (state.status === "completed") return state.result;
if (state.status === "failed") throw new Error(state.error);
if (attempt === 119) throw new Error(apiText("videoTimeout", { provider: task.provider === "seedance" ? "Seedance " : "" }));
await delay(delayMs, options?.signal);
if (attempt === 119) throw new Error(apiText("videoTimeout", { provider: "" }));
await delay(2500, options?.signal);
}
throw new Error(apiText("videoTimeout", { provider: "" }));
}
export async function createVideoGenerationTask(config: AiConfig, prompt: string, references: ReferenceImage[] = [], videoReferences: ReferenceVideo[] = [], audioReferences: ReferenceAudio[] = [], options?: RequestOptions): Promise<VideoGenerationTask> {
export async function createVideoGenerationTask(config: AiConfig, prompt: string, references: ReferenceImage[] = [], options?: RequestOptions): Promise<VideoGenerationTask> {
const selectedModel = (config.model || config.videoModel).trim();
const requestConfig = resolveModelRequestConfig(config, selectedModel);
const script = resolveModelScript(config, selectedModel);
if (script) return createPluginVideoTask(requestConfig, selectedModel, script, prompt, references, options);
assertVideoConfig(requestConfig, requestConfig.model);
if (isSeedanceVideoConfig(requestConfig)) {
return createSeedanceTask(requestConfig, selectedModel, prompt, references, videoReferences, audioReferences, options);
}
if (videoReferences.length || audioReferences.length) {
throw new Error(apiText("videoReferencesUnsupported"));
}
return createOpenAIVideoTask(requestConfig, selectedModel, prompt, references, options);
}
@@ -80,7 +62,7 @@ export async function pollVideoGenerationTask(config: AiConfig, task: VideoGener
}
const requestConfig = resolveModelRequestConfig(config, task.model);
assertVideoConfig(requestConfig, requestConfig.model);
return task.provider === "seedance" ? pollSeedanceTask(requestConfig, task, options) : pollOpenAIVideoTask(requestConfig, task, options);
return pollOpenAIVideoTask(requestConfig, task, options);
}
async function createPluginVideoTask(config: AiConfig, model: string, script: string, prompt: string, references: ReferenceImage[], options?: RequestOptions): Promise<VideoGenerationTask> {
@@ -170,114 +152,6 @@ async function pollOpenAIVideoTask(config: AiConfig, task: VideoGenerationTask,
}
}
async function createSeedanceTask(config: AiConfig, model: string, prompt: string, references: ReferenceImage[], videoReferences: ReferenceVideo[], audioReferences: ReferenceAudio[], options?: RequestOptions): Promise<VideoGenerationTask> {
if (audioReferences.length && !references.length && !videoReferences.length) {
throw new Error(apiText("seedanceAudioRequiresVisual"));
}
assertSeedanceVideoReferences(videoReferences);
assertSeedanceAudioReferences(audioReferences);
const content = await buildSeedanceContent(config, prompt, references, videoReferences, audioReferences);
if (!content.length) throw new Error(apiText("videoPromptRequired"));
const payload = {
model: modelOptionName(model),
content,
ratio: normalizeSeedanceRatio(config.size),
resolution: normalizeSeedanceResolution(config.vquality),
duration: normalizeSeedanceDuration(config.videoSeconds),
generate_audio: boolConfig(config.videoGenerateAudio, true),
watermark: boolConfig(config.videoWatermark, false),
};
try {
const created = unwrapSeedanceTask((await axios.post<ApiEnvelope<SeedanceTask>>(seedanceApiUrl(config), payload, { headers: aiHeaders(config, "application/json"), signal: options?.signal })).data);
if (!created.id) throw new Error(apiText("seedanceNoTaskId"));
return { id: created.id, provider: "seedance", model };
} catch (error) {
throw new Error(readAxiosError(error, apiText("seedanceTaskCreateFailed")));
}
}
async function pollSeedanceTask(config: AiConfig, task: VideoGenerationTask, options?: RequestOptions): Promise<VideoGenerationTaskState> {
try {
const state = unwrapSeedanceTask((await axios.get<ApiEnvelope<SeedanceTask>>(seedanceApiUrl(config, task.id), { headers: aiHeaders(config), signal: options?.signal })).data);
const url = videoResultUrl(state);
if (url) return { status: "completed", result: await videoResultFromUrl(url, options) };
if (state.status === "succeeded" || state.status === "completed") return { status: "failed", error: apiText("seedanceNoVideoUrl") };
if (state.status === "failed" || state.status === "cancelled" || state.status === "expired") return { status: "failed", error: readApiErrorMessage(state.error?.message) || apiText(state.status === "expired" ? "seedanceVideoTimeout" : "seedanceVideoFailed") };
return { status: "pending" };
} catch (error) {
throw new Error(readAxiosError(error, apiText("seedanceTaskQueryFailed")));
}
}
function assertSeedanceVideoReferences(videoReferences: ReferenceVideo[]) {
const error = seedanceVideoReferenceError(videoReferences);
if (error) throw new Error(error);
let total = 0;
for (const video of videoReferences) {
if (!video.durationMs) continue;
if (video.durationMs < 2000 || video.durationMs > 15000) throw new Error(apiText("seedanceVideoDuration"));
total += video.durationMs;
}
if (total > 15000) throw new Error(apiText("seedanceVideoTotalDuration"));
}
function assertSeedanceAudioReferences(audioReferences: ReferenceAudio[]) {
let total = 0;
for (const audio of audioReferences) {
if (!audio.durationMs) continue;
if (audio.durationMs < 2000 || audio.durationMs > 15000) throw new Error(apiText("seedanceAudioDuration"));
total += audio.durationMs;
}
if (total > 15000) throw new Error(apiText("seedanceAudioTotalDuration"));
}
function seedanceApiUrl(config: AiConfig, taskId?: string) {
return buildApiUrl(config.baseUrl, `/contents/generations/tasks${taskId ? `/${encodeURIComponent(taskId)}` : ""}`);
}
async function buildSeedanceContent(config: AiConfig, prompt: string, references: ReferenceImage[], videoReferences: ReferenceVideo[], audioReferences: ReferenceAudio[]) {
const content: Array<Record<string, unknown>> = [];
const text = buildSeedancePromptText(prompt, references, videoReferences, audioReferences);
if (text) content.push({ type: "text", text });
for (const image of references.slice(0, SEEDANCE_REFERENCE_LIMITS.images)) {
content.push({ type: "image_url", image_url: { url: await resolveSeedanceImageUrl(config, image) }, role: "reference_image" });
}
for (const video of videoReferences.slice(0, SEEDANCE_REFERENCE_LIMITS.videos)) {
content.push({ type: "video_url", video_url: { url: await resolveSeedanceVideoUrl(video) }, role: "reference_video" });
}
for (const audio of audioReferences.slice(0, SEEDANCE_REFERENCE_LIMITS.audios)) {
content.push({ type: "audio_url", audio_url: { url: await resolveSeedanceAudioUrl(audio) }, role: "reference_audio" });
}
return content;
}
async function resolveSeedanceImageUrl(config: AiConfig, image: ReferenceImage) {
const directUrl = image.url || image.dataUrl;
if (isPublicMediaUrl(directUrl) || directUrl.startsWith("asset://")) return directUrl;
const dataUrl = await imageToDataUrl(image);
if (!dataUrl) throw new Error(apiText("referenceImageReadFailed"));
return dataUrl;
}
async function resolveSeedanceVideoUrl(video: ReferenceVideo) {
if (isPublicMediaUrl(video.url) || video.url.startsWith("asset://")) return video.url;
let blob: Blob | null = null;
if (video.storageKey) blob = await getMediaBlob(video.storageKey);
if (!blob && video.url?.startsWith("blob:")) blob = await (await fetch(video.url)).blob();
if (!blob) throw new Error(apiText("invalidReferenceVideo"));
return blobToDataUrl(blob);
}
async function resolveSeedanceAudioUrl(audio: ReferenceAudio) {
if (isPublicMediaUrl(audio.url) || audio.url.startsWith("asset://")) return audio.url;
let blob: Blob | null = null;
if (audio.storageKey) blob = await getMediaBlob(audio.storageKey);
if (!blob && audio.url?.startsWith("blob:")) blob = await (await fetch(audio.url)).blob();
if (!blob) throw new Error(apiText("invalidReferenceAudio"));
return blobToDataUrl(blob);
}
async function videoResultFromUrl(url: string, options?: RequestOptions): Promise<VideoGenerationResult> {
try {
const response = await axios.get<Blob>(url, { responseType: "blob", signal: options?.signal });
@@ -319,10 +193,6 @@ function unwrapVideoResponse(payload: ApiVideoResponse) {
return unwrapEnvelope(payload, apiText("noVideoTask"));
}
function unwrapSeedanceTask(payload: ApiEnvelope<SeedanceTask>) {
return unwrapEnvelope(payload, apiText("seedanceNoTask"));
}
function unwrapEnvelope<T>(payload: ApiEnvelope<T>, emptyMessage: string): T {
if (!payload) throw new Error(emptyMessage);
if (typeof payload === "object" && "code" in payload && payload.code !== undefined) {
@@ -333,7 +203,7 @@ function unwrapEnvelope<T>(payload: ApiEnvelope<T>, emptyMessage: string): T {
return payload as T;
}
function videoResultUrl(payload: VideoResponse | SeedanceTask) {
function videoResultUrl(payload: VideoResponse) {
return [payload.video_url, payload.result_url, payload.url, payload.content?.video_url, payload.content?.url].find((url) => typeof url === "string" && (isPublicMediaUrl(url) || /\.mp4(\?|#|$)/i.test(url)));
}
@@ -415,12 +285,3 @@ function delay(ms: number, signal?: AbortSignal) {
);
});
}
function blobToDataUrl(blob: Blob) {
return new Promise<string>((resolve, reject) => {
const reader = new FileReader();
reader.onload = () => resolve(String(reader.result || ""));
reader.onerror = () => reject(new Error(apiText("localAssetReadFailed")));
reader.readAsDataURL(blob);
});
}
+9 -26
View File
@@ -5,7 +5,7 @@ import { nanoid } from "nanoid";
import i18n from "@/i18n";
export type ApiCallFormat = "openai" | "gemini" | "ark";
export type ApiCallFormat = "openai" | "gemini";
export type ModelCapability = "image" | "video" | "text" | "audio";
export type ReasoningEffort = "auto" | "low" | "medium" | "high" | "xhigh";
@@ -66,7 +66,6 @@ export const CONFIG_STORE_KEY = "infinite-canvas:ai_config_store";
const CHANNEL_MODEL_SEPARATOR = "::";
const OPENAI_BASE_URL = "https://api.openai.com";
const GEMINI_BASE_URL = "https://generativelanguage.googleapis.com";
const ARK_BASE_URL = "https://ark.cn-beijing.volces.com/api/v3";
export const defaultConfig: AiConfig = {
channelMode: "local",
@@ -133,7 +132,11 @@ type ConfigStore = {
clearPromptContinue: () => void;
};
const VIDEO_KEYWORDS = ["seedance", "video", "sora", "veo", "kling", "wan", "hailuo"];
const VIDEO_KEYWORDS = ["video", "sora", "veo", "kling", "wan", "hailuo"];
export function boolConfig(value: string, fallback: boolean) {
return value ? value === "true" : fallback;
}
const AUDIO_KEYWORDS = ["audio", "tts", "speech", "voice", "music", "sound"];
const IMAGE_KEYWORDS = ["seedream", "gpt-image", "image", "dall-e", "dalle", "imagen", "flux", "sdxl", "stable-diffusion", "midjourney"];
@@ -372,12 +375,11 @@ function normalizeChannels(config: AiConfig) {
export function defaultBaseUrlForApiFormat(apiFormat: ApiCallFormat) {
if (apiFormat === "gemini") return GEMINI_BASE_URL;
if (apiFormat === "ark") return ARK_BASE_URL;
return OPENAI_BASE_URL;
}
function normalizeApiFormat(apiFormat: unknown): ApiCallFormat {
return apiFormat === "gemini" || apiFormat === "ark" ? apiFormat : "openai";
return apiFormat === "gemini" ? apiFormat : "openai";
}
function uniqueModelOptions(models: string[]) {
@@ -385,27 +387,8 @@ function uniqueModelOptions(models: string[]) {
}
export function buildApiUrl(baseUrl: string, path: string) {
let normalizedBaseUrl = baseUrl.trim().replace(/\/+$/, "");
normalizedBaseUrl = normalizeArkPlanBaseUrl(normalizedBaseUrl);
const normalizedBaseUrl = baseUrl.trim().replace(/\/+$/, "");
const lowerBaseUrl = normalizedBaseUrl.toLowerCase();
const apiBaseUrl = lowerBaseUrl.endsWith("/v1") || lowerBaseUrl.endsWith("/api/v3") || lowerBaseUrl.endsWith("/api/plan/v3") ? normalizedBaseUrl : `${normalizedBaseUrl}/v1`;
const apiBaseUrl = lowerBaseUrl.endsWith("/v1") ? normalizedBaseUrl : `${normalizedBaseUrl}/v1`;
return `${apiBaseUrl}${path}`;
}
function normalizeArkPlanBaseUrl(baseUrl: string) {
try {
const url = new URL(baseUrl);
const path = url.pathname.replace(/\/+$/, "");
const lowerPath = path.toLowerCase();
const arkPlanIndex = lowerPath.indexOf("/api/plan/v3");
if (arkPlanIndex < 0) return baseUrl;
const end = arkPlanIndex + "/api/plan/v3".length;
if (lowerPath.length !== end && lowerPath[end] !== "/") return baseUrl;
url.pathname = path.slice(0, end);
url.search = "";
url.hash = "";
return url.toString().replace(/\/+$/, "");
} catch {
return baseUrl;
}
}