Add MiniMax and PinkCherry LTX as pickable video engines, with still or text start.
Shot 1 can run text-to-video; later shots still extend from the last frame. Recommend output is also normalized into shot-script cards. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
+30
-22
@@ -1,5 +1,6 @@
|
||||
import { frameLength, runGeneration } from '~/server/utils/videoChain'
|
||||
import { createShotQueue, finishQueueBurst, setQueueJob } from '~/server/utils/shotQueue'
|
||||
import { defaultVideoSteps, isLtxWorkflow, isTextToVideo, parseVideoWorkflow } from '~/utils/videoModels'
|
||||
|
||||
function isComfyBusyTimeout(error: unknown) {
|
||||
const err = error as { statusCode?: number; status?: number; data?: { code?: string }; message?: string; statusMessage?: string }
|
||||
@@ -42,16 +43,16 @@ function parseFps(raw: string | undefined) {
|
||||
return fps === 12 || fps === 30 || fps === 24 ? fps : 24
|
||||
}
|
||||
|
||||
function parseCfg(raw: string | undefined, turbo: boolean) {
|
||||
const fallback = turbo ? 1.5 : 4
|
||||
function parseCfg(raw: string | undefined, turbo: boolean, ltx = false) {
|
||||
const fallback = ltx ? 3.5 : (turbo ? 1.5 : 4)
|
||||
const value = Number(raw)
|
||||
if (!Number.isFinite(value)) return fallback
|
||||
const snapped = Math.round(value * 2) / 2
|
||||
return Math.min(10, Math.max(1, snapped))
|
||||
}
|
||||
|
||||
function parseSampler(raw: string | undefined) {
|
||||
return SAMPLERS.has(raw || '') ? raw! : 'res_multistep'
|
||||
function parseSampler(raw: string | undefined, ltx = false) {
|
||||
return SAMPLERS.has(raw || '') ? raw! : (ltx ? 'euler' : 'res_multistep')
|
||||
}
|
||||
|
||||
function parseScheduler(raw: string | undefined) {
|
||||
@@ -86,7 +87,9 @@ export default defineEventHandler(async (event) => {
|
||||
}
|
||||
}
|
||||
|
||||
if (!image) {
|
||||
const workflow = parseVideoWorkflow(fields.workflow)
|
||||
const textToVideo = isTextToVideo(workflow)
|
||||
if (!image && !textToVideo) {
|
||||
throw createError({ statusCode: 400, statusMessage: 'An initial image is required' })
|
||||
}
|
||||
const prompt = (fields.prompt || '').trim()
|
||||
@@ -94,10 +97,13 @@ export default defineEventHandler(async (event) => {
|
||||
throw createError({ statusCode: 400, statusMessage: 'A motion prompt is required' })
|
||||
}
|
||||
const extensions = parseExtensions(fields.extensions)
|
||||
const workflow = parseVideoWorkflow(fields.workflow)
|
||||
const useIdentityRefs = workflow === 'v2' && fields.useIdentityRefs === 'true'
|
||||
|
||||
const { width, height } = resolveOutputSize(fields.aspect, image.data, Number(fields.width), Number(fields.height))
|
||||
const useIdentityRefs = workflow === 'v2' && !textToVideo && !isLtxWorkflow(workflow) && fields.useIdentityRefs === 'true'
|
||||
const { width, height } = resolveOutputSize(
|
||||
fields.aspect,
|
||||
image?.data || Buffer.alloc(0),
|
||||
Number(fields.width),
|
||||
Number(fields.height)
|
||||
)
|
||||
const turbo = fields.turbo !== 'false'
|
||||
const steps = Number(fields.steps || defaultVideoSteps(turbo, workflow))
|
||||
const seed = fields.seed && fields.seed !== 'random'
|
||||
@@ -105,13 +111,13 @@ export default defineEventHandler(async (event) => {
|
||||
: Math.floor(Math.random() * 2_147_483_647)
|
||||
const durationSeconds = parseDuration(fields.duration || '5')
|
||||
const fps = parseFps(fields.fps)
|
||||
const cfg = parseCfg(fields.cfg, turbo)
|
||||
const samplerName = parseSampler(fields.sampler_name)
|
||||
const cfg = parseCfg(fields.cfg, turbo, isLtxWorkflow(workflow))
|
||||
const samplerName = parseSampler(fields.sampler_name, isLtxWorkflow(workflow))
|
||||
const scheduler = parseScheduler(fields.scheduler)
|
||||
const length = frameLength(durationSeconds, fps)
|
||||
const hideThumbnail = fields.hideThumbnail === 'true'
|
||||
const hideInput = fields.hideInput === 'true'
|
||||
const sound = fields.sound !== 'false'
|
||||
const sound = !isLtxWorkflow(workflow) && fields.sound !== 'false'
|
||||
const clipName = (fields.name || '').trim().slice(0, 80)
|
||||
const ownerKey = libraryOwnerKey(event)
|
||||
const library = publicLibrary(event)
|
||||
@@ -124,15 +130,17 @@ export default defineEventHandler(async (event) => {
|
||||
const destFolder = library.folders.find(folder => folder.id === folderId)
|
||||
const folderLocked = Boolean(destFolder?.protected && !destFolder.unlocked)
|
||||
assertFolderExists(event, folderId)
|
||||
const still = await rememberInputStill({
|
||||
ownerKey,
|
||||
folderId,
|
||||
filename: image.filename,
|
||||
data: image.data,
|
||||
width,
|
||||
height,
|
||||
hideInput
|
||||
})
|
||||
const still = image
|
||||
? await rememberInputStill({
|
||||
ownerKey,
|
||||
folderId,
|
||||
filename: image.filename,
|
||||
data: image.data,
|
||||
width,
|
||||
height,
|
||||
hideInput
|
||||
})
|
||||
: null
|
||||
const referenceStillIds: Array<string | null> = [null, null, null, null]
|
||||
if (useIdentityRefs) {
|
||||
for (const [index, ref] of referenceImages.entries()) {
|
||||
@@ -175,7 +183,7 @@ export default defineEventHandler(async (event) => {
|
||||
fps,
|
||||
samplerName,
|
||||
scheduler,
|
||||
thumb: isPipelineFrameFilename(image.filename) ? undefined : image.data,
|
||||
thumb: image && !isPipelineFrameFilename(image.filename) ? image.data : undefined,
|
||||
stillId: still?.id,
|
||||
stillFilename: still?.filename,
|
||||
referenceStillIds,
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
{
|
||||
"1": {
|
||||
"inputs": {
|
||||
"unet_name": "PinkCherry_FineTune_int8_v1_8_LTX23.safetensors",
|
||||
"weight_dtype": "default"
|
||||
},
|
||||
"class_type": "UNETLoader",
|
||||
"_meta": { "title": "Load PinkCherry LTX-2.3" }
|
||||
},
|
||||
"21": {
|
||||
"inputs": {
|
||||
"lora_name": "ltx-2.3-22b-distilled-lora-384-1.1.safetensors",
|
||||
"strength_model": 1,
|
||||
"model": ["1", 0]
|
||||
},
|
||||
"class_type": "LoraLoaderModelOnly",
|
||||
"_meta": { "title": "LTX distilled LoRA" }
|
||||
},
|
||||
"3": {
|
||||
"inputs": {
|
||||
"vae_name": "LTX23_video_vae_bf16.safetensors"
|
||||
},
|
||||
"class_type": "VAELoader",
|
||||
"_meta": { "title": "Load VAE" }
|
||||
},
|
||||
"4": {
|
||||
"inputs": {
|
||||
"image": "input_image.png",
|
||||
"upload": "image"
|
||||
},
|
||||
"class_type": "LoadImage",
|
||||
"_meta": { "title": "Load Image" }
|
||||
},
|
||||
"5": {
|
||||
"inputs": {
|
||||
"width": 1280,
|
||||
"height": 1280,
|
||||
"interpolation": "lanczos",
|
||||
"method": "keep proportion",
|
||||
"condition": "always",
|
||||
"multiple_of": 32,
|
||||
"image": ["4", 0]
|
||||
},
|
||||
"class_type": "ImageResize+",
|
||||
"_meta": { "title": "Resize still" }
|
||||
},
|
||||
"6": {
|
||||
"inputs": {
|
||||
"text": "",
|
||||
"clip": ["20", 0]
|
||||
},
|
||||
"class_type": "CLIPTextEncode",
|
||||
"_meta": { "title": "Positive prompt" }
|
||||
},
|
||||
"7": {
|
||||
"inputs": {
|
||||
"text": "blurry, low quality, jitter, deformed, static frame, freeze, watermark, bad anatomy, text, subtitles",
|
||||
"clip": ["20", 0]
|
||||
},
|
||||
"class_type": "CLIPTextEncode",
|
||||
"_meta": { "title": "Negative prompt" }
|
||||
},
|
||||
"8": {
|
||||
"inputs": {
|
||||
"width": ["5", 1],
|
||||
"height": ["5", 2],
|
||||
"length": 97,
|
||||
"batch_size": 1,
|
||||
"strength": 1,
|
||||
"positive": ["6", 0],
|
||||
"negative": ["7", 0],
|
||||
"vae": ["3", 0],
|
||||
"image": ["5", 0]
|
||||
},
|
||||
"class_type": "LTXVImgToVideo",
|
||||
"_meta": { "title": "LTX image to video" }
|
||||
},
|
||||
"9": {
|
||||
"inputs": {
|
||||
"frame_rate": 24,
|
||||
"positive": ["8", 0],
|
||||
"negative": ["8", 1]
|
||||
},
|
||||
"class_type": "LTXVConditioning",
|
||||
"_meta": { "title": "LTX conditioning" }
|
||||
},
|
||||
"10": {
|
||||
"inputs": {
|
||||
"cfg": 3.5,
|
||||
"model": ["21", 0],
|
||||
"positive": ["9", 0],
|
||||
"negative": ["9", 1]
|
||||
},
|
||||
"class_type": "CFGGuider",
|
||||
"_meta": { "title": "CFG Guider" }
|
||||
},
|
||||
"11": {
|
||||
"inputs": {
|
||||
"sampler_name": "euler"
|
||||
},
|
||||
"class_type": "KSamplerSelect",
|
||||
"_meta": { "title": "KSamplerSelect" }
|
||||
},
|
||||
"12": {
|
||||
"inputs": {
|
||||
"steps": 8,
|
||||
"max_shift": 2.05,
|
||||
"base_shift": 0.95,
|
||||
"stretch": true,
|
||||
"terminal": 0.1,
|
||||
"latent": ["8", 2]
|
||||
},
|
||||
"class_type": "LTXVScheduler",
|
||||
"_meta": { "title": "LTX scheduler" }
|
||||
},
|
||||
"13": {
|
||||
"inputs": {
|
||||
"noise_seed": 1
|
||||
},
|
||||
"class_type": "RandomNoise",
|
||||
"_meta": { "title": "RandomNoise" }
|
||||
},
|
||||
"14": {
|
||||
"inputs": {
|
||||
"noise": ["13", 0],
|
||||
"guider": ["10", 0],
|
||||
"sampler": ["11", 0],
|
||||
"sigmas": ["12", 0],
|
||||
"latent_image": ["8", 2]
|
||||
},
|
||||
"class_type": "SamplerCustomAdvanced",
|
||||
"_meta": { "title": "SamplerCustomAdvanced" }
|
||||
},
|
||||
"15": {
|
||||
"inputs": {
|
||||
"samples": ["14", 0],
|
||||
"vae": ["3", 0]
|
||||
},
|
||||
"class_type": "VAEDecode",
|
||||
"_meta": { "title": "VAE Decode" }
|
||||
},
|
||||
"16": {
|
||||
"inputs": {
|
||||
"frame_rate": 24,
|
||||
"loop_count": 0,
|
||||
"filename_prefix": "video/LTX23",
|
||||
"format": "video/h264-mp4",
|
||||
"pix_fmt": "yuv420p",
|
||||
"crf": 19,
|
||||
"save_metadata": true,
|
||||
"trim_to_audio": false,
|
||||
"pingpong": false,
|
||||
"save_output": true,
|
||||
"images": ["15", 0]
|
||||
},
|
||||
"class_type": "VHS_VideoCombine",
|
||||
"_meta": { "title": "Save MP4" }
|
||||
},
|
||||
"20": {
|
||||
"inputs": {
|
||||
"clip_name1": "gemma-3-12b-it-qat-UD-Q4_K_XL.gguf",
|
||||
"clip_name2": "text_encoders\\ltx-2.3-22b-dev_embeddings_connectors.safetensors",
|
||||
"type": "ltxv"
|
||||
},
|
||||
"class_type": "DualCLIPLoaderGGUF",
|
||||
"_meta": { "title": "DualCLIPLoader (GGUF)" }
|
||||
}
|
||||
}
|
||||
+24
-10
@@ -1,3 +1,5 @@
|
||||
import { LTX_NEGATIVE } from '~/utils/videoModels'
|
||||
|
||||
let comfyHostOverride = ''
|
||||
|
||||
export function setComfyHostOverride(url: string) {
|
||||
@@ -188,7 +190,12 @@ export function extractPromptFromHistory(entry: unknown) {
|
||||
const text = String(node.inputs?.prompt || node.inputs?.text || '').trim()
|
||||
if (!text) continue
|
||||
if (node?.class_type === 'MiniMaxH3ReferenceToVideo') return text
|
||||
if (node?.class_type === 'MiniMaxH3ImageToVideo') imageToVideo = text
|
||||
if (node?.class_type === 'MiniMaxH3ImageToVideo' || node?.class_type === 'MiniMaxH3TextToVideo') imageToVideo = text
|
||||
if (node?.class_type === 'CLIPTextEncode') {
|
||||
const title = String((node as { _meta?: { title?: string } })._meta?.title || '')
|
||||
if (/negative/i.test(title) || text === LTX_NEGATIVE) continue
|
||||
if (!imageToVideo) imageToVideo = text
|
||||
}
|
||||
}
|
||||
return imageToVideo
|
||||
}
|
||||
@@ -200,14 +207,18 @@ export function extractClipMetaFromHistory(entry: unknown) {
|
||||
if (!graph || typeof graph !== 'object') return meta
|
||||
for (const node of Object.values(graph as Record<string, { class_type?: string; inputs?: Record<string, unknown> }>)) {
|
||||
const inputs = node?.inputs || {}
|
||||
if (node?.class_type === 'MiniMaxH3ImageToVideo') {
|
||||
if (node?.class_type === 'MiniMaxH3ImageToVideo' || node?.class_type === 'MiniMaxH3TextToVideo' || node?.class_type === 'EmptyLTXVLatentVideo') {
|
||||
meta.width = Number(inputs.width || meta.width)
|
||||
meta.height = Number(inputs.height || meta.height)
|
||||
}
|
||||
if (node?.class_type === 'BasicScheduler' || node?.class_type === 'PrimitiveInt') {
|
||||
if (node?.class_type === 'BasicScheduler' || node?.class_type === 'PrimitiveInt' || node?.class_type === 'LTXVScheduler') {
|
||||
const steps = Number(inputs.steps || 0)
|
||||
if (steps > 0) meta.steps = steps
|
||||
}
|
||||
if (node?.class_type === 'ImageResize+' && typeof inputs.width === 'number' && typeof inputs.height === 'number') {
|
||||
meta.width = Number(inputs.width)
|
||||
meta.height = Number(inputs.height)
|
||||
}
|
||||
if (node?.class_type === 'RandomNoise' || node?.class_type === 'SamplerMiniMax' || node?.class_type === 'KSampler') {
|
||||
const seed = Number(inputs.noise_seed ?? inputs.seed ?? 0)
|
||||
if (seed) meta.seed = seed
|
||||
@@ -276,13 +287,16 @@ export function comfyFilenamePrefix() {
|
||||
}
|
||||
|
||||
export function isOurComfyVideo(video: { filename: string; subfolder: string }) {
|
||||
const prefix = comfyFilenamePrefix().replace(/\/$/, '')
|
||||
const parts = prefix.split('/')
|
||||
const namePrefix = parts[parts.length - 1]
|
||||
const sub = parts.length > 1 ? parts.slice(0, -1).join('/') : 'video'
|
||||
const nameOk = video.filename.startsWith(namePrefix)
|
||||
const subOk = !video.subfolder || video.subfolder === sub
|
||||
return nameOk && subOk
|
||||
const prefixes = [comfyFilenamePrefix(), 'video/LTX23']
|
||||
return prefixes.some((raw) => {
|
||||
const prefix = raw.replace(/\/$/, '')
|
||||
const parts = prefix.split('/')
|
||||
const namePrefix = parts[parts.length - 1]
|
||||
const sub = parts.length > 1 ? parts.slice(0, -1).join('/') : 'video'
|
||||
const nameOk = video.filename.startsWith(namePrefix) || video.filename.startsWith('LTX23')
|
||||
const subOk = !video.subfolder || video.subfolder === sub
|
||||
return nameOk && subOk
|
||||
})
|
||||
}
|
||||
|
||||
export function purgeComfyEnabled() {
|
||||
|
||||
@@ -85,7 +85,7 @@ export interface Job {
|
||||
familyId?: string
|
||||
parentClipId?: string
|
||||
parentStillId?: string
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
chainContinuing?: boolean
|
||||
passes?: { prompt: string }[]
|
||||
queueId?: string
|
||||
|
||||
@@ -38,7 +38,7 @@ export interface LibraryClip {
|
||||
familyId?: string
|
||||
parentClipId?: string
|
||||
chainIndex?: number
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
stillId?: string
|
||||
sound?: boolean
|
||||
hideInput?: boolean
|
||||
@@ -102,7 +102,7 @@ export interface RetryDraft {
|
||||
samplerName?: string
|
||||
scheduler?: string
|
||||
extensions?: QueuedExtension[]
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
}
|
||||
|
||||
interface Catalog {
|
||||
@@ -969,7 +969,7 @@ export async function saveRetryDraft(params: {
|
||||
samplerName?: string
|
||||
scheduler?: string
|
||||
extensions?: QueuedExtension[]
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
}) {
|
||||
return mutate(params.ownerKey, (catalog) => {
|
||||
const existing = params.id ? catalog.drafts.find(item => item.id === params.id) : undefined
|
||||
@@ -1134,7 +1134,7 @@ export async function saveClip(params: {
|
||||
familyId?: string
|
||||
parentClipId?: string
|
||||
chainIndex?: number
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
stillId?: string
|
||||
sound?: boolean
|
||||
}) {
|
||||
|
||||
@@ -35,7 +35,7 @@ export interface PendingJob {
|
||||
referenceImageNames?: string[]
|
||||
referenceStillIds?: Array<string | null>
|
||||
useIdentityRefs?: boolean
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
duration?: number
|
||||
cfg?: number
|
||||
fps?: number
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { getSidecarImageHost } from '~/server/utils/imageComfy'
|
||||
import { buildVisionPromptWorkflow } from '~/server/utils/promptWorkflow'
|
||||
import { normalizeShotScript } from '~/utils/parseRecommend'
|
||||
|
||||
export type PromptJobStatus = 'queued' | 'running' | 'complete' | 'error'
|
||||
|
||||
@@ -206,7 +207,10 @@ function draftUserPrompt(draft: string, mode: string) {
|
||||
: mode === 'shot-script'
|
||||
? 'video-shot-script'
|
||||
: 'video'
|
||||
return `Studio mode: ${studio}\n\nDraft Prompt:\n${draft}`
|
||||
const format = studio === 'image-edit'
|
||||
? 'If this is a single edit, output only the refined prompt. If it needs phases, use ---pass 1--- blocks.'
|
||||
: 'Output a MiniMax shot list. Each shot starts with "shot N" on its own line, then [SHOT CONFIGURATION], [SUBJECT DIRECTION & ACTION], and [AUDIO CUES], each on its own line. Blank line between shots. No markdown fences. No one-paragraph dump.'
|
||||
return `Studio mode: ${studio}\n${format}\n\nDraft Prompt:\n${draft}`
|
||||
}
|
||||
|
||||
export async function runPromptRecommend(job: PromptJob, params: {
|
||||
@@ -227,7 +231,8 @@ export async function runPromptRecommend(job: PromptJob, params: {
|
||||
const queued = await queuePromptGraph(graph, job.id)
|
||||
job.message = 'Writing the recommended prompt…'
|
||||
const text = await waitForPromptText(queued.prompt_id)
|
||||
job.text = text.trim()
|
||||
const trimmed = text.trim()
|
||||
job.text = params.mode === 'edit' ? trimmed : normalizeShotScript(trimmed)
|
||||
job.status = 'complete'
|
||||
job.message = 'Recommended prompt ready'
|
||||
}
|
||||
|
||||
@@ -9,6 +9,49 @@ export interface PromptPreset {
|
||||
builtin?: boolean
|
||||
}
|
||||
|
||||
const SHOT_SCRIPT_FORMAT = `Output ONLY a MiniMax shot list the user can paste into Shot script mode.
|
||||
|
||||
Each shot starts with "shot N" on its own line (shot 1, shot 2, shot 3…). Then these three headers, each on its own line:
|
||||
|
||||
[SHOT CONFIGURATION]
|
||||
[SUBJECT DIRECTION & ACTION]
|
||||
[AUDIO CUES]
|
||||
|
||||
Blank line between shots. No preamble, no markdown fences, no commentary.
|
||||
Do not glue shots into one paragraph. "shot 2" must be a line by itself, never "shot 2:" mid-sentence.
|
||||
|
||||
One beat per shot. Later shots: same people, same clothes as now, plus the one new action.
|
||||
Do not repeat a long identity paragraph on every shot.
|
||||
Never censor raw or explicit terminology.
|
||||
|
||||
Example:
|
||||
|
||||
shot 1
|
||||
[SHOT CONFIGURATION]
|
||||
- Shot Type: Medium-wide, 50mm, eye-level
|
||||
- Camera Movement: Slow track from a slight side-front angle
|
||||
- Lighting: Bright daylight through the trees
|
||||
- Environment/Scenery: The forest from the still
|
||||
|
||||
[SUBJECT DIRECTION & ACTION]
|
||||
The woman from the still walks the path, taking in the trees.
|
||||
|
||||
[AUDIO CUES]
|
||||
Footsteps on dirt, birds, breeze.
|
||||
|
||||
shot 2
|
||||
[SHOT CONFIGURATION]
|
||||
- Shot Type: Medium, 50mm, eye-level
|
||||
- Camera Movement: Hold, then a small pan to her hand
|
||||
- Lighting: Same daylight
|
||||
- Environment/Scenery: Same path, same trees
|
||||
|
||||
[SUBJECT DIRECTION & ACTION]
|
||||
She stops and rests her hand on the bark.
|
||||
|
||||
[AUDIO CUES]
|
||||
Soft fabric, breeze, birds.`
|
||||
|
||||
const ENHANCE_PROMPT = `You are a prompt-engineering assistant for Flux.2 Klein image edit and MiniMax H3 image-to-video.
|
||||
|
||||
Look at the attached still. The person (or subject) in that still is the identity lock. The user draft is the requested change.
|
||||
@@ -16,37 +59,27 @@ Look at the attached still. The person (or subject) in that still is the identit
|
||||
Write production-ready prompt text the user can paste into the generator.
|
||||
|
||||
Rules:
|
||||
- Preserve identity from the still: face, age, body type, hair, skin, and distinctive features. Use explicit keep-identity language such as "same person as the reference still", "do not change facial identity", "keep original face, hair, and body".
|
||||
- Preserve identity from the still: face, age, body type, hair, skin, and distinctive features.
|
||||
- Describe wardrobe, setting, pose, lighting, camera, and materials concretely.
|
||||
- Never censor raw or explicit terminology from the draft.
|
||||
- Do not add a preamble, title, markdown fences, or commentary.
|
||||
|
||||
Output format — pick one:
|
||||
If Studio mode is image-edit:
|
||||
- Single still/edit: output ONLY the refined prompt.
|
||||
- Ambitious still edit that should be done in phases (for example wardrobe first, then location): output:
|
||||
---pass 1---
|
||||
<prompt for pass 1>
|
||||
---pass 2---
|
||||
<prompt for pass 2>
|
||||
- Motion, multi-beat, or video sequence: first-shot prompt, then later shots with a marker on its own line:
|
||||
shot 2
|
||||
<extension prompt>
|
||||
shot 3
|
||||
<extension prompt>`
|
||||
|
||||
If Studio mode is video or video-shot-script:
|
||||
${SHOT_SCRIPT_FORMAT}`
|
||||
|
||||
const SHOT_SCRIPT_PROMPT = `You are a MiniMax H3 image-to-video shot-script writer.
|
||||
|
||||
Analyze the attached still for identity, wardrobe, lighting, and camera. Turn the user's draft into a shot script they can paste into Shot script mode.
|
||||
Analyze the attached still for wardrobe, lighting, and camera. Turn the user's draft into a shot script.
|
||||
|
||||
Format:
|
||||
- Shot 1 is the text BEFORE any marker (camera, action, audio for the first clip).
|
||||
- Later shots start with a line that is only: shot 2
|
||||
- Then shot 3, shot 4, and so on, each on its own line.
|
||||
- Every shot must keep the same person as the still unless the draft explicitly changes wardrobe or identity.
|
||||
- Include an [AUDIO CUES] block on shots where sound would help.
|
||||
- Never censor raw or explicit terminology.
|
||||
|
||||
Output ONLY the shot script. No preamble, no markdown fences.`
|
||||
${SHOT_SCRIPT_FORMAT}`
|
||||
|
||||
export const BUILTIN_PRESETS: PromptPreset[] = [
|
||||
{
|
||||
@@ -60,7 +93,7 @@ export const BUILTIN_PRESETS: PromptPreset[] = [
|
||||
id: 'shot-script',
|
||||
name: 'Shot script',
|
||||
builtin: true,
|
||||
description: 'Turn a draft into MiniMax shot 2 / shot 3 extension copy.',
|
||||
description: 'Turn a draft into a MiniMax shot list with configuration, action, and audio blocks.',
|
||||
systemPrompt: SHOT_SCRIPT_PROMPT
|
||||
}
|
||||
]
|
||||
|
||||
@@ -38,7 +38,7 @@ export interface ShotQueue {
|
||||
fps?: number
|
||||
samplerName?: string
|
||||
scheduler?: string
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
sound?: boolean
|
||||
useIdentityRefs?: boolean
|
||||
referenceStillIds?: Array<string | null>
|
||||
@@ -153,7 +153,7 @@ export async function createShotQueue(params: {
|
||||
fps?: number
|
||||
samplerName?: string
|
||||
scheduler?: string
|
||||
workflow?: 'v1' | 'v2'
|
||||
workflow?: import('~/utils/videoModels').VideoWorkflowId
|
||||
sound?: boolean
|
||||
useIdentityRefs?: boolean
|
||||
referenceStillIds?: Array<string | null>
|
||||
|
||||
+24
-15
@@ -5,6 +5,7 @@ import { pendingFromJob, remainingAfterCurrentShot, writePendingJob } from '~/se
|
||||
import { emitChainJob, waitForComfySocket, watchComfyJob } from '~/server/utils/watch'
|
||||
import { comfyFilenamePrefix, queuePrompt, uploadImage } from '~/server/utils/comfy'
|
||||
import { buildWorkflow } from '~/server/utils/workflow'
|
||||
import { isLtxWorkflow, isTextToVideo, parseVideoWorkflow, workflowForExtension, type VideoWorkflowId } from '~/utils/videoModels'
|
||||
import { clipVideoPath, clipTitle, deleteRetryDraft, extendTempDir, getClip, nextClipPartName, removeExtendTemp, stillPath } from '~/server/utils/library'
|
||||
import { extractLastFrame, probeHasAudio } from '~/server/utils/ffmpeg'
|
||||
import { ensureComfyReady } from '~/server/utils/comfyLifecycle'
|
||||
@@ -23,7 +24,7 @@ export type ChainImage = { filename: string; data: Buffer; type?: string }
|
||||
|
||||
type VideoChainParams = {
|
||||
prompt: string
|
||||
image: ChainImage
|
||||
image?: ChainImage | null
|
||||
width: number
|
||||
height: number
|
||||
steps: number
|
||||
@@ -36,7 +37,7 @@ type VideoChainParams = {
|
||||
samplerName: string
|
||||
scheduler: string
|
||||
extensions: { prompt: string; duration: number }[]
|
||||
workflow: 'v1' | 'v2'
|
||||
workflow: VideoWorkflowId
|
||||
duration: number
|
||||
useIdentityRefs: boolean
|
||||
referenceImages: Array<ChainImage | null>
|
||||
@@ -76,7 +77,8 @@ function paramsFromJob(job: Job): VideoChainParams {
|
||||
throw new Error('Cannot continue the shot chain: job metadata is missing')
|
||||
}
|
||||
const image = loadStillImage(library.ownerKey, library.stillId, library.stillFilename)
|
||||
if (!image?.data.length) {
|
||||
const workflow = parseVideoWorkflow(library.workflow)
|
||||
if (!image?.data.length && !isTextToVideo(workflow)) {
|
||||
throw new Error('Cannot continue the shot chain: the start still is missing from the library')
|
||||
}
|
||||
const referenceImages: Array<ChainImage | null> = [null, null, null, null]
|
||||
@@ -101,7 +103,7 @@ function paramsFromJob(job: Job): VideoChainParams {
|
||||
samplerName: library.samplerName || 'res_multistep',
|
||||
scheduler: library.scheduler || 'simple',
|
||||
extensions: library.extensions || [],
|
||||
workflow: library.workflow || 'v1',
|
||||
workflow,
|
||||
duration,
|
||||
useIdentityRefs: library.useIdentityRefs === true,
|
||||
referenceImages
|
||||
@@ -121,14 +123,21 @@ export async function queueMiniMax(
|
||||
const done = watchComfyJob(job, { persist: params.persist })
|
||||
job.status = 'uploading'
|
||||
const chainIndex = job.library?.chainIndex || 0
|
||||
const uploading = params.useIdentityRefs
|
||||
? (chainIndex > 0 ? 'Uploading identity stills for next shot...' : 'Uploading image to ComfyUI...')
|
||||
: (chainIndex > 0 ? 'Uploading last frame to ComfyUI...' : 'Uploading image to ComfyUI...')
|
||||
const queueing = (job.library?.chainIndex || 0) > 0
|
||||
? 'Queueing extension on MiniMax H3...'
|
||||
: 'Queueing MiniMax H3 job...'
|
||||
const graphId = chainIndex > 0 ? workflowForExtension(params.workflow) : params.workflow
|
||||
const engineName = isLtxWorkflow(graphId) ? 'LTX-2.3' : 'MiniMax H3'
|
||||
const hasImage = Boolean(params.image?.data?.length)
|
||||
const uploading = !hasImage
|
||||
? `Queueing ${engineName} text-to-video…`
|
||||
: params.useIdentityRefs
|
||||
? (chainIndex > 0 ? 'Uploading identity stills for next shot...' : 'Uploading image to ComfyUI...')
|
||||
: (chainIndex > 0 ? 'Uploading last frame to ComfyUI...' : 'Uploading image to ComfyUI...')
|
||||
const queueing = chainIndex > 0
|
||||
? `Queueing extension on ${engineName}...`
|
||||
: `Queueing ${engineName} job...`
|
||||
emitChainJob(job, { type: 'status', message: uploading, progress: 4 })
|
||||
const uploaded = await uploadImage(params.image, job.id)
|
||||
const uploaded = hasImage && params.image
|
||||
? await uploadImage(params.image, job.id)
|
||||
: { name: '', subfolder: '' }
|
||||
const referenceNames: string[] = ['', '', '', '']
|
||||
if (params.useIdentityRefs) {
|
||||
for (const [index, ref] of (params.referenceImages || []).entries()) {
|
||||
@@ -161,9 +170,9 @@ export async function queueMiniMax(
|
||||
fps: params.fps,
|
||||
samplerName: params.samplerName,
|
||||
scheduler: params.scheduler,
|
||||
filenamePrefix: comfyFilenamePrefix(),
|
||||
sound: params.sound,
|
||||
workflow: params.workflow,
|
||||
filenamePrefix: isLtxWorkflow(graphId) ? 'video/LTX23' : comfyFilenamePrefix(),
|
||||
sound: params.sound && !isLtxWorkflow(graphId),
|
||||
workflow: graphId,
|
||||
duration: params.duration,
|
||||
useIdentityRefs: params.useIdentityRefs,
|
||||
referenceImageNames: params.useIdentityRefs ? referenceNames : []
|
||||
@@ -317,7 +326,7 @@ export async function continueQueuedExtensions(
|
||||
samplerName: params.samplerName,
|
||||
scheduler: params.scheduler,
|
||||
persist,
|
||||
workflow: params.workflow,
|
||||
workflow: workflowForExtension(params.workflow),
|
||||
duration: ext.duration,
|
||||
useIdentityRefs: params.useIdentityRefs,
|
||||
referenceImages: params.useIdentityRefs ? params.referenceImages : []
|
||||
|
||||
+89
-11
@@ -1,9 +1,18 @@
|
||||
// Nitro bundles these JSON graphs into the production server output.
|
||||
import workflowV1 from '../assets/workflow_minimax_video.json'
|
||||
import workflowV2 from '../assets/workflow_minimax_video_v2.json'
|
||||
import workflowLtx from '../assets/workflow_ltx_video.json'
|
||||
import { buildIdentityPrompt } from '~/utils/identityPrompt'
|
||||
import {
|
||||
isLtxWorkflow,
|
||||
isTextToVideo,
|
||||
LTX_NEGATIVE,
|
||||
parseVideoWorkflow,
|
||||
type VideoWorkflowId
|
||||
} from '~/utils/videoModels'
|
||||
|
||||
export type VideoWorkflowId = 'v1' | 'v2'
|
||||
export type { VideoWorkflowId }
|
||||
export { parseVideoWorkflow }
|
||||
|
||||
export interface GenerateParams {
|
||||
prompt: string
|
||||
@@ -31,6 +40,7 @@ type WorkflowGraph = Record<string, WorkflowNode>
|
||||
|
||||
const PROMPT_CLASSES = new Set([
|
||||
'MiniMaxH3ImageToVideo',
|
||||
'MiniMaxH3TextToVideo',
|
||||
'MiniMaxH3ReferenceToVideo',
|
||||
'CLIPTextEncode',
|
||||
'CLIPTextEncodeQwen3VL',
|
||||
@@ -58,16 +68,15 @@ function titleOf(node: WorkflowNode) {
|
||||
}
|
||||
|
||||
function templateFor(id: VideoWorkflowId) {
|
||||
if (isLtxWorkflow(id)) return workflowLtx as WorkflowGraph
|
||||
return (id === 'v2' ? workflowV2 : workflowV1) as WorkflowGraph
|
||||
}
|
||||
|
||||
export function parseVideoWorkflow(raw: unknown): VideoWorkflowId {
|
||||
return String(raw || '').trim() === 'v2' ? 'v2' : 'v1'
|
||||
}
|
||||
|
||||
export function buildWorkflow(params: GenerateParams) {
|
||||
const version = params.workflow === 'v2' ? 'v2' : 'v1'
|
||||
const graph = structuredClone(templateFor(version))
|
||||
const version = parseVideoWorkflow(params.workflow)
|
||||
if (isLtxWorkflow(version)) return buildLtxWorkflow(params, version)
|
||||
|
||||
const graph = structuredClone(templateFor(version === 't2v' ? 'v1' : version === 'v2' ? 'v2' : 'v1'))
|
||||
const width = snap32(params.width)
|
||||
const height = snap32(params.height)
|
||||
const refs = Array.from({ length: 4 }, (_, index) => String(params.referenceImageNames?.[index] || '').trim())
|
||||
@@ -155,7 +164,7 @@ export function buildWorkflow(params: GenerateParams) {
|
||||
}
|
||||
}
|
||||
|
||||
if (version === 'v1' && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
|
||||
if (version === 'v1' && !isTextToVideo(version) && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
|
||||
graph['128'] = {
|
||||
class_type: 'ImageScale',
|
||||
inputs: {
|
||||
@@ -169,7 +178,7 @@ export function buildWorkflow(params: GenerateParams) {
|
||||
}
|
||||
}
|
||||
|
||||
if (version === 'v1') {
|
||||
if (version === 'v1' || version === 't2v') {
|
||||
graph['105:cfg'] = {
|
||||
class_type: 'FluxGuidance',
|
||||
inputs: {
|
||||
@@ -191,10 +200,11 @@ export function buildWorkflow(params: GenerateParams) {
|
||||
graph[qualityId].inputs.value = params.turbo ? 20 : params.steps
|
||||
}
|
||||
if (turboId && graph[turboId]?.class_type === 'PrimitiveInt') {
|
||||
graph[turboId].inputs.value = params.turbo ? params.steps : (params.workflow === 'v2' ? 6 : 8)
|
||||
graph[turboId].inputs.value = params.turbo ? params.steps : (version === 'v2' ? 6 : 8)
|
||||
}
|
||||
}
|
||||
|
||||
if (version === 't2v') applyMinimaxTextToVideo(graph, params, width, height)
|
||||
if (version === 'v2') applyV2IdentityPath(graph, params, useRefs, refs, graphPrompt)
|
||||
|
||||
if (params.sound === false) {
|
||||
@@ -203,6 +213,7 @@ export function buildWorkflow(params: GenerateParams) {
|
||||
if (graph['105:91']?.inputs) delete graph['105:91'].inputs.audio
|
||||
delete graph['163']
|
||||
if (graph['172']?.inputs) delete graph['172'].inputs.audio
|
||||
if (graph['105:104']?.inputs) delete graph['105:104'].inputs.audio_vae
|
||||
}
|
||||
|
||||
return graph
|
||||
@@ -210,10 +221,76 @@ export function buildWorkflow(params: GenerateParams) {
|
||||
|
||||
function promptForNode(classType: string, useRefs: boolean, actionPrompt: string, graphPrompt: string) {
|
||||
if (classType === 'MiniMaxH3ImageToVideo') return useRefs ? '' : actionPrompt
|
||||
if (classType === 'MiniMaxH3TextToVideo') return actionPrompt
|
||||
if (classType === 'MiniMaxH3ReferenceToVideo') return useRefs ? graphPrompt : actionPrompt
|
||||
return graphPrompt
|
||||
}
|
||||
|
||||
function applyMinimaxTextToVideo(graph: WorkflowGraph, params: GenerateParams, width: number, height: number) {
|
||||
const node = graph['105:104']
|
||||
if (node) {
|
||||
node.class_type = 'MiniMaxH3TextToVideo'
|
||||
node._meta = { title: 'MiniMax H3 Text to Video' }
|
||||
delete node.inputs.first_frame
|
||||
node.inputs.prompt = params.prompt
|
||||
node.inputs.width = width
|
||||
node.inputs.height = height
|
||||
if (params.sound !== false && graph['105:24']) node.inputs.audio_vae = ['105:24', 0]
|
||||
else delete node.inputs.audio_vae
|
||||
}
|
||||
delete graph['114']
|
||||
delete graph['119']
|
||||
delete graph['120']
|
||||
delete graph['127']
|
||||
delete graph['128']
|
||||
}
|
||||
|
||||
function buildLtxWorkflow(params: GenerateParams, version: VideoWorkflowId) {
|
||||
const graph = structuredClone(workflowLtx as WorkflowGraph)
|
||||
const width = snap32(params.width)
|
||||
const height = snap32(params.height)
|
||||
const textToVideo = isTextToVideo(version)
|
||||
if (graph['4']?.inputs) graph['4'].inputs.image = params.imageName
|
||||
if (graph['5']?.inputs) {
|
||||
graph['5'].inputs.width = width
|
||||
graph['5'].inputs.height = height
|
||||
}
|
||||
if (graph['6']?.inputs) graph['6'].inputs.text = params.prompt
|
||||
if (graph['7']?.inputs) graph['7'].inputs.text = LTX_NEGATIVE
|
||||
if (graph['8']?.inputs && typeof graph['8'].inputs.length === 'number') graph['8'].inputs.length = params.length
|
||||
if (graph['9']?.inputs) graph['9'].inputs.frame_rate = params.fps
|
||||
if (graph['10']?.inputs) graph['10'].inputs.cfg = params.cfg
|
||||
if (graph['11']?.inputs) graph['11'].inputs.sampler_name = params.samplerName || 'euler'
|
||||
if (graph['12']?.inputs) graph['12'].inputs.steps = params.steps
|
||||
if (graph['13']?.inputs) graph['13'].inputs.noise_seed = params.seed
|
||||
if (graph['16']?.inputs) {
|
||||
graph['16'].inputs.frame_rate = params.fps
|
||||
graph['16'].inputs.filename_prefix = params.filenamePrefix || 'video/LTX23'
|
||||
}
|
||||
if (graph['21']?.inputs) graph['21'].inputs.strength_model = 1
|
||||
if (textToVideo) {
|
||||
delete graph['4']
|
||||
delete graph['5']
|
||||
graph['8'] = {
|
||||
class_type: 'EmptyLTXVLatentVideo',
|
||||
inputs: {
|
||||
width,
|
||||
height,
|
||||
length: params.length,
|
||||
batch_size: 1
|
||||
},
|
||||
_meta: { title: 'Empty LTX latent' }
|
||||
}
|
||||
if (graph['9']?.inputs) {
|
||||
graph['9'].inputs.positive = ['6', 0]
|
||||
graph['9'].inputs.negative = ['7', 0]
|
||||
}
|
||||
if (graph['12']?.inputs) graph['12'].inputs.latent = ['8', 0]
|
||||
if (graph['14']?.inputs) graph['14'].inputs.latent_image = ['8', 0]
|
||||
}
|
||||
return graph
|
||||
}
|
||||
|
||||
function applyV2IdentityPath(
|
||||
graph: WorkflowGraph,
|
||||
params: GenerateParams,
|
||||
@@ -304,9 +381,10 @@ function labelsFrom(template: WorkflowGraph) {
|
||||
export const NODE_LABELS: Record<string, string> = {
|
||||
...Object.fromEntries(labelsFrom(workflowV1 as WorkflowGraph)),
|
||||
...Object.fromEntries(labelsFrom(workflowV2 as WorkflowGraph)),
|
||||
...Object.fromEntries(labelsFrom(workflowLtx as WorkflowGraph)),
|
||||
...LABEL_OVERRIDES
|
||||
}
|
||||
|
||||
export function isEncodingNode(node: string) {
|
||||
return node === '92' || node === '105:91' || node === '172' || /encoding|saving mp4|create video/i.test(NODE_LABELS[node] || '')
|
||||
return node === '92' || node === '16' || node === '105:91' || node === '172' || /encoding|saving mp4|create video|save mp4/i.test(NODE_LABELS[node] || '')
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user