Add MiniMax and PinkCherry LTX as pickable video engines, with still or text start.

Shot 1 can run text-to-video; later shots still extend from the last frame. Recommend output is also normalized into shot-script cards.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-08-27 19:38:53 -05:00
co-authored by Cursor
parent 2bebeea526
commit 40cd17f35c
15 changed files with 676 additions and 120 deletions
+178 -28
View File
@@ -6,7 +6,7 @@
<div class="flex h-10 w-10 items-center justify-center rounded-xl bg-amber-400 text-zinc-950 font-display font-extrabold">A</div>
<div>
<h1 class="font-display text-xl font-bold leading-none">{{ instanceName }}</h1>
<p class="text-[11px] uppercase tracking-[0.22em] text-zinc-500">MiniMax H3 · Flux.2 Klein</p>
<p class="text-[11px] uppercase tracking-[0.22em] text-zinc-500">MiniMax H3 · LTX-2.3 · Flux.2 Klein</p>
</div>
</div>
<div class="flex items-center gap-3 text-sm">
@@ -80,8 +80,8 @@
<button type="button" class="relative mt-3 rounded-full bg-zinc-950/80 px-3 py-1 text-xs" @click.stop="resetImage">Reset</button>
</div>
<div v-else class="flex min-h-44 flex-col items-center justify-center text-center">
<p class="font-medium">{{ studioMode === 'edit' ? 'Image 1' : 'Choose an image' }}</p>
<p class="mt-1 text-sm text-zinc-500">Browse the library or upload a new PNG, JPG, or WEBP</p>
<p class="font-medium">{{ studioMode === 'edit' ? 'Image 1' : (textToVideo ? 'Start still · optional' : 'Choose an image') }}</p>
<p class="mt-1 text-sm text-zinc-500">{{ textToVideo ? 'Text-to-video does not need a still. Shot 2+ will still extend from the last frame.' : 'Browse the library or upload a new PNG, JPG, or WEBP' }}</p>
</div>
</div>
@@ -183,6 +183,16 @@
<p class="text-xs font-medium uppercase tracking-wide text-amber-200/80">Recommended prompt</p>
<p v-if="recommendBusy" class="mt-2 text-sm text-zinc-400">{{ recommendMessage || 'Sidecar Qwen VL is reading the still…' }}</p>
<p v-else-if="recommendError" class="mt-2 text-sm text-red-300">{{ recommendError }}</p>
<div v-else-if="recommendedShots.length > 1" class="mt-2 space-y-2">
<div
v-for="shot in recommendedShots"
:key="shot.n"
class="rounded-xl border border-amber-300/15 bg-zinc-950/50 px-3 py-2"
>
<p class="text-[11px] font-medium uppercase tracking-wide text-amber-200/80">shot {{ shot.n }}</p>
<p class="mt-1 whitespace-pre-wrap text-sm text-zinc-200">{{ shot.prompt }}</p>
</div>
</div>
<p v-else class="mt-2 whitespace-pre-wrap text-sm text-zinc-200">{{ recommendText }}</p>
<div v-if="recommendText && !recommendBusy" class="mt-3 flex flex-wrap gap-2">
<button
@@ -343,7 +353,7 @@
<span class="absolute left-0.5 top-0.5 h-5 w-5 rounded-full bg-white transition peer-checked:translate-x-5" />
</span>
</label>
<label v-if="studioMode === 'video'" class="flex cursor-pointer items-center justify-between gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm">
<label v-if="studioMode === 'video' && !ltxVideo" class="flex cursor-pointer items-center justify-between gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm">
<span>
<span class="block font-medium text-zinc-200">Generate sound</span>
<span class="mt-0.5 block text-xs text-zinc-500">Off skips the audio VAE so Comfy encodes a silent clip.</span>
@@ -371,13 +381,60 @@
<div v-if="studioMode === 'video'" class="space-y-3">
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Video model</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoEngine === 'minimax' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="selectVideoEngine('minimax')"
>
<span class="block font-semibold">MiniMax H3</span>
<span class="text-xs text-zinc-400">Native stereo audio. Identity refs on v2.</span>
</button>
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoEngine === 'ltx' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="selectVideoEngine('ltx')"
>
<span class="block font-semibold">LTX-2.3 PinkCherry</span>
<span class="text-xs text-zinc-400">PinkCherry + distilled LoRA. I2V and T2V.</span>
</button>
</div>
</div>
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Start from</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoStart === 'still' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoStart = 'still'"
>
<span class="block font-semibold">A still</span>
<span class="text-xs text-zinc-400">Image to video. Extensions use the last frame.</span>
</button>
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoStart === 'text' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoStart = 'text'"
>
<span class="block font-semibold">Text only</span>
<span class="text-xs text-zinc-400">No start still. Shot 2+ still continues from last frame.</span>
</button>
</div>
<p class="mt-2 text-xs text-zinc-500">{{ videoModelHint }}</p>
</div>
<div v-if="videoEngine === 'minimax' && videoStart === 'still'">
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">MiniMax graph</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoWorkflow === 'v1' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoWorkflow = 'v1'"
:class="minimaxGraph === 'v1' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="minimaxGraph = 'v1'"
>
<span class="block font-semibold">v1 · validated</span>
<span class="text-xs text-zinc-400">First-frame image to video</span>
@@ -385,8 +442,8 @@
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoWorkflow === 'v2' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoWorkflow = 'v2'"
:class="minimaxGraph === 'v2' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="minimaxGraph = 'v2'"
>
<span class="block font-semibold">v2 · identity refs</span>
<span class="text-xs text-zinc-400">Optional lock with up to 4 stills</span>
@@ -394,7 +451,7 @@
</div>
</div>
<div v-if="videoWorkflow === 'v2'" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 space-y-3">
<div v-if="videoEngine === 'minimax' && videoStart === 'still' && minimaxGraph === 'v2'" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 space-y-3">
<label class="flex cursor-pointer items-center justify-between gap-3 text-sm">
<span>
<span class="block font-medium text-zinc-200">Use identity references</span>
@@ -466,13 +523,13 @@
<span class="text-xs text-zinc-400">{{ option.id === 'auto' ? autoHint : option.hint }}</span>
</button>
</div>
<p class="mt-2 text-xs text-zinc-500">Output {{ outputSize.width }} × {{ outputSize.height }}{{ aspect === 'auto' ? ' · matching the still' : '' }}</p>
<p class="mt-2 text-xs text-zinc-500">Output {{ outputSize.width }} × {{ outputSize.height }}{{ aspect === 'auto' ? (file ? ' · matching the still' : ' · 16:9 until a still is loaded') : '' }}</p>
</div>
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Steps</p>
<div class="grid grid-cols-2 gap-2">
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = true">{{ videoWorkflow === 'v2' ? 6 : 8 }} Turbo · LoRA on</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="!turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = false">20 High-Fidelity · LoRA bypass</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = true">{{ ltxVideo ? '8 Distilled LoRA' : (videoWorkflow === 'v2' ? '6 Turbo · LoRA on' : '8 Turbo · LoRA on') }}</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="!turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = false">{{ ltxVideo ? '12 Quality · LoRA on' : '20 High-Fidelity · LoRA bypass' }}</button>
</div>
</div>
<div class="text-sm">
@@ -499,7 +556,7 @@
@change="onCfgNumber"
>
</div>
<p class="mt-1 text-[11px] text-zinc-500">{{ turbo ? 'Default 1.5 with Turbo LoRA' : 'Default 4.0 with High-Fidelity' }}</p>
<p class="mt-1 text-[11px] text-zinc-500">{{ ltxVideo ? 'Default 3.5 for PinkCherry' : (turbo ? 'Default 1.5 with Turbo LoRA' : 'Default 4.0 with High-Fidelity') }}</p>
</div>
<div class="grid grid-cols-2 gap-3">
<label class="text-sm">
@@ -1256,6 +1313,19 @@
<script setup lang="ts">
import { parseEditPasses, looksLikeShotScript } from '~/utils/parseRecommend'
import { parseShotScript } from '~/utils/parseShots'
import {
composeVideoWorkflow,
isLtxWorkflow,
isTextToVideo,
minimaxGraphOf,
parseVideoWorkflow,
videoEngineOf,
videoStartOf,
type MiniMaxGraphId,
type VideoEngineId,
type VideoStartId
} from '~/utils/videoModels'
const runtimeConfig = useRuntimeConfig()
@@ -1311,7 +1381,7 @@ interface LibraryClip {
familyId?: string
parentClipId?: string
chainIndex?: number
workflow?: 'v1' | 'v2'
workflow?: string
stillId?: string
sound?: boolean
hideInput?: boolean
@@ -1340,7 +1410,7 @@ interface RetryDraft {
samplerName?: string
scheduler?: string
extensions?: { prompt: string; duration: number }[]
workflow?: 'v1' | 'v2'
workflow?: string
}
interface QueuedExtension {
@@ -1517,7 +1587,12 @@ const pickerSlot = ref<'main' | 'editRef' | number>('main')
const libraryFilter = ref<'all' | 'images' | 'videos'>('all')
const selectedKeys = ref<string[]>([])
const lastSelectedKey = ref('')
const videoWorkflow = ref('v1')
const videoEngine = ref<VideoEngineId>('minimax')
const videoStart = ref<VideoStartId>('still')
const minimaxGraph = ref<MiniMaxGraphId>('v1')
const videoWorkflow = computed(() => composeVideoWorkflow(videoEngine.value, videoStart.value, minimaxGraph.value))
const textToVideo = computed(() => isTextToVideo(videoWorkflow.value))
const ltxVideo = computed(() => isLtxWorkflow(videoWorkflow.value))
const useIdentityRefs = ref(false)
const identityRefs = ref<(File | null)[]>([null, null, null, null])
const identityPreviews = ref(['', '', '', ''])
@@ -1653,10 +1728,22 @@ const outputSize = computed(() => {
})
const autoHint = computed(() => {
if (!imageWidth.value || !imageHeight.value) return 'Match the still'
if (!imageWidth.value || !imageHeight.value) {
return textToVideo.value ? '16:9 without a still' : 'Match the still'
}
const size = autoResolution(imageWidth.value, imageHeight.value)
return `${size.width} × ${size.height} · from still`
})
const videoModelHint = computed(() => {
if (ltxVideo.value) {
return textToVideo.value
? 'PinkCherry text-to-video. Distilled LoRA stays on. Silent MP4 — LTX has no audio VAE here.'
: 'PinkCherry image-to-video. Distilled LoRA stays on. Silent MP4 — LTX has no audio VAE here.'
}
if (textToVideo.value) return 'MiniMax H3 text-to-video with native stereo audio. Shot 2+ switches to last-frame image-to-video.'
if (videoWorkflow.value === 'v2') return 'MiniMax H3 image-to-video. Identity refs stay on v2 with a start still.'
return 'MiniMax H3 image-to-video with native stereo audio.'
})
const frameCount = computed(() => Math.max(5, Math.floor(duration.value * fps.value)))
const cfgPct = computed(() => ((cfg.value - CFG_MIN) / (CFG_MAX - CFG_MIN)) * 100)
const currentClip = computed(() => clips.value.find(clip => clip.id === currentClipId.value))
@@ -1671,7 +1758,7 @@ const queueReady = computed(() => {
return extensionQueue.value.every(item => item.prompt.trim())
})
const videoGenerateDisabled = computed(() => {
return videoBusy.value || !file.value || !prompt.value.trim() || !folderId.value || !queueReady.value
return videoBusy.value || (!textToVideo.value && !file.value) || !prompt.value.trim() || !folderId.value || !queueReady.value
})
const recommendDisabled = computed(() => {
if (recommendBusy.value) return true
@@ -1715,6 +1802,10 @@ const editLabel = computed(() => {
return n === 1 ? 'Edit image + 1 pass' : `Edit image + ${n} passes`
})
const parsedShots = computed(() => parseShotScript(prompt.value))
const recommendedShots = computed(() => {
if (!recommendText.value || recommendBusy.value) return []
return parseShotScript(recommendText.value)
})
const identityPrompting = computed(() => studioMode.value === 'video' && videoWorkflow.value === 'v2' && useIdentityRefs.value)
const extraIdentityPictures = computed(() => identityRefs.value.flatMap((item, index) => item ? [index + 2] : []))
const identityActionPrompt = computed(() => {
@@ -2101,7 +2192,20 @@ onMounted(async () => {
hideInputPreview.value = localStorage.getItem('aigen-hide-input-preview') === 'true'
withSound.value = localStorage.getItem('aigen-generate-sound') !== 'false'
autoplayEnabled.value = localStorage.getItem('aigen-autoplay') === 'true'
videoWorkflow.value = localStorage.getItem('aigen-video-workflow') === 'v2' ? 'v2' : 'v1'
videoEngine.value = 'minimax'
videoStart.value = 'still'
minimaxGraph.value = 'v1'
try {
const saved = parseVideoWorkflow(localStorage.getItem('aigen-video-workflow') || '')
videoEngine.value = videoEngineOf(saved)
videoStart.value = videoStartOf(saved)
minimaxGraph.value = minimaxGraphOf(saved)
const engine = localStorage.getItem('aigen-video-engine')
const start = localStorage.getItem('aigen-video-start')
if (engine === 'ltx' || engine === 'minimax') videoEngine.value = engine
if (start === 'text' || start === 'still') videoStart.value = start
if (videoEngine.value === 'ltx') applyEngineDefaults('ltx')
} catch { /* ignore */ }
useIdentityRefs.value = localStorage.getItem('aigen-use-identity-refs') === 'true'
activeQueueId.value = localStorage.getItem(QUEUE_STORE) || ''
const savedPreset = localStorage.getItem('aigen-prompt-preset')
@@ -2190,6 +2294,18 @@ watch(videoWorkflow, (value) => {
} catch { /* ignore */ }
})
watch(videoEngine, (value) => {
try {
localStorage.setItem('aigen-video-engine', value)
} catch { /* ignore */ }
})
watch(videoStart, (value) => {
try {
localStorage.setItem('aigen-video-start', value)
} catch { /* ignore */ }
})
watch(useIdentityRefs, (value) => {
try {
localStorage.setItem('aigen-use-identity-refs', String(value))
@@ -2197,7 +2313,8 @@ watch(useIdentityRefs, (value) => {
})
watch(turbo, (on) => {
if (!cfgTouched.value) cfg.value = on ? CFG_TURBO : CFG_QUALITY
if (cfgTouched.value) return
cfg.value = ltxVideo.value ? 3.5 : (on ? CFG_TURBO : CFG_QUALITY)
})
function clampDuration(value: number) {
@@ -2336,7 +2453,7 @@ function applyRecommend() {
}
} else {
prompt.value = text
if (looksLikeShotScript(text)) shotScriptMode.value = true
if (looksLikeShotScript(text) || parseShotScript(text).length > 1) shotScriptMode.value = true
}
recommendApplied.value = true
toast('Recommended prompt applied')
@@ -2709,7 +2826,7 @@ async function rerun(item: LibraryClip, collection = false) {
if (showPrivacyToggles.value) hideInputPreview.value = target.hideInput === true
folderId.value = target.folderId
browseFolderId.value = target.folderId
videoWorkflow.value = target.workflow === 'v2' ? 'v2' : 'v1'
applyVideoWorkflow(target.workflow)
if (typeof target.sound === 'boolean') withSound.value = target.sound
shotScriptMode.value = restoreAll
useIdentityRefs.value = false
@@ -2722,7 +2839,8 @@ async function rerun(item: LibraryClip, collection = false) {
}))
: []
try {
await loadRerunImage(target, parts)
const initialTextToVideo = textToVideo.value && !target.parentClipId && !(target.chainIndex)
if (!(initialTextToVideo && !target.stillId)) await loadRerunImage(target, parts)
statusMessage.value = restoreAll
? 'Collection loaded into the form. Generate when you are ready.'
: 'Clip loaded into the form. Generate when you are ready.'
@@ -2730,8 +2848,12 @@ async function rerun(item: LibraryClip, collection = false) {
? 'Shot script and settings restored. Make any changes, then generate.'
: 'Inputs restored. Make any changes, then generate.')
} catch {
statusMessage.value = 'Clip settings loaded. Re-drop the start still if it is missing.'
toast('Settings restored, but the start still could not be loaded. Pick it again.')
statusMessage.value = textToVideo.value
? 'Clip settings loaded. Add a still only if you want image-to-video.'
: 'Clip settings loaded. Re-drop the start still if it is missing.'
toast(textToVideo.value
? 'Settings restored. Text-to-video does not need a start still.'
: 'Settings restored, but the start still could not be loaded. Pick it again.')
}
}
@@ -2746,7 +2868,7 @@ async function loadDraft(draft: RetryDraft) {
withSound.value = draft.sound !== false
if (draft.hideInput && showPrivacyToggles.value) hideInputPreview.value = true
duration.value = clampDuration(Number(draft.duration))
if (draft.workflow === 'v2') videoWorkflow.value = 'v2'
applyVideoWorkflow(draft.workflow)
extensionQueue.value = (draft.extensions || []).map(item => ({
id: crypto.randomUUID(),
prompt: String(item.prompt || ''),
@@ -2758,9 +2880,13 @@ async function loadDraft(draft: RetryDraft) {
readFile(new File([blob], filename, { type: blob.type || 'image/png' }), { keepThumbnailPref: true })
statusMessage.value = 'Held job restored. Generate when ComfyUI is free.'
} catch {
if (textToVideo.value) {
statusMessage.value = 'Held job restored. Generate when ComfyUI is free.'
} else {
statusMessage.value = 'Held job settings restored. Re-drop the still if the image is missing.'
toast('Settings restored. Unlock the folder or re-drop the still if the image is missing.')
}
}
}
async function dismissDraft(draft: RetryDraft) {
@@ -3363,8 +3489,32 @@ async function useEditAsInput() {
}
}
function applyEngineDefaults(engine: VideoEngineId) {
if (engine === 'ltx') {
samplerName.value = 'euler'
if (!cfgTouched.value) cfg.value = 3.5
return
}
if (samplerName.value === 'euler') samplerName.value = 'res_multistep'
if (!cfgTouched.value) cfg.value = turbo.value ? CFG_TURBO : CFG_QUALITY
}
function selectVideoEngine(engine: VideoEngineId) {
if (videoEngine.value === engine) return
cfgTouched.value = false
videoEngine.value = engine
applyEngineDefaults(engine)
}
function applyVideoWorkflow(raw?: string) {
const parsed = parseVideoWorkflow(raw)
videoEngine.value = videoEngineOf(parsed)
videoStart.value = videoStartOf(parsed)
minimaxGraph.value = minimaxGraphOf(parsed)
}
async function generate() {
if (!file.value || !prompt.value.trim()) return
if ((!file.value && !textToVideo.value) || !prompt.value.trim()) return
if (!queueReady.value) {
toast(shotScriptMode.value ? 'Each shot in the script needs a prompt.' : 'Each queued extension needs a prompt.')
return
@@ -3401,7 +3551,7 @@ async function generate() {
startTimer('video')
try {
const body = new FormData()
body.append('image', file.value)
if (file.value) body.append('image', file.value)
body.append('prompt', initialPrompt)
body.append('aspect', aspect.value)
body.append('width', String(outputSize.value.width))
+22 -14
View File
@@ -1,5 +1,6 @@
import { frameLength, runGeneration } from '~/server/utils/videoChain'
import { createShotQueue, finishQueueBurst, setQueueJob } from '~/server/utils/shotQueue'
import { defaultVideoSteps, isLtxWorkflow, isTextToVideo, parseVideoWorkflow } from '~/utils/videoModels'
function isComfyBusyTimeout(error: unknown) {
const err = error as { statusCode?: number; status?: number; data?: { code?: string }; message?: string; statusMessage?: string }
@@ -42,16 +43,16 @@ function parseFps(raw: string | undefined) {
return fps === 12 || fps === 30 || fps === 24 ? fps : 24
}
function parseCfg(raw: string | undefined, turbo: boolean) {
const fallback = turbo ? 1.5 : 4
function parseCfg(raw: string | undefined, turbo: boolean, ltx = false) {
const fallback = ltx ? 3.5 : (turbo ? 1.5 : 4)
const value = Number(raw)
if (!Number.isFinite(value)) return fallback
const snapped = Math.round(value * 2) / 2
return Math.min(10, Math.max(1, snapped))
}
function parseSampler(raw: string | undefined) {
return SAMPLERS.has(raw || '') ? raw! : 'res_multistep'
function parseSampler(raw: string | undefined, ltx = false) {
return SAMPLERS.has(raw || '') ? raw! : (ltx ? 'euler' : 'res_multistep')
}
function parseScheduler(raw: string | undefined) {
@@ -86,7 +87,9 @@ export default defineEventHandler(async (event) => {
}
}
if (!image) {
const workflow = parseVideoWorkflow(fields.workflow)
const textToVideo = isTextToVideo(workflow)
if (!image && !textToVideo) {
throw createError({ statusCode: 400, statusMessage: 'An initial image is required' })
}
const prompt = (fields.prompt || '').trim()
@@ -94,10 +97,13 @@ export default defineEventHandler(async (event) => {
throw createError({ statusCode: 400, statusMessage: 'A motion prompt is required' })
}
const extensions = parseExtensions(fields.extensions)
const workflow = parseVideoWorkflow(fields.workflow)
const useIdentityRefs = workflow === 'v2' && fields.useIdentityRefs === 'true'
const { width, height } = resolveOutputSize(fields.aspect, image.data, Number(fields.width), Number(fields.height))
const useIdentityRefs = workflow === 'v2' && !textToVideo && !isLtxWorkflow(workflow) && fields.useIdentityRefs === 'true'
const { width, height } = resolveOutputSize(
fields.aspect,
image?.data || Buffer.alloc(0),
Number(fields.width),
Number(fields.height)
)
const turbo = fields.turbo !== 'false'
const steps = Number(fields.steps || defaultVideoSteps(turbo, workflow))
const seed = fields.seed && fields.seed !== 'random'
@@ -105,13 +111,13 @@ export default defineEventHandler(async (event) => {
: Math.floor(Math.random() * 2_147_483_647)
const durationSeconds = parseDuration(fields.duration || '5')
const fps = parseFps(fields.fps)
const cfg = parseCfg(fields.cfg, turbo)
const samplerName = parseSampler(fields.sampler_name)
const cfg = parseCfg(fields.cfg, turbo, isLtxWorkflow(workflow))
const samplerName = parseSampler(fields.sampler_name, isLtxWorkflow(workflow))
const scheduler = parseScheduler(fields.scheduler)
const length = frameLength(durationSeconds, fps)
const hideThumbnail = fields.hideThumbnail === 'true'
const hideInput = fields.hideInput === 'true'
const sound = fields.sound !== 'false'
const sound = !isLtxWorkflow(workflow) && fields.sound !== 'false'
const clipName = (fields.name || '').trim().slice(0, 80)
const ownerKey = libraryOwnerKey(event)
const library = publicLibrary(event)
@@ -124,7 +130,8 @@ export default defineEventHandler(async (event) => {
const destFolder = library.folders.find(folder => folder.id === folderId)
const folderLocked = Boolean(destFolder?.protected && !destFolder.unlocked)
assertFolderExists(event, folderId)
const still = await rememberInputStill({
const still = image
? await rememberInputStill({
ownerKey,
folderId,
filename: image.filename,
@@ -133,6 +140,7 @@ export default defineEventHandler(async (event) => {
height,
hideInput
})
: null
const referenceStillIds: Array<string | null> = [null, null, null, null]
if (useIdentityRefs) {
for (const [index, ref] of referenceImages.entries()) {
@@ -175,7 +183,7 @@ export default defineEventHandler(async (event) => {
fps,
samplerName,
scheduler,
thumb: isPipelineFrameFilename(image.filename) ? undefined : image.data,
thumb: image && !isPipelineFrameFilename(image.filename) ? image.data : undefined,
stillId: still?.id,
stillFilename: still?.filename,
referenceStillIds,
+168
View File
@@ -0,0 +1,168 @@
{
"1": {
"inputs": {
"unet_name": "PinkCherry_FineTune_int8_v1_8_LTX23.safetensors",
"weight_dtype": "default"
},
"class_type": "UNETLoader",
"_meta": { "title": "Load PinkCherry LTX-2.3" }
},
"21": {
"inputs": {
"lora_name": "ltx-2.3-22b-distilled-lora-384-1.1.safetensors",
"strength_model": 1,
"model": ["1", 0]
},
"class_type": "LoraLoaderModelOnly",
"_meta": { "title": "LTX distilled LoRA" }
},
"3": {
"inputs": {
"vae_name": "LTX23_video_vae_bf16.safetensors"
},
"class_type": "VAELoader",
"_meta": { "title": "Load VAE" }
},
"4": {
"inputs": {
"image": "input_image.png",
"upload": "image"
},
"class_type": "LoadImage",
"_meta": { "title": "Load Image" }
},
"5": {
"inputs": {
"width": 1280,
"height": 1280,
"interpolation": "lanczos",
"method": "keep proportion",
"condition": "always",
"multiple_of": 32,
"image": ["4", 0]
},
"class_type": "ImageResize+",
"_meta": { "title": "Resize still" }
},
"6": {
"inputs": {
"text": "",
"clip": ["20", 0]
},
"class_type": "CLIPTextEncode",
"_meta": { "title": "Positive prompt" }
},
"7": {
"inputs": {
"text": "blurry, low quality, jitter, deformed, static frame, freeze, watermark, bad anatomy, text, subtitles",
"clip": ["20", 0]
},
"class_type": "CLIPTextEncode",
"_meta": { "title": "Negative prompt" }
},
"8": {
"inputs": {
"width": ["5", 1],
"height": ["5", 2],
"length": 97,
"batch_size": 1,
"strength": 1,
"positive": ["6", 0],
"negative": ["7", 0],
"vae": ["3", 0],
"image": ["5", 0]
},
"class_type": "LTXVImgToVideo",
"_meta": { "title": "LTX image to video" }
},
"9": {
"inputs": {
"frame_rate": 24,
"positive": ["8", 0],
"negative": ["8", 1]
},
"class_type": "LTXVConditioning",
"_meta": { "title": "LTX conditioning" }
},
"10": {
"inputs": {
"cfg": 3.5,
"model": ["21", 0],
"positive": ["9", 0],
"negative": ["9", 1]
},
"class_type": "CFGGuider",
"_meta": { "title": "CFG Guider" }
},
"11": {
"inputs": {
"sampler_name": "euler"
},
"class_type": "KSamplerSelect",
"_meta": { "title": "KSamplerSelect" }
},
"12": {
"inputs": {
"steps": 8,
"max_shift": 2.05,
"base_shift": 0.95,
"stretch": true,
"terminal": 0.1,
"latent": ["8", 2]
},
"class_type": "LTXVScheduler",
"_meta": { "title": "LTX scheduler" }
},
"13": {
"inputs": {
"noise_seed": 1
},
"class_type": "RandomNoise",
"_meta": { "title": "RandomNoise" }
},
"14": {
"inputs": {
"noise": ["13", 0],
"guider": ["10", 0],
"sampler": ["11", 0],
"sigmas": ["12", 0],
"latent_image": ["8", 2]
},
"class_type": "SamplerCustomAdvanced",
"_meta": { "title": "SamplerCustomAdvanced" }
},
"15": {
"inputs": {
"samples": ["14", 0],
"vae": ["3", 0]
},
"class_type": "VAEDecode",
"_meta": { "title": "VAE Decode" }
},
"16": {
"inputs": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "video/LTX23",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"images": ["15", 0]
},
"class_type": "VHS_VideoCombine",
"_meta": { "title": "Save MP4" }
},
"20": {
"inputs": {
"clip_name1": "gemma-3-12b-it-qat-UD-Q4_K_XL.gguf",
"clip_name2": "text_encoders\\ltx-2.3-22b-dev_embeddings_connectors.safetensors",
"type": "ltxv"
},
"class_type": "DualCLIPLoaderGGUF",
"_meta": { "title": "DualCLIPLoader (GGUF)" }
}
}
+19 -5
View File
@@ -1,3 +1,5 @@
import { LTX_NEGATIVE } from '~/utils/videoModels'
let comfyHostOverride = ''
export function setComfyHostOverride(url: string) {
@@ -188,7 +190,12 @@ export function extractPromptFromHistory(entry: unknown) {
const text = String(node.inputs?.prompt || node.inputs?.text || '').trim()
if (!text) continue
if (node?.class_type === 'MiniMaxH3ReferenceToVideo') return text
if (node?.class_type === 'MiniMaxH3ImageToVideo') imageToVideo = text
if (node?.class_type === 'MiniMaxH3ImageToVideo' || node?.class_type === 'MiniMaxH3TextToVideo') imageToVideo = text
if (node?.class_type === 'CLIPTextEncode') {
const title = String((node as { _meta?: { title?: string } })._meta?.title || '')
if (/negative/i.test(title) || text === LTX_NEGATIVE) continue
if (!imageToVideo) imageToVideo = text
}
}
return imageToVideo
}
@@ -200,14 +207,18 @@ export function extractClipMetaFromHistory(entry: unknown) {
if (!graph || typeof graph !== 'object') return meta
for (const node of Object.values(graph as Record<string, { class_type?: string; inputs?: Record<string, unknown> }>)) {
const inputs = node?.inputs || {}
if (node?.class_type === 'MiniMaxH3ImageToVideo') {
if (node?.class_type === 'MiniMaxH3ImageToVideo' || node?.class_type === 'MiniMaxH3TextToVideo' || node?.class_type === 'EmptyLTXVLatentVideo') {
meta.width = Number(inputs.width || meta.width)
meta.height = Number(inputs.height || meta.height)
}
if (node?.class_type === 'BasicScheduler' || node?.class_type === 'PrimitiveInt') {
if (node?.class_type === 'BasicScheduler' || node?.class_type === 'PrimitiveInt' || node?.class_type === 'LTXVScheduler') {
const steps = Number(inputs.steps || 0)
if (steps > 0) meta.steps = steps
}
if (node?.class_type === 'ImageResize+' && typeof inputs.width === 'number' && typeof inputs.height === 'number') {
meta.width = Number(inputs.width)
meta.height = Number(inputs.height)
}
if (node?.class_type === 'RandomNoise' || node?.class_type === 'SamplerMiniMax' || node?.class_type === 'KSampler') {
const seed = Number(inputs.noise_seed ?? inputs.seed ?? 0)
if (seed) meta.seed = seed
@@ -276,13 +287,16 @@ export function comfyFilenamePrefix() {
}
export function isOurComfyVideo(video: { filename: string; subfolder: string }) {
const prefix = comfyFilenamePrefix().replace(/\/$/, '')
const prefixes = [comfyFilenamePrefix(), 'video/LTX23']
return prefixes.some((raw) => {
const prefix = raw.replace(/\/$/, '')
const parts = prefix.split('/')
const namePrefix = parts[parts.length - 1]
const sub = parts.length > 1 ? parts.slice(0, -1).join('/') : 'video'
const nameOk = video.filename.startsWith(namePrefix)
const nameOk = video.filename.startsWith(namePrefix) || video.filename.startsWith('LTX23')
const subOk = !video.subfolder || video.subfolder === sub
return nameOk && subOk
})
}
export function purgeComfyEnabled() {
+1 -1
View File
@@ -85,7 +85,7 @@ export interface Job {
familyId?: string
parentClipId?: string
parentStillId?: string
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
chainContinuing?: boolean
passes?: { prompt: string }[]
queueId?: string
+4 -4
View File
@@ -38,7 +38,7 @@ export interface LibraryClip {
familyId?: string
parentClipId?: string
chainIndex?: number
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
stillId?: string
sound?: boolean
hideInput?: boolean
@@ -102,7 +102,7 @@ export interface RetryDraft {
samplerName?: string
scheduler?: string
extensions?: QueuedExtension[]
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
}
interface Catalog {
@@ -969,7 +969,7 @@ export async function saveRetryDraft(params: {
samplerName?: string
scheduler?: string
extensions?: QueuedExtension[]
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
}) {
return mutate(params.ownerKey, (catalog) => {
const existing = params.id ? catalog.drafts.find(item => item.id === params.id) : undefined
@@ -1134,7 +1134,7 @@ export async function saveClip(params: {
familyId?: string
parentClipId?: string
chainIndex?: number
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
stillId?: string
sound?: boolean
}) {
+1 -1
View File
@@ -35,7 +35,7 @@ export interface PendingJob {
referenceImageNames?: string[]
referenceStillIds?: Array<string | null>
useIdentityRefs?: boolean
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
duration?: number
cfg?: number
fps?: number
+7 -2
View File
@@ -1,5 +1,6 @@
import { getSidecarImageHost } from '~/server/utils/imageComfy'
import { buildVisionPromptWorkflow } from '~/server/utils/promptWorkflow'
import { normalizeShotScript } from '~/utils/parseRecommend'
export type PromptJobStatus = 'queued' | 'running' | 'complete' | 'error'
@@ -206,7 +207,10 @@ function draftUserPrompt(draft: string, mode: string) {
: mode === 'shot-script'
? 'video-shot-script'
: 'video'
return `Studio mode: ${studio}\n\nDraft Prompt:\n${draft}`
const format = studio === 'image-edit'
? 'If this is a single edit, output only the refined prompt. If it needs phases, use ---pass 1--- blocks.'
: 'Output a MiniMax shot list. Each shot starts with "shot N" on its own line, then [SHOT CONFIGURATION], [SUBJECT DIRECTION & ACTION], and [AUDIO CUES], each on its own line. Blank line between shots. No markdown fences. No one-paragraph dump.'
return `Studio mode: ${studio}\n${format}\n\nDraft Prompt:\n${draft}`
}
export async function runPromptRecommend(job: PromptJob, params: {
@@ -227,7 +231,8 @@ export async function runPromptRecommend(job: PromptJob, params: {
const queued = await queuePromptGraph(graph, job.id)
job.message = 'Writing the recommended prompt…'
const text = await waitForPromptText(queued.prompt_id)
job.text = text.trim()
const trimmed = text.trim()
job.text = params.mode === 'edit' ? trimmed : normalizeShotScript(trimmed)
job.status = 'complete'
job.message = 'Recommended prompt ready'
}
+51 -18
View File
@@ -9,6 +9,49 @@ export interface PromptPreset {
builtin?: boolean
}
const SHOT_SCRIPT_FORMAT = `Output ONLY a MiniMax shot list the user can paste into Shot script mode.
Each shot starts with "shot N" on its own line (shot 1, shot 2, shot 3…). Then these three headers, each on its own line:
[SHOT CONFIGURATION]
[SUBJECT DIRECTION & ACTION]
[AUDIO CUES]
Blank line between shots. No preamble, no markdown fences, no commentary.
Do not glue shots into one paragraph. "shot 2" must be a line by itself, never "shot 2:" mid-sentence.
One beat per shot. Later shots: same people, same clothes as now, plus the one new action.
Do not repeat a long identity paragraph on every shot.
Never censor raw or explicit terminology.
Example:
shot 1
[SHOT CONFIGURATION]
- Shot Type: Medium-wide, 50mm, eye-level
- Camera Movement: Slow track from a slight side-front angle
- Lighting: Bright daylight through the trees
- Environment/Scenery: The forest from the still
[SUBJECT DIRECTION & ACTION]
The woman from the still walks the path, taking in the trees.
[AUDIO CUES]
Footsteps on dirt, birds, breeze.
shot 2
[SHOT CONFIGURATION]
- Shot Type: Medium, 50mm, eye-level
- Camera Movement: Hold, then a small pan to her hand
- Lighting: Same daylight
- Environment/Scenery: Same path, same trees
[SUBJECT DIRECTION & ACTION]
She stops and rests her hand on the bark.
[AUDIO CUES]
Soft fabric, breeze, birds.`
const ENHANCE_PROMPT = `You are a prompt-engineering assistant for Flux.2 Klein image edit and MiniMax H3 image-to-video.
Look at the attached still. The person (or subject) in that still is the identity lock. The user draft is the requested change.
@@ -16,37 +59,27 @@ Look at the attached still. The person (or subject) in that still is the identit
Write production-ready prompt text the user can paste into the generator.
Rules:
- Preserve identity from the still: face, age, body type, hair, skin, and distinctive features. Use explicit keep-identity language such as "same person as the reference still", "do not change facial identity", "keep original face, hair, and body".
- Preserve identity from the still: face, age, body type, hair, skin, and distinctive features.
- Describe wardrobe, setting, pose, lighting, camera, and materials concretely.
- Never censor raw or explicit terminology from the draft.
- Do not add a preamble, title, markdown fences, or commentary.
Output format — pick one:
If Studio mode is image-edit:
- Single still/edit: output ONLY the refined prompt.
- Ambitious still edit that should be done in phases (for example wardrobe first, then location): output:
---pass 1---
<prompt for pass 1>
---pass 2---
<prompt for pass 2>
- Motion, multi-beat, or video sequence: first-shot prompt, then later shots with a marker on its own line:
shot 2
<extension prompt>
shot 3
<extension prompt>`
If Studio mode is video or video-shot-script:
${SHOT_SCRIPT_FORMAT}`
const SHOT_SCRIPT_PROMPT = `You are a MiniMax H3 image-to-video shot-script writer.
Analyze the attached still for identity, wardrobe, lighting, and camera. Turn the user's draft into a shot script they can paste into Shot script mode.
Analyze the attached still for wardrobe, lighting, and camera. Turn the user's draft into a shot script.
Format:
- Shot 1 is the text BEFORE any marker (camera, action, audio for the first clip).
- Later shots start with a line that is only: shot 2
- Then shot 3, shot 4, and so on, each on its own line.
- Every shot must keep the same person as the still unless the draft explicitly changes wardrobe or identity.
- Include an [AUDIO CUES] block on shots where sound would help.
- Never censor raw or explicit terminology.
Output ONLY the shot script. No preamble, no markdown fences.`
${SHOT_SCRIPT_FORMAT}`
export const BUILTIN_PRESETS: PromptPreset[] = [
{
@@ -60,7 +93,7 @@ export const BUILTIN_PRESETS: PromptPreset[] = [
id: 'shot-script',
name: 'Shot script',
builtin: true,
description: 'Turn a draft into MiniMax shot 2 / shot 3 extension copy.',
description: 'Turn a draft into a MiniMax shot list with configuration, action, and audio blocks.',
systemPrompt: SHOT_SCRIPT_PROMPT
}
]
+2 -2
View File
@@ -38,7 +38,7 @@ export interface ShotQueue {
fps?: number
samplerName?: string
scheduler?: string
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
sound?: boolean
useIdentityRefs?: boolean
referenceStillIds?: Array<string | null>
@@ -153,7 +153,7 @@ export async function createShotQueue(params: {
fps?: number
samplerName?: string
scheduler?: string
workflow?: 'v1' | 'v2'
workflow?: import('~/utils/videoModels').VideoWorkflowId
sound?: boolean
useIdentityRefs?: boolean
referenceStillIds?: Array<string | null>
+22 -13
View File
@@ -5,6 +5,7 @@ import { pendingFromJob, remainingAfterCurrentShot, writePendingJob } from '~/se
import { emitChainJob, waitForComfySocket, watchComfyJob } from '~/server/utils/watch'
import { comfyFilenamePrefix, queuePrompt, uploadImage } from '~/server/utils/comfy'
import { buildWorkflow } from '~/server/utils/workflow'
import { isLtxWorkflow, isTextToVideo, parseVideoWorkflow, workflowForExtension, type VideoWorkflowId } from '~/utils/videoModels'
import { clipVideoPath, clipTitle, deleteRetryDraft, extendTempDir, getClip, nextClipPartName, removeExtendTemp, stillPath } from '~/server/utils/library'
import { extractLastFrame, probeHasAudio } from '~/server/utils/ffmpeg'
import { ensureComfyReady } from '~/server/utils/comfyLifecycle'
@@ -23,7 +24,7 @@ export type ChainImage = { filename: string; data: Buffer; type?: string }
type VideoChainParams = {
prompt: string
image: ChainImage
image?: ChainImage | null
width: number
height: number
steps: number
@@ -36,7 +37,7 @@ type VideoChainParams = {
samplerName: string
scheduler: string
extensions: { prompt: string; duration: number }[]
workflow: 'v1' | 'v2'
workflow: VideoWorkflowId
duration: number
useIdentityRefs: boolean
referenceImages: Array<ChainImage | null>
@@ -76,7 +77,8 @@ function paramsFromJob(job: Job): VideoChainParams {
throw new Error('Cannot continue the shot chain: job metadata is missing')
}
const image = loadStillImage(library.ownerKey, library.stillId, library.stillFilename)
if (!image?.data.length) {
const workflow = parseVideoWorkflow(library.workflow)
if (!image?.data.length && !isTextToVideo(workflow)) {
throw new Error('Cannot continue the shot chain: the start still is missing from the library')
}
const referenceImages: Array<ChainImage | null> = [null, null, null, null]
@@ -101,7 +103,7 @@ function paramsFromJob(job: Job): VideoChainParams {
samplerName: library.samplerName || 'res_multistep',
scheduler: library.scheduler || 'simple',
extensions: library.extensions || [],
workflow: library.workflow || 'v1',
workflow,
duration,
useIdentityRefs: library.useIdentityRefs === true,
referenceImages
@@ -121,14 +123,21 @@ export async function queueMiniMax(
const done = watchComfyJob(job, { persist: params.persist })
job.status = 'uploading'
const chainIndex = job.library?.chainIndex || 0
const uploading = params.useIdentityRefs
const graphId = chainIndex > 0 ? workflowForExtension(params.workflow) : params.workflow
const engineName = isLtxWorkflow(graphId) ? 'LTX-2.3' : 'MiniMax H3'
const hasImage = Boolean(params.image?.data?.length)
const uploading = !hasImage
? `Queueing ${engineName} text-to-video…`
: params.useIdentityRefs
? (chainIndex > 0 ? 'Uploading identity stills for next shot...' : 'Uploading image to ComfyUI...')
: (chainIndex > 0 ? 'Uploading last frame to ComfyUI...' : 'Uploading image to ComfyUI...')
const queueing = (job.library?.chainIndex || 0) > 0
? 'Queueing extension on MiniMax H3...'
: 'Queueing MiniMax H3 job...'
const queueing = chainIndex > 0
? `Queueing extension on ${engineName}...`
: `Queueing ${engineName} job...`
emitChainJob(job, { type: 'status', message: uploading, progress: 4 })
const uploaded = await uploadImage(params.image, job.id)
const uploaded = hasImage && params.image
? await uploadImage(params.image, job.id)
: { name: '', subfolder: '' }
const referenceNames: string[] = ['', '', '', '']
if (params.useIdentityRefs) {
for (const [index, ref] of (params.referenceImages || []).entries()) {
@@ -161,9 +170,9 @@ export async function queueMiniMax(
fps: params.fps,
samplerName: params.samplerName,
scheduler: params.scheduler,
filenamePrefix: comfyFilenamePrefix(),
sound: params.sound,
workflow: params.workflow,
filenamePrefix: isLtxWorkflow(graphId) ? 'video/LTX23' : comfyFilenamePrefix(),
sound: params.sound && !isLtxWorkflow(graphId),
workflow: graphId,
duration: params.duration,
useIdentityRefs: params.useIdentityRefs,
referenceImageNames: params.useIdentityRefs ? referenceNames : []
@@ -317,7 +326,7 @@ export async function continueQueuedExtensions(
samplerName: params.samplerName,
scheduler: params.scheduler,
persist,
workflow: params.workflow,
workflow: workflowForExtension(params.workflow),
duration: ext.duration,
useIdentityRefs: params.useIdentityRefs,
referenceImages: params.useIdentityRefs ? params.referenceImages : []
+89 -11
View File
@@ -1,9 +1,18 @@
// Nitro bundles these JSON graphs into the production server output.
import workflowV1 from '../assets/workflow_minimax_video.json'
import workflowV2 from '../assets/workflow_minimax_video_v2.json'
import workflowLtx from '../assets/workflow_ltx_video.json'
import { buildIdentityPrompt } from '~/utils/identityPrompt'
import {
isLtxWorkflow,
isTextToVideo,
LTX_NEGATIVE,
parseVideoWorkflow,
type VideoWorkflowId
} from '~/utils/videoModels'
export type VideoWorkflowId = 'v1' | 'v2'
export type { VideoWorkflowId }
export { parseVideoWorkflow }
export interface GenerateParams {
prompt: string
@@ -31,6 +40,7 @@ type WorkflowGraph = Record<string, WorkflowNode>
const PROMPT_CLASSES = new Set([
'MiniMaxH3ImageToVideo',
'MiniMaxH3TextToVideo',
'MiniMaxH3ReferenceToVideo',
'CLIPTextEncode',
'CLIPTextEncodeQwen3VL',
@@ -58,16 +68,15 @@ function titleOf(node: WorkflowNode) {
}
function templateFor(id: VideoWorkflowId) {
if (isLtxWorkflow(id)) return workflowLtx as WorkflowGraph
return (id === 'v2' ? workflowV2 : workflowV1) as WorkflowGraph
}
export function parseVideoWorkflow(raw: unknown): VideoWorkflowId {
return String(raw || '').trim() === 'v2' ? 'v2' : 'v1'
}
export function buildWorkflow(params: GenerateParams) {
const version = params.workflow === 'v2' ? 'v2' : 'v1'
const graph = structuredClone(templateFor(version))
const version = parseVideoWorkflow(params.workflow)
if (isLtxWorkflow(version)) return buildLtxWorkflow(params, version)
const graph = structuredClone(templateFor(version === 't2v' ? 'v1' : version === 'v2' ? 'v2' : 'v1'))
const width = snap32(params.width)
const height = snap32(params.height)
const refs = Array.from({ length: 4 }, (_, index) => String(params.referenceImageNames?.[index] || '').trim())
@@ -155,7 +164,7 @@ export function buildWorkflow(params: GenerateParams) {
}
}
if (version === 'v1' && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
if (version === 'v1' && !isTextToVideo(version) && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
graph['128'] = {
class_type: 'ImageScale',
inputs: {
@@ -169,7 +178,7 @@ export function buildWorkflow(params: GenerateParams) {
}
}
if (version === 'v1') {
if (version === 'v1' || version === 't2v') {
graph['105:cfg'] = {
class_type: 'FluxGuidance',
inputs: {
@@ -191,10 +200,11 @@ export function buildWorkflow(params: GenerateParams) {
graph[qualityId].inputs.value = params.turbo ? 20 : params.steps
}
if (turboId && graph[turboId]?.class_type === 'PrimitiveInt') {
graph[turboId].inputs.value = params.turbo ? params.steps : (params.workflow === 'v2' ? 6 : 8)
graph[turboId].inputs.value = params.turbo ? params.steps : (version === 'v2' ? 6 : 8)
}
}
if (version === 't2v') applyMinimaxTextToVideo(graph, params, width, height)
if (version === 'v2') applyV2IdentityPath(graph, params, useRefs, refs, graphPrompt)
if (params.sound === false) {
@@ -203,6 +213,7 @@ export function buildWorkflow(params: GenerateParams) {
if (graph['105:91']?.inputs) delete graph['105:91'].inputs.audio
delete graph['163']
if (graph['172']?.inputs) delete graph['172'].inputs.audio
if (graph['105:104']?.inputs) delete graph['105:104'].inputs.audio_vae
}
return graph
@@ -210,10 +221,76 @@ export function buildWorkflow(params: GenerateParams) {
function promptForNode(classType: string, useRefs: boolean, actionPrompt: string, graphPrompt: string) {
if (classType === 'MiniMaxH3ImageToVideo') return useRefs ? '' : actionPrompt
if (classType === 'MiniMaxH3TextToVideo') return actionPrompt
if (classType === 'MiniMaxH3ReferenceToVideo') return useRefs ? graphPrompt : actionPrompt
return graphPrompt
}
function applyMinimaxTextToVideo(graph: WorkflowGraph, params: GenerateParams, width: number, height: number) {
const node = graph['105:104']
if (node) {
node.class_type = 'MiniMaxH3TextToVideo'
node._meta = { title: 'MiniMax H3 Text to Video' }
delete node.inputs.first_frame
node.inputs.prompt = params.prompt
node.inputs.width = width
node.inputs.height = height
if (params.sound !== false && graph['105:24']) node.inputs.audio_vae = ['105:24', 0]
else delete node.inputs.audio_vae
}
delete graph['114']
delete graph['119']
delete graph['120']
delete graph['127']
delete graph['128']
}
function buildLtxWorkflow(params: GenerateParams, version: VideoWorkflowId) {
const graph = structuredClone(workflowLtx as WorkflowGraph)
const width = snap32(params.width)
const height = snap32(params.height)
const textToVideo = isTextToVideo(version)
if (graph['4']?.inputs) graph['4'].inputs.image = params.imageName
if (graph['5']?.inputs) {
graph['5'].inputs.width = width
graph['5'].inputs.height = height
}
if (graph['6']?.inputs) graph['6'].inputs.text = params.prompt
if (graph['7']?.inputs) graph['7'].inputs.text = LTX_NEGATIVE
if (graph['8']?.inputs && typeof graph['8'].inputs.length === 'number') graph['8'].inputs.length = params.length
if (graph['9']?.inputs) graph['9'].inputs.frame_rate = params.fps
if (graph['10']?.inputs) graph['10'].inputs.cfg = params.cfg
if (graph['11']?.inputs) graph['11'].inputs.sampler_name = params.samplerName || 'euler'
if (graph['12']?.inputs) graph['12'].inputs.steps = params.steps
if (graph['13']?.inputs) graph['13'].inputs.noise_seed = params.seed
if (graph['16']?.inputs) {
graph['16'].inputs.frame_rate = params.fps
graph['16'].inputs.filename_prefix = params.filenamePrefix || 'video/LTX23'
}
if (graph['21']?.inputs) graph['21'].inputs.strength_model = 1
if (textToVideo) {
delete graph['4']
delete graph['5']
graph['8'] = {
class_type: 'EmptyLTXVLatentVideo',
inputs: {
width,
height,
length: params.length,
batch_size: 1
},
_meta: { title: 'Empty LTX latent' }
}
if (graph['9']?.inputs) {
graph['9'].inputs.positive = ['6', 0]
graph['9'].inputs.negative = ['7', 0]
}
if (graph['12']?.inputs) graph['12'].inputs.latent = ['8', 0]
if (graph['14']?.inputs) graph['14'].inputs.latent_image = ['8', 0]
}
return graph
}
function applyV2IdentityPath(
graph: WorkflowGraph,
params: GenerateParams,
@@ -304,9 +381,10 @@ function labelsFrom(template: WorkflowGraph) {
export const NODE_LABELS: Record<string, string> = {
...Object.fromEntries(labelsFrom(workflowV1 as WorkflowGraph)),
...Object.fromEntries(labelsFrom(workflowV2 as WorkflowGraph)),
...Object.fromEntries(labelsFrom(workflowLtx as WorkflowGraph)),
...LABEL_OVERRIDES
}
export function isEncodingNode(node: string) {
return node === '92' || node === '105:91' || node === '172' || /encoding|saving mp4|create video/i.test(NODE_LABELS[node] || '')
return node === '92' || node === '16' || node === '105:91' || node === '172' || /encoding|saving mp4|create video|save mp4/i.test(NODE_LABELS[node] || '')
}
+39
View File
@@ -1,3 +1,5 @@
import { parseShotScript } from './parseShots'
export function parseEditPasses(text: string): string[] | null {
const src = String(text || '').replace(/\r\n/g, '\n')
const marker = /^[ \t]*---pass\s+(\d+)---[ \t]*$/gim
@@ -21,3 +23,40 @@ export function parseEditPasses(text: string): string[] | null {
export function looksLikeShotScript(text: string) {
return /^[ \t]*shot\s+[2-9]\d*\b/im.test(String(text || ''))
}
function stripFences(text: string) {
return String(text || '')
.replace(/^\s*```(?:[\w-]+)?\s*\n?/, '')
.replace(/\n?```\s*$/, '')
.trim()
}
function splitInlineShotMarkers(text: string) {
return String(text || '')
.replace(/\r\n/g, '\n')
.replace(/(^|[^A-Za-z])shot\s+(\d+)\s*[:.)\-]?\s*/gi, '$1\n\nshot $2\n')
.replace(/\n{3,}/g, '\n\n')
.trim()
}
function formatShotBody(body: string) {
let text = String(body || '').trim()
text = text.replace(/^\s*[:.\-–)]\s*/, '')
text = text.replace(/\s*\[SHOT CONFIGURATION\]\s*/gi, '\n\n[SHOT CONFIGURATION]\n')
text = text.replace(/\s*\[SUBJECT DIRECTION(?:\s*&\s*ACTION)?\]\s*/gi, '\n\n[SUBJECT DIRECTION & ACTION]\n')
text = text.replace(/\s*\[AUDIO CUES\]\s*/gi, '\n\n[AUDIO CUES]\n')
text = text.replace(/[ \t]+\n/g, '\n').replace(/\n{3,}/g, '\n\n').trim()
return text
}
export function normalizeShotScript(text: string) {
const src = splitInlineShotMarkers(stripFences(text))
const shots = parseShotScript(src)
if (!shots.length) return src
if (shots.length === 1 && !/^[ \t]*shot\s+\d+\b/im.test(src)) {
return formatShotBody(shots[0].prompt)
}
return shots
.map(shot => `shot ${shot.n}\n${formatShotBody(shot.prompt)}`.trim())
.join('\n\n')
}
+56
View File
@@ -0,0 +1,56 @@
export type VideoEngineId = 'minimax' | 'ltx'
export type VideoStartId = 'still' | 'text'
export type MiniMaxGraphId = 'v1' | 'v2'
export type VideoWorkflowId = 'v1' | 'v2' | 't2v' | 'ltx' | 'ltx-t2v'
export function parseVideoWorkflow(raw: unknown): VideoWorkflowId {
const value = String(raw || '').trim()
if (value === 'v2') return 'v2'
if (value === 't2v') return 't2v'
if (value === 'ltx' || value === 'ltx-i2v') return 'ltx'
if (value === 'ltx-t2v') return 'ltx-t2v'
return 'v1'
}
export function videoEngineOf(id: VideoWorkflowId): VideoEngineId {
return id.startsWith('ltx') ? 'ltx' : 'minimax'
}
export function videoStartOf(id: VideoWorkflowId): VideoStartId {
return id === 't2v' || id === 'ltx-t2v' ? 'text' : 'still'
}
export function isLtxWorkflow(id: VideoWorkflowId) {
return videoEngineOf(id) === 'ltx'
}
export function isTextToVideo(id: VideoWorkflowId) {
return videoStartOf(id) === 'text'
}
export function minimaxGraphOf(id: VideoWorkflowId): MiniMaxGraphId {
return id === 'v2' ? 'v2' : 'v1'
}
export function workflowForExtension(id: VideoWorkflowId): VideoWorkflowId {
if (id === 't2v') return 'v1'
if (id === 'ltx-t2v') return 'ltx'
return id
}
export function composeVideoWorkflow(engine: VideoEngineId, start: VideoStartId, graph: MiniMaxGraphId = 'v1'): VideoWorkflowId {
if (engine === 'ltx') return start === 'text' ? 'ltx-t2v' : 'ltx'
if (start === 'text') return 't2v'
return graph === 'v2' ? 'v2' : 'v1'
}
export function defaultVideoSteps(turbo: boolean, workflow: string = 'v1') {
const id = parseVideoWorkflow(workflow)
if (!turbo) return isLtxWorkflow(id) ? 12 : 20
if (isLtxWorkflow(id)) return 8
return id === 'v2' ? 6 : 8
}
export const LTX_UNET = 'PinkCherry_FineTune_int8_v1_8_LTX23.safetensors'
export const LTX_DISTILLED_LORA = 'ltx-2.3-22b-distilled-lora-384-1.1.safetensors'
export const LTX_NEGATIVE = 'blurry, low quality, jitter, deformed, static frame, freeze, watermark, bad anatomy, text, subtitles'
-4
View File
@@ -1,4 +0,0 @@
export function defaultVideoSteps(turbo: boolean, workflow = 'v1') {
if (!turbo) return 20
return workflow === 'v2' ? 6 : 8
}