Add MiniMax and PinkCherry LTX as pickable video engines, with still or text start.

Shot 1 can run text-to-video; later shots still extend from the last frame. Recommend output is also normalized into shot-script cards.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-08-27 19:38:53 -05:00
co-authored by Cursor
parent 2bebeea526
commit 40cd17f35c
15 changed files with 676 additions and 120 deletions
+178 -28
View File
@@ -6,7 +6,7 @@
<div class="flex h-10 w-10 items-center justify-center rounded-xl bg-amber-400 text-zinc-950 font-display font-extrabold">A</div> <div class="flex h-10 w-10 items-center justify-center rounded-xl bg-amber-400 text-zinc-950 font-display font-extrabold">A</div>
<div> <div>
<h1 class="font-display text-xl font-bold leading-none">{{ instanceName }}</h1> <h1 class="font-display text-xl font-bold leading-none">{{ instanceName }}</h1>
<p class="text-[11px] uppercase tracking-[0.22em] text-zinc-500">MiniMax H3 · Flux.2 Klein</p> <p class="text-[11px] uppercase tracking-[0.22em] text-zinc-500">MiniMax H3 · LTX-2.3 · Flux.2 Klein</p>
</div> </div>
</div> </div>
<div class="flex items-center gap-3 text-sm"> <div class="flex items-center gap-3 text-sm">
@@ -80,8 +80,8 @@
<button type="button" class="relative mt-3 rounded-full bg-zinc-950/80 px-3 py-1 text-xs" @click.stop="resetImage">Reset</button> <button type="button" class="relative mt-3 rounded-full bg-zinc-950/80 px-3 py-1 text-xs" @click.stop="resetImage">Reset</button>
</div> </div>
<div v-else class="flex min-h-44 flex-col items-center justify-center text-center"> <div v-else class="flex min-h-44 flex-col items-center justify-center text-center">
<p class="font-medium">{{ studioMode === 'edit' ? 'Image 1' : 'Choose an image' }}</p> <p class="font-medium">{{ studioMode === 'edit' ? 'Image 1' : (textToVideo ? 'Start still · optional' : 'Choose an image') }}</p>
<p class="mt-1 text-sm text-zinc-500">Browse the library or upload a new PNG, JPG, or WEBP</p> <p class="mt-1 text-sm text-zinc-500">{{ textToVideo ? 'Text-to-video does not need a still. Shot 2+ will still extend from the last frame.' : 'Browse the library or upload a new PNG, JPG, or WEBP' }}</p>
</div> </div>
</div> </div>
@@ -183,6 +183,16 @@
<p class="text-xs font-medium uppercase tracking-wide text-amber-200/80">Recommended prompt</p> <p class="text-xs font-medium uppercase tracking-wide text-amber-200/80">Recommended prompt</p>
<p v-if="recommendBusy" class="mt-2 text-sm text-zinc-400">{{ recommendMessage || 'Sidecar Qwen VL is reading the still…' }}</p> <p v-if="recommendBusy" class="mt-2 text-sm text-zinc-400">{{ recommendMessage || 'Sidecar Qwen VL is reading the still…' }}</p>
<p v-else-if="recommendError" class="mt-2 text-sm text-red-300">{{ recommendError }}</p> <p v-else-if="recommendError" class="mt-2 text-sm text-red-300">{{ recommendError }}</p>
<div v-else-if="recommendedShots.length > 1" class="mt-2 space-y-2">
<div
v-for="shot in recommendedShots"
:key="shot.n"
class="rounded-xl border border-amber-300/15 bg-zinc-950/50 px-3 py-2"
>
<p class="text-[11px] font-medium uppercase tracking-wide text-amber-200/80">shot {{ shot.n }}</p>
<p class="mt-1 whitespace-pre-wrap text-sm text-zinc-200">{{ shot.prompt }}</p>
</div>
</div>
<p v-else class="mt-2 whitespace-pre-wrap text-sm text-zinc-200">{{ recommendText }}</p> <p v-else class="mt-2 whitespace-pre-wrap text-sm text-zinc-200">{{ recommendText }}</p>
<div v-if="recommendText && !recommendBusy" class="mt-3 flex flex-wrap gap-2"> <div v-if="recommendText && !recommendBusy" class="mt-3 flex flex-wrap gap-2">
<button <button
@@ -343,7 +353,7 @@
<span class="absolute left-0.5 top-0.5 h-5 w-5 rounded-full bg-white transition peer-checked:translate-x-5" /> <span class="absolute left-0.5 top-0.5 h-5 w-5 rounded-full bg-white transition peer-checked:translate-x-5" />
</span> </span>
</label> </label>
<label v-if="studioMode === 'video'" class="flex cursor-pointer items-center justify-between gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm"> <label v-if="studioMode === 'video' && !ltxVideo" class="flex cursor-pointer items-center justify-between gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm">
<span> <span>
<span class="block font-medium text-zinc-200">Generate sound</span> <span class="block font-medium text-zinc-200">Generate sound</span>
<span class="mt-0.5 block text-xs text-zinc-500">Off skips the audio VAE so Comfy encodes a silent clip.</span> <span class="mt-0.5 block text-xs text-zinc-500">Off skips the audio VAE so Comfy encodes a silent clip.</span>
@@ -371,13 +381,60 @@
<div v-if="studioMode === 'video'" class="space-y-3"> <div v-if="studioMode === 'video'" class="space-y-3">
<div> <div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Video model</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoEngine === 'minimax' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="selectVideoEngine('minimax')"
>
<span class="block font-semibold">MiniMax H3</span>
<span class="text-xs text-zinc-400">Native stereo audio. Identity refs on v2.</span>
</button>
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoEngine === 'ltx' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="selectVideoEngine('ltx')"
>
<span class="block font-semibold">LTX-2.3 PinkCherry</span>
<span class="text-xs text-zinc-400">PinkCherry + distilled LoRA. I2V and T2V.</span>
</button>
</div>
</div>
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Start from</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoStart === 'still' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoStart = 'still'"
>
<span class="block font-semibold">A still</span>
<span class="text-xs text-zinc-400">Image to video. Extensions use the last frame.</span>
</button>
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoStart === 'text' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoStart = 'text'"
>
<span class="block font-semibold">Text only</span>
<span class="text-xs text-zinc-400">No start still. Shot 2+ still continues from last frame.</span>
</button>
</div>
<p class="mt-2 text-xs text-zinc-500">{{ videoModelHint }}</p>
</div>
<div v-if="videoEngine === 'minimax' && videoStart === 'still'">
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">MiniMax graph</p> <p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">MiniMax graph</p>
<div class="grid grid-cols-2 gap-2"> <div class="grid grid-cols-2 gap-2">
<button <button
type="button" type="button"
class="rounded-xl border px-3 py-2 text-left text-sm" class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoWorkflow === 'v1' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'" :class="minimaxGraph === 'v1' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoWorkflow = 'v1'" @click="minimaxGraph = 'v1'"
> >
<span class="block font-semibold">v1 · validated</span> <span class="block font-semibold">v1 · validated</span>
<span class="text-xs text-zinc-400">First-frame image to video</span> <span class="text-xs text-zinc-400">First-frame image to video</span>
@@ -385,8 +442,8 @@
<button <button
type="button" type="button"
class="rounded-xl border px-3 py-2 text-left text-sm" class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoWorkflow === 'v2' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'" :class="minimaxGraph === 'v2' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoWorkflow = 'v2'" @click="minimaxGraph = 'v2'"
> >
<span class="block font-semibold">v2 · identity refs</span> <span class="block font-semibold">v2 · identity refs</span>
<span class="text-xs text-zinc-400">Optional lock with up to 4 stills</span> <span class="text-xs text-zinc-400">Optional lock with up to 4 stills</span>
@@ -394,7 +451,7 @@
</div> </div>
</div> </div>
<div v-if="videoWorkflow === 'v2'" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 space-y-3"> <div v-if="videoEngine === 'minimax' && videoStart === 'still' && minimaxGraph === 'v2'" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 space-y-3">
<label class="flex cursor-pointer items-center justify-between gap-3 text-sm"> <label class="flex cursor-pointer items-center justify-between gap-3 text-sm">
<span> <span>
<span class="block font-medium text-zinc-200">Use identity references</span> <span class="block font-medium text-zinc-200">Use identity references</span>
@@ -466,13 +523,13 @@
<span class="text-xs text-zinc-400">{{ option.id === 'auto' ? autoHint : option.hint }}</span> <span class="text-xs text-zinc-400">{{ option.id === 'auto' ? autoHint : option.hint }}</span>
</button> </button>
</div> </div>
<p class="mt-2 text-xs text-zinc-500">Output {{ outputSize.width }} × {{ outputSize.height }}{{ aspect === 'auto' ? ' · matching the still' : '' }}</p> <p class="mt-2 text-xs text-zinc-500">Output {{ outputSize.width }} × {{ outputSize.height }}{{ aspect === 'auto' ? (file ? ' · matching the still' : ' · 16:9 until a still is loaded') : '' }}</p>
</div> </div>
<div> <div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Steps</p> <p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Steps</p>
<div class="grid grid-cols-2 gap-2"> <div class="grid grid-cols-2 gap-2">
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = true">{{ videoWorkflow === 'v2' ? 6 : 8 }} Turbo · LoRA on</button> <button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = true">{{ ltxVideo ? '8 Distilled LoRA' : (videoWorkflow === 'v2' ? '6 Turbo · LoRA on' : '8 Turbo · LoRA on') }}</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="!turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = false">20 High-Fidelity · LoRA bypass</button> <button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="!turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = false">{{ ltxVideo ? '12 Quality · LoRA on' : '20 High-Fidelity · LoRA bypass' }}</button>
</div> </div>
</div> </div>
<div class="text-sm"> <div class="text-sm">
@@ -499,7 +556,7 @@
@change="onCfgNumber" @change="onCfgNumber"
> >
</div> </div>
<p class="mt-1 text-[11px] text-zinc-500">{{ turbo ? 'Default 1.5 with Turbo LoRA' : 'Default 4.0 with High-Fidelity' }}</p> <p class="mt-1 text-[11px] text-zinc-500">{{ ltxVideo ? 'Default 3.5 for PinkCherry' : (turbo ? 'Default 1.5 with Turbo LoRA' : 'Default 4.0 with High-Fidelity') }}</p>
</div> </div>
<div class="grid grid-cols-2 gap-3"> <div class="grid grid-cols-2 gap-3">
<label class="text-sm"> <label class="text-sm">
@@ -1256,6 +1313,19 @@
<script setup lang="ts"> <script setup lang="ts">
import { parseEditPasses, looksLikeShotScript } from '~/utils/parseRecommend' import { parseEditPasses, looksLikeShotScript } from '~/utils/parseRecommend'
import { parseShotScript } from '~/utils/parseShots'
import {
composeVideoWorkflow,
isLtxWorkflow,
isTextToVideo,
minimaxGraphOf,
parseVideoWorkflow,
videoEngineOf,
videoStartOf,
type MiniMaxGraphId,
type VideoEngineId,
type VideoStartId
} from '~/utils/videoModels'
const runtimeConfig = useRuntimeConfig() const runtimeConfig = useRuntimeConfig()
@@ -1311,7 +1381,7 @@ interface LibraryClip {
familyId?: string familyId?: string
parentClipId?: string parentClipId?: string
chainIndex?: number chainIndex?: number
workflow?: 'v1' | 'v2' workflow?: string
stillId?: string stillId?: string
sound?: boolean sound?: boolean
hideInput?: boolean hideInput?: boolean
@@ -1340,7 +1410,7 @@ interface RetryDraft {
samplerName?: string samplerName?: string
scheduler?: string scheduler?: string
extensions?: { prompt: string; duration: number }[] extensions?: { prompt: string; duration: number }[]
workflow?: 'v1' | 'v2' workflow?: string
} }
interface QueuedExtension { interface QueuedExtension {
@@ -1517,7 +1587,12 @@ const pickerSlot = ref<'main' | 'editRef' | number>('main')
const libraryFilter = ref<'all' | 'images' | 'videos'>('all') const libraryFilter = ref<'all' | 'images' | 'videos'>('all')
const selectedKeys = ref<string[]>([]) const selectedKeys = ref<string[]>([])
const lastSelectedKey = ref('') const lastSelectedKey = ref('')
const videoWorkflow = ref('v1') const videoEngine = ref<VideoEngineId>('minimax')
const videoStart = ref<VideoStartId>('still')
const minimaxGraph = ref<MiniMaxGraphId>('v1')
const videoWorkflow = computed(() => composeVideoWorkflow(videoEngine.value, videoStart.value, minimaxGraph.value))
const textToVideo = computed(() => isTextToVideo(videoWorkflow.value))
const ltxVideo = computed(() => isLtxWorkflow(videoWorkflow.value))
const useIdentityRefs = ref(false) const useIdentityRefs = ref(false)
const identityRefs = ref<(File | null)[]>([null, null, null, null]) const identityRefs = ref<(File | null)[]>([null, null, null, null])
const identityPreviews = ref(['', '', '', '']) const identityPreviews = ref(['', '', '', ''])
@@ -1653,10 +1728,22 @@ const outputSize = computed(() => {
}) })
const autoHint = computed(() => { const autoHint = computed(() => {
if (!imageWidth.value || !imageHeight.value) return 'Match the still' if (!imageWidth.value || !imageHeight.value) {
return textToVideo.value ? '16:9 without a still' : 'Match the still'
}
const size = autoResolution(imageWidth.value, imageHeight.value) const size = autoResolution(imageWidth.value, imageHeight.value)
return `${size.width} × ${size.height} · from still` return `${size.width} × ${size.height} · from still`
}) })
const videoModelHint = computed(() => {
if (ltxVideo.value) {
return textToVideo.value
? 'PinkCherry text-to-video. Distilled LoRA stays on. Silent MP4 — LTX has no audio VAE here.'
: 'PinkCherry image-to-video. Distilled LoRA stays on. Silent MP4 — LTX has no audio VAE here.'
}
if (textToVideo.value) return 'MiniMax H3 text-to-video with native stereo audio. Shot 2+ switches to last-frame image-to-video.'
if (videoWorkflow.value === 'v2') return 'MiniMax H3 image-to-video. Identity refs stay on v2 with a start still.'
return 'MiniMax H3 image-to-video with native stereo audio.'
})
const frameCount = computed(() => Math.max(5, Math.floor(duration.value * fps.value))) const frameCount = computed(() => Math.max(5, Math.floor(duration.value * fps.value)))
const cfgPct = computed(() => ((cfg.value - CFG_MIN) / (CFG_MAX - CFG_MIN)) * 100) const cfgPct = computed(() => ((cfg.value - CFG_MIN) / (CFG_MAX - CFG_MIN)) * 100)
const currentClip = computed(() => clips.value.find(clip => clip.id === currentClipId.value)) const currentClip = computed(() => clips.value.find(clip => clip.id === currentClipId.value))
@@ -1671,7 +1758,7 @@ const queueReady = computed(() => {
return extensionQueue.value.every(item => item.prompt.trim()) return extensionQueue.value.every(item => item.prompt.trim())
}) })
const videoGenerateDisabled = computed(() => { const videoGenerateDisabled = computed(() => {
return videoBusy.value || !file.value || !prompt.value.trim() || !folderId.value || !queueReady.value return videoBusy.value || (!textToVideo.value && !file.value) || !prompt.value.trim() || !folderId.value || !queueReady.value
}) })
const recommendDisabled = computed(() => { const recommendDisabled = computed(() => {
if (recommendBusy.value) return true if (recommendBusy.value) return true
@@ -1715,6 +1802,10 @@ const editLabel = computed(() => {
return n === 1 ? 'Edit image + 1 pass' : `Edit image + ${n} passes` return n === 1 ? 'Edit image + 1 pass' : `Edit image + ${n} passes`
}) })
const parsedShots = computed(() => parseShotScript(prompt.value)) const parsedShots = computed(() => parseShotScript(prompt.value))
const recommendedShots = computed(() => {
if (!recommendText.value || recommendBusy.value) return []
return parseShotScript(recommendText.value)
})
const identityPrompting = computed(() => studioMode.value === 'video' && videoWorkflow.value === 'v2' && useIdentityRefs.value) const identityPrompting = computed(() => studioMode.value === 'video' && videoWorkflow.value === 'v2' && useIdentityRefs.value)
const extraIdentityPictures = computed(() => identityRefs.value.flatMap((item, index) => item ? [index + 2] : [])) const extraIdentityPictures = computed(() => identityRefs.value.flatMap((item, index) => item ? [index + 2] : []))
const identityActionPrompt = computed(() => { const identityActionPrompt = computed(() => {
@@ -2101,7 +2192,20 @@ onMounted(async () => {
hideInputPreview.value = localStorage.getItem('aigen-hide-input-preview') === 'true' hideInputPreview.value = localStorage.getItem('aigen-hide-input-preview') === 'true'
withSound.value = localStorage.getItem('aigen-generate-sound') !== 'false' withSound.value = localStorage.getItem('aigen-generate-sound') !== 'false'
autoplayEnabled.value = localStorage.getItem('aigen-autoplay') === 'true' autoplayEnabled.value = localStorage.getItem('aigen-autoplay') === 'true'
videoWorkflow.value = localStorage.getItem('aigen-video-workflow') === 'v2' ? 'v2' : 'v1' videoEngine.value = 'minimax'
videoStart.value = 'still'
minimaxGraph.value = 'v1'
try {
const saved = parseVideoWorkflow(localStorage.getItem('aigen-video-workflow') || '')
videoEngine.value = videoEngineOf(saved)
videoStart.value = videoStartOf(saved)
minimaxGraph.value = minimaxGraphOf(saved)
const engine = localStorage.getItem('aigen-video-engine')
const start = localStorage.getItem('aigen-video-start')
if (engine === 'ltx' || engine === 'minimax') videoEngine.value = engine
if (start === 'text' || start === 'still') videoStart.value = start
if (videoEngine.value === 'ltx') applyEngineDefaults('ltx')
} catch { /* ignore */ }
useIdentityRefs.value = localStorage.getItem('aigen-use-identity-refs') === 'true' useIdentityRefs.value = localStorage.getItem('aigen-use-identity-refs') === 'true'
activeQueueId.value = localStorage.getItem(QUEUE_STORE) || '' activeQueueId.value = localStorage.getItem(QUEUE_STORE) || ''
const savedPreset = localStorage.getItem('aigen-prompt-preset') const savedPreset = localStorage.getItem('aigen-prompt-preset')
@@ -2190,6 +2294,18 @@ watch(videoWorkflow, (value) => {
} catch { /* ignore */ } } catch { /* ignore */ }
}) })
watch(videoEngine, (value) => {
try {
localStorage.setItem('aigen-video-engine', value)
} catch { /* ignore */ }
})
watch(videoStart, (value) => {
try {
localStorage.setItem('aigen-video-start', value)
} catch { /* ignore */ }
})
watch(useIdentityRefs, (value) => { watch(useIdentityRefs, (value) => {
try { try {
localStorage.setItem('aigen-use-identity-refs', String(value)) localStorage.setItem('aigen-use-identity-refs', String(value))
@@ -2197,7 +2313,8 @@ watch(useIdentityRefs, (value) => {
}) })
watch(turbo, (on) => { watch(turbo, (on) => {
if (!cfgTouched.value) cfg.value = on ? CFG_TURBO : CFG_QUALITY if (cfgTouched.value) return
cfg.value = ltxVideo.value ? 3.5 : (on ? CFG_TURBO : CFG_QUALITY)
}) })
function clampDuration(value: number) { function clampDuration(value: number) {
@@ -2336,7 +2453,7 @@ function applyRecommend() {
} }
} else { } else {
prompt.value = text prompt.value = text
if (looksLikeShotScript(text)) shotScriptMode.value = true if (looksLikeShotScript(text) || parseShotScript(text).length > 1) shotScriptMode.value = true
} }
recommendApplied.value = true recommendApplied.value = true
toast('Recommended prompt applied') toast('Recommended prompt applied')
@@ -2709,7 +2826,7 @@ async function rerun(item: LibraryClip, collection = false) {
if (showPrivacyToggles.value) hideInputPreview.value = target.hideInput === true if (showPrivacyToggles.value) hideInputPreview.value = target.hideInput === true
folderId.value = target.folderId folderId.value = target.folderId
browseFolderId.value = target.folderId browseFolderId.value = target.folderId
videoWorkflow.value = target.workflow === 'v2' ? 'v2' : 'v1' applyVideoWorkflow(target.workflow)
if (typeof target.sound === 'boolean') withSound.value = target.sound if (typeof target.sound === 'boolean') withSound.value = target.sound
shotScriptMode.value = restoreAll shotScriptMode.value = restoreAll
useIdentityRefs.value = false useIdentityRefs.value = false
@@ -2722,7 +2839,8 @@ async function rerun(item: LibraryClip, collection = false) {
})) }))
: [] : []
try { try {
await loadRerunImage(target, parts) const initialTextToVideo = textToVideo.value && !target.parentClipId && !(target.chainIndex)
if (!(initialTextToVideo && !target.stillId)) await loadRerunImage(target, parts)
statusMessage.value = restoreAll statusMessage.value = restoreAll
? 'Collection loaded into the form. Generate when you are ready.' ? 'Collection loaded into the form. Generate when you are ready.'
: 'Clip loaded into the form. Generate when you are ready.' : 'Clip loaded into the form. Generate when you are ready.'
@@ -2730,8 +2848,12 @@ async function rerun(item: LibraryClip, collection = false) {
? 'Shot script and settings restored. Make any changes, then generate.' ? 'Shot script and settings restored. Make any changes, then generate.'
: 'Inputs restored. Make any changes, then generate.') : 'Inputs restored. Make any changes, then generate.')
} catch { } catch {
statusMessage.value = 'Clip settings loaded. Re-drop the start still if it is missing.' statusMessage.value = textToVideo.value
toast('Settings restored, but the start still could not be loaded. Pick it again.') ? 'Clip settings loaded. Add a still only if you want image-to-video.'
: 'Clip settings loaded. Re-drop the start still if it is missing.'
toast(textToVideo.value
? 'Settings restored. Text-to-video does not need a start still.'
: 'Settings restored, but the start still could not be loaded. Pick it again.')
} }
} }
@@ -2746,7 +2868,7 @@ async function loadDraft(draft: RetryDraft) {
withSound.value = draft.sound !== false withSound.value = draft.sound !== false
if (draft.hideInput && showPrivacyToggles.value) hideInputPreview.value = true if (draft.hideInput && showPrivacyToggles.value) hideInputPreview.value = true
duration.value = clampDuration(Number(draft.duration)) duration.value = clampDuration(Number(draft.duration))
if (draft.workflow === 'v2') videoWorkflow.value = 'v2' applyVideoWorkflow(draft.workflow)
extensionQueue.value = (draft.extensions || []).map(item => ({ extensionQueue.value = (draft.extensions || []).map(item => ({
id: crypto.randomUUID(), id: crypto.randomUUID(),
prompt: String(item.prompt || ''), prompt: String(item.prompt || ''),
@@ -2758,9 +2880,13 @@ async function loadDraft(draft: RetryDraft) {
readFile(new File([blob], filename, { type: blob.type || 'image/png' }), { keepThumbnailPref: true }) readFile(new File([blob], filename, { type: blob.type || 'image/png' }), { keepThumbnailPref: true })
statusMessage.value = 'Held job restored. Generate when ComfyUI is free.' statusMessage.value = 'Held job restored. Generate when ComfyUI is free.'
} catch { } catch {
if (textToVideo.value) {
statusMessage.value = 'Held job restored. Generate when ComfyUI is free.'
} else {
statusMessage.value = 'Held job settings restored. Re-drop the still if the image is missing.' statusMessage.value = 'Held job settings restored. Re-drop the still if the image is missing.'
toast('Settings restored. Unlock the folder or re-drop the still if the image is missing.') toast('Settings restored. Unlock the folder or re-drop the still if the image is missing.')
} }
}
} }
async function dismissDraft(draft: RetryDraft) { async function dismissDraft(draft: RetryDraft) {
@@ -3363,8 +3489,32 @@ async function useEditAsInput() {
} }
} }
function applyEngineDefaults(engine: VideoEngineId) {
if (engine === 'ltx') {
samplerName.value = 'euler'
if (!cfgTouched.value) cfg.value = 3.5
return
}
if (samplerName.value === 'euler') samplerName.value = 'res_multistep'
if (!cfgTouched.value) cfg.value = turbo.value ? CFG_TURBO : CFG_QUALITY
}
function selectVideoEngine(engine: VideoEngineId) {
if (videoEngine.value === engine) return
cfgTouched.value = false
videoEngine.value = engine
applyEngineDefaults(engine)
}
function applyVideoWorkflow(raw?: string) {
const parsed = parseVideoWorkflow(raw)
videoEngine.value = videoEngineOf(parsed)
videoStart.value = videoStartOf(parsed)
minimaxGraph.value = minimaxGraphOf(parsed)
}
async function generate() { async function generate() {
if (!file.value || !prompt.value.trim()) return if ((!file.value && !textToVideo.value) || !prompt.value.trim()) return
if (!queueReady.value) { if (!queueReady.value) {
toast(shotScriptMode.value ? 'Each shot in the script needs a prompt.' : 'Each queued extension needs a prompt.') toast(shotScriptMode.value ? 'Each shot in the script needs a prompt.' : 'Each queued extension needs a prompt.')
return return
@@ -3401,7 +3551,7 @@ async function generate() {
startTimer('video') startTimer('video')
try { try {
const body = new FormData() const body = new FormData()
body.append('image', file.value) if (file.value) body.append('image', file.value)
body.append('prompt', initialPrompt) body.append('prompt', initialPrompt)
body.append('aspect', aspect.value) body.append('aspect', aspect.value)
body.append('width', String(outputSize.value.width)) body.append('width', String(outputSize.value.width))
+22 -14
View File
@@ -1,5 +1,6 @@
import { frameLength, runGeneration } from '~/server/utils/videoChain' import { frameLength, runGeneration } from '~/server/utils/videoChain'
import { createShotQueue, finishQueueBurst, setQueueJob } from '~/server/utils/shotQueue' import { createShotQueue, finishQueueBurst, setQueueJob } from '~/server/utils/shotQueue'
import { defaultVideoSteps, isLtxWorkflow, isTextToVideo, parseVideoWorkflow } from '~/utils/videoModels'
function isComfyBusyTimeout(error: unknown) { function isComfyBusyTimeout(error: unknown) {
const err = error as { statusCode?: number; status?: number; data?: { code?: string }; message?: string; statusMessage?: string } const err = error as { statusCode?: number; status?: number; data?: { code?: string }; message?: string; statusMessage?: string }
@@ -42,16 +43,16 @@ function parseFps(raw: string | undefined) {
return fps === 12 || fps === 30 || fps === 24 ? fps : 24 return fps === 12 || fps === 30 || fps === 24 ? fps : 24
} }
function parseCfg(raw: string | undefined, turbo: boolean) { function parseCfg(raw: string | undefined, turbo: boolean, ltx = false) {
const fallback = turbo ? 1.5 : 4 const fallback = ltx ? 3.5 : (turbo ? 1.5 : 4)
const value = Number(raw) const value = Number(raw)
if (!Number.isFinite(value)) return fallback if (!Number.isFinite(value)) return fallback
const snapped = Math.round(value * 2) / 2 const snapped = Math.round(value * 2) / 2
return Math.min(10, Math.max(1, snapped)) return Math.min(10, Math.max(1, snapped))
} }
function parseSampler(raw: string | undefined) { function parseSampler(raw: string | undefined, ltx = false) {
return SAMPLERS.has(raw || '') ? raw! : 'res_multistep' return SAMPLERS.has(raw || '') ? raw! : (ltx ? 'euler' : 'res_multistep')
} }
function parseScheduler(raw: string | undefined) { function parseScheduler(raw: string | undefined) {
@@ -86,7 +87,9 @@ export default defineEventHandler(async (event) => {
} }
} }
if (!image) { const workflow = parseVideoWorkflow(fields.workflow)
const textToVideo = isTextToVideo(workflow)
if (!image && !textToVideo) {
throw createError({ statusCode: 400, statusMessage: 'An initial image is required' }) throw createError({ statusCode: 400, statusMessage: 'An initial image is required' })
} }
const prompt = (fields.prompt || '').trim() const prompt = (fields.prompt || '').trim()
@@ -94,10 +97,13 @@ export default defineEventHandler(async (event) => {
throw createError({ statusCode: 400, statusMessage: 'A motion prompt is required' }) throw createError({ statusCode: 400, statusMessage: 'A motion prompt is required' })
} }
const extensions = parseExtensions(fields.extensions) const extensions = parseExtensions(fields.extensions)
const workflow = parseVideoWorkflow(fields.workflow) const useIdentityRefs = workflow === 'v2' && !textToVideo && !isLtxWorkflow(workflow) && fields.useIdentityRefs === 'true'
const useIdentityRefs = workflow === 'v2' && fields.useIdentityRefs === 'true' const { width, height } = resolveOutputSize(
fields.aspect,
const { width, height } = resolveOutputSize(fields.aspect, image.data, Number(fields.width), Number(fields.height)) image?.data || Buffer.alloc(0),
Number(fields.width),
Number(fields.height)
)
const turbo = fields.turbo !== 'false' const turbo = fields.turbo !== 'false'
const steps = Number(fields.steps || defaultVideoSteps(turbo, workflow)) const steps = Number(fields.steps || defaultVideoSteps(turbo, workflow))
const seed = fields.seed && fields.seed !== 'random' const seed = fields.seed && fields.seed !== 'random'
@@ -105,13 +111,13 @@ export default defineEventHandler(async (event) => {
: Math.floor(Math.random() * 2_147_483_647) : Math.floor(Math.random() * 2_147_483_647)
const durationSeconds = parseDuration(fields.duration || '5') const durationSeconds = parseDuration(fields.duration || '5')
const fps = parseFps(fields.fps) const fps = parseFps(fields.fps)
const cfg = parseCfg(fields.cfg, turbo) const cfg = parseCfg(fields.cfg, turbo, isLtxWorkflow(workflow))
const samplerName = parseSampler(fields.sampler_name) const samplerName = parseSampler(fields.sampler_name, isLtxWorkflow(workflow))
const scheduler = parseScheduler(fields.scheduler) const scheduler = parseScheduler(fields.scheduler)
const length = frameLength(durationSeconds, fps) const length = frameLength(durationSeconds, fps)
const hideThumbnail = fields.hideThumbnail === 'true' const hideThumbnail = fields.hideThumbnail === 'true'
const hideInput = fields.hideInput === 'true' const hideInput = fields.hideInput === 'true'
const sound = fields.sound !== 'false' const sound = !isLtxWorkflow(workflow) && fields.sound !== 'false'
const clipName = (fields.name || '').trim().slice(0, 80) const clipName = (fields.name || '').trim().slice(0, 80)
const ownerKey = libraryOwnerKey(event) const ownerKey = libraryOwnerKey(event)
const library = publicLibrary(event) const library = publicLibrary(event)
@@ -124,7 +130,8 @@ export default defineEventHandler(async (event) => {
const destFolder = library.folders.find(folder => folder.id === folderId) const destFolder = library.folders.find(folder => folder.id === folderId)
const folderLocked = Boolean(destFolder?.protected && !destFolder.unlocked) const folderLocked = Boolean(destFolder?.protected && !destFolder.unlocked)
assertFolderExists(event, folderId) assertFolderExists(event, folderId)
const still = await rememberInputStill({ const still = image
? await rememberInputStill({
ownerKey, ownerKey,
folderId, folderId,
filename: image.filename, filename: image.filename,
@@ -133,6 +140,7 @@ export default defineEventHandler(async (event) => {
height, height,
hideInput hideInput
}) })
: null
const referenceStillIds: Array<string | null> = [null, null, null, null] const referenceStillIds: Array<string | null> = [null, null, null, null]
if (useIdentityRefs) { if (useIdentityRefs) {
for (const [index, ref] of referenceImages.entries()) { for (const [index, ref] of referenceImages.entries()) {
@@ -175,7 +183,7 @@ export default defineEventHandler(async (event) => {
fps, fps,
samplerName, samplerName,
scheduler, scheduler,
thumb: isPipelineFrameFilename(image.filename) ? undefined : image.data, thumb: image && !isPipelineFrameFilename(image.filename) ? image.data : undefined,
stillId: still?.id, stillId: still?.id,
stillFilename: still?.filename, stillFilename: still?.filename,
referenceStillIds, referenceStillIds,
+168
View File
@@ -0,0 +1,168 @@
{
"1": {
"inputs": {
"unet_name": "PinkCherry_FineTune_int8_v1_8_LTX23.safetensors",
"weight_dtype": "default"
},
"class_type": "UNETLoader",
"_meta": { "title": "Load PinkCherry LTX-2.3" }
},
"21": {
"inputs": {
"lora_name": "ltx-2.3-22b-distilled-lora-384-1.1.safetensors",
"strength_model": 1,
"model": ["1", 0]
},
"class_type": "LoraLoaderModelOnly",
"_meta": { "title": "LTX distilled LoRA" }
},
"3": {
"inputs": {
"vae_name": "LTX23_video_vae_bf16.safetensors"
},
"class_type": "VAELoader",
"_meta": { "title": "Load VAE" }
},
"4": {
"inputs": {
"image": "input_image.png",
"upload": "image"
},
"class_type": "LoadImage",
"_meta": { "title": "Load Image" }
},
"5": {
"inputs": {
"width": 1280,
"height": 1280,
"interpolation": "lanczos",
"method": "keep proportion",
"condition": "always",
"multiple_of": 32,
"image": ["4", 0]
},
"class_type": "ImageResize+",
"_meta": { "title": "Resize still" }
},
"6": {
"inputs": {
"text": "",
"clip": ["20", 0]
},
"class_type": "CLIPTextEncode",
"_meta": { "title": "Positive prompt" }
},
"7": {
"inputs": {
"text": "blurry, low quality, jitter, deformed, static frame, freeze, watermark, bad anatomy, text, subtitles",
"clip": ["20", 0]
},
"class_type": "CLIPTextEncode",
"_meta": { "title": "Negative prompt" }
},
"8": {
"inputs": {
"width": ["5", 1],
"height": ["5", 2],
"length": 97,
"batch_size": 1,
"strength": 1,
"positive": ["6", 0],
"negative": ["7", 0],
"vae": ["3", 0],
"image": ["5", 0]
},
"class_type": "LTXVImgToVideo",
"_meta": { "title": "LTX image to video" }
},
"9": {
"inputs": {
"frame_rate": 24,
"positive": ["8", 0],
"negative": ["8", 1]
},
"class_type": "LTXVConditioning",
"_meta": { "title": "LTX conditioning" }
},
"10": {
"inputs": {
"cfg": 3.5,
"model": ["21", 0],
"positive": ["9", 0],
"negative": ["9", 1]
},
"class_type": "CFGGuider",
"_meta": { "title": "CFG Guider" }
},
"11": {
"inputs": {
"sampler_name": "euler"
},
"class_type": "KSamplerSelect",
"_meta": { "title": "KSamplerSelect" }
},
"12": {
"inputs": {
"steps": 8,
"max_shift": 2.05,
"base_shift": 0.95,
"stretch": true,
"terminal": 0.1,
"latent": ["8", 2]
},
"class_type": "LTXVScheduler",
"_meta": { "title": "LTX scheduler" }
},
"13": {
"inputs": {
"noise_seed": 1
},
"class_type": "RandomNoise",
"_meta": { "title": "RandomNoise" }
},
"14": {
"inputs": {
"noise": ["13", 0],
"guider": ["10", 0],
"sampler": ["11", 0],
"sigmas": ["12", 0],
"latent_image": ["8", 2]
},
"class_type": "SamplerCustomAdvanced",
"_meta": { "title": "SamplerCustomAdvanced" }
},
"15": {
"inputs": {
"samples": ["14", 0],
"vae": ["3", 0]
},
"class_type": "VAEDecode",
"_meta": { "title": "VAE Decode" }
},
"16": {
"inputs": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "video/LTX23",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"images": ["15", 0]
},
"class_type": "VHS_VideoCombine",
"_meta": { "title": "Save MP4" }
},
"20": {
"inputs": {
"clip_name1": "gemma-3-12b-it-qat-UD-Q4_K_XL.gguf",
"clip_name2": "text_encoders\\ltx-2.3-22b-dev_embeddings_connectors.safetensors",
"type": "ltxv"
},
"class_type": "DualCLIPLoaderGGUF",
"_meta": { "title": "DualCLIPLoader (GGUF)" }
}
}
+19 -5
View File
@@ -1,3 +1,5 @@
import { LTX_NEGATIVE } from '~/utils/videoModels'
let comfyHostOverride = '' let comfyHostOverride = ''
export function setComfyHostOverride(url: string) { export function setComfyHostOverride(url: string) {
@@ -188,7 +190,12 @@ export function extractPromptFromHistory(entry: unknown) {
const text = String(node.inputs?.prompt || node.inputs?.text || '').trim() const text = String(node.inputs?.prompt || node.inputs?.text || '').trim()
if (!text) continue if (!text) continue
if (node?.class_type === 'MiniMaxH3ReferenceToVideo') return text if (node?.class_type === 'MiniMaxH3ReferenceToVideo') return text
if (node?.class_type === 'MiniMaxH3ImageToVideo') imageToVideo = text if (node?.class_type === 'MiniMaxH3ImageToVideo' || node?.class_type === 'MiniMaxH3TextToVideo') imageToVideo = text
if (node?.class_type === 'CLIPTextEncode') {
const title = String((node as { _meta?: { title?: string } })._meta?.title || '')
if (/negative/i.test(title) || text === LTX_NEGATIVE) continue
if (!imageToVideo) imageToVideo = text
}
} }
return imageToVideo return imageToVideo
} }
@@ -200,14 +207,18 @@ export function extractClipMetaFromHistory(entry: unknown) {
if (!graph || typeof graph !== 'object') return meta if (!graph || typeof graph !== 'object') return meta
for (const node of Object.values(graph as Record<string, { class_type?: string; inputs?: Record<string, unknown> }>)) { for (const node of Object.values(graph as Record<string, { class_type?: string; inputs?: Record<string, unknown> }>)) {
const inputs = node?.inputs || {} const inputs = node?.inputs || {}
if (node?.class_type === 'MiniMaxH3ImageToVideo') { if (node?.class_type === 'MiniMaxH3ImageToVideo' || node?.class_type === 'MiniMaxH3TextToVideo' || node?.class_type === 'EmptyLTXVLatentVideo') {
meta.width = Number(inputs.width || meta.width) meta.width = Number(inputs.width || meta.width)
meta.height = Number(inputs.height || meta.height) meta.height = Number(inputs.height || meta.height)
} }
if (node?.class_type === 'BasicScheduler' || node?.class_type === 'PrimitiveInt') { if (node?.class_type === 'BasicScheduler' || node?.class_type === 'PrimitiveInt' || node?.class_type === 'LTXVScheduler') {
const steps = Number(inputs.steps || 0) const steps = Number(inputs.steps || 0)
if (steps > 0) meta.steps = steps if (steps > 0) meta.steps = steps
} }
if (node?.class_type === 'ImageResize+' && typeof inputs.width === 'number' && typeof inputs.height === 'number') {
meta.width = Number(inputs.width)
meta.height = Number(inputs.height)
}
if (node?.class_type === 'RandomNoise' || node?.class_type === 'SamplerMiniMax' || node?.class_type === 'KSampler') { if (node?.class_type === 'RandomNoise' || node?.class_type === 'SamplerMiniMax' || node?.class_type === 'KSampler') {
const seed = Number(inputs.noise_seed ?? inputs.seed ?? 0) const seed = Number(inputs.noise_seed ?? inputs.seed ?? 0)
if (seed) meta.seed = seed if (seed) meta.seed = seed
@@ -276,13 +287,16 @@ export function comfyFilenamePrefix() {
} }
export function isOurComfyVideo(video: { filename: string; subfolder: string }) { export function isOurComfyVideo(video: { filename: string; subfolder: string }) {
const prefix = comfyFilenamePrefix().replace(/\/$/, '') const prefixes = [comfyFilenamePrefix(), 'video/LTX23']
return prefixes.some((raw) => {
const prefix = raw.replace(/\/$/, '')
const parts = prefix.split('/') const parts = prefix.split('/')
const namePrefix = parts[parts.length - 1] const namePrefix = parts[parts.length - 1]
const sub = parts.length > 1 ? parts.slice(0, -1).join('/') : 'video' const sub = parts.length > 1 ? parts.slice(0, -1).join('/') : 'video'
const nameOk = video.filename.startsWith(namePrefix) const nameOk = video.filename.startsWith(namePrefix) || video.filename.startsWith('LTX23')
const subOk = !video.subfolder || video.subfolder === sub const subOk = !video.subfolder || video.subfolder === sub
return nameOk && subOk return nameOk && subOk
})
} }
export function purgeComfyEnabled() { export function purgeComfyEnabled() {
+1 -1
View File
@@ -85,7 +85,7 @@ export interface Job {
familyId?: string familyId?: string
parentClipId?: string parentClipId?: string
parentStillId?: string parentStillId?: string
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
chainContinuing?: boolean chainContinuing?: boolean
passes?: { prompt: string }[] passes?: { prompt: string }[]
queueId?: string queueId?: string
+4 -4
View File
@@ -38,7 +38,7 @@ export interface LibraryClip {
familyId?: string familyId?: string
parentClipId?: string parentClipId?: string
chainIndex?: number chainIndex?: number
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
stillId?: string stillId?: string
sound?: boolean sound?: boolean
hideInput?: boolean hideInput?: boolean
@@ -102,7 +102,7 @@ export interface RetryDraft {
samplerName?: string samplerName?: string
scheduler?: string scheduler?: string
extensions?: QueuedExtension[] extensions?: QueuedExtension[]
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
} }
interface Catalog { interface Catalog {
@@ -969,7 +969,7 @@ export async function saveRetryDraft(params: {
samplerName?: string samplerName?: string
scheduler?: string scheduler?: string
extensions?: QueuedExtension[] extensions?: QueuedExtension[]
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
}) { }) {
return mutate(params.ownerKey, (catalog) => { return mutate(params.ownerKey, (catalog) => {
const existing = params.id ? catalog.drafts.find(item => item.id === params.id) : undefined const existing = params.id ? catalog.drafts.find(item => item.id === params.id) : undefined
@@ -1134,7 +1134,7 @@ export async function saveClip(params: {
familyId?: string familyId?: string
parentClipId?: string parentClipId?: string
chainIndex?: number chainIndex?: number
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
stillId?: string stillId?: string
sound?: boolean sound?: boolean
}) { }) {
+1 -1
View File
@@ -35,7 +35,7 @@ export interface PendingJob {
referenceImageNames?: string[] referenceImageNames?: string[]
referenceStillIds?: Array<string | null> referenceStillIds?: Array<string | null>
useIdentityRefs?: boolean useIdentityRefs?: boolean
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
duration?: number duration?: number
cfg?: number cfg?: number
fps?: number fps?: number
+7 -2
View File
@@ -1,5 +1,6 @@
import { getSidecarImageHost } from '~/server/utils/imageComfy' import { getSidecarImageHost } from '~/server/utils/imageComfy'
import { buildVisionPromptWorkflow } from '~/server/utils/promptWorkflow' import { buildVisionPromptWorkflow } from '~/server/utils/promptWorkflow'
import { normalizeShotScript } from '~/utils/parseRecommend'
export type PromptJobStatus = 'queued' | 'running' | 'complete' | 'error' export type PromptJobStatus = 'queued' | 'running' | 'complete' | 'error'
@@ -206,7 +207,10 @@ function draftUserPrompt(draft: string, mode: string) {
: mode === 'shot-script' : mode === 'shot-script'
? 'video-shot-script' ? 'video-shot-script'
: 'video' : 'video'
return `Studio mode: ${studio}\n\nDraft Prompt:\n${draft}` const format = studio === 'image-edit'
? 'If this is a single edit, output only the refined prompt. If it needs phases, use ---pass 1--- blocks.'
: 'Output a MiniMax shot list. Each shot starts with "shot N" on its own line, then [SHOT CONFIGURATION], [SUBJECT DIRECTION & ACTION], and [AUDIO CUES], each on its own line. Blank line between shots. No markdown fences. No one-paragraph dump.'
return `Studio mode: ${studio}\n${format}\n\nDraft Prompt:\n${draft}`
} }
export async function runPromptRecommend(job: PromptJob, params: { export async function runPromptRecommend(job: PromptJob, params: {
@@ -227,7 +231,8 @@ export async function runPromptRecommend(job: PromptJob, params: {
const queued = await queuePromptGraph(graph, job.id) const queued = await queuePromptGraph(graph, job.id)
job.message = 'Writing the recommended prompt…' job.message = 'Writing the recommended prompt…'
const text = await waitForPromptText(queued.prompt_id) const text = await waitForPromptText(queued.prompt_id)
job.text = text.trim() const trimmed = text.trim()
job.text = params.mode === 'edit' ? trimmed : normalizeShotScript(trimmed)
job.status = 'complete' job.status = 'complete'
job.message = 'Recommended prompt ready' job.message = 'Recommended prompt ready'
} }
+51 -18
View File
@@ -9,6 +9,49 @@ export interface PromptPreset {
builtin?: boolean builtin?: boolean
} }
const SHOT_SCRIPT_FORMAT = `Output ONLY a MiniMax shot list the user can paste into Shot script mode.
Each shot starts with "shot N" on its own line (shot 1, shot 2, shot 3…). Then these three headers, each on its own line:
[SHOT CONFIGURATION]
[SUBJECT DIRECTION & ACTION]
[AUDIO CUES]
Blank line between shots. No preamble, no markdown fences, no commentary.
Do not glue shots into one paragraph. "shot 2" must be a line by itself, never "shot 2:" mid-sentence.
One beat per shot. Later shots: same people, same clothes as now, plus the one new action.
Do not repeat a long identity paragraph on every shot.
Never censor raw or explicit terminology.
Example:
shot 1
[SHOT CONFIGURATION]
- Shot Type: Medium-wide, 50mm, eye-level
- Camera Movement: Slow track from a slight side-front angle
- Lighting: Bright daylight through the trees
- Environment/Scenery: The forest from the still
[SUBJECT DIRECTION & ACTION]
The woman from the still walks the path, taking in the trees.
[AUDIO CUES]
Footsteps on dirt, birds, breeze.
shot 2
[SHOT CONFIGURATION]
- Shot Type: Medium, 50mm, eye-level
- Camera Movement: Hold, then a small pan to her hand
- Lighting: Same daylight
- Environment/Scenery: Same path, same trees
[SUBJECT DIRECTION & ACTION]
She stops and rests her hand on the bark.
[AUDIO CUES]
Soft fabric, breeze, birds.`
const ENHANCE_PROMPT = `You are a prompt-engineering assistant for Flux.2 Klein image edit and MiniMax H3 image-to-video. const ENHANCE_PROMPT = `You are a prompt-engineering assistant for Flux.2 Klein image edit and MiniMax H3 image-to-video.
Look at the attached still. The person (or subject) in that still is the identity lock. The user draft is the requested change. Look at the attached still. The person (or subject) in that still is the identity lock. The user draft is the requested change.
@@ -16,37 +59,27 @@ Look at the attached still. The person (or subject) in that still is the identit
Write production-ready prompt text the user can paste into the generator. Write production-ready prompt text the user can paste into the generator.
Rules: Rules:
- Preserve identity from the still: face, age, body type, hair, skin, and distinctive features. Use explicit keep-identity language such as "same person as the reference still", "do not change facial identity", "keep original face, hair, and body". - Preserve identity from the still: face, age, body type, hair, skin, and distinctive features.
- Describe wardrobe, setting, pose, lighting, camera, and materials concretely. - Describe wardrobe, setting, pose, lighting, camera, and materials concretely.
- Never censor raw or explicit terminology from the draft. - Never censor raw or explicit terminology from the draft.
- Do not add a preamble, title, markdown fences, or commentary. - Do not add a preamble, title, markdown fences, or commentary.
Output format — pick one: If Studio mode is image-edit:
- Single still/edit: output ONLY the refined prompt. - Single still/edit: output ONLY the refined prompt.
- Ambitious still edit that should be done in phases (for example wardrobe first, then location): output: - Ambitious still edit that should be done in phases (for example wardrobe first, then location): output:
---pass 1--- ---pass 1---
<prompt for pass 1> <prompt for pass 1>
---pass 2--- ---pass 2---
<prompt for pass 2> <prompt for pass 2>
- Motion, multi-beat, or video sequence: first-shot prompt, then later shots with a marker on its own line:
shot 2 If Studio mode is video or video-shot-script:
<extension prompt> ${SHOT_SCRIPT_FORMAT}`
shot 3
<extension prompt>`
const SHOT_SCRIPT_PROMPT = `You are a MiniMax H3 image-to-video shot-script writer. const SHOT_SCRIPT_PROMPT = `You are a MiniMax H3 image-to-video shot-script writer.
Analyze the attached still for identity, wardrobe, lighting, and camera. Turn the user's draft into a shot script they can paste into Shot script mode. Analyze the attached still for wardrobe, lighting, and camera. Turn the user's draft into a shot script.
Format: ${SHOT_SCRIPT_FORMAT}`
- Shot 1 is the text BEFORE any marker (camera, action, audio for the first clip).
- Later shots start with a line that is only: shot 2
- Then shot 3, shot 4, and so on, each on its own line.
- Every shot must keep the same person as the still unless the draft explicitly changes wardrobe or identity.
- Include an [AUDIO CUES] block on shots where sound would help.
- Never censor raw or explicit terminology.
Output ONLY the shot script. No preamble, no markdown fences.`
export const BUILTIN_PRESETS: PromptPreset[] = [ export const BUILTIN_PRESETS: PromptPreset[] = [
{ {
@@ -60,7 +93,7 @@ export const BUILTIN_PRESETS: PromptPreset[] = [
id: 'shot-script', id: 'shot-script',
name: 'Shot script', name: 'Shot script',
builtin: true, builtin: true,
description: 'Turn a draft into MiniMax shot 2 / shot 3 extension copy.', description: 'Turn a draft into a MiniMax shot list with configuration, action, and audio blocks.',
systemPrompt: SHOT_SCRIPT_PROMPT systemPrompt: SHOT_SCRIPT_PROMPT
} }
] ]
+2 -2
View File
@@ -38,7 +38,7 @@ export interface ShotQueue {
fps?: number fps?: number
samplerName?: string samplerName?: string
scheduler?: string scheduler?: string
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
sound?: boolean sound?: boolean
useIdentityRefs?: boolean useIdentityRefs?: boolean
referenceStillIds?: Array<string | null> referenceStillIds?: Array<string | null>
@@ -153,7 +153,7 @@ export async function createShotQueue(params: {
fps?: number fps?: number
samplerName?: string samplerName?: string
scheduler?: string scheduler?: string
workflow?: 'v1' | 'v2' workflow?: import('~/utils/videoModels').VideoWorkflowId
sound?: boolean sound?: boolean
useIdentityRefs?: boolean useIdentityRefs?: boolean
referenceStillIds?: Array<string | null> referenceStillIds?: Array<string | null>
+22 -13
View File
@@ -5,6 +5,7 @@ import { pendingFromJob, remainingAfterCurrentShot, writePendingJob } from '~/se
import { emitChainJob, waitForComfySocket, watchComfyJob } from '~/server/utils/watch' import { emitChainJob, waitForComfySocket, watchComfyJob } from '~/server/utils/watch'
import { comfyFilenamePrefix, queuePrompt, uploadImage } from '~/server/utils/comfy' import { comfyFilenamePrefix, queuePrompt, uploadImage } from '~/server/utils/comfy'
import { buildWorkflow } from '~/server/utils/workflow' import { buildWorkflow } from '~/server/utils/workflow'
import { isLtxWorkflow, isTextToVideo, parseVideoWorkflow, workflowForExtension, type VideoWorkflowId } from '~/utils/videoModels'
import { clipVideoPath, clipTitle, deleteRetryDraft, extendTempDir, getClip, nextClipPartName, removeExtendTemp, stillPath } from '~/server/utils/library' import { clipVideoPath, clipTitle, deleteRetryDraft, extendTempDir, getClip, nextClipPartName, removeExtendTemp, stillPath } from '~/server/utils/library'
import { extractLastFrame, probeHasAudio } from '~/server/utils/ffmpeg' import { extractLastFrame, probeHasAudio } from '~/server/utils/ffmpeg'
import { ensureComfyReady } from '~/server/utils/comfyLifecycle' import { ensureComfyReady } from '~/server/utils/comfyLifecycle'
@@ -23,7 +24,7 @@ export type ChainImage = { filename: string; data: Buffer; type?: string }
type VideoChainParams = { type VideoChainParams = {
prompt: string prompt: string
image: ChainImage image?: ChainImage | null
width: number width: number
height: number height: number
steps: number steps: number
@@ -36,7 +37,7 @@ type VideoChainParams = {
samplerName: string samplerName: string
scheduler: string scheduler: string
extensions: { prompt: string; duration: number }[] extensions: { prompt: string; duration: number }[]
workflow: 'v1' | 'v2' workflow: VideoWorkflowId
duration: number duration: number
useIdentityRefs: boolean useIdentityRefs: boolean
referenceImages: Array<ChainImage | null> referenceImages: Array<ChainImage | null>
@@ -76,7 +77,8 @@ function paramsFromJob(job: Job): VideoChainParams {
throw new Error('Cannot continue the shot chain: job metadata is missing') throw new Error('Cannot continue the shot chain: job metadata is missing')
} }
const image = loadStillImage(library.ownerKey, library.stillId, library.stillFilename) const image = loadStillImage(library.ownerKey, library.stillId, library.stillFilename)
if (!image?.data.length) { const workflow = parseVideoWorkflow(library.workflow)
if (!image?.data.length && !isTextToVideo(workflow)) {
throw new Error('Cannot continue the shot chain: the start still is missing from the library') throw new Error('Cannot continue the shot chain: the start still is missing from the library')
} }
const referenceImages: Array<ChainImage | null> = [null, null, null, null] const referenceImages: Array<ChainImage | null> = [null, null, null, null]
@@ -101,7 +103,7 @@ function paramsFromJob(job: Job): VideoChainParams {
samplerName: library.samplerName || 'res_multistep', samplerName: library.samplerName || 'res_multistep',
scheduler: library.scheduler || 'simple', scheduler: library.scheduler || 'simple',
extensions: library.extensions || [], extensions: library.extensions || [],
workflow: library.workflow || 'v1', workflow,
duration, duration,
useIdentityRefs: library.useIdentityRefs === true, useIdentityRefs: library.useIdentityRefs === true,
referenceImages referenceImages
@@ -121,14 +123,21 @@ export async function queueMiniMax(
const done = watchComfyJob(job, { persist: params.persist }) const done = watchComfyJob(job, { persist: params.persist })
job.status = 'uploading' job.status = 'uploading'
const chainIndex = job.library?.chainIndex || 0 const chainIndex = job.library?.chainIndex || 0
const uploading = params.useIdentityRefs const graphId = chainIndex > 0 ? workflowForExtension(params.workflow) : params.workflow
const engineName = isLtxWorkflow(graphId) ? 'LTX-2.3' : 'MiniMax H3'
const hasImage = Boolean(params.image?.data?.length)
const uploading = !hasImage
? `Queueing ${engineName} text-to-video…`
: params.useIdentityRefs
? (chainIndex > 0 ? 'Uploading identity stills for next shot...' : 'Uploading image to ComfyUI...') ? (chainIndex > 0 ? 'Uploading identity stills for next shot...' : 'Uploading image to ComfyUI...')
: (chainIndex > 0 ? 'Uploading last frame to ComfyUI...' : 'Uploading image to ComfyUI...') : (chainIndex > 0 ? 'Uploading last frame to ComfyUI...' : 'Uploading image to ComfyUI...')
const queueing = (job.library?.chainIndex || 0) > 0 const queueing = chainIndex > 0
? 'Queueing extension on MiniMax H3...' ? `Queueing extension on ${engineName}...`
: 'Queueing MiniMax H3 job...' : `Queueing ${engineName} job...`
emitChainJob(job, { type: 'status', message: uploading, progress: 4 }) emitChainJob(job, { type: 'status', message: uploading, progress: 4 })
const uploaded = await uploadImage(params.image, job.id) const uploaded = hasImage && params.image
? await uploadImage(params.image, job.id)
: { name: '', subfolder: '' }
const referenceNames: string[] = ['', '', '', ''] const referenceNames: string[] = ['', '', '', '']
if (params.useIdentityRefs) { if (params.useIdentityRefs) {
for (const [index, ref] of (params.referenceImages || []).entries()) { for (const [index, ref] of (params.referenceImages || []).entries()) {
@@ -161,9 +170,9 @@ export async function queueMiniMax(
fps: params.fps, fps: params.fps,
samplerName: params.samplerName, samplerName: params.samplerName,
scheduler: params.scheduler, scheduler: params.scheduler,
filenamePrefix: comfyFilenamePrefix(), filenamePrefix: isLtxWorkflow(graphId) ? 'video/LTX23' : comfyFilenamePrefix(),
sound: params.sound, sound: params.sound && !isLtxWorkflow(graphId),
workflow: params.workflow, workflow: graphId,
duration: params.duration, duration: params.duration,
useIdentityRefs: params.useIdentityRefs, useIdentityRefs: params.useIdentityRefs,
referenceImageNames: params.useIdentityRefs ? referenceNames : [] referenceImageNames: params.useIdentityRefs ? referenceNames : []
@@ -317,7 +326,7 @@ export async function continueQueuedExtensions(
samplerName: params.samplerName, samplerName: params.samplerName,
scheduler: params.scheduler, scheduler: params.scheduler,
persist, persist,
workflow: params.workflow, workflow: workflowForExtension(params.workflow),
duration: ext.duration, duration: ext.duration,
useIdentityRefs: params.useIdentityRefs, useIdentityRefs: params.useIdentityRefs,
referenceImages: params.useIdentityRefs ? params.referenceImages : [] referenceImages: params.useIdentityRefs ? params.referenceImages : []
+89 -11
View File
@@ -1,9 +1,18 @@
// Nitro bundles these JSON graphs into the production server output. // Nitro bundles these JSON graphs into the production server output.
import workflowV1 from '../assets/workflow_minimax_video.json' import workflowV1 from '../assets/workflow_minimax_video.json'
import workflowV2 from '../assets/workflow_minimax_video_v2.json' import workflowV2 from '../assets/workflow_minimax_video_v2.json'
import workflowLtx from '../assets/workflow_ltx_video.json'
import { buildIdentityPrompt } from '~/utils/identityPrompt' import { buildIdentityPrompt } from '~/utils/identityPrompt'
import {
isLtxWorkflow,
isTextToVideo,
LTX_NEGATIVE,
parseVideoWorkflow,
type VideoWorkflowId
} from '~/utils/videoModels'
export type VideoWorkflowId = 'v1' | 'v2' export type { VideoWorkflowId }
export { parseVideoWorkflow }
export interface GenerateParams { export interface GenerateParams {
prompt: string prompt: string
@@ -31,6 +40,7 @@ type WorkflowGraph = Record<string, WorkflowNode>
const PROMPT_CLASSES = new Set([ const PROMPT_CLASSES = new Set([
'MiniMaxH3ImageToVideo', 'MiniMaxH3ImageToVideo',
'MiniMaxH3TextToVideo',
'MiniMaxH3ReferenceToVideo', 'MiniMaxH3ReferenceToVideo',
'CLIPTextEncode', 'CLIPTextEncode',
'CLIPTextEncodeQwen3VL', 'CLIPTextEncodeQwen3VL',
@@ -58,16 +68,15 @@ function titleOf(node: WorkflowNode) {
} }
function templateFor(id: VideoWorkflowId) { function templateFor(id: VideoWorkflowId) {
if (isLtxWorkflow(id)) return workflowLtx as WorkflowGraph
return (id === 'v2' ? workflowV2 : workflowV1) as WorkflowGraph return (id === 'v2' ? workflowV2 : workflowV1) as WorkflowGraph
} }
export function parseVideoWorkflow(raw: unknown): VideoWorkflowId {
return String(raw || '').trim() === 'v2' ? 'v2' : 'v1'
}
export function buildWorkflow(params: GenerateParams) { export function buildWorkflow(params: GenerateParams) {
const version = params.workflow === 'v2' ? 'v2' : 'v1' const version = parseVideoWorkflow(params.workflow)
const graph = structuredClone(templateFor(version)) if (isLtxWorkflow(version)) return buildLtxWorkflow(params, version)
const graph = structuredClone(templateFor(version === 't2v' ? 'v1' : version === 'v2' ? 'v2' : 'v1'))
const width = snap32(params.width) const width = snap32(params.width)
const height = snap32(params.height) const height = snap32(params.height)
const refs = Array.from({ length: 4 }, (_, index) => String(params.referenceImageNames?.[index] || '').trim()) const refs = Array.from({ length: 4 }, (_, index) => String(params.referenceImageNames?.[index] || '').trim())
@@ -155,7 +164,7 @@ export function buildWorkflow(params: GenerateParams) {
} }
} }
if (version === 'v1' && graph['128']?.class_type === 'ImageScaleToTotalPixels') { if (version === 'v1' && !isTextToVideo(version) && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
graph['128'] = { graph['128'] = {
class_type: 'ImageScale', class_type: 'ImageScale',
inputs: { inputs: {
@@ -169,7 +178,7 @@ export function buildWorkflow(params: GenerateParams) {
} }
} }
if (version === 'v1') { if (version === 'v1' || version === 't2v') {
graph['105:cfg'] = { graph['105:cfg'] = {
class_type: 'FluxGuidance', class_type: 'FluxGuidance',
inputs: { inputs: {
@@ -191,10 +200,11 @@ export function buildWorkflow(params: GenerateParams) {
graph[qualityId].inputs.value = params.turbo ? 20 : params.steps graph[qualityId].inputs.value = params.turbo ? 20 : params.steps
} }
if (turboId && graph[turboId]?.class_type === 'PrimitiveInt') { if (turboId && graph[turboId]?.class_type === 'PrimitiveInt') {
graph[turboId].inputs.value = params.turbo ? params.steps : (params.workflow === 'v2' ? 6 : 8) graph[turboId].inputs.value = params.turbo ? params.steps : (version === 'v2' ? 6 : 8)
} }
} }
if (version === 't2v') applyMinimaxTextToVideo(graph, params, width, height)
if (version === 'v2') applyV2IdentityPath(graph, params, useRefs, refs, graphPrompt) if (version === 'v2') applyV2IdentityPath(graph, params, useRefs, refs, graphPrompt)
if (params.sound === false) { if (params.sound === false) {
@@ -203,6 +213,7 @@ export function buildWorkflow(params: GenerateParams) {
if (graph['105:91']?.inputs) delete graph['105:91'].inputs.audio if (graph['105:91']?.inputs) delete graph['105:91'].inputs.audio
delete graph['163'] delete graph['163']
if (graph['172']?.inputs) delete graph['172'].inputs.audio if (graph['172']?.inputs) delete graph['172'].inputs.audio
if (graph['105:104']?.inputs) delete graph['105:104'].inputs.audio_vae
} }
return graph return graph
@@ -210,10 +221,76 @@ export function buildWorkflow(params: GenerateParams) {
function promptForNode(classType: string, useRefs: boolean, actionPrompt: string, graphPrompt: string) { function promptForNode(classType: string, useRefs: boolean, actionPrompt: string, graphPrompt: string) {
if (classType === 'MiniMaxH3ImageToVideo') return useRefs ? '' : actionPrompt if (classType === 'MiniMaxH3ImageToVideo') return useRefs ? '' : actionPrompt
if (classType === 'MiniMaxH3TextToVideo') return actionPrompt
if (classType === 'MiniMaxH3ReferenceToVideo') return useRefs ? graphPrompt : actionPrompt if (classType === 'MiniMaxH3ReferenceToVideo') return useRefs ? graphPrompt : actionPrompt
return graphPrompt return graphPrompt
} }
function applyMinimaxTextToVideo(graph: WorkflowGraph, params: GenerateParams, width: number, height: number) {
const node = graph['105:104']
if (node) {
node.class_type = 'MiniMaxH3TextToVideo'
node._meta = { title: 'MiniMax H3 Text to Video' }
delete node.inputs.first_frame
node.inputs.prompt = params.prompt
node.inputs.width = width
node.inputs.height = height
if (params.sound !== false && graph['105:24']) node.inputs.audio_vae = ['105:24', 0]
else delete node.inputs.audio_vae
}
delete graph['114']
delete graph['119']
delete graph['120']
delete graph['127']
delete graph['128']
}
function buildLtxWorkflow(params: GenerateParams, version: VideoWorkflowId) {
const graph = structuredClone(workflowLtx as WorkflowGraph)
const width = snap32(params.width)
const height = snap32(params.height)
const textToVideo = isTextToVideo(version)
if (graph['4']?.inputs) graph['4'].inputs.image = params.imageName
if (graph['5']?.inputs) {
graph['5'].inputs.width = width
graph['5'].inputs.height = height
}
if (graph['6']?.inputs) graph['6'].inputs.text = params.prompt
if (graph['7']?.inputs) graph['7'].inputs.text = LTX_NEGATIVE
if (graph['8']?.inputs && typeof graph['8'].inputs.length === 'number') graph['8'].inputs.length = params.length
if (graph['9']?.inputs) graph['9'].inputs.frame_rate = params.fps
if (graph['10']?.inputs) graph['10'].inputs.cfg = params.cfg
if (graph['11']?.inputs) graph['11'].inputs.sampler_name = params.samplerName || 'euler'
if (graph['12']?.inputs) graph['12'].inputs.steps = params.steps
if (graph['13']?.inputs) graph['13'].inputs.noise_seed = params.seed
if (graph['16']?.inputs) {
graph['16'].inputs.frame_rate = params.fps
graph['16'].inputs.filename_prefix = params.filenamePrefix || 'video/LTX23'
}
if (graph['21']?.inputs) graph['21'].inputs.strength_model = 1
if (textToVideo) {
delete graph['4']
delete graph['5']
graph['8'] = {
class_type: 'EmptyLTXVLatentVideo',
inputs: {
width,
height,
length: params.length,
batch_size: 1
},
_meta: { title: 'Empty LTX latent' }
}
if (graph['9']?.inputs) {
graph['9'].inputs.positive = ['6', 0]
graph['9'].inputs.negative = ['7', 0]
}
if (graph['12']?.inputs) graph['12'].inputs.latent = ['8', 0]
if (graph['14']?.inputs) graph['14'].inputs.latent_image = ['8', 0]
}
return graph
}
function applyV2IdentityPath( function applyV2IdentityPath(
graph: WorkflowGraph, graph: WorkflowGraph,
params: GenerateParams, params: GenerateParams,
@@ -304,9 +381,10 @@ function labelsFrom(template: WorkflowGraph) {
export const NODE_LABELS: Record<string, string> = { export const NODE_LABELS: Record<string, string> = {
...Object.fromEntries(labelsFrom(workflowV1 as WorkflowGraph)), ...Object.fromEntries(labelsFrom(workflowV1 as WorkflowGraph)),
...Object.fromEntries(labelsFrom(workflowV2 as WorkflowGraph)), ...Object.fromEntries(labelsFrom(workflowV2 as WorkflowGraph)),
...Object.fromEntries(labelsFrom(workflowLtx as WorkflowGraph)),
...LABEL_OVERRIDES ...LABEL_OVERRIDES
} }
export function isEncodingNode(node: string) { export function isEncodingNode(node: string) {
return node === '92' || node === '105:91' || node === '172' || /encoding|saving mp4|create video/i.test(NODE_LABELS[node] || '') return node === '92' || node === '16' || node === '105:91' || node === '172' || /encoding|saving mp4|create video|save mp4/i.test(NODE_LABELS[node] || '')
} }
+39
View File
@@ -1,3 +1,5 @@
import { parseShotScript } from './parseShots'
export function parseEditPasses(text: string): string[] | null { export function parseEditPasses(text: string): string[] | null {
const src = String(text || '').replace(/\r\n/g, '\n') const src = String(text || '').replace(/\r\n/g, '\n')
const marker = /^[ \t]*---pass\s+(\d+)---[ \t]*$/gim const marker = /^[ \t]*---pass\s+(\d+)---[ \t]*$/gim
@@ -21,3 +23,40 @@ export function parseEditPasses(text: string): string[] | null {
export function looksLikeShotScript(text: string) { export function looksLikeShotScript(text: string) {
return /^[ \t]*shot\s+[2-9]\d*\b/im.test(String(text || '')) return /^[ \t]*shot\s+[2-9]\d*\b/im.test(String(text || ''))
} }
function stripFences(text: string) {
return String(text || '')
.replace(/^\s*```(?:[\w-]+)?\s*\n?/, '')
.replace(/\n?```\s*$/, '')
.trim()
}
function splitInlineShotMarkers(text: string) {
return String(text || '')
.replace(/\r\n/g, '\n')
.replace(/(^|[^A-Za-z])shot\s+(\d+)\s*[:.)\-]?\s*/gi, '$1\n\nshot $2\n')
.replace(/\n{3,}/g, '\n\n')
.trim()
}
function formatShotBody(body: string) {
let text = String(body || '').trim()
text = text.replace(/^\s*[:.\-–)]\s*/, '')
text = text.replace(/\s*\[SHOT CONFIGURATION\]\s*/gi, '\n\n[SHOT CONFIGURATION]\n')
text = text.replace(/\s*\[SUBJECT DIRECTION(?:\s*&\s*ACTION)?\]\s*/gi, '\n\n[SUBJECT DIRECTION & ACTION]\n')
text = text.replace(/\s*\[AUDIO CUES\]\s*/gi, '\n\n[AUDIO CUES]\n')
text = text.replace(/[ \t]+\n/g, '\n').replace(/\n{3,}/g, '\n\n').trim()
return text
}
export function normalizeShotScript(text: string) {
const src = splitInlineShotMarkers(stripFences(text))
const shots = parseShotScript(src)
if (!shots.length) return src
if (shots.length === 1 && !/^[ \t]*shot\s+\d+\b/im.test(src)) {
return formatShotBody(shots[0].prompt)
}
return shots
.map(shot => `shot ${shot.n}\n${formatShotBody(shot.prompt)}`.trim())
.join('\n\n')
}
+56
View File
@@ -0,0 +1,56 @@
export type VideoEngineId = 'minimax' | 'ltx'
export type VideoStartId = 'still' | 'text'
export type MiniMaxGraphId = 'v1' | 'v2'
export type VideoWorkflowId = 'v1' | 'v2' | 't2v' | 'ltx' | 'ltx-t2v'
export function parseVideoWorkflow(raw: unknown): VideoWorkflowId {
const value = String(raw || '').trim()
if (value === 'v2') return 'v2'
if (value === 't2v') return 't2v'
if (value === 'ltx' || value === 'ltx-i2v') return 'ltx'
if (value === 'ltx-t2v') return 'ltx-t2v'
return 'v1'
}
export function videoEngineOf(id: VideoWorkflowId): VideoEngineId {
return id.startsWith('ltx') ? 'ltx' : 'minimax'
}
export function videoStartOf(id: VideoWorkflowId): VideoStartId {
return id === 't2v' || id === 'ltx-t2v' ? 'text' : 'still'
}
export function isLtxWorkflow(id: VideoWorkflowId) {
return videoEngineOf(id) === 'ltx'
}
export function isTextToVideo(id: VideoWorkflowId) {
return videoStartOf(id) === 'text'
}
export function minimaxGraphOf(id: VideoWorkflowId): MiniMaxGraphId {
return id === 'v2' ? 'v2' : 'v1'
}
export function workflowForExtension(id: VideoWorkflowId): VideoWorkflowId {
if (id === 't2v') return 'v1'
if (id === 'ltx-t2v') return 'ltx'
return id
}
export function composeVideoWorkflow(engine: VideoEngineId, start: VideoStartId, graph: MiniMaxGraphId = 'v1'): VideoWorkflowId {
if (engine === 'ltx') return start === 'text' ? 'ltx-t2v' : 'ltx'
if (start === 'text') return 't2v'
return graph === 'v2' ? 'v2' : 'v1'
}
export function defaultVideoSteps(turbo: boolean, workflow: string = 'v1') {
const id = parseVideoWorkflow(workflow)
if (!turbo) return isLtxWorkflow(id) ? 12 : 20
if (isLtxWorkflow(id)) return 8
return id === 'v2' ? 6 : 8
}
export const LTX_UNET = 'PinkCherry_FineTune_int8_v1_8_LTX23.safetensors'
export const LTX_DISTILLED_LORA = 'ltx-2.3-22b-distilled-lora-384-1.1.safetensors'
export const LTX_NEGATIVE = 'blurry, low quality, jitter, deformed, static frame, freeze, watermark, bad anatomy, text, subtitles'
-4
View File
@@ -1,4 +0,0 @@
export function defaultVideoSteps(turbo: boolean, workflow = 'v1') {
if (!turbo) return 20
return workflow === 'v2' ? 6 : 8
}