Add MiniMax and PinkCherry LTX as pickable video engines, with still or text start.

Shot 1 can run text-to-video; later shots still extend from the last frame. Recommend output is also normalized into shot-script cards.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-08-27 19:38:53 -05:00
co-authored by Cursor
parent 2bebeea526
commit 40cd17f35c
15 changed files with 676 additions and 120 deletions
+180 -30
View File
@@ -6,7 +6,7 @@
<div class="flex h-10 w-10 items-center justify-center rounded-xl bg-amber-400 text-zinc-950 font-display font-extrabold">A</div>
<div>
<h1 class="font-display text-xl font-bold leading-none">{{ instanceName }}</h1>
<p class="text-[11px] uppercase tracking-[0.22em] text-zinc-500">MiniMax H3 · Flux.2 Klein</p>
<p class="text-[11px] uppercase tracking-[0.22em] text-zinc-500">MiniMax H3 · LTX-2.3 · Flux.2 Klein</p>
</div>
</div>
<div class="flex items-center gap-3 text-sm">
@@ -80,8 +80,8 @@
<button type="button" class="relative mt-3 rounded-full bg-zinc-950/80 px-3 py-1 text-xs" @click.stop="resetImage">Reset</button>
</div>
<div v-else class="flex min-h-44 flex-col items-center justify-center text-center">
<p class="font-medium">{{ studioMode === 'edit' ? 'Image 1' : 'Choose an image' }}</p>
<p class="mt-1 text-sm text-zinc-500">Browse the library or upload a new PNG, JPG, or WEBP</p>
<p class="font-medium">{{ studioMode === 'edit' ? 'Image 1' : (textToVideo ? 'Start still · optional' : 'Choose an image') }}</p>
<p class="mt-1 text-sm text-zinc-500">{{ textToVideo ? 'Text-to-video does not need a still. Shot 2+ will still extend from the last frame.' : 'Browse the library or upload a new PNG, JPG, or WEBP' }}</p>
</div>
</div>
@@ -183,6 +183,16 @@
<p class="text-xs font-medium uppercase tracking-wide text-amber-200/80">Recommended prompt</p>
<p v-if="recommendBusy" class="mt-2 text-sm text-zinc-400">{{ recommendMessage || 'Sidecar Qwen VL is reading the still…' }}</p>
<p v-else-if="recommendError" class="mt-2 text-sm text-red-300">{{ recommendError }}</p>
<div v-else-if="recommendedShots.length > 1" class="mt-2 space-y-2">
<div
v-for="shot in recommendedShots"
:key="shot.n"
class="rounded-xl border border-amber-300/15 bg-zinc-950/50 px-3 py-2"
>
<p class="text-[11px] font-medium uppercase tracking-wide text-amber-200/80">shot {{ shot.n }}</p>
<p class="mt-1 whitespace-pre-wrap text-sm text-zinc-200">{{ shot.prompt }}</p>
</div>
</div>
<p v-else class="mt-2 whitespace-pre-wrap text-sm text-zinc-200">{{ recommendText }}</p>
<div v-if="recommendText && !recommendBusy" class="mt-3 flex flex-wrap gap-2">
<button
@@ -343,7 +353,7 @@
<span class="absolute left-0.5 top-0.5 h-5 w-5 rounded-full bg-white transition peer-checked:translate-x-5" />
</span>
</label>
<label v-if="studioMode === 'video'" class="flex cursor-pointer items-center justify-between gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm">
<label v-if="studioMode === 'video' && !ltxVideo" class="flex cursor-pointer items-center justify-between gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm">
<span>
<span class="block font-medium text-zinc-200">Generate sound</span>
<span class="mt-0.5 block text-xs text-zinc-500">Off skips the audio VAE so Comfy encodes a silent clip.</span>
@@ -371,13 +381,60 @@
<div v-if="studioMode === 'video'" class="space-y-3">
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Video model</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoEngine === 'minimax' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="selectVideoEngine('minimax')"
>
<span class="block font-semibold">MiniMax H3</span>
<span class="text-xs text-zinc-400">Native stereo audio. Identity refs on v2.</span>
</button>
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoEngine === 'ltx' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="selectVideoEngine('ltx')"
>
<span class="block font-semibold">LTX-2.3 PinkCherry</span>
<span class="text-xs text-zinc-400">PinkCherry + distilled LoRA. I2V and T2V.</span>
</button>
</div>
</div>
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Start from</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoStart === 'still' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoStart = 'still'"
>
<span class="block font-semibold">A still</span>
<span class="text-xs text-zinc-400">Image to video. Extensions use the last frame.</span>
</button>
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoStart === 'text' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoStart = 'text'"
>
<span class="block font-semibold">Text only</span>
<span class="text-xs text-zinc-400">No start still. Shot 2+ still continues from last frame.</span>
</button>
</div>
<p class="mt-2 text-xs text-zinc-500">{{ videoModelHint }}</p>
</div>
<div v-if="videoEngine === 'minimax' && videoStart === 'still'">
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">MiniMax graph</p>
<div class="grid grid-cols-2 gap-2">
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoWorkflow === 'v1' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoWorkflow = 'v1'"
:class="minimaxGraph === 'v1' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="minimaxGraph = 'v1'"
>
<span class="block font-semibold">v1 · validated</span>
<span class="text-xs text-zinc-400">First-frame image to video</span>
@@ -385,8 +442,8 @@
<button
type="button"
class="rounded-xl border px-3 py-2 text-left text-sm"
:class="videoWorkflow === 'v2' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="videoWorkflow = 'v2'"
:class="minimaxGraph === 'v2' ? 'border-amber-300 bg-amber-400/10 text-amber-100' : 'border-white/10 hover:border-white/20'"
@click="minimaxGraph = 'v2'"
>
<span class="block font-semibold">v2 · identity refs</span>
<span class="text-xs text-zinc-400">Optional lock with up to 4 stills</span>
@@ -394,7 +451,7 @@
</div>
</div>
<div v-if="videoWorkflow === 'v2'" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 space-y-3">
<div v-if="videoEngine === 'minimax' && videoStart === 'still' && minimaxGraph === 'v2'" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 space-y-3">
<label class="flex cursor-pointer items-center justify-between gap-3 text-sm">
<span>
<span class="block font-medium text-zinc-200">Use identity references</span>
@@ -466,13 +523,13 @@
<span class="text-xs text-zinc-400">{{ option.id === 'auto' ? autoHint : option.hint }}</span>
</button>
</div>
<p class="mt-2 text-xs text-zinc-500">Output {{ outputSize.width }} × {{ outputSize.height }}{{ aspect === 'auto' ? ' · matching the still' : '' }}</p>
<p class="mt-2 text-xs text-zinc-500">Output {{ outputSize.width }} × {{ outputSize.height }}{{ aspect === 'auto' ? (file ? ' · matching the still' : ' · 16:9 until a still is loaded') : '' }}</p>
</div>
<div>
<p class="mb-2 text-xs uppercase tracking-wider text-zinc-500">Steps</p>
<div class="grid grid-cols-2 gap-2">
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = true">{{ videoWorkflow === 'v2' ? 6 : 8 }} Turbo · LoRA on</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="!turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = false">20 High-Fidelity · LoRA bypass</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = true">{{ ltxVideo ? '8 Distilled LoRA' : (videoWorkflow === 'v2' ? '6 Turbo · LoRA on' : '8 Turbo · LoRA on') }}</button>
<button type="button" class="rounded-xl border px-3 py-2 text-sm" :class="!turbo ? 'border-amber-300 bg-amber-400/10' : 'border-white/10'" @click="turbo = false">{{ ltxVideo ? '12 Quality · LoRA on' : '20 High-Fidelity · LoRA bypass' }}</button>
</div>
</div>
<div class="text-sm">
@@ -499,7 +556,7 @@
@change="onCfgNumber"
>
</div>
<p class="mt-1 text-[11px] text-zinc-500">{{ turbo ? 'Default 1.5 with Turbo LoRA' : 'Default 4.0 with High-Fidelity' }}</p>
<p class="mt-1 text-[11px] text-zinc-500">{{ ltxVideo ? 'Default 3.5 for PinkCherry' : (turbo ? 'Default 1.5 with Turbo LoRA' : 'Default 4.0 with High-Fidelity') }}</p>
</div>
<div class="grid grid-cols-2 gap-3">
<label class="text-sm">
@@ -1256,6 +1313,19 @@
<script setup lang="ts">
import { parseEditPasses, looksLikeShotScript } from '~/utils/parseRecommend'
import { parseShotScript } from '~/utils/parseShots'
import {
composeVideoWorkflow,
isLtxWorkflow,
isTextToVideo,
minimaxGraphOf,
parseVideoWorkflow,
videoEngineOf,
videoStartOf,
type MiniMaxGraphId,
type VideoEngineId,
type VideoStartId
} from '~/utils/videoModels'
const runtimeConfig = useRuntimeConfig()
@@ -1311,7 +1381,7 @@ interface LibraryClip {
familyId?: string
parentClipId?: string
chainIndex?: number
workflow?: 'v1' | 'v2'
workflow?: string
stillId?: string
sound?: boolean
hideInput?: boolean
@@ -1340,7 +1410,7 @@ interface RetryDraft {
samplerName?: string
scheduler?: string
extensions?: { prompt: string; duration: number }[]
workflow?: 'v1' | 'v2'
workflow?: string
}
interface QueuedExtension {
@@ -1517,7 +1587,12 @@ const pickerSlot = ref<'main' | 'editRef' | number>('main')
const libraryFilter = ref<'all' | 'images' | 'videos'>('all')
const selectedKeys = ref<string[]>([])
const lastSelectedKey = ref('')
const videoWorkflow = ref('v1')
const videoEngine = ref<VideoEngineId>('minimax')
const videoStart = ref<VideoStartId>('still')
const minimaxGraph = ref<MiniMaxGraphId>('v1')
const videoWorkflow = computed(() => composeVideoWorkflow(videoEngine.value, videoStart.value, minimaxGraph.value))
const textToVideo = computed(() => isTextToVideo(videoWorkflow.value))
const ltxVideo = computed(() => isLtxWorkflow(videoWorkflow.value))
const useIdentityRefs = ref(false)
const identityRefs = ref<(File | null)[]>([null, null, null, null])
const identityPreviews = ref(['', '', '', ''])
@@ -1653,10 +1728,22 @@ const outputSize = computed(() => {
})
const autoHint = computed(() => {
if (!imageWidth.value || !imageHeight.value) return 'Match the still'
if (!imageWidth.value || !imageHeight.value) {
return textToVideo.value ? '16:9 without a still' : 'Match the still'
}
const size = autoResolution(imageWidth.value, imageHeight.value)
return `${size.width} × ${size.height} · from still`
})
const videoModelHint = computed(() => {
if (ltxVideo.value) {
return textToVideo.value
? 'PinkCherry text-to-video. Distilled LoRA stays on. Silent MP4 — LTX has no audio VAE here.'
: 'PinkCherry image-to-video. Distilled LoRA stays on. Silent MP4 — LTX has no audio VAE here.'
}
if (textToVideo.value) return 'MiniMax H3 text-to-video with native stereo audio. Shot 2+ switches to last-frame image-to-video.'
if (videoWorkflow.value === 'v2') return 'MiniMax H3 image-to-video. Identity refs stay on v2 with a start still.'
return 'MiniMax H3 image-to-video with native stereo audio.'
})
const frameCount = computed(() => Math.max(5, Math.floor(duration.value * fps.value)))
const cfgPct = computed(() => ((cfg.value - CFG_MIN) / (CFG_MAX - CFG_MIN)) * 100)
const currentClip = computed(() => clips.value.find(clip => clip.id === currentClipId.value))
@@ -1671,7 +1758,7 @@ const queueReady = computed(() => {
return extensionQueue.value.every(item => item.prompt.trim())
})
const videoGenerateDisabled = computed(() => {
return videoBusy.value || !file.value || !prompt.value.trim() || !folderId.value || !queueReady.value
return videoBusy.value || (!textToVideo.value && !file.value) || !prompt.value.trim() || !folderId.value || !queueReady.value
})
const recommendDisabled = computed(() => {
if (recommendBusy.value) return true
@@ -1715,6 +1802,10 @@ const editLabel = computed(() => {
return n === 1 ? 'Edit image + 1 pass' : `Edit image + ${n} passes`
})
const parsedShots = computed(() => parseShotScript(prompt.value))
const recommendedShots = computed(() => {
if (!recommendText.value || recommendBusy.value) return []
return parseShotScript(recommendText.value)
})
const identityPrompting = computed(() => studioMode.value === 'video' && videoWorkflow.value === 'v2' && useIdentityRefs.value)
const extraIdentityPictures = computed(() => identityRefs.value.flatMap((item, index) => item ? [index + 2] : []))
const identityActionPrompt = computed(() => {
@@ -2101,7 +2192,20 @@ onMounted(async () => {
hideInputPreview.value = localStorage.getItem('aigen-hide-input-preview') === 'true'
withSound.value = localStorage.getItem('aigen-generate-sound') !== 'false'
autoplayEnabled.value = localStorage.getItem('aigen-autoplay') === 'true'
videoWorkflow.value = localStorage.getItem('aigen-video-workflow') === 'v2' ? 'v2' : 'v1'
videoEngine.value = 'minimax'
videoStart.value = 'still'
minimaxGraph.value = 'v1'
try {
const saved = parseVideoWorkflow(localStorage.getItem('aigen-video-workflow') || '')
videoEngine.value = videoEngineOf(saved)
videoStart.value = videoStartOf(saved)
minimaxGraph.value = minimaxGraphOf(saved)
const engine = localStorage.getItem('aigen-video-engine')
const start = localStorage.getItem('aigen-video-start')
if (engine === 'ltx' || engine === 'minimax') videoEngine.value = engine
if (start === 'text' || start === 'still') videoStart.value = start
if (videoEngine.value === 'ltx') applyEngineDefaults('ltx')
} catch { /* ignore */ }
useIdentityRefs.value = localStorage.getItem('aigen-use-identity-refs') === 'true'
activeQueueId.value = localStorage.getItem(QUEUE_STORE) || ''
const savedPreset = localStorage.getItem('aigen-prompt-preset')
@@ -2190,6 +2294,18 @@ watch(videoWorkflow, (value) => {
} catch { /* ignore */ }
})
watch(videoEngine, (value) => {
try {
localStorage.setItem('aigen-video-engine', value)
} catch { /* ignore */ }
})
watch(videoStart, (value) => {
try {
localStorage.setItem('aigen-video-start', value)
} catch { /* ignore */ }
})
watch(useIdentityRefs, (value) => {
try {
localStorage.setItem('aigen-use-identity-refs', String(value))
@@ -2197,7 +2313,8 @@ watch(useIdentityRefs, (value) => {
})
watch(turbo, (on) => {
if (!cfgTouched.value) cfg.value = on ? CFG_TURBO : CFG_QUALITY
if (cfgTouched.value) return
cfg.value = ltxVideo.value ? 3.5 : (on ? CFG_TURBO : CFG_QUALITY)
})
function clampDuration(value: number) {
@@ -2336,7 +2453,7 @@ function applyRecommend() {
}
} else {
prompt.value = text
if (looksLikeShotScript(text)) shotScriptMode.value = true
if (looksLikeShotScript(text) || parseShotScript(text).length > 1) shotScriptMode.value = true
}
recommendApplied.value = true
toast('Recommended prompt applied')
@@ -2709,7 +2826,7 @@ async function rerun(item: LibraryClip, collection = false) {
if (showPrivacyToggles.value) hideInputPreview.value = target.hideInput === true
folderId.value = target.folderId
browseFolderId.value = target.folderId
videoWorkflow.value = target.workflow === 'v2' ? 'v2' : 'v1'
applyVideoWorkflow(target.workflow)
if (typeof target.sound === 'boolean') withSound.value = target.sound
shotScriptMode.value = restoreAll
useIdentityRefs.value = false
@@ -2722,7 +2839,8 @@ async function rerun(item: LibraryClip, collection = false) {
}))
: []
try {
await loadRerunImage(target, parts)
const initialTextToVideo = textToVideo.value && !target.parentClipId && !(target.chainIndex)
if (!(initialTextToVideo && !target.stillId)) await loadRerunImage(target, parts)
statusMessage.value = restoreAll
? 'Collection loaded into the form. Generate when you are ready.'
: 'Clip loaded into the form. Generate when you are ready.'
@@ -2730,8 +2848,12 @@ async function rerun(item: LibraryClip, collection = false) {
? 'Shot script and settings restored. Make any changes, then generate.'
: 'Inputs restored. Make any changes, then generate.')
} catch {
statusMessage.value = 'Clip settings loaded. Re-drop the start still if it is missing.'
toast('Settings restored, but the start still could not be loaded. Pick it again.')
statusMessage.value = textToVideo.value
? 'Clip settings loaded. Add a still only if you want image-to-video.'
: 'Clip settings loaded. Re-drop the start still if it is missing.'
toast(textToVideo.value
? 'Settings restored. Text-to-video does not need a start still.'
: 'Settings restored, but the start still could not be loaded. Pick it again.')
}
}
@@ -2746,7 +2868,7 @@ async function loadDraft(draft: RetryDraft) {
withSound.value = draft.sound !== false
if (draft.hideInput && showPrivacyToggles.value) hideInputPreview.value = true
duration.value = clampDuration(Number(draft.duration))
if (draft.workflow === 'v2') videoWorkflow.value = 'v2'
applyVideoWorkflow(draft.workflow)
extensionQueue.value = (draft.extensions || []).map(item => ({
id: crypto.randomUUID(),
prompt: String(item.prompt || ''),
@@ -2758,8 +2880,12 @@ async function loadDraft(draft: RetryDraft) {
readFile(new File([blob], filename, { type: blob.type || 'image/png' }), { keepThumbnailPref: true })
statusMessage.value = 'Held job restored. Generate when ComfyUI is free.'
} catch {
statusMessage.value = 'Held job settings restored. Re-drop the still if the image is missing.'
toast('Settings restored. Unlock the folder or re-drop the still if the image is missing.')
if (textToVideo.value) {
statusMessage.value = 'Held job restored. Generate when ComfyUI is free.'
} else {
statusMessage.value = 'Held job settings restored. Re-drop the still if the image is missing.'
toast('Settings restored. Unlock the folder or re-drop the still if the image is missing.')
}
}
}
@@ -3363,8 +3489,32 @@ async function useEditAsInput() {
}
}
function applyEngineDefaults(engine: VideoEngineId) {
if (engine === 'ltx') {
samplerName.value = 'euler'
if (!cfgTouched.value) cfg.value = 3.5
return
}
if (samplerName.value === 'euler') samplerName.value = 'res_multistep'
if (!cfgTouched.value) cfg.value = turbo.value ? CFG_TURBO : CFG_QUALITY
}
function selectVideoEngine(engine: VideoEngineId) {
if (videoEngine.value === engine) return
cfgTouched.value = false
videoEngine.value = engine
applyEngineDefaults(engine)
}
function applyVideoWorkflow(raw?: string) {
const parsed = parseVideoWorkflow(raw)
videoEngine.value = videoEngineOf(parsed)
videoStart.value = videoStartOf(parsed)
minimaxGraph.value = minimaxGraphOf(parsed)
}
async function generate() {
if (!file.value || !prompt.value.trim()) return
if ((!file.value && !textToVideo.value) || !prompt.value.trim()) return
if (!queueReady.value) {
toast(shotScriptMode.value ? 'Each shot in the script needs a prompt.' : 'Each queued extension needs a prompt.')
return
@@ -3401,7 +3551,7 @@ async function generate() {
startTimer('video')
try {
const body = new FormData()
body.append('image', file.value)
if (file.value) body.append('image', file.value)
body.append('prompt', initialPrompt)
body.append('aspect', aspect.value)
body.append('width', String(outputSize.value.width))