Add MiniMax and PinkCherry LTX as pickable video engines, with still or text start.

Shot 1 can run text-to-video; later shots still extend from the last frame. Recommend output is also normalized into shot-script cards.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-08-27 19:38:53 -05:00
co-authored by Cursor
parent 2bebeea526
commit 40cd17f35c
15 changed files with 676 additions and 120 deletions
+89 -11
View File
@@ -1,9 +1,18 @@
// Nitro bundles these JSON graphs into the production server output.
import workflowV1 from '../assets/workflow_minimax_video.json'
import workflowV2 from '../assets/workflow_minimax_video_v2.json'
import workflowLtx from '../assets/workflow_ltx_video.json'
import { buildIdentityPrompt } from '~/utils/identityPrompt'
import {
isLtxWorkflow,
isTextToVideo,
LTX_NEGATIVE,
parseVideoWorkflow,
type VideoWorkflowId
} from '~/utils/videoModels'
export type VideoWorkflowId = 'v1' | 'v2'
export type { VideoWorkflowId }
export { parseVideoWorkflow }
export interface GenerateParams {
prompt: string
@@ -31,6 +40,7 @@ type WorkflowGraph = Record<string, WorkflowNode>
const PROMPT_CLASSES = new Set([
'MiniMaxH3ImageToVideo',
'MiniMaxH3TextToVideo',
'MiniMaxH3ReferenceToVideo',
'CLIPTextEncode',
'CLIPTextEncodeQwen3VL',
@@ -58,16 +68,15 @@ function titleOf(node: WorkflowNode) {
}
function templateFor(id: VideoWorkflowId) {
if (isLtxWorkflow(id)) return workflowLtx as WorkflowGraph
return (id === 'v2' ? workflowV2 : workflowV1) as WorkflowGraph
}
export function parseVideoWorkflow(raw: unknown): VideoWorkflowId {
return String(raw || '').trim() === 'v2' ? 'v2' : 'v1'
}
export function buildWorkflow(params: GenerateParams) {
const version = params.workflow === 'v2' ? 'v2' : 'v1'
const graph = structuredClone(templateFor(version))
const version = parseVideoWorkflow(params.workflow)
if (isLtxWorkflow(version)) return buildLtxWorkflow(params, version)
const graph = structuredClone(templateFor(version === 't2v' ? 'v1' : version === 'v2' ? 'v2' : 'v1'))
const width = snap32(params.width)
const height = snap32(params.height)
const refs = Array.from({ length: 4 }, (_, index) => String(params.referenceImageNames?.[index] || '').trim())
@@ -155,7 +164,7 @@ export function buildWorkflow(params: GenerateParams) {
}
}
if (version === 'v1' && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
if (version === 'v1' && !isTextToVideo(version) && graph['128']?.class_type === 'ImageScaleToTotalPixels') {
graph['128'] = {
class_type: 'ImageScale',
inputs: {
@@ -169,7 +178,7 @@ export function buildWorkflow(params: GenerateParams) {
}
}
if (version === 'v1') {
if (version === 'v1' || version === 't2v') {
graph['105:cfg'] = {
class_type: 'FluxGuidance',
inputs: {
@@ -191,10 +200,11 @@ export function buildWorkflow(params: GenerateParams) {
graph[qualityId].inputs.value = params.turbo ? 20 : params.steps
}
if (turboId && graph[turboId]?.class_type === 'PrimitiveInt') {
graph[turboId].inputs.value = params.turbo ? params.steps : (params.workflow === 'v2' ? 6 : 8)
graph[turboId].inputs.value = params.turbo ? params.steps : (version === 'v2' ? 6 : 8)
}
}
if (version === 't2v') applyMinimaxTextToVideo(graph, params, width, height)
if (version === 'v2') applyV2IdentityPath(graph, params, useRefs, refs, graphPrompt)
if (params.sound === false) {
@@ -203,6 +213,7 @@ export function buildWorkflow(params: GenerateParams) {
if (graph['105:91']?.inputs) delete graph['105:91'].inputs.audio
delete graph['163']
if (graph['172']?.inputs) delete graph['172'].inputs.audio
if (graph['105:104']?.inputs) delete graph['105:104'].inputs.audio_vae
}
return graph
@@ -210,10 +221,76 @@ export function buildWorkflow(params: GenerateParams) {
function promptForNode(classType: string, useRefs: boolean, actionPrompt: string, graphPrompt: string) {
if (classType === 'MiniMaxH3ImageToVideo') return useRefs ? '' : actionPrompt
if (classType === 'MiniMaxH3TextToVideo') return actionPrompt
if (classType === 'MiniMaxH3ReferenceToVideo') return useRefs ? graphPrompt : actionPrompt
return graphPrompt
}
function applyMinimaxTextToVideo(graph: WorkflowGraph, params: GenerateParams, width: number, height: number) {
const node = graph['105:104']
if (node) {
node.class_type = 'MiniMaxH3TextToVideo'
node._meta = { title: 'MiniMax H3 Text to Video' }
delete node.inputs.first_frame
node.inputs.prompt = params.prompt
node.inputs.width = width
node.inputs.height = height
if (params.sound !== false && graph['105:24']) node.inputs.audio_vae = ['105:24', 0]
else delete node.inputs.audio_vae
}
delete graph['114']
delete graph['119']
delete graph['120']
delete graph['127']
delete graph['128']
}
function buildLtxWorkflow(params: GenerateParams, version: VideoWorkflowId) {
const graph = structuredClone(workflowLtx as WorkflowGraph)
const width = snap32(params.width)
const height = snap32(params.height)
const textToVideo = isTextToVideo(version)
if (graph['4']?.inputs) graph['4'].inputs.image = params.imageName
if (graph['5']?.inputs) {
graph['5'].inputs.width = width
graph['5'].inputs.height = height
}
if (graph['6']?.inputs) graph['6'].inputs.text = params.prompt
if (graph['7']?.inputs) graph['7'].inputs.text = LTX_NEGATIVE
if (graph['8']?.inputs && typeof graph['8'].inputs.length === 'number') graph['8'].inputs.length = params.length
if (graph['9']?.inputs) graph['9'].inputs.frame_rate = params.fps
if (graph['10']?.inputs) graph['10'].inputs.cfg = params.cfg
if (graph['11']?.inputs) graph['11'].inputs.sampler_name = params.samplerName || 'euler'
if (graph['12']?.inputs) graph['12'].inputs.steps = params.steps
if (graph['13']?.inputs) graph['13'].inputs.noise_seed = params.seed
if (graph['16']?.inputs) {
graph['16'].inputs.frame_rate = params.fps
graph['16'].inputs.filename_prefix = params.filenamePrefix || 'video/LTX23'
}
if (graph['21']?.inputs) graph['21'].inputs.strength_model = 1
if (textToVideo) {
delete graph['4']
delete graph['5']
graph['8'] = {
class_type: 'EmptyLTXVLatentVideo',
inputs: {
width,
height,
length: params.length,
batch_size: 1
},
_meta: { title: 'Empty LTX latent' }
}
if (graph['9']?.inputs) {
graph['9'].inputs.positive = ['6', 0]
graph['9'].inputs.negative = ['7', 0]
}
if (graph['12']?.inputs) graph['12'].inputs.latent = ['8', 0]
if (graph['14']?.inputs) graph['14'].inputs.latent_image = ['8', 0]
}
return graph
}
function applyV2IdentityPath(
graph: WorkflowGraph,
params: GenerateParams,
@@ -304,9 +381,10 @@ function labelsFrom(template: WorkflowGraph) {
export const NODE_LABELS: Record<string, string> = {
...Object.fromEntries(labelsFrom(workflowV1 as WorkflowGraph)),
...Object.fromEntries(labelsFrom(workflowV2 as WorkflowGraph)),
...Object.fromEntries(labelsFrom(workflowLtx as WorkflowGraph)),
...LABEL_OVERRIDES
}
export function isEncodingNode(node: string) {
return node === '92' || node === '105:91' || node === '172' || /encoding|saving mp4|create video/i.test(NODE_LABELS[node] || '')
return node === '92' || node === '16' || node === '105:91' || node === '172' || /encoding|saving mp4|create video|save mp4/i.test(NODE_LABELS[node] || '')
}