Send V2 identity prompts to Reference to Video instead of Image to Video.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-08-26 23:08:43 -05:00
co-authored by Cursor
parent e334cb7a7a
commit d280985553
3 changed files with 33 additions and 17 deletions
+14 -8
View File
@@ -109,10 +109,10 @@
</div>
<p v-if="shotScriptMode && studioMode === 'video'" class="mb-2 text-xs text-zinc-500">
Paste a block. Text before the first marker is shot 1. Put <span class="font-medium text-zinc-300">shot 2</span>, <span class="font-medium text-zinc-300">shot 3</span> on their own lines to queue extensions — a line like <span class="font-medium text-zinc-300">Shot Type</span> will not start a new shot. Every shot uses the duration slider.
<span v-if="identityPrompting"> Identity boilerplate is prepended to the first shot only.</span>
<span v-if="identityPrompting"> Identity boilerplate is prepended to the first shot only, then concatenated into Reference to Video.</span>
</p>
<p v-else-if="identityPrompting" class="mb-2 text-xs text-zinc-500">
Write camera, action, and audio. We prepend the <span class="font-medium text-zinc-300">&lt;Picture 1&gt;</span> identity lock automatically.
Write camera, action, and audio. That is the dynamic action prompt. We concatenate the <span class="font-medium text-zinc-300">&lt;Picture 1&gt;</span> identity prefix in front of it and send the result to <span class="font-medium text-zinc-300">Reference to Video</span>, not Image to Video.
</p>
<textarea
v-model="prompt"
@@ -121,8 +121,13 @@
:placeholder="promptPlaceholder"
/>
<details v-if="identityPrompting && prompt.trim()" class="mt-2 rounded-xl border border-white/10 bg-zinc-950/40 px-3 py-2">
<summary class="cursor-pointer text-xs text-zinc-400">Prompt sent to MiniMax</summary>
<p class="mt-2 whitespace-pre-wrap text-xs text-zinc-300">{{ composedIdentityPrompt }}</p>
<summary class="cursor-pointer text-xs text-zinc-400">Prompt sent to Reference to Video</summary>
<p class="mt-2 text-[10px] uppercase tracking-wide text-zinc-500">Identity prefix</p>
<p class="mt-1 whitespace-pre-wrap text-xs text-zinc-400">{{ identityBoilerplate(extraIdentityPictures) }}</p>
<p class="mt-3 text-[10px] uppercase tracking-wide text-zinc-500">Dynamic action prompt</p>
<p class="mt-1 whitespace-pre-wrap text-xs text-zinc-300">{{ identityActionPrompt }}</p>
<p class="mt-3 text-[10px] uppercase tracking-wide text-zinc-500">Concatenated result</p>
<p class="mt-1 whitespace-pre-wrap text-xs text-zinc-300">{{ composedIdentityPrompt }}</p>
</details>
<div v-if="shotScriptMode && studioMode === 'video'" class="mt-3 space-y-2">
<p v-if="!parsedShots.length" class="text-xs text-zinc-500">Add a first shot, then <span class="text-zinc-300">shot 2</span> for an extension.</p>
@@ -1296,11 +1301,12 @@ const generateLabel = computed(() => {
const parsedShots = computed(() => parseShotScript(prompt.value))
const identityPrompting = computed(() => studioMode.value === 'video' && videoWorkflow.value === 'v2' && useIdentityRefs.value)
const extraIdentityPictures = computed(() => identityRefs.value.flatMap((item, index) => item ? [index + 2] : []))
const identityActionPrompt = computed(() => {
if (shotScriptMode.value) return parsedShots.value[0]?.prompt || prompt.value
return prompt.value
})
const composedIdentityPrompt = computed(() => {
const action = shotScriptMode.value
? (parsedShots.value[0]?.prompt || prompt.value)
: prompt.value
return buildIdentityPrompt(action, extraIdentityPictures.value)
return buildIdentityPrompt(identityActionPrompt.value, extraIdentityPictures.value)
})
const promptPlaceholder = computed(() => {
if (studioMode.value === 'edit') {
+6 -4
View File
@@ -163,12 +163,14 @@ export function extractPromptFromHistory(entry: unknown) {
const prompt = (entry as { prompt?: unknown[] })?.prompt
const graph = Array.isArray(prompt) ? prompt[2] : null
if (!graph || typeof graph !== 'object') return ''
let imageToVideo = ''
for (const node of Object.values(graph as Record<string, { class_type?: string; inputs?: Record<string, unknown> }>)) {
if (node?.class_type === 'MiniMaxH3ImageToVideo') {
return String(node.inputs?.prompt || node.inputs?.text || '').trim()
}
const text = String(node.inputs?.prompt || node.inputs?.text || '').trim()
if (!text) continue
if (node?.class_type === 'MiniMaxH3ReferenceToVideo') return text
if (node?.class_type === 'MiniMaxH3ImageToVideo') imageToVideo = text
}
return ''
return imageToVideo
}
export function extractClipMetaFromHistory(entry: unknown) {
+13 -5
View File
@@ -73,7 +73,8 @@ export function buildWorkflow(params: GenerateParams) {
const refs = Array.from({ length: 4 }, (_, index) => String(params.referenceImageNames?.[index] || '').trim())
const extraPictures = refs.flatMap((name, index) => name ? [index + 2] : [])
const useRefs = version === 'v2' && params.useIdentityRefs === true
const graphPrompt = useRefs ? buildIdentityPrompt(params.prompt, extraPictures) : params.prompt
const actionPrompt = params.prompt
const graphPrompt = useRefs ? buildIdentityPrompt(actionPrompt, extraPictures) : actionPrompt
for (const [id, node] of Object.entries(graph)) {
if (IMAGE_CLASSES.has(node.class_type) && 'image' in node.inputs && id === '114') {
@@ -81,19 +82,20 @@ export function buildWorkflow(params: GenerateParams) {
}
if (PROMPT_CLASSES.has(node.class_type) || node.class_type.includes('Qwen3')) {
if ('prompt' in node.inputs) node.inputs.prompt = graphPrompt
if ('text' in node.inputs) node.inputs.text = graphPrompt
const nodePrompt = promptForNode(node.class_type, useRefs, actionPrompt, graphPrompt)
if ('prompt' in node.inputs) node.inputs.prompt = nodePrompt
if ('text' in node.inputs) node.inputs.text = nodePrompt
}
if (node.class_type === 'MiniMaxH3ImageToVideo') {
node.inputs.prompt = graphPrompt
node.inputs.prompt = useRefs ? '' : actionPrompt
if (typeof node.inputs.width === 'number') node.inputs.width = width
if (typeof node.inputs.height === 'number') node.inputs.height = height
if (typeof node.inputs.length === 'number') node.inputs.length = params.length
}
if (node.class_type === 'MiniMaxH3ReferenceToVideo') {
node.inputs.prompt = graphPrompt
node.inputs.prompt = useRefs ? graphPrompt : actionPrompt
if (!Array.isArray(node.inputs.length)) node.inputs.length = ['174', 1]
}
@@ -206,6 +208,12 @@ export function buildWorkflow(params: GenerateParams) {
return graph
}
function promptForNode(classType: string, useRefs: boolean, actionPrompt: string, graphPrompt: string) {
if (classType === 'MiniMaxH3ImageToVideo') return useRefs ? '' : actionPrompt
if (classType === 'MiniMaxH3ReferenceToVideo') return useRefs ? graphPrompt : actionPrompt
return graphPrompt
}
function applyV2IdentityPath(
graph: WorkflowGraph,
params: GenerateParams,