Choose generation mode automatically from starting image

This commit is contained in:
Towsty
2026-09-06 17:42:53 -05:00
parent 666c922f40
commit bb91b4fd3f
4 changed files with 60 additions and 59 deletions
+1 -26
View File
@@ -38,26 +38,7 @@
</button> </button>
</div> </div>
<p v-else class="border-t border-white/10 px-4 py-2 text-xs text-zinc-500">MiniMax H3</p> <p v-else class="border-t border-white/10 px-4 py-2 text-xs text-zinc-500">MiniMax H3</p>
<div class="grid grid-cols-2 border-t border-white/10"> <p class="border-t border-white/10 px-4 py-2 text-xs text-zinc-400">{{ videoStart === 'still' ? 'Image to video' : 'Text to video' }} · automatic</p>
<button
type="button"
class="border-r border-white/10 px-3 py-2.5 text-left text-sm transition"
:class="engineBtnClass(current === 'video' && videoStart === 'still')"
@click="pickStart('still')"
>
<span class="block font-semibold">Still</span>
<span class="text-xs opacity-80">Image to video</span>
</button>
<button
type="button"
class="px-3 py-2.5 text-left text-sm transition"
:class="engineBtnClass(current === 'video' && videoStart === 'text')"
@click="pickStart('text')"
>
<span class="block font-semibold">Text</span>
<span class="text-xs opacity-80">No start still</span>
</button>
</div>
</div> </div>
<div <div
@@ -157,7 +138,6 @@ const props = withDefaults(defineProps<{
const emit = defineEmits<{ const emit = defineEmits<{
select: [kind: 'video' | 'image'] select: [kind: 'video' | 'image']
'video-engine': [engine: 'minimax' | 'ltx'] 'video-engine': [engine: 'minimax' | 'ltx']
'video-start': [start: 'still' | 'text']
'image-engine': [engine: 'flux' | 'krea'] 'image-engine': [engine: 'flux' | 'krea']
'music-engine': [engine: 'ace' | 'yue'] 'music-engine': [engine: 'ace' | 'yue']
}>() }>()
@@ -172,11 +152,6 @@ function pickVideo(engine: 'minimax' | 'ltx') {
if (props.current !== 'video') emit('select', 'video') if (props.current !== 'video') emit('select', 'video')
} }
function pickStart(start: 'still' | 'text') {
emit('video-start', start)
if (props.current !== 'video') emit('select', 'video')
}
function pickImage(engine: 'flux' | 'krea') { function pickImage(engine: 'flux' | 'krea') {
emit('image-engine', engine) emit('image-engine', engine)
if (props.current !== 'image') emit('select', 'image') if (props.current !== 'image') emit('select', 'image')
+17 -33
View File
@@ -16,7 +16,6 @@
:image-engine="v2Engine" :image-engine="v2Engine"
@select="selectStudioKind" @select="selectStudioKind"
@video-engine="selectVideoEngine" @video-engine="selectVideoEngine"
@video-start="selectVideoStart"
@image-engine="setV2Engine" @image-engine="setV2Engine"
/> />
<div class="grid min-w-0 gap-6 lg:grid-cols-[minmax(0,1fr)_minmax(0,1fr)] lg:items-stretch"> <div class="grid min-w-0 gap-6 lg:grid-cols-[minmax(0,1fr)_minmax(0,1fr)] lg:items-stretch">
@@ -25,7 +24,7 @@
<h2 class="font-display text-2xl font-bold">Input</h2> <h2 class="font-display text-2xl font-bold">Input</h2>
<p class="text-sm text-zinc-400">{{ studioMode === 'editv2' <p class="text-sm text-zinc-400">{{ studioMode === 'editv2'
? (v2Mode === 'generate' ? (v2Mode === 'generate'
? 'Generate: no reference image. Prompt only.' ? 'Text-to-image: describe what you want. Add a starting image to switch to image-to-image.'
: v2Mode === 'iterate' : v2Mode === 'iterate'
? (file ? (file
? (editRefFile ? (editRefFile
@@ -41,7 +40,7 @@
? 'Drop the still to edit. An optional second still applies image 2 onto image 1. Files leave the desktop after save.' ? 'Drop the still to edit. An optional second still applies image 2 onto image 1. Files leave the desktop after save.'
: (textToVideo : (textToVideo
? 'Text-to-video: write a prompt and generate. No start still required. Shot 2+ still continues from the last frame.' ? 'Text-to-video: write a prompt and generate. No start still required. Shot 2+ still continues from the last frame.'
: 'Drop a still, describe the motion, then send the job. Switch the Video card to Text for text-to-video.') }}</p> : 'Describe the motion and generate. Add a starting image for image-to-video; remove it for text-to-video.') }}</p>
</div> </div>
<div <div
@@ -85,10 +84,11 @@
<span class="mt-1 block text-zinc-500">Text-to-video does not need one. Shot 2+ still extends from the last frame.</span> <span class="mt-1 block text-zinc-500">Text-to-video does not need one. Shot 2+ still extends from the last frame.</span>
</div> </div>
<div <div
v-else-if="studioMode === 'editv2' && v2Mode === 'generate' && (file || inputStillId)" v-else-if="studioMode === 'editv2' && v2Mode === 'generate'"
class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm text-zinc-400" class="rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3 text-sm text-zinc-400"
> >
Still A is loaded for Edit / Compose / Refine / Iterate. Generate ignores it and uses the prompt only. <button type="button" class="font-medium text-amber-100 hover:underline" @click="openPicker('main')">Optional · add a starting image</button>
<span class="mt-1 block text-zinc-400">No image: text-to-image. Add an image to edit it instead.</span>
</div> </div>
<div <div
v-else-if="studioMode === 'editv2' && v2Mode === 'iterate'" v-else-if="studioMode === 'editv2' && v2Mode === 'iterate'"
@@ -131,12 +131,12 @@
<button <button
type="button" type="button"
class="rounded-full border px-3 py-1 text-xs font-medium disabled:cursor-not-allowed disabled:opacity-40" class="rounded-full border px-3 py-1 text-xs font-medium disabled:cursor-not-allowed disabled:opacity-40"
:class="v2Mode === 'edit' ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'" :class="(v2Mode === 'edit' || v2Mode === 'generate') ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'"
:disabled="v2ModeLocked && v2Mode !== 'edit'" :disabled="v2ModeLocked && v2Mode !== 'edit' && v2Mode !== 'generate'"
:title="v2ModeLocked && v2Mode !== 'edit' ? 'Remove extra passes to change graphs' : ''" :title="v2ModeLocked && v2Mode !== 'edit' && v2Mode !== 'generate' ? 'Remove extra passes to change graphs' : 'Uses your image when one is provided; otherwise generates from text'"
@click="setV2Mode('edit')" @click="setV2Mode('edit')"
> >
Edit Image
</button> </button>
<button <button
type="button" type="button"
@@ -158,16 +158,6 @@
> >
Refine Refine
</button> </button>
<button
type="button"
class="rounded-full border px-3 py-1 text-xs font-medium disabled:cursor-not-allowed disabled:opacity-40"
:class="v2Mode === 'generate' ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'"
:disabled="v2ModeLocked && v2Mode !== 'generate'"
:title="v2ModeLocked && v2Mode !== 'generate' ? 'Remove extra passes to change graphs' : ''"
@click="setV2Mode('generate')"
>
Generate
</button>
<button <button
type="button" type="button"
class="rounded-full border px-3 py-1 text-xs font-medium disabled:cursor-not-allowed disabled:opacity-40" class="rounded-full border px-3 py-1 text-xs font-medium disabled:cursor-not-allowed disabled:opacity-40"
@@ -2253,6 +2243,7 @@
</template> </template>
<script setup lang="ts"> <script setup lang="ts">
import { imageModeForInput, videoStartForInput } from '~/utils/generationInput'
import { normalizeImageIterations, type ImageIteration } from '~/utils/imageIterations' import { normalizeImageIterations, type ImageIteration } from '~/utils/imageIterations'
import { composePromptParts, formatPromptParts, resolvePromptWrappers, restorePromptParts } from '~/utils/promptParts' import { composePromptParts, formatPromptParts, resolvePromptWrappers, restorePromptParts } from '~/utils/promptParts'
import { parseEditPasses, looksLikeShotScript } from '~/utils/parseRecommend' import { parseEditPasses, looksLikeShotScript } from '~/utils/parseRecommend'
@@ -2373,7 +2364,6 @@ import {
minimaxGraphOf, minimaxGraphOf,
parseVideoWorkflow, parseVideoWorkflow,
videoEngineOf, videoEngineOf,
videoStartOf,
type MiniMaxGraphId, type MiniMaxGraphId,
type VideoEngineId, type VideoEngineId,
type VideoStartId type VideoStartId
@@ -2720,7 +2710,11 @@ const editScaleToTotalPixels = ref(false)
const editScaleMegapixels = ref(1) const editScaleMegapixels = ref(1)
const editRefFile = ref<File | null>(null) const editRefFile = ref<File | null>(null)
const editRefPreview = ref('') const editRefPreview = ref('')
const v2Mode = ref<ImageV2Mode>('edit') const selectedV2Mode = ref<ImageV2Mode>('edit')
const v2Mode = computed<ImageV2Mode>({
get: () => imageModeForInput(selectedV2Mode.value, Boolean(file.value), editPassQueue.value.length > 0),
set: mode => { selectedV2Mode.value = mode }
})
const v2Engine = ref<ImageV2Engine>('flux') const v2Engine = ref<ImageV2Engine>('flux')
const v2Task = ref<ImageV2Task>('scene') const v2Task = ref<ImageV2Task>('scene')
const v2Negative = ref('') const v2Negative = ref('')
@@ -2869,7 +2863,7 @@ const lastSelectedKey = ref('')
const pendingLibraryDelete = ref<PendingLibraryDelete | null>(null) const pendingLibraryDelete = ref<PendingLibraryDelete | null>(null)
const libraryDeleteBusy = ref(false) const libraryDeleteBusy = ref(false)
const videoEngine = ref<VideoEngineId>('minimax') const videoEngine = ref<VideoEngineId>('minimax')
const videoStart = ref<VideoStartId>('still') const videoStart = computed<VideoStartId>(() => videoStartForInput(Boolean(file.value)))
const minimaxGraph = ref<MiniMaxGraphId>('v2') const minimaxGraph = ref<MiniMaxGraphId>('v2')
const videoWorkflow = computed(() => composeVideoWorkflow(videoEngine.value, videoStart.value, minimaxGraph.value)) const videoWorkflow = computed(() => composeVideoWorkflow(videoEngine.value, videoStart.value, minimaxGraph.value))
const textToVideo = computed(() => isTextToVideo(videoWorkflow.value)) const textToVideo = computed(() => isTextToVideo(videoWorkflow.value))
@@ -4028,17 +4022,13 @@ onMounted(async () => {
withSound.value = localStorage.getItem('aigen-generate-sound') !== 'false' withSound.value = localStorage.getItem('aigen-generate-sound') !== 'false'
autoplayEnabled.value = localStorage.getItem('aigen-autoplay') === 'true' autoplayEnabled.value = localStorage.getItem('aigen-autoplay') === 'true'
videoEngine.value = 'minimax' videoEngine.value = 'minimax'
videoStart.value = 'still'
minimaxGraph.value = 'v2' minimaxGraph.value = 'v2'
try { try {
const saved = parseVideoWorkflow(localStorage.getItem('aigen-video-workflow') || '') const saved = parseVideoWorkflow(localStorage.getItem('aigen-video-workflow') || '')
videoEngine.value = videoEngineOf(saved) videoEngine.value = videoEngineOf(saved)
videoStart.value = videoStartOf(saved)
minimaxGraph.value = minimaxGraphOf(saved) === 'v1' ? 'v2' : minimaxGraphOf(saved) minimaxGraph.value = minimaxGraphOf(saved) === 'v1' ? 'v2' : minimaxGraphOf(saved)
const engine = localStorage.getItem('aigen-video-engine') const engine = localStorage.getItem('aigen-video-engine')
const start = localStorage.getItem('aigen-video-start')
if (engine === 'ltx' || engine === 'minimax') videoEngine.value = engine if (engine === 'ltx' || engine === 'minimax') videoEngine.value = engine
if (start === 'text' || start === 'still') videoStart.value = start
if (!ltxEnabled.value && videoEngine.value === 'ltx') { if (!ltxEnabled.value && videoEngine.value === 'ltx') {
videoEngine.value = 'minimax' videoEngine.value = 'minimax'
applyEngineDefaults('minimax') applyEngineDefaults('minimax')
@@ -7171,7 +7161,7 @@ async function editImage() {
} }
function setV2Mode(mode: ImageV2Mode) { function setV2Mode(mode: ImageV2Mode) {
if (v2Mode.value === mode) return if (v2Mode.value === imageModeForInput(mode, Boolean(file.value), editPassQueue.value.length > 0)) return
if (editPassQueue.value.length && v2PassChaining.value) { if (editPassQueue.value.length && v2PassChaining.value) {
toast('Remove extra passes to change graphs. Extra passes stay in the current mode.') toast('Remove extra passes to change graphs. Extra passes stay in the current mode.')
return return
@@ -7491,15 +7481,9 @@ function selectVideoEngine(engine: VideoEngineId) {
applyEngineDefaults(engine) applyEngineDefaults(engine)
} }
function selectVideoStart(start: VideoStartId) {
videoStart.value = start
if (studioMode.value !== 'video') selectStudioKind('video')
}
function applyVideoWorkflow(raw?: string) { function applyVideoWorkflow(raw?: string) {
const parsed = coerceVideoWorkflow(raw, ltxEnabled.value) const parsed = coerceVideoWorkflow(raw, ltxEnabled.value)
videoEngine.value = videoEngineOf(parsed) videoEngine.value = videoEngineOf(parsed)
videoStart.value = videoStartOf(parsed)
minimaxGraph.value = minimaxGraphOf(parsed) === 'v1' ? 'v2' : minimaxGraphOf(parsed) minimaxGraph.value = minimaxGraphOf(parsed) === 'v1' ? 'v2' : minimaxGraphOf(parsed)
} }
+31
View File
@@ -0,0 +1,31 @@
import test from 'node:test'
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import ts from 'typescript'
const source = readFileSync(new URL('../utils/generationInput.ts', import.meta.url), 'utf8')
const code = ts.transpileModule(source, {compilerOptions:{module:ts.ModuleKind.CommonJS}}).outputText
const api = {}
new Function('exports', code)(api)
test('image generation follows adding and removing the starting image, regardless of saved basic mode', () => {
for (const saved of ['edit','generate']) {
assert.equal(api.imageModeForInput(saved,false),'generate')
assert.equal(api.imageModeForInput(saved,true),'edit')
assert.equal(api.imageModeForInput(saved,false),'generate')
}
})
test('video generation follows the starting image', () => {
assert.equal(api.videoStartForInput(false),'text')
assert.equal(api.videoStartForInput(true),'still')
assert.equal(api.videoStartForInput(false),'text')
})
test('removing a batch input preserves the batch as text-based iterations', () => {
assert.equal(api.imageModeForInput('edit',false,true),'iterate')
assert.equal(api.imageModeForInput('edit',true,true),'edit')
})
test('explicit compose, refine and iterate modes remain explicit', () => {
for (const mode of ['compose','refine','iterate']) {
assert.equal(api.imageModeForInput(mode,false),mode)
assert.equal(api.imageModeForInput(mode,true),mode)
}
})
+11
View File
@@ -0,0 +1,11 @@
import type { ImageV2Mode } from './imageV2'
export function imageModeForInput(mode: ImageV2Mode, hasImage: boolean, hasPasses = false): ImageV2Mode {
if (mode === 'compose' || mode === 'refine' || mode === 'iterate') return mode
if (!hasImage && hasPasses) return 'iterate'
return hasImage ? 'edit' : 'generate'
}
export function videoStartForInput(hasImage: boolean): 'still' | 'text' {
return hasImage ? 'still' : 'text'
}