Restore automatic video batch continuation and prepare stable Comfy loading

This commit is contained in:
Towsty
2026-09-09 06:58:35 -05:00
parent 4bf992210d
commit badf59d914
10 changed files with 83 additions and 21 deletions
+12
View File
@@ -0,0 +1,12 @@
# Video batch continuation and HostBuffer failure
The September 9 investigation found two separate problems:
1. Automatic continuation defaulted off and its checkbox had moved inside Generation Settings. Paused batches hid their “Run remaining” controls. Resume reused the previous one-shot budget, so it paused again after the next shot. Resume also restored a live flag before the shot queue overwrote it with `autoRun=false`.
2. The local `aigen-headless.log` traceback reports `comfy_aimdo/host_buffer.py` → `comfy/memory_management.py` while the Qwen text encoder is loading weights. This is not FFmpeg frame extraction. Commits `da8d0e4` and `4bf9922` change FFmpeg retries and do not address that model-loader failure.
New video batches now default to automatic continuation. Its checkbox stays outside the settings drawer. Paused batches show “Run remaining”; the bounded action says “Run only shot N”. Resume restores automatic continuation in both the Studio row and shot queue and passes the refreshed run-all policy to the dispatcher. Explicit one-shot / next-three actions remain bounded. Errors still stop the batch instead of silently skipping failed shots.
For the observed Windows HostBuffer failure, the connector now prepares Comfy's supported `--disable-dynamic-vram` launch flag when that installed version supports it. This selects estimate-based loading instead of AIMDO dynamic loading. It changes memory management, not the generation graph, handoff frame, refinement, model precision, or source segments. It may change loading speed and memory use; GPU inference has not been tested. `COMFY_DYNAMIC_VRAM=true` on the connector opts back into upstream defaults.
The memory change takes effect only after the connector is reloaded and Comfy is next started. Do not restart either during an active render. No running job or existing queue was changed during diagnosis. The checkbox default applies to new batches; use “Run remaining” on an existing paused batch after deploying the UI fix.
+12 -11
View File
@@ -1099,16 +1099,6 @@
</details>
</div>
<label
v-if="studioMode === 'video' && queuedExtensionCount > 0"
class="flex cursor-pointer items-start gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3"
>
<input v-model="queueAutoRun" type="checkbox" class="mt-1">
<span>
<span class="block text-sm font-medium text-zinc-200">Run remaining shots automatically</span>
<span class="mt-0.5 block text-xs text-zinc-500">Overnight set-and-forget. Leave this off to edit prompts on Queue (under this job) and process the next shot when you are ready.</span>
</span>
</label>
<div v-if="studioMode === 'video'" class="space-y-3">
<label
class="flex cursor-pointer items-start gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3"
@@ -1145,6 +1135,17 @@
</label>
</div>
</GenerationSettings>
<label
v-if="studioMode === 'video' && queuedExtensionCount > 0"
class="flex cursor-pointer items-start gap-3 rounded-2xl border border-white/10 bg-zinc-950/40 px-4 py-3"
>
<input v-model="queueAutoRun" type="checkbox" class="mt-1">
<span>
<span class="block text-sm font-medium text-zinc-200">Run remaining shots automatically</span>
<span class="mt-0.5 block text-xs text-zinc-500">Overnight set-and-forget. Leave this off to edit prompts on Queue (under this job) and process the next shot when you are ready.</span>
</span>
</label>
<div class="sticky bottom-3 z-10 flex flex-wrap gap-3 rounded-2xl bg-zinc-950/95 p-2 shadow-lg">
<button
type="button"
@@ -2867,7 +2868,7 @@ const renamingFamily = ref(false)
const renameDraft = ref('')
const renameInput = ref<HTMLInputElement | null>(null)
const unlockPassword = ref('')
const queueAutoRun = ref(false)
const queueAutoRun = ref(true)
const refineExtensionFrame = ref(true)
const saveLosslessAnchor = ref(true)
const refinementDenoise = ref(REFINEMENT_DENOISE_DEFAULT)
+6 -6
View File
@@ -94,7 +94,7 @@
:disabled="Boolean(busyId)"
@click="cutIn(job)"
>
{{ job.cutIn ? 'Next after current shot' : (canCutIn ? 'Run next' : 'Start now') }}
{{ job.cutIn ? 'Next after current shot' : (canCutIn ? 'Prioritize batch' : 'Start batch') }}
</button>
<button
v-if="job.status === 'error'"
@@ -712,7 +712,7 @@ function jobLine(job: StudioJobRow) {
function continueLabel(queue: Queue) {
const next = nextSegment(queue)
if (!next) return 'Process next'
return `Continue shot ${next.index + 1}`
return `Run only shot ${next.index + 1}`
}
function segmentCaption(queue: Queue, segment: Segment) {
@@ -1134,11 +1134,11 @@ function queueIsArmed(queue: Queue) {
}
function canProcessShots(job: StudioJobRow) {
return Boolean(job.shots) && !jobIsGenerating(job) && !jobIsHeldPaused(job)
return Boolean(job.shots) && !jobIsGenerating(job)
}
function canProcessQueue(queue: Queue) {
return queue.status !== 'running' && !queueIsHeldPaused(queue)
return queue.status !== 'running'
}
function canPauseJob(job: StudioJobRow) {
@@ -1151,7 +1151,7 @@ function jobIsArmed(job: StudioJobRow) {
function jobPauseLabel(job: StudioJobRow) {
if (jobIsLatched(job)) return 'Will pause after this'
if (jobIsHeldPaused(job)) return 'Paused'
if (jobIsHeldPaused(job)) return 'Resume remaining'
return 'Pause after this'
}
@@ -1163,7 +1163,7 @@ function jobPauseTitle(job: StudioJobRow) {
function queuePauseLabel(queue: Queue) {
if (queueIsLatched(queue)) return 'Will pause after this'
if (queueIsHeldPaused(queue)) return 'Paused'
if (queueIsHeldPaused(queue)) return 'Resume remaining'
return 'Pause after this'
}
+4
View File
@@ -1,4 +1,5 @@
import { createUpscaleHost } from './upscale-host.mjs'
import { stableMemoryArgs } from './comfy-memory-policy.mjs'
import { createGpuReservation } from './gpu-reservation.mjs'
import { createGpuProxy } from './gpu-proxy.mjs'
import { createYueGpHost } from './yuegp-host.mjs'
@@ -265,6 +266,9 @@ function resolveHeadlessLaunch() {
]
const extraArgs = String(inst.launchArgs || '').trim()
if (extraArgs) args.push(...extraArgs.split(/\s+/).filter(Boolean))
const cliPath = join(inst.installPath, 'ComfyUI', 'comfy', 'cli_args.py')
const memoryArgs = stableMemoryArgs(args, { cliSource: existsSync(cliPath) ? readFileSync(cliPath, 'utf8') : '' })
args.splice(0, args.length, ...memoryArgs)
if (extraConfig) args.push('--extra-model-paths-config', extraConfig)
if (existsSync(roots.input)) args.push('--input-directory', roots.input)
if (existsSync(roots.output)) args.push('--output-directory', roots.output)
+5
View File
@@ -0,0 +1,5 @@
/** Avoid the failing AIMDO disk-to-HostBuffer path on this Windows GPU host. */
export function stableMemoryArgs(args, { platform = process.platform, cliSource = '', dynamicVram = process.env.COMFY_DYNAMIC_VRAM } = {}) {
if (platform !== 'win32' || dynamicVram === 'true' || !cliSource.includes('--disable-dynamic-vram')) return [...args]
return [...args.filter(arg => !['--enable-dynamic-vram', '--disable-dynamic-vram', '--fast-disk'].includes(arg)), '--disable-dynamic-vram']
}
+1 -1
View File
@@ -4,7 +4,7 @@ import { toggleStudioQueuePause } from '~/server/utils/studioQueue'
export default defineEventHandler(async (event) => {
const { owner } = assertLibraryOwner(event)
const id = String(getRouterParam(event, 'id') || '')
const body = await readBody<{ resume?: boolean }>().catch(() => ({} as { resume?: boolean }))
const body = await readBody<{ resume?: boolean }>(event).catch(() => ({} as { resume?: boolean }))
const queue = getShotQueue(owner, id)
if (!queue) throw createError({ statusCode: 404, statusMessage: 'Queue not found' })
const result = await toggleStudioQueuePause(owner, {
+1
View File
@@ -445,6 +445,7 @@ export async function setShotQueuePause(owner: string, id: string, pause: boolea
else if (queue.status !== 'complete') queue.status = 'paused'
} else {
queue.stopAfterCurrent = false
if (pendingSegmentCount(queue) > 0) queue.autoRun = true
if (live?.library) live.library.queueAutoRun = queue.autoRun === true
if (queue.status === 'paused') queue.status = 'idle'
}
+9 -1
View File
@@ -1808,6 +1808,10 @@ export async function toggleStudioQueuePause(owner: string, options: {
if (options.queueId && job.shotQueueId !== options.queueId) continue
job.pauseAfterCurrent = false
job.pausedByUser = false
if (job.shotQueueId && job.status !== 'complete' && job.status !== 'cancelled') {
job.resumeAutoRun = true
job.payload.queueAutoRun = true
}
}
syncPausedFlag(store)
return structuredClone(store)
@@ -1831,7 +1835,7 @@ export async function toggleStudioQueuePause(owner: string, options: {
}
const busy = await videoJobsBusy()
if (heldPaused[0] && !busy) {
await resumeHeldStudioJob(heldPaused[0]).catch(() => null)
await resumeHeldStudioJob(rows.find(row => row.id === heldPaused[0].id) || heldPaused[0]).catch(() => null)
} else {
if (heldPaused.length && busy) {
await mutate(owner, (jobs) => {
@@ -1936,6 +1940,10 @@ export async function onStudioQueueBurstStarted(owner: string, queueId: string,
if (queue?.autoRun === true) {
row.resumeAutoRun = true
row.payload.queueAutoRun = true
} else if (queue) {
// An explicit one-shot / next-three request must replace an earlier run-all policy.
row.resumeAutoRun = false
row.payload.queueAutoRun = false
}
row.updatedAt = Date.now()
syncPausedFlag(store)
+11
View File
@@ -0,0 +1,11 @@
import test from 'node:test'
import assert from 'node:assert/strict'
import { stableMemoryArgs } from '../scripts/comfy-memory-policy.mjs'
test('Windows uses the supported non-dynamic loader without changing workflow or precision flags', () => {
const args = ['main.py', '--enable-dynamic-vram', '--fast-disk', '--enable-manager', '--fp16-vae']
assert.deepEqual(stableMemoryArgs(args, { platform: 'win32', cliSource: '--disable-dynamic-vram', dynamicVram: 'false' }), ['main.py', '--enable-manager', '--fp16-vae', '--disable-dynamic-vram'])
assert.deepEqual(stableMemoryArgs(args, { platform: 'win32', cliSource: '' }), args)
assert.deepEqual(stableMemoryArgs(args, { platform: 'linux', cliSource: '--disable-dynamic-vram' }), args)
assert.deepEqual(stableMemoryArgs(args, { platform: 'win32', cliSource: '--disable-dynamic-vram', dynamicVram: 'true' }), args)
})
+22 -2
View File
@@ -18,7 +18,7 @@ const compiled = ts.transpileModule(source, {
// Execute the real queue module against temporary stores and a fake GPU.
// No production library, network, Comfy process, or render is touched.
function fixture({ rows = [], lives = [], queue = { running: 0, pending: 0 }, history = {} } = {}) {
function fixture({ rows = [], lives = [], queue = { running: 0, pending: 0 }, history = {}, shotQueues = [] } = {}) {
const root = mkdtempSync(join(tmpdir(), 'aigen-queue-test-'))
const owner = 'test-owner'
mkdirSync(join(root, 'users', owner), { recursive: true })
@@ -39,7 +39,9 @@ function fixture({ rows = [], lives = [], queue = { running: 0, pending: 0 }, hi
patchPendingJob: () => {}, deletePendingJob: () => {}
},
'~/server/utils/shotQueue': {
getShotQueue: () => null, pendingSegmentCount: () => 0
getShotQueue: (_owner, id) => shotQueues.find(q => q.id === id), pendingSegmentCount: queue => queue?.pendingCount || 0,
isShotQueueCancelled: () => false, listShotQueues: () => shotQueues,
setShotQueuePause: async (_owner, id, pause) => { const q = shotQueues.find(q => q.id === id); if (q) { q.stopAfterCurrent = pause; q.autoRun = !pause } }, pauseShotQueue: async () => {}
},
'~/server/utils/comfy': { fetchLiveQueue: async () => queue, fetchHistory: async () => {
if (history instanceof Error) throw history
@@ -55,6 +57,7 @@ function fixture({ rows = [], lives = [], queue = { running: 0, pending: 0 }, hi
'~/utils/imageV2': { imageV2StackSpecials: () => ({}) },
'~/utils/globalLocks': { allowIdentityRefs: () => false },
'~/server/utils/generationLog': { updateGenerationLogByStudioJob: async () => {} },
'~/server/utils/videoChain': { startQueueBurst: async (_owner, id, count) => { started.push({ id, count }); return { jobId: 'resumed-video' } } },
'~/server/utils/musicChain': {
startMusicJob: async () => {
const live = liveJob(`started-${started.length}`, 'music')
@@ -191,3 +194,20 @@ for (const history of [new Error('history timed out'), null]) {
assert.equal(f.started.length, 0)
})
}
test('Resume a manually held video batch runs all remaining shots instead of one', async () => {
const item = { ...row('paused-video', 'held', undefined), kind: 'video', shotQueueId: 'shots', pausedByUser: true, payload: { prompt: 'video', queueAutoRun: false, extensions: [] } }
const shots = { id: 'shots', autoRun: false, stopAfterCurrent: true, pendingCount: 9, status: 'paused' }
const f = fixture({ rows: [item], shotQueues: [shots] })
await f.api.toggleStudioQueuePause('test-owner', { jobId: item.id, resume: true })
assert.deepEqual(f.started, [{ id: 'shots', count: 'all' }])
assert.equal(f.rows()[0].payload.queueAutoRun, true)
assert.equal(shots.autoRun, true)
})
test('explicit bounded burst clears an earlier run-all policy', async () => {
const item = { ...row('bounded', 'held', undefined), kind: 'video', shotQueueId: 'shots', resumeAutoRun: true, payload: { prompt: 'video', queueAutoRun: true, extensions: [] } }
const f = fixture({ rows: [item], shotQueues: [{ id: 'shots', autoRun: false, pendingCount: 9 }] })
await f.api.onStudioQueueBurstStarted('test-owner', 'shots', 'one-shot-live')
assert.equal(f.rows()[0].resumeAutoRun, false)
assert.equal(f.rows()[0].payload.queueAutoRun, false)
})