Add image-to-text Describe on the bench via exclusive llama.cpp caption jobs.

This commit is contained in:
Towsty
2026-09-30 22:50:19 -05:00
parent 9481791ac5
commit 15ef891b26
15 changed files with 937 additions and 35 deletions
+299
View File
@@ -0,0 +1,299 @@
import { spawn } from 'node:child_process'
import { createServer } from 'node:net'
import { existsSync, mkdirSync, readFileSync, writeFileSync, renameSync, appendFileSync, readdirSync, copyFileSync, createWriteStream } from 'node:fs'
import { join, resolve, extname } from 'node:path'
import { pipeline } from 'node:stream/promises'
import { Transform } from 'node:stream'
import { captionStylePrompt, CAPTION_STYLES } from '../shared/caption.mjs'
const MODEL_NAME = 'Qwen2.5-VL-7B-NSFW-Caption-V4.Q5_K_M.gguf'
const MMPROJ_NAME = 'Qwen2.5-VL-7B-NSFW-Caption-V4.mmproj-f16.gguf'
function defaultModelsDir() {
const shared = process.env.COMFY_MODELS_ROOT
|| join(process.env.LOCALAPPDATA || '', 'Comfy-Desktop', 'ComfyUI-Shared', 'models')
return resolve(process.env.CAPTION_MODELS || join(shared, 'caption', 'qwen25vl-7b-nsfw-v4'))
}
/** Prefer winget ggml.llamacpp; never download. Process exit after one caption = keep_alive 0. */
function resolveLlamaServer(explicit) {
if (explicit) return resolve(explicit)
if (process.env.CAPTION_LLAMA_SERVER) return resolve(process.env.CAPTION_LLAMA_SERVER)
const local = process.env.LOCALAPPDATA || ''
const candidates = [
join(local, 'Microsoft', 'WinGet', 'Packages', 'ggml.llamacpp_Microsoft.Winget.Source_8wekyb3d8bbwe', 'llama-server.exe'),
join(local, 'llama.cpp', 'llama-server.exe'),
join('C:\\', 'llama.cpp', 'llama-server.exe')
]
for (const path of candidates) {
if (existsSync(path)) return path
}
return 'llama-server'
}
function freePort() {
return new Promise((resolvePort, reject) => {
const server = createServer()
server.listen(0, '127.0.0.1', () => {
const address = server.address()
const port = typeof address === 'object' && address ? address.port : 0
server.close(error => error ? reject(error) : resolvePort(port))
})
server.on('error', reject)
})
}
async function waitForServer(port, signal, timeoutMs = 120_000, fetchImpl = fetch) {
const started = Date.now()
while (Date.now() - started < timeoutMs) {
if (signal?.aborted) throw new Error('Caption cancelled while loading the model.')
try {
const response = await fetchImpl(`http://127.0.0.1:${port}/health`, { signal: AbortSignal.timeout(2000) })
if (response.ok) return
} catch { /* booting */ }
await new Promise(r => setTimeout(r, 500))
}
throw new Error('llama-server did not become ready in time.')
}
function mimeFor(path) {
const ext = extname(path).toLowerCase()
if (ext === '.jpg' || ext === '.jpeg') return 'image/jpeg'
if (ext === '.webp') return 'image/webp'
if (ext === '.gif') return 'image/gif'
return 'image/png'
}
function killTree(child) {
return new Promise(resolveKill => {
if (!child?.pid) return resolveKill()
let finished = false
const done = () => { if (finished) return; finished = true; resolveKill() }
child.once('close', done)
try { if (typeof child.kill === 'function') child.kill() } catch { /* ignore */ }
if (process.platform === 'win32' && child.pid > 0) {
try {
const killer = spawn('taskkill', ['/PID', String(child.pid), '/T', '/F'], { windowsHide: true, stdio: 'ignore' })
killer.once('close', done)
killer.once('error', done)
} catch { done() }
}
setTimeout(done, 3000)
})
}
export function validateCaptionHostRequest(body) {
if (!/^[a-zA-Z0-9-]{12,80}$/.test(body?.id || '')) throw new Error('Invalid caption job ID.')
const style = String(body.style || body.captionStyle || 'descriptive')
if (!CAPTION_STYLES.includes(style)) throw new Error('Unknown caption style.')
return { id: body.id, style, imagePath: body.imagePath ? String(body.imagePath) : '' }
}
/** One llama-server process per caption. Process exit unloads VRAM (keep_alive 0). */
export function createCaptionHost({
prepare,
leaseValid,
spawnProcess = spawn,
modelsDir,
llamaServer,
dataDir,
fetchImpl = fetch,
serverWaitMs = 120_000
} = {}) {
const models = resolve(modelsDir || defaultModelsDir())
const executable = resolveLlamaServer(llamaServer)
const data = resolve(dataDir || process.env.CAPTION_JOBS_DIR || join(models, 'aigen-jobs'))
let active = null
let holdUntil = 0
if (existsSync(data)) {
for (const id of readdirSync(data)) {
try {
const path = join(data, id, 'status.json')
if (!existsSync(path)) continue
const state = JSON.parse(readFileSync(path, 'utf8'))
if (['running', 'starting', 'cancelling'].includes(state.status)) {
holdUntil = Date.now() + 10000
state.status = 'error'
state.error = 'Caption host restarted. GPU model unloaded.'
writeFileSync(path + '.tmp', JSON.stringify(state))
renameSync(path + '.tmp', path)
}
} catch { /* ignore */ }
}
}
const dir = id => {
if (!/^[a-zA-Z0-9-]{12,80}$/.test(id || '')) throw new Error('Invalid caption job ID.')
return join(data, id)
}
const persist = job => {
const target = join(dir(job.id), 'status.json')
writeFileSync(target + '.tmp', JSON.stringify(job))
renameSync(target + '.tmp', target)
}
const read = id => {
if (active?.job.id === id) return { ...active.job }
const path = join(dir(id), 'status.json')
return existsSync(path) ? JSON.parse(readFileSync(path, 'utf8')) : null
}
const modelPath = () => join(models, MODEL_NAME)
const mmprojPath = () => join(models, MMPROJ_NAME)
const configured = () => {
try {
if (!existsSync(modelPath()) || !existsSync(mmprojPath())) return false
if (executable.includes('\\') || executable.includes('/')) return existsSync(executable)
return true
} catch { return false }
}
return {
busy: () => Boolean(active) || Date.now() < holdUntil,
configured,
modelsDir: () => models,
read,
async upload(id, stream) {
if (read(id)?.status === 'complete') throw new Error('This caption job has already finished.')
const jobDir = dir(id)
mkdirSync(jobDir, { recursive: true })
const dest = join(jobDir, 'input.upload')
let bytes = 0
await pipeline(stream, new Transform({
transform(chunk, _, callback) {
bytes += chunk.length
callback(bytes > 40 * 1024 * 1024 ? new Error('Image exceeds the 40 MB upload limit.') : null, chunk)
}
}), createWriteStream(dest, { flags: 'w' }))
if (bytes < 32) throw new Error('Image file is empty.')
renameSync(dest, join(jobDir, 'input.png'))
},
async start(body, lease) {
const request = validateCaptionHostRequest(body)
const previous = read(request.id)
if (previous) return previous
if (active || Date.now() < holdUntil) throw Object.assign(new Error('Caption is already running or releasing VRAM.'), { statusCode: 409 })
if (!configured()) throw new Error(`Caption model missing under ${models}. Run scripts/setup-caption.ps1.`)
if (!leaseValid(lease)) throw new Error('GPU reservation expired.')
let imagePath = request.imagePath
if (imagePath) {
if (!existsSync(imagePath)) throw new Error('imagePath does not exist on the GPU host.')
mkdirSync(dir(request.id), { recursive: true })
const dest = join(dir(request.id), `input${extname(imagePath) || '.png'}`)
copyFileSync(imagePath, dest)
imagePath = dest
} else {
imagePath = join(dir(request.id), 'input.png')
if (!existsSync(imagePath)) {
const alt = existsSync(dir(request.id)) && readdirSync(dir(request.id)).find(name => /^input\./i.test(name))
if (!alt) throw new Error('Upload an image first or pass imagePath.')
imagePath = join(dir(request.id), alt)
}
}
const job = {
id: request.id, status: 'starting', message: 'Preparing GPU for caption', progress: 0,
style: request.style, startedAt: Date.now(), checkedAt: Date.now()
}
active = { job, child: null, cancelled: false, abort: new AbortController() }
const run = active
mkdirSync(dir(job.id), { recursive: true })
writeFileSync(join(dir(job.id), 'request.json'), JSON.stringify({ ...request, imagePath, prompt: captionStylePrompt(request.style) }))
persist(job)
try {
if (prepare) await prepare()
if (run.cancelled || !leaseValid(lease)) throw new Error('Caption start cancelled or GPU reservation expired.')
job.status = 'running'; job.message = 'Loading Qwen2.5-VL caption model'; job.progress = 5; persist(job)
const port = await freePort()
// One caption per process. killTree after response unloads VRAM (keep_alive 0).
const child = spawnProcess(executable, [
'-m', modelPath(), '--mmproj', mmprojPath(),
'--host', '127.0.0.1', '--port', String(port),
'-ngl', String(process.env.CAPTION_NGL || '99'),
'-c', String(process.env.CAPTION_CTX || '4096'),
'--jinja'
], { windowsHide: true, shell: false, stdio: ['ignore', 'pipe', 'pipe'], env: { ...process.env } })
run.child = child
const log = chunk => appendFileSync(join(dir(job.id), 'worker.log'), chunk)
child.stdout.on('data', log)
child.stderr.on('data', log)
const watchdog = setInterval(() => {
if (!leaseValid(lease)) { job.error = 'GPU reservation expired; caption stopped.'; run.abort.abort(); void killTree(child) }
}, 5000)
try {
await waitForServer(port, run.abort.signal, serverWaitMs, fetchImpl)
if (run.cancelled) throw new Error('Cancelled')
job.message = 'Captioning'; job.progress = 40; persist(job)
const bytes = readFileSync(imagePath)
const dataUrl = `data:${mimeFor(imagePath)};base64,${bytes.toString('base64')}`
const response = await fetchImpl(`http://127.0.0.1:${port}/v1/chat/completions`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
signal: run.abort.signal.aborted ? run.abort.signal : AbortSignal.timeout(Number(process.env.CAPTION_TIMEOUT_MS || 180_000)),
body: JSON.stringify({
temperature: 0.2,
max_tokens: 512,
messages: [{
role: 'user',
content: [
{ type: 'text', text: captionStylePrompt(request.style) },
{ type: 'image_url', image_url: { url: dataUrl } }
]
}]
})
})
if (run.abort.signal.aborted) throw new Error('Cancelled')
if (!response.ok) {
const detail = await response.text().catch(() => '')
throw new Error(`llama-server caption failed (${response.status}): ${detail.slice(0, 400)}`)
}
const payload = await response.json()
const text = String(payload?.choices?.[0]?.message?.content || '').trim()
if (!text) throw new Error('Caption model returned empty text.')
job.text = text
job.status = run.cancelled ? 'cancelled' : 'complete'
job.message = job.status === 'complete' ? 'Caption ready' : 'Cancelled'
job.progress = 100
job.checkedAt = Date.now()
persist(job)
return { ...job }
} finally {
clearInterval(watchdog)
await killTree(child)
holdUntil = Date.now() + 2000
if (active === run) active = null
}
} catch (error) {
job.status = run.cancelled ? 'cancelled' : 'error'
job.error = error.message || String(error)
job.message = job.error
job.checkedAt = Date.now()
persist(job)
if (run.child) await killTree(run.child)
if (active === run) active = null
holdUntil = Date.now() + 2000
throw error
}
},
async cancel(id) {
if (active?.job.id !== id) return read(id)
const run = active
run.cancelled = true
run.job.status = 'cancelling'
run.job.message = 'Cancelling'
persist(run.job)
run.abort.abort()
if (run.child) await killTree(run.child)
return read(id)
},
async captionOnce(body, lease) {
const id = body.id || `caption-${Date.now()}-${Math.random().toString(36).slice(2, 10)}`
const result = await this.start({ ...body, id }, lease)
if (result.status === 'complete') return { text: result.text, id: result.id, style: result.style }
throw new Error(result.error || 'Caption failed.')
}
}
}
export const captionModelFiles = { model: MODEL_NAME, mmproj: MMPROJ_NAME }
+58 -19
View File
@@ -4,6 +4,7 @@ import { stableMemoryArgs } from './comfy-memory-policy.mjs'
import { createGpuReservation } from './gpu-reservation.mjs'
import { createGpuProxy } from './gpu-proxy.mjs'
import { createYue2Host } from './yue2-host.mjs'
import { createCaptionHost } from './caption-host.mjs'
import http from 'node:http'
import net from 'node:net'
import { execFile, spawn } from 'node:child_process'
@@ -147,7 +148,7 @@ let proxyTarget = 0
function ensureProxyListening() {
if (proxyServer) return
proxyServer = createGpuProxy({ target: () => proxyTarget, reservation: gpuReservation, authorized, markWork, externalBusy: () => yue2.busy() || upscale.busy() })
proxyServer = createGpuProxy({ target: () => proxyTarget, reservation: gpuReservation, authorized, markWork, externalBusy: () => yue2.busy() || upscale.busy() || caption.busy() })
proxyServer.on('error', (error) => {
console.log(JSON.stringify({ src: 'comfy-host-agent', event: 'proxy-error', error: String(error.message || error) }))
})
@@ -819,7 +820,7 @@ function purgeDesktopFiles(body) {
}
const gpuReservation = createGpuReservation({ idle: async () => {
if (yue2.busy() || upscale.busy()) return false
if (yue2.busy() || upscale.busy() || caption.busy()) return false
if ((await trainingLock()).busy) return false
const healthy = await syncProxy()
if (healthy) {
@@ -830,29 +831,65 @@ const gpuReservation = createGpuReservation({ idle: async () => {
return !(await processUp()) && !(await pythonMainUp().catch(() => true))
} })
const yue2 = createYue2Host({
leaseValid: lease => gpuReservation.isOwner(lease),
prepare: async () => {
if ((await trainingLock()).busy) throw new Error('GPU is busy with training.')
const healthy = await syncProxy()
if (healthy) {
const queue = await fetchLocalQueue(healthy)
if (!queue.ok || queue.running || queue.pending) throw new Error('Comfy is busy; YuE2 cannot start.')
}
// Stop Comfy and refuse to launch while its Python still owns VRAM.
if (healthy || await processUp() || await pythonMainUp().catch(() => false)) {
await stopComfyProcesses()
markAsleep()
}
if (await pythonMainUp()) throw new Error('Comfy has not stopped; retry after the GPU is free.')
async function prepareExclusiveGpu(label) {
if ((await trainingLock()).busy) throw new Error('GPU is busy with training.')
if (label !== 'YuE2' && yue2.busy()) throw new Error('YuE2 is using the GPU.')
if (label !== 'caption' && caption.busy()) throw new Error('Caption is using the GPU.')
if (upscale.busy()) throw new Error('Local upscale is using the GPU.')
const healthy = await syncProxy()
if (healthy) {
const queue = await fetchLocalQueue(healthy)
if (!queue.ok || queue.running || queue.pending) throw new Error(`Comfy is busy; ${label} cannot start.`)
}
})
if (healthy || await processUp() || await pythonMainUp().catch(() => false)) {
await stopComfyProcesses()
markAsleep()
}
if (await pythonMainUp()) throw new Error('Comfy has not stopped; retry after the GPU is free.')
}
const upscale = createUpscaleHost({ leaseValid: token => gpuReservation.isOwner(token) })
const yue2 = createYue2Host({
leaseValid: lease => gpuReservation.isOwner(lease),
prepare: async () => prepareExclusiveGpu('YuE2')
})
const caption = createCaptionHost({
leaseValid: lease => gpuReservation.isOwner(lease),
prepare: async () => prepareExclusiveGpu('caption')
})
async function handleControl(req, res) {
if (!authorized(req)) return json(res, 401, { ok: false, error: 'unauthorized' })
const url = new URL(req.url || '/', 'http://localhost')
if (url.pathname.startsWith('/caption')) {
const match = url.pathname.match(/^\/caption\/jobs\/([a-zA-Z0-9-]{12,80})(\/input|\/cancel)?$/)
if (req.method === 'GET' && url.pathname === '/caption/status') {
return json(res, 200, { configured: caption.configured(), busy: caption.busy(), backend: 'llama.cpp', modelsDir: caption.modelsDir() })
}
if (req.method === 'POST' && url.pathname === '/caption') {
try {
const body = await readJson(req)
return json(res, 200, await caption.captionOnce(body, String(req.headers['x-aigen-gpu-lease'] || '')))
} catch (error) {
return json(res, error.statusCode || 400, { error: error.message || 'Caption failed' })
}
}
if (req.method === 'POST' && url.pathname === '/caption/jobs') {
return json(res, 200, await caption.start(await readJson(req), String(req.headers['x-aigen-gpu-lease'] || '')))
}
if (match && req.method === 'PUT' && match[2] === '/input') {
await caption.upload(match[1], req)
return json(res, 200, { uploaded: true })
}
if (match && req.method === 'POST' && match[2] === '/cancel') return json(res, 200, await caption.cancel(match[1]))
if (match && req.method === 'GET' && !match[2]) {
const job = caption.read(match[1])
return json(res, job ? 200 : 404, job || { error: 'Caption job not found' })
}
return json(res, 404, { error: 'Unknown caption endpoint' })
}
if (url.pathname.startsWith('/upscale/')) {
const match = url.pathname.match(/^\/upscale\/jobs\/([a-zA-Z0-9-]{12,80})(\/input|\/video|\/cancel)?$/)
if (req.method === 'POST' && url.pathname === '/upscale/jobs') return json(res, 200, upscale.start(await readJson(req), String(req.headers['x-aigen-gpu-lease'] || '')))
@@ -909,12 +946,14 @@ async function handleControl(req, res) {
gpu: gpuReservation.availability(),
training: { busy: lastTraining.busy },
yue2: { busy: yue2.busy(), configured: yue2.configured() },
caption: { busy: caption.busy(), configured: caption.configured() },
upscale: { busy: upscale.busy(), engine: 'realesrgan-rife', local: true }
})
}
if (req.method === 'POST' && url.pathname === '/start') {
if (upscale.busy()) return json(res, 409, { message: 'Local upscale is using the GPU.' })
if (yue2.busy()) return json(res, 409, { message: 'YuE2 is using the GPU.' })
if (caption.busy()) return json(res, 409, { message: 'Caption is using the GPU.' })
const training = await trainingLock()
if (training.busy) {
return json(res, 409, {
@@ -1016,7 +1055,7 @@ const server = http.createServer(async (req, res) => {
} else await handleControl(req, res)
} catch (error) {
req.resume()
if ((String(req.url || '').startsWith('/yue2/') || String(req.url || '').startsWith('/upscale/')) && !res.headersSent) return json(res, error.statusCode || 400, { error: error.message || 'Music host request failed' })
if ((String(req.url || '').startsWith('/yue2/') || String(req.url || '').startsWith('/upscale/') || String(req.url || '').startsWith('/caption')) && !res.headersSent) return json(res, error.statusCode || 400, { error: error.message || 'Host request failed' })
if (!res.headersSent) json(res, error.statusCode || 400, { ok: false, message: error.statusCode === 409 ? 'GPU is in use. Waiting for availability.' : 'GPU coordination request failed.' })
}
})
+54
View File
@@ -0,0 +1,54 @@
# Caption (Qwen2.5-VL NSFW Caption V4 GGUF) — verify host files only.
# Models are already on the 5080 host. Do not download. Do not search Hugging Face.
$ErrorActionPreference = 'Stop'
function Resolve-ModelsRoot {
if ($env:COMFY_MODELS_ROOT) { return $env:COMFY_MODELS_ROOT }
$shared = Join-Path $env:LOCALAPPDATA 'Comfy-Desktop\ComfyUI-Shared\models'
if (Test-Path $shared) { return $shared }
throw 'Set COMFY_MODELS_ROOT or install Comfy Desktop Shared models.'
}
$root = Join-Path (Resolve-ModelsRoot) 'caption\qwen25vl-7b-nsfw-v4'
if (-not (Test-Path $root)) {
throw "Caption models folder missing: $root"
}
$files = @(
'Qwen2.5-VL-7B-NSFW-Caption-V4.Q5_K_M.gguf',
'Qwen2.5-VL-7B-NSFW-Caption-V4.mmproj-f16.gguf'
)
foreach ($name in $files) {
$path = Join-Path $root $name
if (-not (Test-Path $path)) {
throw "Missing $name under $root. Do not download; use the files already on this host."
}
Write-Host "OK $name"
}
$llama = $null
if ($env:CAPTION_LLAMA_SERVER -and (Test-Path $env:CAPTION_LLAMA_SERVER)) {
$llama = Get-Item $env:CAPTION_LLAMA_SERVER
} else {
$winget = Join-Path $env:LOCALAPPDATA 'Microsoft\WinGet\Packages\ggml.llamacpp_Microsoft.Winget.Source_8wekyb3d8bbwe\llama-server.exe'
if (Test-Path $winget) { $llama = Get-Item $winget }
else { $llama = Get-Command llama-server -ErrorAction SilentlyContinue }
}
if (-not $llama) {
throw 'llama-server not found. Install with: winget install ggml.llamacpp'
}
$llamaPath = if ($llama.PSObject.Properties['FullName'] -and $llama.FullName) { $llama.FullName } elseif ($llama.PSObject.Properties['Source'] -and $llama.Source) { $llama.Source } else { [string]$llama }
Write-Host "llama-server: $llamaPath"
$ready = @{
model = 'Qwen2.5-VL-7B-NSFW-Caption-V4.Q5_K_M.gguf'
mmproj = 'Qwen2.5-VL-7B-NSFW-Caption-V4.mmproj-f16.gguf'
modelsDir = $root
llamaServer = $llamaPath
readyAt = (Get-Date).ToUniversalTime().ToString('o')
} | ConvertTo-Json
Set-Content -Path (Join-Path $root 'aigen-ready.json') -Value $ready -Encoding utf8
Write-Host "Caption ready: $root"
+7
View File
@@ -18,6 +18,12 @@ Get-NetTCPConnection -LocalPort 8199 -State Listen -ErrorAction SilentlyContinue
Start-Sleep -Milliseconds 400
# Caption VLM — point at existing host GGUFs (no download).
$captionModels = Join-Path $env:LOCALAPPDATA 'Comfy-Desktop\ComfyUI-Shared\models\caption\qwen25vl-7b-nsfw-v4'
$llamaServer = Join-Path $env:LOCALAPPDATA 'Microsoft\WinGet\Packages\ggml.llamacpp_Microsoft.Winget.Source_8wekyb3d8bbwe\llama-server.exe'
if (Test-Path $captionModels) { $env:CAPTION_MODELS = $captionModels }
if (Test-Path $llamaServer) { $env:CAPTION_LLAMA_SERVER = $llamaServer }
if ($Wait) {
Set-Location $root
& $node $agent
@@ -31,4 +37,5 @@ $psi.WorkingDirectory = $root
$psi.UseShellExecute = $false
$psi.CreateNoWindow = $true
$psi.WindowStyle = [System.Diagnostics.ProcessWindowStyle]::Hidden
# CAPTION_* already set on this process; child inherits when UseShellExecute is false
[void][System.Diagnostics.Process]::Start($psi)