From a2895b8054abd8f128b86c986dd10643dec74cf6 Mon Sep 17 00:00:00 2001 From: Towsty Date: Fri, 28 Aug 2026 22:03:20 -0500 Subject: [PATCH] Add Image v2 Refine for on-device regional edits. Mask painter and denoise strength sit beside Edit and Compose. The canned hand/chest prompt helper is gone so that text stays yours. Co-authored-by: Cursor --- components/ImageV2MaskPainter.vue | 219 +++++++++++++++++++++++++++++ docs/image-v2.md | 47 +++++-- pages/index.vue | 200 ++++++++++++++++++++++++-- server/api/v2/generate.post.ts | 44 ++++-- server/assets/klein_v2_refine.json | 204 +++++++++++++++++++++++++++ server/utils/generationPresets.ts | 7 +- server/utils/imageChainV2.ts | 35 +++-- server/utils/imageWorkflowV2.ts | 61 +++++++- server/utils/studioQueue.ts | 26 +++- utils/imageV2.ts | 38 ++++- 10 files changed, 825 insertions(+), 56 deletions(-) create mode 100644 components/ImageV2MaskPainter.vue create mode 100644 server/assets/klein_v2_refine.json diff --git a/components/ImageV2MaskPainter.vue b/components/ImageV2MaskPainter.vue new file mode 100644 index 0000000..d149072 --- /dev/null +++ b/components/ImageV2MaskPainter.vue @@ -0,0 +1,219 @@ + + + diff --git a/docs/image-v2.md b/docs/image-v2.md index 2ad41be..7c4e844 100644 --- a/docs/image-v2.md +++ b/docs/image-v2.md @@ -6,8 +6,9 @@ Image edit (v1) and Image v2 are siblings. v1 is unchanged. Do not mute-fix `wor |---|---|---| | Tab | Image edit | Image v2 | | Route | `POST /api/edit` | `POST /api/v2/generate` and `POST /v2/generate` | -| Graphs | One file, prune unused branch | `klein_v2_edit.json` or `klein_v2_compose.json` | +| Graphs | One file, prune unused branch | `klein_v2_edit.json`, `klein_v2_compose.json`, or `klein_v2_refine.json` | | Second still | Optional; switches branch | Compose only; required. Mismatch is a 400 | +| Mask | — | Refine only; required. Missing mask is a 400, not an Edit fallback | | LoRAs | Optional user picker | SNOFS + Consistency, four strengths | | Defaults | 20 steps, CFG 1 | 24 steps, CFG 4 (turbo: 8 / 1) | @@ -15,8 +16,9 @@ Poll v2 jobs the same way as v1: `GET /api/generate/{id}/stream` and `GET /api/g ## Modes -- **edit** — one image + text. Task is `scene`. Sending `image_b` is rejected. -- **compose** — two images + text. `image_b` is required. No silent one-image fallback. +- **edit** — one image + text. Task is `scene`. Sending `image_b` is rejected. Use this to change lighting, background, or the whole frame from a single still. +- **compose** — two images + text. `image_b` is required. No silent one-image fallback. Use this to lock a person from A and pull a face or outfit from B. +- **refine** — one canvas + a painted mask + text. `image_a` and `mask` are required. `image_b` is ignored. Prompt only what should change inside the mask. Strength is denoise in that region (not CFG). Use this for face / hand / chest fixes without opening Comfy. Compose tasks: @@ -25,11 +27,16 @@ Compose tasks: - **face_lock** — body/pose/scene from A; face from B - **scene** — A is the edit image; B is style/background -Dummy test: Compose + two unrelated photos + “keep everything the same” must change the output. If it matches old v1 gens, still B is not connected. +Refine does not prepend Edit/Compose role headers. + +Dummy tests: + +- Compose + two unrelated photos + “keep everything the same” must change the output. If it matches old v1 gens, still B is not connected. +- Refine + last good gen + mask on the left breast + “red X painted on the left breast” at strength 0.4. Pass = X on the breast, face/pose/background stay. Fail = whole image regenerates, or nothing changes. ## Nodes the mapper patches -Both graphs: +Edit and Compose: | Node | Role | |------|------| @@ -37,7 +44,7 @@ Both graphs: | `2` / `23` | Scale to MP, lanczos (`megapixels`) | | `7` | SNOFS LoRA (`lora_name`, `strength_model`, `strength_clip`) | | `8` | Consistency LoRA (`lora_name`, `strength_model`, `strength_clip`) | -| `9` | Positive `CLIPTextEncode.text` (role header + user prompt) | +| `9` | Positive `CLIPTextEncode.text` (role header + user prompt on Compose) | | `10` | Negative `CLIPTextEncode.text` | | `15` | Seed | | `17` | Steps (Flux2Scheduler; Klein-native, euler sampler) | @@ -53,7 +60,19 @@ Compose only: | `24` | VAE encode B | | `25` / `26` | Second ReferenceLatent (B fused into pos/neg) | -If `image_b` is sent and the executed graph has fewer than two `LoadImage` nodes, the job fails. +Refine only (`klein_v2_refine.json`): + +| Node | Role | +|------|------| +| `1` | Load canvas | +| `30` | Load mask (white = edit, black = keep) | +| `2` | Scale canvas to MP | +| `31` / `32` / `34` | Resize mask to canvas, take red channel, grow slightly | +| `33` | `SetLatentNoiseMask` on the encoded canvas | +| `17` | `BasicScheduler.denoise` — this is Strength, not CFG | +| `19` | Sampler starts from the masked canvas latent, not an empty Flux2 latent | + +If `image_b` is sent on Edit/Compose and the executed graph has fewer than two `LoadImage` nodes, the job fails. If Refine runs without a mask input or without denoise on the scheduler, the job fails. ## Default sliders @@ -66,13 +85,23 @@ If `image_b` is sent and the executed graph has fewer than two `LoadImage` nodes | Steps | 24 (turbo 8) | | CFG | 4 (turbo 1) | | Scale to MP | 1.0 lanczos | +| Strength (denoise) | 0.35 on Refine only. Hidden on Edit/Compose. Range 0.15–0.75 | -There is no Denoise slider. This is reference-latent Klein, not inpaint. +### When to use each Refine strength + +| Preset | Strength | SNOFS model / CLIP | Consistency | Use for | +|--------|----------|--------------------|-------------|---------| +| Face | 0.25–0.35 (chip 0.28) | 0.55 / 0.25 | 0.75 / 0.80 | Glasses, cheeks, jaw. Keep the rest of the canvas. | +| Hand / chest | 0.35–0.45 (chip 0.40) | 0.65 / 0.30 | 0.70 / 0.70 | Hands, breasts, local contact. Dummy test uses 0.40. | +| Heavy | 0.55 | current sliders | current sliders | Stubborn region that barely moved at 0.40. | + +Do not raise Strength to rewrite the whole frame. If the face or background moves, the mask is too big or Strength is too high. ### NSFW identity vs outfit swap - **identity / face_lock** — keep Consistency at 0.70 / 0.70 so the face from A (identity) or B (face_lock) holds. Do not drop SNOFS CLIP below ~0.30 or the skin/body read falls apart. - **outfit** — keep SNOFS Model ~0.65 so cloth reads; do not raise it past ~0.85 or it starts rewriting the body from A. Consistency stays at 0.70 so the person in A does not become the person in B. - **scene / edit** — start at the table defaults. Turbo is for drafts only. +- **refine** — prompt only the masked change. Face vs hand/chest chips set Strength and LoRAs together. -Presets on this tab store sliders + mode + task only. They do not load v1 LoRA stacks or a cached latent. +Presets on this tab store sliders + mode + task (+ Strength when Refine). They do not load v1 LoRA stacks or a cached latent. diff --git a/pages/index.vue b/pages/index.vue index 01b8b85..2d91d7e 100644 --- a/pages/index.vue +++ b/pages/index.vue @@ -65,7 +65,9 @@

Input

{{ studioMode === 'editv2' - ? (v2Mode === 'compose' + ? (v2Mode === 'refine' + ? 'Refine: paint a mask on Still A. Prompt only what should change in the painted area. Strength is denoise in that region.' + : v2Mode === 'compose' ? 'Compose: still A is the person/body. Still B is the face or outfit. Two stills required — no silent one-image fallback.' : 'Edit: one still plus text. Compose is a separate graph if you need a second still.') : studioMode === 'edit' @@ -81,7 +83,16 @@ @drop.prevent="onDrop" @click="openPicker('main')" > -

+
+ + +
+
Source still
@@ -92,7 +103,7 @@
-

{{ studioMode === 'edit' || studioMode === 'editv2' ? (studioMode === 'editv2' ? 'Still A' : 'Image 1') : (textToVideo ? 'Start still · optional' : 'Choose an image') }}

+

{{ studioMode === 'edit' || studioMode === 'editv2' ? (studioMode === 'editv2' ? (v2Mode === 'refine' ? 'Still A · canvas' : 'Still A') : 'Image 1') : (textToVideo ? 'Start still · optional' : 'Choose an image') }}

{{ textToVideo ? 'Text-to-video does not need a still. Shot 2+ will still extend from the last frame.' : 'Browse the library or upload a new PNG, JPG, or WEBP' }}

@@ -102,7 +113,7 @@ type="button" class="rounded-full border px-3 py-1 text-xs font-medium" :class="v2Mode === 'edit' ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'" - @click="v2Mode = 'edit'; v2Task = 'scene'" + @click="setV2Mode('edit')" > Edit @@ -110,10 +121,18 @@ type="button" class="rounded-full border px-3 py-1 text-xs font-medium" :class="v2Mode === 'compose' ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'" - @click="v2Mode = 'compose'" + @click="setV2Mode('compose')" > Compose + +
+ + + +
+
@@ -1784,9 +1860,18 @@ import { IMAGE_V2_CFG_DEFAULT, IMAGE_V2_CONSISTENCY_CLIP, IMAGE_V2_CONSISTENCY_MODEL, + IMAGE_V2_DENOISE_DEFAULT, + IMAGE_V2_DENOISE_MAX, + IMAGE_V2_DENOISE_MIN, + IMAGE_V2_DENOISE_STEP, + IMAGE_V2_REFINE_FACE, + IMAGE_V2_REFINE_FACE_PROMPT, + IMAGE_V2_REFINE_HAND, + IMAGE_V2_REFINE_HEAVY, IMAGE_V2_SNOFS_CLIP, IMAGE_V2_SNOFS_MODEL, IMAGE_V2_STEPS_DEFAULT, + clampImageV2Denoise, clampImageV2Strength, type ImageV2Mode, type ImageV2PresetSettings, @@ -2068,6 +2153,11 @@ const v2Steps = ref(IMAGE_V2_STEPS_DEFAULT) const v2Cfg = ref(IMAGE_V2_CFG_DEFAULT) const v2Megapixels = ref(1) const v2Turbo = ref(false) +const v2Denoise = ref(IMAGE_V2_DENOISE_DEFAULT) +const refineMaskDirty = ref(false) +const refinePainter = ref<{ exportPng: () => Promise; clear: () => void } | null>(null) +const lastV2StillId = ref('') +const v2JobPending = ref(false) const editChainStep = ref(1) const editChainTotal = ref(1) const editChainLabel = ref('') @@ -2444,12 +2534,14 @@ const stillBHint = computed(() => { const editV2Blocked = computed(() => { if (!file.value || !prompt.value.trim() || !folderId.value) return true if (v2Mode.value === 'compose' && !editRefFile.value) return true + if (v2Mode.value === 'refine' && !refineMaskDirty.value) return true if (!comfyOk.value && !imageComfyOk.value) return true return false }) const editV2BlockReason = computed(() => { if (!file.value) return 'Load still A first.' if (v2Mode.value === 'compose' && !editRefFile.value) return 'Compose requires still B. This will not fall back to one-image generation.' + if (v2Mode.value === 'refine' && !refineMaskDirty.value) return 'Paint a mask on Still A first. Refine will not fall back to Edit.' if (!prompt.value.trim()) return 'Write a prompt first.' if (!folderId.value) return 'Choose a library folder before generating.' if (!imageComfyConfigured.value && !comfyOk.value) return 'Image v2 is not configured. Set COMFY_HOST.' @@ -2458,7 +2550,7 @@ const editV2BlockReason = computed(() => { }) const editV2SubmitLabel = computed(() => { const occupied = videoBusy.value || editBusy.value || studioJobs.value.some(job => job.status === 'running') - const label = v2Mode.value === 'compose' ? 'Compose image' : 'Edit image' + const label = v2Mode.value === 'refine' ? 'Refine image' : v2Mode.value === 'compose' ? 'Compose image' : 'Edit image' return occupied ? `Queue ${label.toLowerCase()}` : label }) const generateLabel = computed(() => { @@ -2516,6 +2608,7 @@ const composedIdentityPrompt = computed(() => { }) const promptPlaceholder = computed(() => { if (studioMode.value === 'editv2') { + if (v2Mode.value === 'refine') return 'Prompt only what should change in the painted area.' if (v2Mode.value === 'compose' && v2Task.value === 'outfit') return 'Keep everything the same except the clothes from still B.' if (v2Mode.value === 'compose' && v2Task.value === 'face_lock') return 'Keep the body and scene from still A. Use the face from still B.' if (v2Mode.value === 'compose' && v2Task.value === 'identity') return 'Same person as still A. Change the pose and scene.' @@ -3254,7 +3347,7 @@ function currentPresetSnapshot() { loraStack: [], settings: { mode: v2Mode.value, - task: v2Mode.value === 'compose' ? v2Task.value : 'scene', + task: v2Mode.value === 'compose' ? v2Task.value : v2Mode.value === 'refine' ? 'refine' : 'scene', negative: v2Negative.value, snofsModel: clampImageV2Strength(v2SnofsModel.value, IMAGE_V2_SNOFS_MODEL), snofsClip: clampImageV2Strength(v2SnofsClip.value, IMAGE_V2_SNOFS_CLIP), @@ -3263,7 +3356,8 @@ function currentPresetSnapshot() { steps: clampImageSteps(v2Steps.value, IMAGE_V2_STEPS_DEFAULT), cfg: clampImageCfg(v2Cfg.value, IMAGE_V2_CFG_DEFAULT), megapixels: clampImageScaleMegapixels(v2Megapixels.value, 1), - turbo: v2Turbo.value === true + turbo: v2Turbo.value === true, + strength: v2Mode.value === 'refine' ? clampImageV2Denoise(v2Denoise.value) : undefined } } } @@ -3302,8 +3396,8 @@ function applyGenerationPreset(preset: GenerationPreset) { const skipped = preset.loraStack.length - stack.length if (preset.kind === 'imagev2') { const settings = preset.settings as ImageV2PresetSettings - v2Mode.value = settings.mode === 'compose' ? 'compose' : 'edit' - v2Task.value = v2Mode.value === 'compose' ? (settings.task || 'scene') : 'scene' + v2Mode.value = settings.mode === 'compose' ? 'compose' : settings.mode === 'refine' ? 'refine' : 'edit' + v2Task.value = v2Mode.value === 'compose' ? (settings.task || 'scene') : v2Mode.value === 'refine' ? 'refine' : 'scene' if (typeof settings.negative === 'string') v2Negative.value = settings.negative v2SnofsModel.value = clampImageV2Strength(settings.snofsModel, IMAGE_V2_SNOFS_MODEL) v2SnofsClip.value = clampImageV2Strength(settings.snofsClip, IMAGE_V2_SNOFS_CLIP) @@ -3313,6 +3407,8 @@ function applyGenerationPreset(preset: GenerationPreset) { v2Cfg.value = clampImageCfg(settings.cfg, IMAGE_V2_CFG_DEFAULT) v2Megapixels.value = clampImageScaleMegapixels(settings.megapixels, 1) v2Turbo.value = settings.turbo === true + if (v2Mode.value === 'refine') v2Denoise.value = clampImageV2Denoise(settings.strength) + void ensureRefineCanvas() imageV2PresetId.value = preset.id imageV2PresetName.value = preset.name } else if (preset.kind === 'image') { @@ -3615,6 +3711,8 @@ function resetImage() { preview.value = '' imageWidth.value = 0 imageHeight.value = 0 + refineMaskDirty.value = false + refinePainter.value?.clear() } function clearVideoForm() { @@ -5097,6 +5195,62 @@ async function editImage() { } } +function setV2Mode(mode: ImageV2Mode) { + v2Mode.value = mode + if (mode === 'edit') v2Task.value = 'scene' + if (mode === 'refine') { + v2Task.value = 'refine' + void ensureRefineCanvas() + } +} + +function applyRefinePreset(kind: 'face' | 'hand' | 'heavy') { + if (kind === 'face') { + v2Denoise.value = IMAGE_V2_REFINE_FACE.strength + v2SnofsModel.value = IMAGE_V2_REFINE_FACE.snofsModel + v2SnofsClip.value = IMAGE_V2_REFINE_FACE.snofsClip + v2ConsistencyModel.value = IMAGE_V2_REFINE_FACE.consistencyModel + v2ConsistencyClip.value = IMAGE_V2_REFINE_FACE.consistencyClip + if (!prompt.value.trim()) prompt.value = IMAGE_V2_REFINE_FACE_PROMPT + return + } + if (kind === 'hand') { + v2Denoise.value = IMAGE_V2_REFINE_HAND.strength + v2SnofsModel.value = IMAGE_V2_REFINE_HAND.snofsModel + v2SnofsClip.value = IMAGE_V2_REFINE_HAND.snofsClip + v2ConsistencyModel.value = IMAGE_V2_REFINE_HAND.consistencyModel + v2ConsistencyClip.value = IMAGE_V2_REFINE_HAND.consistencyClip + return + } + v2Denoise.value = IMAGE_V2_REFINE_HEAVY.strength +} + +async function ensureRefineCanvas() { + if (file.value) return + const stillId = lastV2StillId.value || stillIdFromLibraryUrl(editResultUrl.value) + if (!stillId) return + try { + const blob = await $fetch(`/api/library/stills/${stillId}`, { responseType: 'blob' }) + if (!blob?.size) return + readFile(new File([blob], 'refine-canvas.png', { type: blob.type || 'image/png' }), { stillId }) + } catch { + // leave Still A empty if the last output is gone + } +} + +async function sendToRefine() { + if (!editResultUrl.value) return + studioMode.value = 'editv2' + outputFocus.value = 'edit' + try { + await useEditAsInput() + setV2Mode('refine') + toast('Output is Still A. Paint a mask, then refine.') + } catch { + toast('Could not send that still to Refine') + } +} + async function editImageV2() { if (editV2Blocked.value) { toast(editV2BlockReason.value || 'Load still A first.') @@ -5106,13 +5260,26 @@ async function editImageV2() { toast('Compose requires still B. This will not fall back to one-image generation.') return } + if (v2Mode.value === 'refine' && !refineMaskDirty.value) { + toast('Paint a mask on Still A first. Refine will not fall back to Edit.') + return + } const hideOut = hideThumbnail.value try { const body = new FormData() body.append('mode', v2Mode.value) - body.append('task', v2Mode.value === 'compose' ? v2Task.value : 'scene') + body.append('task', v2Mode.value === 'refine' ? 'refine' : v2Mode.value === 'compose' ? v2Task.value : 'scene') body.append('image_a', file.value as File) if (v2Mode.value === 'compose' && editRefFile.value) body.append('image_b', editRefFile.value) + if (v2Mode.value === 'refine') { + const maskBlob = await refinePainter.value?.exportPng() + if (!maskBlob) { + toast('Paint a mask on Still A first. Refine will not fall back to Edit.') + return + } + body.append('mask', new File([maskBlob], 'refine-mask.png', { type: 'image/png' })) + body.append('strength', String(clampImageV2Denoise(v2Denoise.value))) + } body.append('prompt', prompt.value.trim()) body.append('negative', v2Negative.value) body.append('snofs_model', String(clampImageV2Strength(v2SnofsModel.value, IMAGE_V2_SNOFS_MODEL))) @@ -5161,12 +5328,13 @@ async function editImageV2() { editChainLabel.value = '' editOverallProgress.value = 0 editCompletedChainStep.value = 0 - editActiveChainPlan.value = [{ label: v2Mode.value === 'compose' ? 'Compose' : 'Edit', prompt: prompt.value.trim() }] + editActiveChainPlan.value = [{ label: v2Mode.value === 'refine' ? 'Refine' : v2Mode.value === 'compose' ? 'Compose' : 'Edit', prompt: prompt.value.trim() }] startTimer('edit') editDownloadName.value = downloadName editJobId.value = started.jobId outputStudioJobId.value = started.studioJobId || '' persistActiveJob('edit', started.jobId, started.hideThumbnail === true || hideOut, started.folderLocked === true) + v2JobPending.value = true listen('edit', started.jobId, started.hideThumbnail === true || hideOut, started.folderLocked === true) void pollComfyHealth() void refreshStudioQueue() @@ -5458,6 +5626,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc if (payload.stillId && !locked) { editResultUrl.value = `/api/library/stills/${payload.stillId}` + if (studioMode.value === 'editv2' || v2JobPending.value) lastV2StillId.value = payload.stillId editLockedSave.value = false if (payload.hideThumbnail || hidden) { concealEditOutput.value = true @@ -5501,6 +5670,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc editStatusMessage.value = 'Saved to the locked folder. Unlock it to view.' } else { editResultUrl.value = `/api/library/stills/${payload.stillId}` + if (studioMode.value === 'editv2' || v2JobPending.value) lastV2StillId.value = payload.stillId concealEditOutput.value = Boolean(payload.hideThumbnail || hidden) editAwaitingReveal.value = concealEditOutput.value editLockedSave.value = false @@ -5509,6 +5679,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc } editBusy.value = false editStatusBusy.value = false + v2JobPending.value = false stopTimer('edit') stopListen('edit') clearActiveJob('edit') @@ -5521,6 +5692,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc if (failed) { editSettledUi = true editBusy.value = false + v2JobPending.value = false editStatusBusy.value = false stopTimer('edit') stopListen('edit') diff --git a/server/api/v2/generate.post.ts b/server/api/v2/generate.post.ts index 314e9b9..b32045e 100644 --- a/server/api/v2/generate.post.ts +++ b/server/api/v2/generate.post.ts @@ -11,6 +11,8 @@ import { IMAGE_V2_STEPS_DEFAULT, IMAGE_V2_TURBO_CFG, IMAGE_V2_TURBO_STEPS, + IMAGE_V2_DENOISE_DEFAULT, + clampImageV2Denoise, clampImageV2Strength, parseImageV2Mode, parseImageV2Task, @@ -37,7 +39,7 @@ async function fileFromUrl(url: string): Promise { } const mime = String(res.headers.get('content-type') || 'image/png').split(';')[0] if (!/^image\//i.test(mime)) { - throw createError({ statusCode: 400, statusMessage: 'image_a / image_b URL must be an image' }) + throw createError({ statusCode: 400, statusMessage: 'image_a / image_b / mask URL must be an image' }) } const data = Buffer.from(await res.arrayBuffer()) if (data.length > 40 * 1024 * 1024) { @@ -72,16 +74,19 @@ function readMultipart(parts: Array<{ name?: string; filename?: string; type?: s const fields: Record = {} let imageA: ImageFile | null = null let imageB: ImageFile | null = null + let mask: ImageFile | null = null for (const part of parts || []) { if ((part.name === 'image_a' || part.name === 'image') && part.filename && part.data?.length) { imageA = { filename: part.filename, data: part.data, type: part.type } } else if ((part.name === 'image_b' || part.name === 'image2') && part.filename && part.data?.length) { imageB = { filename: part.filename, data: part.data, type: part.type } + } else if (part.name === 'mask' && part.filename && part.data?.length) { + mask = { filename: part.filename, data: part.data, type: part.type } } else if (part.name && part.data) { fields[part.name] = part.data.toString('utf8') } } - return { fields, imageA, imageB } + return { fields, imageA, imageB, mask } } export default defineEventHandler(async (event) => { @@ -89,6 +94,7 @@ export default defineEventHandler(async (event) => { let fields: Record = {} let uploadedA: ImageFile | null = null let uploadedB: ImageFile | null = null + let uploadedMask: ImageFile | null = null if (contentType.includes('multipart/form-data')) { const form = await readMultipartFormData(event).catch(() => null) @@ -96,15 +102,16 @@ export default defineEventHandler(async (event) => { fields = parsed.fields uploadedA = parsed.imageA uploadedB = parsed.imageB + uploadedMask = parsed.mask } else { fields = await readBody>(event).catch(() => ({})) } const mode = parseImageV2Mode(fields.mode) if (!mode) { - throw createError({ statusCode: 400, statusMessage: 'mode must be edit or compose' }) + throw createError({ statusCode: 400, statusMessage: 'mode must be edit, compose, or refine' }) } - const task = parseImageV2Task(fields.task, 'scene') + const task = parseImageV2Task(fields.task, mode === 'refine' ? 'refine' : 'scene') const prompt = String(fields.prompt || '').trim() if (!prompt) { throw createError({ statusCode: 400, statusMessage: 'A prompt is required' }) @@ -125,11 +132,18 @@ export default defineEventHandler(async (event) => { const ownerKey = libraryOwnerKey(event) const imageA = await resolveImageRef(ownerKey, fields.image_a, uploadedA) - const imageB = await resolveImageRef(ownerKey, fields.image_b, uploadedB) + const imageB = mode === 'refine' ? null : await resolveImageRef(ownerKey, fields.image_b, uploadedB) + const mask = mode === 'refine' ? await resolveImageRef(ownerKey, fields.mask, uploadedMask) : null if (!imageA) { throw createError({ statusCode: 400, statusMessage: 'image_a is required' }) } + if (mode === 'refine' && !mask) { + throw createError({ + statusCode: 400, + statusMessage: 'Refine requires a mask. Refusing to fall back to Edit.' + }) + } if (mode === 'edit' && imageB) { throw createError({ statusCode: 400, @@ -166,7 +180,8 @@ export default defineEventHandler(async (event) => { const size = imageDimensions(imageA.data) const clipName = String(fields.name || '').trim().slice(0, 80) const v2Mode = mode as ImageV2Mode - const v2Task = (mode === 'edit' ? 'scene' : task) as ImageV2Task + const v2Task = (mode === 'refine' ? 'refine' : mode === 'edit' ? 'scene' : task) as ImageV2Task + const refineStrength = mode === 'refine' ? clampImageV2Denoise(fields.strength, IMAGE_V2_DENOISE_DEFAULT) : undefined const still = await rememberInputStill({ ownerKey, @@ -186,6 +201,15 @@ export default defineEventHandler(async (event) => { hideInput }) : null + const savedMask = mask + ? await rememberInputStill({ + ownerKey, + folderId, + filename: mask.filename || 'refine-mask.png', + data: mask.data, + hideInput + }) + : null const studio = await addStudioJob({ ownerKey, @@ -228,7 +252,10 @@ export default defineEventHandler(async (event) => { snofsModel: clampImageV2Strength(fields.snofs_model, IMAGE_V2_SNOFS_MODEL), snofsClip: clampImageV2Strength(fields.snofs_clip, IMAGE_V2_SNOFS_CLIP), consistencyModel: clampImageV2Strength(fields.consistency_model, IMAGE_V2_CONSISTENCY_MODEL), - consistencyClip: clampImageV2Strength(fields.consistency_clip, IMAGE_V2_CONSISTENCY_CLIP) + consistencyClip: clampImageV2Strength(fields.consistency_clip, IMAGE_V2_CONSISTENCY_CLIP), + maskStillId: savedMask?.id, + maskStillFilename: savedMask?.filename, + refineStrength } }) await kickStudioQueue() @@ -248,7 +275,8 @@ export default defineEventHandler(async (event) => { cfg, mode: v2Mode, task: v2Task, - workflow: v2Mode === 'compose' ? 'klein_v2_compose.json' : 'klein_v2_edit.json', + workflow: v2Mode === 'refine' ? 'klein_v2_refine.json' : v2Mode === 'compose' ? 'klein_v2_compose.json' : 'klein_v2_edit.json', + strength: refineStrength, hideThumbnail, folderLocked } diff --git a/server/assets/klein_v2_refine.json b/server/assets/klein_v2_refine.json new file mode 100644 index 0000000..80c0f07 --- /dev/null +++ b/server/assets/klein_v2_refine.json @@ -0,0 +1,204 @@ +{ + "1": { + "inputs": { "image": "" }, + "class_type": "LoadImage", + "_meta": { "title": "Load Canvas" } + }, + "30": { + "inputs": { "image": "" }, + "class_type": "LoadImage", + "_meta": { "title": "Load Mask" } + }, + "2": { + "inputs": { + "upscale_method": "lanczos", + "megapixels": 1, + "resolution_steps": 1, + "image": ["1", 0] + }, + "class_type": "ImageScaleToTotalPixels", + "_meta": { "title": "Scale Canvas" } + }, + "3": { + "inputs": { "image": ["2", 0] }, + "class_type": "GetImageSize", + "_meta": { "title": "Get Canvas Size" } + }, + "31": { + "inputs": { + "upscale_method": "lanczos", + "width": ["3", 0], + "height": ["3", 1], + "crop": "disabled", + "image": ["30", 0] + }, + "class_type": "ImageScale", + "_meta": { "title": "Resize Mask to Canvas" } + }, + "32": { + "inputs": { + "channel": "red", + "image": ["31", 0] + }, + "class_type": "ImageToMask", + "_meta": { "title": "Mask Channel" } + }, + "34": { + "inputs": { + "expand": 6, + "tapered_corners": true, + "mask": ["32", 0] + }, + "class_type": "GrowMask", + "_meta": { "title": "Grow Mask" } + }, + "4": { + "inputs": { + "unet_name": "flux-2-klein-base-9b-fp8.safetensors", + "weight_dtype": "default" + }, + "class_type": "UNETLoader", + "_meta": { "title": "Load Flux.2 Klein 9B Base" } + }, + "5": { + "inputs": { + "clip_name": "qwen_3_8b_fp8mixed.safetensors", + "type": "flux2", + "device": "default" + }, + "class_type": "CLIPLoader", + "_meta": { "title": "Load Qwen 3 8B CLIP" } + }, + "6": { + "inputs": { "vae_name": "full_encoder_small_decoder.safetensors" }, + "class_type": "VAELoader", + "_meta": { "title": "Load Klein VAE" } + }, + "7": { + "inputs": { + "lora_name": "klein_snofs_v1_4.safetensors", + "strength_model": 0.65, + "strength_clip": 0.3, + "model": ["4", 0], + "clip": ["5", 0] + }, + "class_type": "LoraLoader", + "_meta": { "title": "SNOFS" } + }, + "8": { + "inputs": { + "lora_name": "Flux2-Klein-9B-consistency-V2.safetensors", + "strength_model": 0.7, + "strength_clip": 0.7, + "model": ["7", 0], + "clip": ["7", 1] + }, + "class_type": "LoraLoader", + "_meta": { "title": "Consistency" } + }, + "9": { + "inputs": { + "text": "", + "clip": ["8", 1] + }, + "class_type": "CLIPTextEncode", + "_meta": { "title": "Positive Prompt" } + }, + "10": { + "inputs": { + "text": "", + "clip": ["8", 1] + }, + "class_type": "CLIPTextEncode", + "_meta": { "title": "Negative Prompt" } + }, + "11": { + "inputs": { + "pixels": ["2", 0], + "vae": ["6", 0] + }, + "class_type": "VAEEncode", + "_meta": { "title": "VAE Encode Canvas" } + }, + "12": { + "inputs": { + "conditioning": ["9", 0], + "latent": ["11", 0] + }, + "class_type": "ReferenceLatent", + "_meta": { "title": "Reference Latent +" } + }, + "13": { + "inputs": { + "conditioning": ["10", 0], + "latent": ["11", 0] + }, + "class_type": "ReferenceLatent", + "_meta": { "title": "Reference Latent -" } + }, + "33": { + "inputs": { + "samples": ["11", 0], + "mask": ["34", 0] + }, + "class_type": "SetLatentNoiseMask", + "_meta": { "title": "Mask Latent Noise" } + }, + "15": { + "inputs": { "noise_seed": 1 }, + "class_type": "RandomNoise", + "_meta": { "title": "RandomNoise" } + }, + "16": { + "inputs": { "sampler_name": "euler" }, + "class_type": "KSamplerSelect", + "_meta": { "title": "KSamplerSelect" } + }, + "17": { + "inputs": { + "scheduler": "simple", + "steps": 24, + "denoise": 0.35, + "model": ["8", 0] + }, + "class_type": "BasicScheduler", + "_meta": { "title": "BasicScheduler" } + }, + "18": { + "inputs": { + "cfg": 4, + "model": ["8", 0], + "positive": ["12", 0], + "negative": ["13", 0] + }, + "class_type": "CFGGuider", + "_meta": { "title": "CFG Guider" } + }, + "19": { + "inputs": { + "noise": ["15", 0], + "guider": ["18", 0], + "sampler": ["16", 0], + "sigmas": ["17", 0], + "latent_image": ["33", 0] + }, + "class_type": "SamplerCustomAdvanced", + "_meta": { "title": "SamplerCustomAdvanced" } + }, + "20": { + "inputs": { + "samples": ["19", 0], + "vae": ["6", 0] + }, + "class_type": "VAEDecode", + "_meta": { "title": "VAE Decode" } + }, + "21": { + "inputs": { + "filename_prefix": "aigen-v2-refine", + "images": ["20", 0] + }, + "class_type": "SaveImage", + "_meta": { "title": "Save Image" } + } +} diff --git a/server/utils/generationPresets.ts b/server/utils/generationPresets.ts index 3ab3c6a..40b4785 100644 --- a/server/utils/generationPresets.ts +++ b/server/utils/generationPresets.ts @@ -21,6 +21,8 @@ import { IMAGE_V2_SNOFS_CLIP, IMAGE_V2_SNOFS_MODEL, IMAGE_V2_STEPS_DEFAULT, + IMAGE_V2_DENOISE_DEFAULT, + clampImageV2Denoise, clampImageV2Strength, parseImageV2Mode, parseImageV2Task, @@ -114,7 +116,7 @@ function sanitizeImageV2Settings(raw: unknown): ImageV2PresetSettings { const turbo = rec.turbo === true return { mode, - task: mode === 'compose' ? parseImageV2Task(rec.task, 'scene') : 'scene', + task: mode === 'compose' ? parseImageV2Task(rec.task, 'scene') : mode === 'refine' ? 'refine' : 'scene', negative: String(rec.negative || '').slice(0, 2000), snofsModel: clampImageV2Strength(rec.snofsModel ?? rec.snofs_model, IMAGE_V2_SNOFS_MODEL), snofsClip: clampImageV2Strength(rec.snofsClip ?? rec.snofs_clip, IMAGE_V2_SNOFS_CLIP), @@ -123,7 +125,8 @@ function sanitizeImageV2Settings(raw: unknown): ImageV2PresetSettings { steps: turbo ? 8 : clampImageSteps(rec.steps, IMAGE_V2_STEPS_DEFAULT), cfg: turbo ? 1 : clampImageCfg(rec.cfg, IMAGE_V2_CFG_DEFAULT), megapixels: clampImageScaleMegapixels(rec.megapixels ?? rec.scaleMegapixels, 1), - turbo + turbo, + strength: mode === 'refine' ? clampImageV2Denoise(rec.strength, IMAGE_V2_DENOISE_DEFAULT) : undefined } } diff --git a/server/utils/imageChainV2.ts b/server/utils/imageChainV2.ts index e288ee8..e56b62f 100644 --- a/server/utils/imageChainV2.ts +++ b/server/utils/imageChainV2.ts @@ -15,6 +15,7 @@ export type EditV2RunParams = { task: ImageV2Task image: EditImageFile reference: EditImageFile | null + mask?: EditImageFile | null prompt: string negative: string steps: number @@ -25,11 +26,15 @@ export type EditV2RunParams = { consistencyModel: number consistencyClip: number megapixels: number + strength?: number } export async function runEditV2(job: Job, params: EditV2RunParams) { const library = job.library if (!library) throw new Error('Edit job is missing library metadata') + if (params.mode === 'refine' && !params.mask) { + throw new Error('Refine requires a mask. Refusing to fall back to Edit.') + } if (params.mode === 'compose' && !params.reference) { throw new Error('Compose requires image B. Refusing to fall back to one-image generation.') } @@ -52,23 +57,35 @@ export async function runEditV2(job: Job, params: EditV2RunParams) { await withImageComfyHost(job.imageComfyHost, async () => { emitChainJob(job, { type: 'status', - message: params.mode === 'compose' ? 'Uploading stills A and B to Beast...' : 'Uploading still A to Beast...', + message: params.mode === 'refine' + ? 'Uploading canvas and mask to Beast...' + : params.mode === 'compose' ? 'Uploading stills A and B to Beast...' : 'Uploading still A to Beast...', progress: 8 }) const uploaded = await uploadImage(params.image, job.id) - const uploadedRef = params.reference + const uploadedRef = params.mode === 'refine' + ? null + : params.reference + ? await uploadImage({ + ...params.reference, + filename: `ref_${params.reference.filename || 'image_b.png'}` + }, job.id) + : null + const uploadedMask = params.mode === 'refine' && params.mask ? await uploadImage({ - ...params.reference, - filename: `ref_${params.reference.filename || 'image_b.png'}` + ...params.mask, + filename: `mask_${params.mask.filename || 'refine-mask.png'}` }, job.id) : null if (job.status === 'cancelled') throw new Error('Job interrupted.') emitChainJob(job, { type: 'status', - message: params.mode === 'compose' - ? `Queueing Klein v2 compose (${params.task}) on Beast...` - : 'Queueing Klein v2 edit on Beast...', + message: params.mode === 'refine' + ? 'Queueing Klein v2 refine on Beast...' + : params.mode === 'compose' + ? `Queueing Klein v2 compose (${params.task}) on Beast...` + : 'Queueing Klein v2 edit on Beast...', progress: 12 }) await ensureComfyLoraNames('image') @@ -80,6 +97,8 @@ export async function runEditV2(job: Job, params: EditV2RunParams) { negative: params.negative, imageAName: uploaded.name, imageBName: uploadedRef?.name, + maskName: uploadedMask?.name, + strength: params.strength, snofsModel: params.snofsModel, snofsClip: params.snofsClip, consistencyModel: params.consistencyModel, @@ -137,7 +156,7 @@ export async function runEditV2(job: Job, params: EditV2RunParams) { video: { filename: output.filename, subfolder: output.subfolder, type: output.type }, imageName: uploaded.name, imageSubfolder: uploaded.subfolder, - extraImageNames: uploadedRef?.name ? [uploadedRef.name] : [], + extraImageNames: [uploadedRef?.name, uploadedMask?.name].filter((name): name is string => Boolean(name)), promptId: job.promptId }) diff --git a/server/utils/imageWorkflowV2.ts b/server/utils/imageWorkflowV2.ts index 36d91b2..4059341 100644 --- a/server/utils/imageWorkflowV2.ts +++ b/server/utils/imageWorkflowV2.ts @@ -1,5 +1,6 @@ import editTemplate from '../assets/klein_v2_edit.json' import composeTemplate from '../assets/klein_v2_compose.json' +import refineTemplate from '../assets/klein_v2_refine.json' import { IMAGE_SCALE_TO_TOTAL_PIXELS } from '~/server/utils/comfy' import { cachedComfyLoraNames } from '~/server/utils/loras' import { resolveComfyLoraName, loraIdentityKey } from '~/utils/loras' @@ -11,6 +12,8 @@ import { IMAGE_V2_SNOFS_CLIP, IMAGE_V2_SNOFS_LORA, IMAGE_V2_SNOFS_MODEL, + IMAGE_V2_DENOISE_DEFAULT, + clampImageV2Denoise, clampImageV2Strength, composeImageV2Prompt, type ImageV2Mode, @@ -22,6 +25,8 @@ type WorkflowGraph = Record const LOAD_A = '1' const LOAD_B = '22' +const LOAD_MASK = '30' +const SCHEDULER_DENOISE = '17' const SCALE_A = '2' const SCALE_B = '23' const PROMPT = '9' @@ -35,6 +40,7 @@ const CONSISTENCY = '8' export const IMAGE_V2_EDIT_WORKFLOW = 'klein_v2_edit.json' export const IMAGE_V2_COMPOSE_WORKFLOW = 'klein_v2_compose.json' +export const IMAGE_V2_REFINE_WORKFLOW = 'klein_v2_refine.json' export interface ImageV2BuildParams { mode: ImageV2Mode @@ -43,6 +49,8 @@ export interface ImageV2BuildParams { negative?: string imageAName: string imageBName?: string + maskName?: string + strength?: number snofsModel?: number snofsClip?: number consistencyModel?: number @@ -87,8 +95,36 @@ function patchScaleMegapixels(graph: WorkflowGraph, megapixels: number) { } } +function graphHasMaskInput(graph: WorkflowGraph) { + return Object.values(graph).some((node) => { + const mask = node.inputs?.mask + return mask !== undefined && mask !== null && mask !== '' + }) +} + export function assertImageV2Graph(graph: WorkflowGraph, mode: ImageV2Mode, imageBName?: string) { const loaders = loadImageNames(graph) + if (mode === 'refine') { + const mask = graph[LOAD_MASK] + if (!mask || mask.class_type !== 'LoadImage' || !String(mask.inputs.image || '').trim()) { + throw createError({ + statusCode: 500, + statusMessage: 'Refine job is missing the mask image. Refusing to run.' + }) + } + if (!graphHasMaskInput(graph)) { + throw createError({ + statusCode: 500, + statusMessage: 'Refine graph has no mask input. Refusing to run.' + }) + } + if (!('denoise' in (graph[SCHEDULER]?.inputs || {}))) { + throw createError({ + statusCode: 500, + statusMessage: 'Refine graph has no denoise on the sampler. Refusing to run.' + }) + } + } if (mode === 'compose') { if (loaders.length < 2) { throw createError({ @@ -104,7 +140,7 @@ export function assertImageV2Graph(graph: WorkflowGraph, mode: ImageV2Mode, imag }) } } - if (imageBName && loaders.length < 2) { + if (mode !== 'refine' && imageBName && loaders.length < 2) { throw createError({ statusCode: 500, statusMessage: 'image_b was sent but the executed graph has no second image input.' @@ -124,7 +160,11 @@ export function assertImageV2Graph(graph: WorkflowGraph, mode: ImageV2Mode, imag export function buildImageV2Workflow(params: ImageV2BuildParams) { const compose = params.mode === 'compose' - const graph = structuredClone(compose ? composeTemplate : editTemplate) as WorkflowGraph + const refine = params.mode === 'refine' + if (refine && !String(params.maskName || '').trim()) { + throw createError({ statusCode: 400, statusMessage: 'Refine requires a mask. Refusing to fall back to Edit.' }) + } + const graph = structuredClone(refine ? refineTemplate : compose ? composeTemplate : editTemplate) as WorkflowGraph const prompt = composeImageV2Prompt(params.mode, params.task, params.prompt) const negative = String(params.negative || '') const snofsModel = clampImageV2Strength(params.snofsModel, IMAGE_V2_SNOFS_MODEL) @@ -133,10 +173,15 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) { const consistencyClip = clampImageV2Strength(params.consistencyClip, IMAGE_V2_CONSISTENCY_CLIP) const steps = clampImageSteps(params.steps, 24) const cfg = clampImageCfg(params.cfg, 4) - const workflowFile = compose ? IMAGE_V2_COMPOSE_WORKFLOW : IMAGE_V2_EDIT_WORKFLOW + const strength = refine ? clampImageV2Denoise(params.strength, IMAGE_V2_DENOISE_DEFAULT) : undefined + const workflowFile = refine ? IMAGE_V2_REFINE_WORKFLOW : compose ? IMAGE_V2_COMPOSE_WORKFLOW : IMAGE_V2_EDIT_WORKFLOW setInput(graph, LOAD_A, 'image', params.imageAName) if (compose) setInput(graph, LOAD_B, 'image', params.imageBName || '') + if (refine) { + setInput(graph, LOAD_MASK, 'image', params.maskName || '') + setInput(graph, SCHEDULER_DENOISE, 'denoise', strength) + } setInput(graph, PROMPT, 'text', prompt) setInput(graph, NEGATIVE, 'text', negative) setInput(graph, NOISE, 'noise_seed', params.seed) @@ -152,7 +197,7 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) { setInput(graph, CONSISTENCY, 'strength_model', consistencyModel) setInput(graph, CONSISTENCY, 'strength_clip', consistencyClip) - assertImageV2Graph(graph, params.mode, params.imageBName) + assertImageV2Graph(graph, params.mode, refine ? undefined : params.imageBName) const loaders = loadImageNames(graph) console.log(JSON.stringify({ @@ -160,6 +205,9 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) { workflow: workflowFile, mode: params.mode, task: params.task, + canvas: { id: LOAD_A, file: graph[LOAD_A]?.inputs.image }, + mask: refine ? { id: LOAD_MASK, file: graph[LOAD_MASK]?.inputs.image } : undefined, + strength, loadImage: Object.fromEntries(loaders.map(item => [item.id, { title: item.title, file: item.image }])), loras: { snofs: { name: graph[SNOFS]?.inputs.lora_name, model: snofsModel, clip: snofsClip }, @@ -171,14 +219,16 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) { megapixels: clampImageScaleMegapixels(params.megapixels ?? 1) })) - return { graph, workflowFile, loaders, prompt } + return { graph, workflowFile, loaders, prompt, strength } } export const IMAGE_V2_NODE_LABELS: Record = { '1': 'Loading image A', '22': 'Loading image B', + '30': 'Loading mask', '2': 'Scaling image A', '23': 'Scaling image B', + '31': 'Resizing mask', '4': 'Loading Flux.2 Klein 9B', '5': 'Loading CLIP', '6': 'Loading VAE', @@ -187,6 +237,7 @@ export const IMAGE_V2_NODE_LABELS: Record = { '9': 'Encoding prompt', '11': 'Encoding image A', '24': 'Encoding image B', + '33': 'Applying mask', '19': 'Sampling Klein v2', '20': 'Decoding still', '21': 'Saving still' diff --git a/server/utils/studioQueue.ts b/server/utils/studioQueue.ts index 3e78574..265fd5b 100644 --- a/server/utils/studioQueue.ts +++ b/server/utils/studioQueue.ts @@ -52,12 +52,15 @@ export interface StudioJobPayload { scaleToTotalPixels?: boolean scaleMegapixels?: number imagePipeline?: 'v1' | 'v2' - v2Mode?: 'edit' | 'compose' - v2Task?: 'scene' | 'identity' | 'outfit' | 'face_lock' + v2Mode?: 'edit' | 'compose' | 'refine' + v2Task?: 'scene' | 'identity' | 'outfit' | 'face_lock' | 'refine' snofsModel?: number snofsClip?: number consistencyModel?: number consistencyClip?: number + maskStillId?: string + maskStillFilename?: string + refineStrength?: number } export interface StudioJob { @@ -694,7 +697,18 @@ async function startStudioEditJob(item: StudioJob) { : null if (payload.imagePipeline === 'v2') { - const mode = payload.v2Mode === 'compose' ? 'compose' : 'edit' + const mode = payload.v2Mode === 'compose' ? 'compose' : payload.v2Mode === 'refine' ? 'refine' : 'edit' + const maskId = payload.maskStillId + const mask = mode === 'refine' && maskId && existsSync(stillPath(item.ownerKey, maskId)) + ? { + filename: payload.maskStillFilename || 'refine-mask.png', + data: readFileSync(stillPath(item.ownerKey, maskId)), + type: 'image/png' + } + : null + if (mode === 'refine' && !mask) { + throw new Error('Refine requires a mask. Refusing to fall back to Edit.') + } if (mode === 'compose' && !reference) { throw new Error('Compose requires image B. Refusing to fall back to one-image generation.') } @@ -732,7 +746,8 @@ async function startStudioEditJob(item: StudioJob) { mode, task: payload.v2Task || 'scene', image, - reference, + reference: mode === 'refine' ? null : reference, + mask, prompt: payload.prompt, negative: payload.negative || '', steps: payload.steps, @@ -742,7 +757,8 @@ async function startStudioEditJob(item: StudioJob) { snofsClip: payload.snofsClip ?? 0.35, consistencyModel: payload.consistencyModel ?? 0.7, consistencyClip: payload.consistencyClip ?? 0.7, - megapixels: payload.scaleMegapixels ?? 1 + megapixels: payload.scaleMegapixels ?? 1, + strength: payload.refineStrength }).catch((error) => { const message = error instanceof Error ? error.message : String(error) if (live && live.status !== 'error' && live.status !== 'cancelled' && live.status !== 'deferred') { diff --git a/utils/imageV2.ts b/utils/imageV2.ts index b91cda8..63fa535 100644 --- a/utils/imageV2.ts +++ b/utils/imageV2.ts @@ -1,5 +1,5 @@ -export const IMAGE_V2_MODES = ['edit', 'compose'] as const -export const IMAGE_V2_TASKS = ['scene', 'identity', 'outfit', 'face_lock'] as const +export const IMAGE_V2_MODES = ['edit', 'compose', 'refine'] as const +export const IMAGE_V2_TASKS = ['scene', 'identity', 'outfit', 'face_lock', 'refine'] as const export type ImageV2Mode = (typeof IMAGE_V2_MODES)[number] export type ImageV2Task = (typeof IMAGE_V2_TASKS)[number] @@ -18,8 +18,28 @@ export const IMAGE_V2_STRENGTH_STEP = 0.05 export const IMAGE_V2_SNOFS_LORA = 'klein_snofs_v1_4.safetensors' export const IMAGE_V2_CONSISTENCY_LORA = 'Flux2-Klein-9B-consistency-V2.safetensors' +export const IMAGE_V2_DENOISE_DEFAULT = 0.35 +export const IMAGE_V2_DENOISE_MIN = 0.15 +export const IMAGE_V2_DENOISE_MAX = 0.75 +export const IMAGE_V2_DENOISE_STEP = 0.01 +export const IMAGE_V2_REFINE_FACE = { + strength: 0.28, + snofsModel: 0.55, + snofsClip: 0.25, + consistencyModel: 0.75, + consistencyClip: 0.8 +} +export const IMAGE_V2_REFINE_HAND = { + strength: 0.4, + snofsModel: 0.65, + snofsClip: 0.3, + consistencyModel: 0.7, + consistencyClip: 0.7 +} +export const IMAGE_V2_REFINE_HEAVY = { strength: 0.55 } +export const IMAGE_V2_REFINE_FACE_PROMPT = 'same face as the canvas, same glasses, same cheeks and jaw. Change only the face in the masked area.' -export const IMAGE_V2_ROLE_HEADERS: Record, string> = { +export const IMAGE_V2_ROLE_HEADERS: Record, string> = { outfit: 'Person, face, body, pose, and background from image 1. Clothing only from image 2. Fit the outfit from image 2 to the body in image 1. Do not copy image 2’s face, body shape, or pose.', face_lock: 'Body, pose, and scene from image 1. Exact face from image 2.', identity: 'Same person as image 1. Use image 2 only to reinforce the face. Follow the user’s pose/scene prompt.' @@ -42,10 +62,17 @@ export function clampImageV2Strength(raw: unknown, fallback: number) { return Math.min(IMAGE_V2_STRENGTH_MAX, Math.max(IMAGE_V2_STRENGTH_MIN, Math.round(snapped * 100) / 100)) } +export function clampImageV2Denoise(raw: unknown, fallback = IMAGE_V2_DENOISE_DEFAULT) { + const value = Number(raw) + if (!Number.isFinite(value)) return fallback + const snapped = Math.round(value / IMAGE_V2_DENOISE_STEP) * IMAGE_V2_DENOISE_STEP + return Math.min(IMAGE_V2_DENOISE_MAX, Math.max(IMAGE_V2_DENOISE_MIN, Math.round(snapped * 100) / 100)) +} + export function composeImageV2Prompt(mode: ImageV2Mode, task: ImageV2Task, prompt: string) { const body = String(prompt || '').trim() - if (mode !== 'compose' || task === 'scene') return body - const header = IMAGE_V2_ROLE_HEADERS[task] + if (mode === 'refine' || mode !== 'compose' || task === 'scene') return body + const header = IMAGE_V2_ROLE_HEADERS[task as Exclude] return header ? `${header}\n\n${body}` : body } @@ -61,4 +88,5 @@ export type ImageV2PresetSettings = { cfg?: number megapixels?: number turbo?: boolean + strength?: number }