diff --git a/components/ImageV2MaskPainter.vue b/components/ImageV2MaskPainter.vue
new file mode 100644
index 0000000..d149072
--- /dev/null
+++ b/components/ImageV2MaskPainter.vue
@@ -0,0 +1,219 @@
+
+
+
+
+
+
+
Canvas loaded · paint the mask
+
+
+
+
+
+ Brush
+
+
+ Eraser
+
+
+ Size
+
+ {{ size }}
+
+
+ Clear
+
+
+
White paint = edit. Black = keep. Prompt only what should change in the painted area.
+
+
+
+
diff --git a/docs/image-v2.md b/docs/image-v2.md
index 2ad41be..7c4e844 100644
--- a/docs/image-v2.md
+++ b/docs/image-v2.md
@@ -6,8 +6,9 @@ Image edit (v1) and Image v2 are siblings. v1 is unchanged. Do not mute-fix `wor
|---|---|---|
| Tab | Image edit | Image v2 |
| Route | `POST /api/edit` | `POST /api/v2/generate` and `POST /v2/generate` |
-| Graphs | One file, prune unused branch | `klein_v2_edit.json` or `klein_v2_compose.json` |
+| Graphs | One file, prune unused branch | `klein_v2_edit.json`, `klein_v2_compose.json`, or `klein_v2_refine.json` |
| Second still | Optional; switches branch | Compose only; required. Mismatch is a 400 |
+| Mask | — | Refine only; required. Missing mask is a 400, not an Edit fallback |
| LoRAs | Optional user picker | SNOFS + Consistency, four strengths |
| Defaults | 20 steps, CFG 1 | 24 steps, CFG 4 (turbo: 8 / 1) |
@@ -15,8 +16,9 @@ Poll v2 jobs the same way as v1: `GET /api/generate/{id}/stream` and `GET /api/g
## Modes
-- **edit** — one image + text. Task is `scene`. Sending `image_b` is rejected.
-- **compose** — two images + text. `image_b` is required. No silent one-image fallback.
+- **edit** — one image + text. Task is `scene`. Sending `image_b` is rejected. Use this to change lighting, background, or the whole frame from a single still.
+- **compose** — two images + text. `image_b` is required. No silent one-image fallback. Use this to lock a person from A and pull a face or outfit from B.
+- **refine** — one canvas + a painted mask + text. `image_a` and `mask` are required. `image_b` is ignored. Prompt only what should change inside the mask. Strength is denoise in that region (not CFG). Use this for face / hand / chest fixes without opening Comfy.
Compose tasks:
@@ -25,11 +27,16 @@ Compose tasks:
- **face_lock** — body/pose/scene from A; face from B
- **scene** — A is the edit image; B is style/background
-Dummy test: Compose + two unrelated photos + “keep everything the same” must change the output. If it matches old v1 gens, still B is not connected.
+Refine does not prepend Edit/Compose role headers.
+
+Dummy tests:
+
+- Compose + two unrelated photos + “keep everything the same” must change the output. If it matches old v1 gens, still B is not connected.
+- Refine + last good gen + mask on the left breast + “red X painted on the left breast” at strength 0.4. Pass = X on the breast, face/pose/background stay. Fail = whole image regenerates, or nothing changes.
## Nodes the mapper patches
-Both graphs:
+Edit and Compose:
| Node | Role |
|------|------|
@@ -37,7 +44,7 @@ Both graphs:
| `2` / `23` | Scale to MP, lanczos (`megapixels`) |
| `7` | SNOFS LoRA (`lora_name`, `strength_model`, `strength_clip`) |
| `8` | Consistency LoRA (`lora_name`, `strength_model`, `strength_clip`) |
-| `9` | Positive `CLIPTextEncode.text` (role header + user prompt) |
+| `9` | Positive `CLIPTextEncode.text` (role header + user prompt on Compose) |
| `10` | Negative `CLIPTextEncode.text` |
| `15` | Seed |
| `17` | Steps (Flux2Scheduler; Klein-native, euler sampler) |
@@ -53,7 +60,19 @@ Compose only:
| `24` | VAE encode B |
| `25` / `26` | Second ReferenceLatent (B fused into pos/neg) |
-If `image_b` is sent and the executed graph has fewer than two `LoadImage` nodes, the job fails.
+Refine only (`klein_v2_refine.json`):
+
+| Node | Role |
+|------|------|
+| `1` | Load canvas |
+| `30` | Load mask (white = edit, black = keep) |
+| `2` | Scale canvas to MP |
+| `31` / `32` / `34` | Resize mask to canvas, take red channel, grow slightly |
+| `33` | `SetLatentNoiseMask` on the encoded canvas |
+| `17` | `BasicScheduler.denoise` — this is Strength, not CFG |
+| `19` | Sampler starts from the masked canvas latent, not an empty Flux2 latent |
+
+If `image_b` is sent on Edit/Compose and the executed graph has fewer than two `LoadImage` nodes, the job fails. If Refine runs without a mask input or without denoise on the scheduler, the job fails.
## Default sliders
@@ -66,13 +85,23 @@ If `image_b` is sent and the executed graph has fewer than two `LoadImage` nodes
| Steps | 24 (turbo 8) |
| CFG | 4 (turbo 1) |
| Scale to MP | 1.0 lanczos |
+| Strength (denoise) | 0.35 on Refine only. Hidden on Edit/Compose. Range 0.15–0.75 |
-There is no Denoise slider. This is reference-latent Klein, not inpaint.
+### When to use each Refine strength
+
+| Preset | Strength | SNOFS model / CLIP | Consistency | Use for |
+|--------|----------|--------------------|-------------|---------|
+| Face | 0.25–0.35 (chip 0.28) | 0.55 / 0.25 | 0.75 / 0.80 | Glasses, cheeks, jaw. Keep the rest of the canvas. |
+| Hand / chest | 0.35–0.45 (chip 0.40) | 0.65 / 0.30 | 0.70 / 0.70 | Hands, breasts, local contact. Dummy test uses 0.40. |
+| Heavy | 0.55 | current sliders | current sliders | Stubborn region that barely moved at 0.40. |
+
+Do not raise Strength to rewrite the whole frame. If the face or background moves, the mask is too big or Strength is too high.
### NSFW identity vs outfit swap
- **identity / face_lock** — keep Consistency at 0.70 / 0.70 so the face from A (identity) or B (face_lock) holds. Do not drop SNOFS CLIP below ~0.30 or the skin/body read falls apart.
- **outfit** — keep SNOFS Model ~0.65 so cloth reads; do not raise it past ~0.85 or it starts rewriting the body from A. Consistency stays at 0.70 so the person in A does not become the person in B.
- **scene / edit** — start at the table defaults. Turbo is for drafts only.
+- **refine** — prompt only the masked change. Face vs hand/chest chips set Strength and LoRAs together.
-Presets on this tab store sliders + mode + task only. They do not load v1 LoRA stacks or a cached latent.
+Presets on this tab store sliders + mode + task (+ Strength when Refine). They do not load v1 LoRA stacks or a cached latent.
diff --git a/pages/index.vue b/pages/index.vue
index 01b8b85..2d91d7e 100644
--- a/pages/index.vue
+++ b/pages/index.vue
@@ -65,7 +65,9 @@
Input
{{ studioMode === 'editv2'
- ? (v2Mode === 'compose'
+ ? (v2Mode === 'refine'
+ ? 'Refine: paint a mask on Still A. Prompt only what should change in the painted area. Strength is denoise in that region.'
+ : v2Mode === 'compose'
? 'Compose: still A is the person/body. Still B is the face or outfit. Two stills required — no silent one-image fallback.'
: 'Edit: one still plus text. Compose is a separate graph if you need a second still.')
: studioMode === 'edit'
@@ -81,7 +83,16 @@
@drop.prevent="onDrop"
@click="openPicker('main')"
>
-
+
+
+ Reset
+
+
Reset
@@ -92,7 +103,7 @@
Reset
-
{{ studioMode === 'edit' || studioMode === 'editv2' ? (studioMode === 'editv2' ? 'Still A' : 'Image 1') : (textToVideo ? 'Start still · optional' : 'Choose an image') }}
+
{{ studioMode === 'edit' || studioMode === 'editv2' ? (studioMode === 'editv2' ? (v2Mode === 'refine' ? 'Still A · canvas' : 'Still A') : 'Image 1') : (textToVideo ? 'Start still · optional' : 'Choose an image') }}
{{ textToVideo ? 'Text-to-video does not need a still. Shot 2+ will still extend from the last frame.' : 'Browse the library or upload a new PNG, JPG, or WEBP' }}
@@ -102,7 +113,7 @@
type="button"
class="rounded-full border px-3 py-1 text-xs font-medium"
:class="v2Mode === 'edit' ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'"
- @click="v2Mode = 'edit'; v2Task = 'scene'"
+ @click="setV2Mode('edit')"
>
Edit
@@ -110,10 +121,18 @@
type="button"
class="rounded-full border px-3 py-1 text-xs font-medium"
:class="v2Mode === 'compose' ? 'border-amber-300/70 bg-amber-400/10 text-amber-100' : 'border-white/10 text-zinc-400 hover:text-white'"
- @click="v2Mode = 'compose'"
+ @click="setV2Mode('compose')"
>
Compose
+
+ Refine
+
Task
-
{{ studioMode === 'edit' || studioMode === 'editv2' ? 'Edit prompt' : 'Motion & scene prompt' }}
+
{{ studioMode === 'editv2' && v2Mode === 'refine' ? 'Refine prompt' : studioMode === 'edit' || studioMode === 'editv2' ? 'Edit prompt' : 'Motion & scene prompt' }}
+
+
Prompt only what should change in the painted area.
+
+
+ Face prompt
+
+
+
{{ parsedShots.length }} shot{{ parsedShots.length === 1 ? '' : 's' }} in this script. Open the prompt to scroll through them.
@@ -671,7 +698,49 @@
/>
-
Klein v2 · 9B Base · euler. Edit and Compose are separate graphs. Shares the desktop GPU with video.
+
Klein v2 · 9B Base · euler. Edit, Compose, and Refine are separate graphs. Shares the desktop GPU with video.
+
+
+ Strength (denoise)
+
+
+ {{ v2Denoise }}
+
+
+
+
+ Face {{ IMAGE_V2_REFINE_FACE.strength.toFixed(2) }}
+
+
+ Hand-chest {{ IMAGE_V2_REFINE_HAND.strength.toFixed(2) }}
+
+
+ Heavy {{ IMAGE_V2_REFINE_HEAVY.strength.toFixed(2) }}
+
+
+
SNOFS Model
@@ -1258,6 +1327,13 @@
>
Use as input still
+
+ Send to Refine
+
@@ -1784,9 +1860,18 @@ import {
IMAGE_V2_CFG_DEFAULT,
IMAGE_V2_CONSISTENCY_CLIP,
IMAGE_V2_CONSISTENCY_MODEL,
+ IMAGE_V2_DENOISE_DEFAULT,
+ IMAGE_V2_DENOISE_MAX,
+ IMAGE_V2_DENOISE_MIN,
+ IMAGE_V2_DENOISE_STEP,
+ IMAGE_V2_REFINE_FACE,
+ IMAGE_V2_REFINE_FACE_PROMPT,
+ IMAGE_V2_REFINE_HAND,
+ IMAGE_V2_REFINE_HEAVY,
IMAGE_V2_SNOFS_CLIP,
IMAGE_V2_SNOFS_MODEL,
IMAGE_V2_STEPS_DEFAULT,
+ clampImageV2Denoise,
clampImageV2Strength,
type ImageV2Mode,
type ImageV2PresetSettings,
@@ -2068,6 +2153,11 @@ const v2Steps = ref(IMAGE_V2_STEPS_DEFAULT)
const v2Cfg = ref(IMAGE_V2_CFG_DEFAULT)
const v2Megapixels = ref(1)
const v2Turbo = ref(false)
+const v2Denoise = ref(IMAGE_V2_DENOISE_DEFAULT)
+const refineMaskDirty = ref(false)
+const refinePainter = ref<{ exportPng: () => Promise; clear: () => void } | null>(null)
+const lastV2StillId = ref('')
+const v2JobPending = ref(false)
const editChainStep = ref(1)
const editChainTotal = ref(1)
const editChainLabel = ref('')
@@ -2444,12 +2534,14 @@ const stillBHint = computed(() => {
const editV2Blocked = computed(() => {
if (!file.value || !prompt.value.trim() || !folderId.value) return true
if (v2Mode.value === 'compose' && !editRefFile.value) return true
+ if (v2Mode.value === 'refine' && !refineMaskDirty.value) return true
if (!comfyOk.value && !imageComfyOk.value) return true
return false
})
const editV2BlockReason = computed(() => {
if (!file.value) return 'Load still A first.'
if (v2Mode.value === 'compose' && !editRefFile.value) return 'Compose requires still B. This will not fall back to one-image generation.'
+ if (v2Mode.value === 'refine' && !refineMaskDirty.value) return 'Paint a mask on Still A first. Refine will not fall back to Edit.'
if (!prompt.value.trim()) return 'Write a prompt first.'
if (!folderId.value) return 'Choose a library folder before generating.'
if (!imageComfyConfigured.value && !comfyOk.value) return 'Image v2 is not configured. Set COMFY_HOST.'
@@ -2458,7 +2550,7 @@ const editV2BlockReason = computed(() => {
})
const editV2SubmitLabel = computed(() => {
const occupied = videoBusy.value || editBusy.value || studioJobs.value.some(job => job.status === 'running')
- const label = v2Mode.value === 'compose' ? 'Compose image' : 'Edit image'
+ const label = v2Mode.value === 'refine' ? 'Refine image' : v2Mode.value === 'compose' ? 'Compose image' : 'Edit image'
return occupied ? `Queue ${label.toLowerCase()}` : label
})
const generateLabel = computed(() => {
@@ -2516,6 +2608,7 @@ const composedIdentityPrompt = computed(() => {
})
const promptPlaceholder = computed(() => {
if (studioMode.value === 'editv2') {
+ if (v2Mode.value === 'refine') return 'Prompt only what should change in the painted area.'
if (v2Mode.value === 'compose' && v2Task.value === 'outfit') return 'Keep everything the same except the clothes from still B.'
if (v2Mode.value === 'compose' && v2Task.value === 'face_lock') return 'Keep the body and scene from still A. Use the face from still B.'
if (v2Mode.value === 'compose' && v2Task.value === 'identity') return 'Same person as still A. Change the pose and scene.'
@@ -3254,7 +3347,7 @@ function currentPresetSnapshot() {
loraStack: [],
settings: {
mode: v2Mode.value,
- task: v2Mode.value === 'compose' ? v2Task.value : 'scene',
+ task: v2Mode.value === 'compose' ? v2Task.value : v2Mode.value === 'refine' ? 'refine' : 'scene',
negative: v2Negative.value,
snofsModel: clampImageV2Strength(v2SnofsModel.value, IMAGE_V2_SNOFS_MODEL),
snofsClip: clampImageV2Strength(v2SnofsClip.value, IMAGE_V2_SNOFS_CLIP),
@@ -3263,7 +3356,8 @@ function currentPresetSnapshot() {
steps: clampImageSteps(v2Steps.value, IMAGE_V2_STEPS_DEFAULT),
cfg: clampImageCfg(v2Cfg.value, IMAGE_V2_CFG_DEFAULT),
megapixels: clampImageScaleMegapixels(v2Megapixels.value, 1),
- turbo: v2Turbo.value === true
+ turbo: v2Turbo.value === true,
+ strength: v2Mode.value === 'refine' ? clampImageV2Denoise(v2Denoise.value) : undefined
}
}
}
@@ -3302,8 +3396,8 @@ function applyGenerationPreset(preset: GenerationPreset) {
const skipped = preset.loraStack.length - stack.length
if (preset.kind === 'imagev2') {
const settings = preset.settings as ImageV2PresetSettings
- v2Mode.value = settings.mode === 'compose' ? 'compose' : 'edit'
- v2Task.value = v2Mode.value === 'compose' ? (settings.task || 'scene') : 'scene'
+ v2Mode.value = settings.mode === 'compose' ? 'compose' : settings.mode === 'refine' ? 'refine' : 'edit'
+ v2Task.value = v2Mode.value === 'compose' ? (settings.task || 'scene') : v2Mode.value === 'refine' ? 'refine' : 'scene'
if (typeof settings.negative === 'string') v2Negative.value = settings.negative
v2SnofsModel.value = clampImageV2Strength(settings.snofsModel, IMAGE_V2_SNOFS_MODEL)
v2SnofsClip.value = clampImageV2Strength(settings.snofsClip, IMAGE_V2_SNOFS_CLIP)
@@ -3313,6 +3407,8 @@ function applyGenerationPreset(preset: GenerationPreset) {
v2Cfg.value = clampImageCfg(settings.cfg, IMAGE_V2_CFG_DEFAULT)
v2Megapixels.value = clampImageScaleMegapixels(settings.megapixels, 1)
v2Turbo.value = settings.turbo === true
+ if (v2Mode.value === 'refine') v2Denoise.value = clampImageV2Denoise(settings.strength)
+ void ensureRefineCanvas()
imageV2PresetId.value = preset.id
imageV2PresetName.value = preset.name
} else if (preset.kind === 'image') {
@@ -3615,6 +3711,8 @@ function resetImage() {
preview.value = ''
imageWidth.value = 0
imageHeight.value = 0
+ refineMaskDirty.value = false
+ refinePainter.value?.clear()
}
function clearVideoForm() {
@@ -5097,6 +5195,62 @@ async function editImage() {
}
}
+function setV2Mode(mode: ImageV2Mode) {
+ v2Mode.value = mode
+ if (mode === 'edit') v2Task.value = 'scene'
+ if (mode === 'refine') {
+ v2Task.value = 'refine'
+ void ensureRefineCanvas()
+ }
+}
+
+function applyRefinePreset(kind: 'face' | 'hand' | 'heavy') {
+ if (kind === 'face') {
+ v2Denoise.value = IMAGE_V2_REFINE_FACE.strength
+ v2SnofsModel.value = IMAGE_V2_REFINE_FACE.snofsModel
+ v2SnofsClip.value = IMAGE_V2_REFINE_FACE.snofsClip
+ v2ConsistencyModel.value = IMAGE_V2_REFINE_FACE.consistencyModel
+ v2ConsistencyClip.value = IMAGE_V2_REFINE_FACE.consistencyClip
+ if (!prompt.value.trim()) prompt.value = IMAGE_V2_REFINE_FACE_PROMPT
+ return
+ }
+ if (kind === 'hand') {
+ v2Denoise.value = IMAGE_V2_REFINE_HAND.strength
+ v2SnofsModel.value = IMAGE_V2_REFINE_HAND.snofsModel
+ v2SnofsClip.value = IMAGE_V2_REFINE_HAND.snofsClip
+ v2ConsistencyModel.value = IMAGE_V2_REFINE_HAND.consistencyModel
+ v2ConsistencyClip.value = IMAGE_V2_REFINE_HAND.consistencyClip
+ return
+ }
+ v2Denoise.value = IMAGE_V2_REFINE_HEAVY.strength
+}
+
+async function ensureRefineCanvas() {
+ if (file.value) return
+ const stillId = lastV2StillId.value || stillIdFromLibraryUrl(editResultUrl.value)
+ if (!stillId) return
+ try {
+ const blob = await $fetch(`/api/library/stills/${stillId}`, { responseType: 'blob' })
+ if (!blob?.size) return
+ readFile(new File([blob], 'refine-canvas.png', { type: blob.type || 'image/png' }), { stillId })
+ } catch {
+ // leave Still A empty if the last output is gone
+ }
+}
+
+async function sendToRefine() {
+ if (!editResultUrl.value) return
+ studioMode.value = 'editv2'
+ outputFocus.value = 'edit'
+ try {
+ await useEditAsInput()
+ setV2Mode('refine')
+ toast('Output is Still A. Paint a mask, then refine.')
+ } catch {
+ toast('Could not send that still to Refine')
+ }
+}
+
async function editImageV2() {
if (editV2Blocked.value) {
toast(editV2BlockReason.value || 'Load still A first.')
@@ -5106,13 +5260,26 @@ async function editImageV2() {
toast('Compose requires still B. This will not fall back to one-image generation.')
return
}
+ if (v2Mode.value === 'refine' && !refineMaskDirty.value) {
+ toast('Paint a mask on Still A first. Refine will not fall back to Edit.')
+ return
+ }
const hideOut = hideThumbnail.value
try {
const body = new FormData()
body.append('mode', v2Mode.value)
- body.append('task', v2Mode.value === 'compose' ? v2Task.value : 'scene')
+ body.append('task', v2Mode.value === 'refine' ? 'refine' : v2Mode.value === 'compose' ? v2Task.value : 'scene')
body.append('image_a', file.value as File)
if (v2Mode.value === 'compose' && editRefFile.value) body.append('image_b', editRefFile.value)
+ if (v2Mode.value === 'refine') {
+ const maskBlob = await refinePainter.value?.exportPng()
+ if (!maskBlob) {
+ toast('Paint a mask on Still A first. Refine will not fall back to Edit.')
+ return
+ }
+ body.append('mask', new File([maskBlob], 'refine-mask.png', { type: 'image/png' }))
+ body.append('strength', String(clampImageV2Denoise(v2Denoise.value)))
+ }
body.append('prompt', prompt.value.trim())
body.append('negative', v2Negative.value)
body.append('snofs_model', String(clampImageV2Strength(v2SnofsModel.value, IMAGE_V2_SNOFS_MODEL)))
@@ -5161,12 +5328,13 @@ async function editImageV2() {
editChainLabel.value = ''
editOverallProgress.value = 0
editCompletedChainStep.value = 0
- editActiveChainPlan.value = [{ label: v2Mode.value === 'compose' ? 'Compose' : 'Edit', prompt: prompt.value.trim() }]
+ editActiveChainPlan.value = [{ label: v2Mode.value === 'refine' ? 'Refine' : v2Mode.value === 'compose' ? 'Compose' : 'Edit', prompt: prompt.value.trim() }]
startTimer('edit')
editDownloadName.value = downloadName
editJobId.value = started.jobId
outputStudioJobId.value = started.studioJobId || ''
persistActiveJob('edit', started.jobId, started.hideThumbnail === true || hideOut, started.folderLocked === true)
+ v2JobPending.value = true
listen('edit', started.jobId, started.hideThumbnail === true || hideOut, started.folderLocked === true)
void pollComfyHealth()
void refreshStudioQueue()
@@ -5458,6 +5626,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc
if (payload.stillId && !locked) {
editResultUrl.value = `/api/library/stills/${payload.stillId}`
+ if (studioMode.value === 'editv2' || v2JobPending.value) lastV2StillId.value = payload.stillId
editLockedSave.value = false
if (payload.hideThumbnail || hidden) {
concealEditOutput.value = true
@@ -5501,6 +5670,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc
editStatusMessage.value = 'Saved to the locked folder. Unlock it to view.'
} else {
editResultUrl.value = `/api/library/stills/${payload.stillId}`
+ if (studioMode.value === 'editv2' || v2JobPending.value) lastV2StillId.value = payload.stillId
concealEditOutput.value = Boolean(payload.hideThumbnail || hidden)
editAwaitingReveal.value = concealEditOutput.value
editLockedSave.value = false
@@ -5509,6 +5679,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc
}
editBusy.value = false
editStatusBusy.value = false
+ v2JobPending.value = false
stopTimer('edit')
stopListen('edit')
clearActiveJob('edit')
@@ -5521,6 +5692,7 @@ function applyEditEvent(payload: Record, hidden: boolean, folderLoc
if (failed) {
editSettledUi = true
editBusy.value = false
+ v2JobPending.value = false
editStatusBusy.value = false
stopTimer('edit')
stopListen('edit')
diff --git a/server/api/v2/generate.post.ts b/server/api/v2/generate.post.ts
index 314e9b9..b32045e 100644
--- a/server/api/v2/generate.post.ts
+++ b/server/api/v2/generate.post.ts
@@ -11,6 +11,8 @@ import {
IMAGE_V2_STEPS_DEFAULT,
IMAGE_V2_TURBO_CFG,
IMAGE_V2_TURBO_STEPS,
+ IMAGE_V2_DENOISE_DEFAULT,
+ clampImageV2Denoise,
clampImageV2Strength,
parseImageV2Mode,
parseImageV2Task,
@@ -37,7 +39,7 @@ async function fileFromUrl(url: string): Promise {
}
const mime = String(res.headers.get('content-type') || 'image/png').split(';')[0]
if (!/^image\//i.test(mime)) {
- throw createError({ statusCode: 400, statusMessage: 'image_a / image_b URL must be an image' })
+ throw createError({ statusCode: 400, statusMessage: 'image_a / image_b / mask URL must be an image' })
}
const data = Buffer.from(await res.arrayBuffer())
if (data.length > 40 * 1024 * 1024) {
@@ -72,16 +74,19 @@ function readMultipart(parts: Array<{ name?: string; filename?: string; type?: s
const fields: Record = {}
let imageA: ImageFile | null = null
let imageB: ImageFile | null = null
+ let mask: ImageFile | null = null
for (const part of parts || []) {
if ((part.name === 'image_a' || part.name === 'image') && part.filename && part.data?.length) {
imageA = { filename: part.filename, data: part.data, type: part.type }
} else if ((part.name === 'image_b' || part.name === 'image2') && part.filename && part.data?.length) {
imageB = { filename: part.filename, data: part.data, type: part.type }
+ } else if (part.name === 'mask' && part.filename && part.data?.length) {
+ mask = { filename: part.filename, data: part.data, type: part.type }
} else if (part.name && part.data) {
fields[part.name] = part.data.toString('utf8')
}
}
- return { fields, imageA, imageB }
+ return { fields, imageA, imageB, mask }
}
export default defineEventHandler(async (event) => {
@@ -89,6 +94,7 @@ export default defineEventHandler(async (event) => {
let fields: Record = {}
let uploadedA: ImageFile | null = null
let uploadedB: ImageFile | null = null
+ let uploadedMask: ImageFile | null = null
if (contentType.includes('multipart/form-data')) {
const form = await readMultipartFormData(event).catch(() => null)
@@ -96,15 +102,16 @@ export default defineEventHandler(async (event) => {
fields = parsed.fields
uploadedA = parsed.imageA
uploadedB = parsed.imageB
+ uploadedMask = parsed.mask
} else {
fields = await readBody>(event).catch(() => ({}))
}
const mode = parseImageV2Mode(fields.mode)
if (!mode) {
- throw createError({ statusCode: 400, statusMessage: 'mode must be edit or compose' })
+ throw createError({ statusCode: 400, statusMessage: 'mode must be edit, compose, or refine' })
}
- const task = parseImageV2Task(fields.task, 'scene')
+ const task = parseImageV2Task(fields.task, mode === 'refine' ? 'refine' : 'scene')
const prompt = String(fields.prompt || '').trim()
if (!prompt) {
throw createError({ statusCode: 400, statusMessage: 'A prompt is required' })
@@ -125,11 +132,18 @@ export default defineEventHandler(async (event) => {
const ownerKey = libraryOwnerKey(event)
const imageA = await resolveImageRef(ownerKey, fields.image_a, uploadedA)
- const imageB = await resolveImageRef(ownerKey, fields.image_b, uploadedB)
+ const imageB = mode === 'refine' ? null : await resolveImageRef(ownerKey, fields.image_b, uploadedB)
+ const mask = mode === 'refine' ? await resolveImageRef(ownerKey, fields.mask, uploadedMask) : null
if (!imageA) {
throw createError({ statusCode: 400, statusMessage: 'image_a is required' })
}
+ if (mode === 'refine' && !mask) {
+ throw createError({
+ statusCode: 400,
+ statusMessage: 'Refine requires a mask. Refusing to fall back to Edit.'
+ })
+ }
if (mode === 'edit' && imageB) {
throw createError({
statusCode: 400,
@@ -166,7 +180,8 @@ export default defineEventHandler(async (event) => {
const size = imageDimensions(imageA.data)
const clipName = String(fields.name || '').trim().slice(0, 80)
const v2Mode = mode as ImageV2Mode
- const v2Task = (mode === 'edit' ? 'scene' : task) as ImageV2Task
+ const v2Task = (mode === 'refine' ? 'refine' : mode === 'edit' ? 'scene' : task) as ImageV2Task
+ const refineStrength = mode === 'refine' ? clampImageV2Denoise(fields.strength, IMAGE_V2_DENOISE_DEFAULT) : undefined
const still = await rememberInputStill({
ownerKey,
@@ -186,6 +201,15 @@ export default defineEventHandler(async (event) => {
hideInput
})
: null
+ const savedMask = mask
+ ? await rememberInputStill({
+ ownerKey,
+ folderId,
+ filename: mask.filename || 'refine-mask.png',
+ data: mask.data,
+ hideInput
+ })
+ : null
const studio = await addStudioJob({
ownerKey,
@@ -228,7 +252,10 @@ export default defineEventHandler(async (event) => {
snofsModel: clampImageV2Strength(fields.snofs_model, IMAGE_V2_SNOFS_MODEL),
snofsClip: clampImageV2Strength(fields.snofs_clip, IMAGE_V2_SNOFS_CLIP),
consistencyModel: clampImageV2Strength(fields.consistency_model, IMAGE_V2_CONSISTENCY_MODEL),
- consistencyClip: clampImageV2Strength(fields.consistency_clip, IMAGE_V2_CONSISTENCY_CLIP)
+ consistencyClip: clampImageV2Strength(fields.consistency_clip, IMAGE_V2_CONSISTENCY_CLIP),
+ maskStillId: savedMask?.id,
+ maskStillFilename: savedMask?.filename,
+ refineStrength
}
})
await kickStudioQueue()
@@ -248,7 +275,8 @@ export default defineEventHandler(async (event) => {
cfg,
mode: v2Mode,
task: v2Task,
- workflow: v2Mode === 'compose' ? 'klein_v2_compose.json' : 'klein_v2_edit.json',
+ workflow: v2Mode === 'refine' ? 'klein_v2_refine.json' : v2Mode === 'compose' ? 'klein_v2_compose.json' : 'klein_v2_edit.json',
+ strength: refineStrength,
hideThumbnail,
folderLocked
}
diff --git a/server/assets/klein_v2_refine.json b/server/assets/klein_v2_refine.json
new file mode 100644
index 0000000..80c0f07
--- /dev/null
+++ b/server/assets/klein_v2_refine.json
@@ -0,0 +1,204 @@
+{
+ "1": {
+ "inputs": { "image": "" },
+ "class_type": "LoadImage",
+ "_meta": { "title": "Load Canvas" }
+ },
+ "30": {
+ "inputs": { "image": "" },
+ "class_type": "LoadImage",
+ "_meta": { "title": "Load Mask" }
+ },
+ "2": {
+ "inputs": {
+ "upscale_method": "lanczos",
+ "megapixels": 1,
+ "resolution_steps": 1,
+ "image": ["1", 0]
+ },
+ "class_type": "ImageScaleToTotalPixels",
+ "_meta": { "title": "Scale Canvas" }
+ },
+ "3": {
+ "inputs": { "image": ["2", 0] },
+ "class_type": "GetImageSize",
+ "_meta": { "title": "Get Canvas Size" }
+ },
+ "31": {
+ "inputs": {
+ "upscale_method": "lanczos",
+ "width": ["3", 0],
+ "height": ["3", 1],
+ "crop": "disabled",
+ "image": ["30", 0]
+ },
+ "class_type": "ImageScale",
+ "_meta": { "title": "Resize Mask to Canvas" }
+ },
+ "32": {
+ "inputs": {
+ "channel": "red",
+ "image": ["31", 0]
+ },
+ "class_type": "ImageToMask",
+ "_meta": { "title": "Mask Channel" }
+ },
+ "34": {
+ "inputs": {
+ "expand": 6,
+ "tapered_corners": true,
+ "mask": ["32", 0]
+ },
+ "class_type": "GrowMask",
+ "_meta": { "title": "Grow Mask" }
+ },
+ "4": {
+ "inputs": {
+ "unet_name": "flux-2-klein-base-9b-fp8.safetensors",
+ "weight_dtype": "default"
+ },
+ "class_type": "UNETLoader",
+ "_meta": { "title": "Load Flux.2 Klein 9B Base" }
+ },
+ "5": {
+ "inputs": {
+ "clip_name": "qwen_3_8b_fp8mixed.safetensors",
+ "type": "flux2",
+ "device": "default"
+ },
+ "class_type": "CLIPLoader",
+ "_meta": { "title": "Load Qwen 3 8B CLIP" }
+ },
+ "6": {
+ "inputs": { "vae_name": "full_encoder_small_decoder.safetensors" },
+ "class_type": "VAELoader",
+ "_meta": { "title": "Load Klein VAE" }
+ },
+ "7": {
+ "inputs": {
+ "lora_name": "klein_snofs_v1_4.safetensors",
+ "strength_model": 0.65,
+ "strength_clip": 0.3,
+ "model": ["4", 0],
+ "clip": ["5", 0]
+ },
+ "class_type": "LoraLoader",
+ "_meta": { "title": "SNOFS" }
+ },
+ "8": {
+ "inputs": {
+ "lora_name": "Flux2-Klein-9B-consistency-V2.safetensors",
+ "strength_model": 0.7,
+ "strength_clip": 0.7,
+ "model": ["7", 0],
+ "clip": ["7", 1]
+ },
+ "class_type": "LoraLoader",
+ "_meta": { "title": "Consistency" }
+ },
+ "9": {
+ "inputs": {
+ "text": "",
+ "clip": ["8", 1]
+ },
+ "class_type": "CLIPTextEncode",
+ "_meta": { "title": "Positive Prompt" }
+ },
+ "10": {
+ "inputs": {
+ "text": "",
+ "clip": ["8", 1]
+ },
+ "class_type": "CLIPTextEncode",
+ "_meta": { "title": "Negative Prompt" }
+ },
+ "11": {
+ "inputs": {
+ "pixels": ["2", 0],
+ "vae": ["6", 0]
+ },
+ "class_type": "VAEEncode",
+ "_meta": { "title": "VAE Encode Canvas" }
+ },
+ "12": {
+ "inputs": {
+ "conditioning": ["9", 0],
+ "latent": ["11", 0]
+ },
+ "class_type": "ReferenceLatent",
+ "_meta": { "title": "Reference Latent +" }
+ },
+ "13": {
+ "inputs": {
+ "conditioning": ["10", 0],
+ "latent": ["11", 0]
+ },
+ "class_type": "ReferenceLatent",
+ "_meta": { "title": "Reference Latent -" }
+ },
+ "33": {
+ "inputs": {
+ "samples": ["11", 0],
+ "mask": ["34", 0]
+ },
+ "class_type": "SetLatentNoiseMask",
+ "_meta": { "title": "Mask Latent Noise" }
+ },
+ "15": {
+ "inputs": { "noise_seed": 1 },
+ "class_type": "RandomNoise",
+ "_meta": { "title": "RandomNoise" }
+ },
+ "16": {
+ "inputs": { "sampler_name": "euler" },
+ "class_type": "KSamplerSelect",
+ "_meta": { "title": "KSamplerSelect" }
+ },
+ "17": {
+ "inputs": {
+ "scheduler": "simple",
+ "steps": 24,
+ "denoise": 0.35,
+ "model": ["8", 0]
+ },
+ "class_type": "BasicScheduler",
+ "_meta": { "title": "BasicScheduler" }
+ },
+ "18": {
+ "inputs": {
+ "cfg": 4,
+ "model": ["8", 0],
+ "positive": ["12", 0],
+ "negative": ["13", 0]
+ },
+ "class_type": "CFGGuider",
+ "_meta": { "title": "CFG Guider" }
+ },
+ "19": {
+ "inputs": {
+ "noise": ["15", 0],
+ "guider": ["18", 0],
+ "sampler": ["16", 0],
+ "sigmas": ["17", 0],
+ "latent_image": ["33", 0]
+ },
+ "class_type": "SamplerCustomAdvanced",
+ "_meta": { "title": "SamplerCustomAdvanced" }
+ },
+ "20": {
+ "inputs": {
+ "samples": ["19", 0],
+ "vae": ["6", 0]
+ },
+ "class_type": "VAEDecode",
+ "_meta": { "title": "VAE Decode" }
+ },
+ "21": {
+ "inputs": {
+ "filename_prefix": "aigen-v2-refine",
+ "images": ["20", 0]
+ },
+ "class_type": "SaveImage",
+ "_meta": { "title": "Save Image" }
+ }
+}
diff --git a/server/utils/generationPresets.ts b/server/utils/generationPresets.ts
index 3ab3c6a..40b4785 100644
--- a/server/utils/generationPresets.ts
+++ b/server/utils/generationPresets.ts
@@ -21,6 +21,8 @@ import {
IMAGE_V2_SNOFS_CLIP,
IMAGE_V2_SNOFS_MODEL,
IMAGE_V2_STEPS_DEFAULT,
+ IMAGE_V2_DENOISE_DEFAULT,
+ clampImageV2Denoise,
clampImageV2Strength,
parseImageV2Mode,
parseImageV2Task,
@@ -114,7 +116,7 @@ function sanitizeImageV2Settings(raw: unknown): ImageV2PresetSettings {
const turbo = rec.turbo === true
return {
mode,
- task: mode === 'compose' ? parseImageV2Task(rec.task, 'scene') : 'scene',
+ task: mode === 'compose' ? parseImageV2Task(rec.task, 'scene') : mode === 'refine' ? 'refine' : 'scene',
negative: String(rec.negative || '').slice(0, 2000),
snofsModel: clampImageV2Strength(rec.snofsModel ?? rec.snofs_model, IMAGE_V2_SNOFS_MODEL),
snofsClip: clampImageV2Strength(rec.snofsClip ?? rec.snofs_clip, IMAGE_V2_SNOFS_CLIP),
@@ -123,7 +125,8 @@ function sanitizeImageV2Settings(raw: unknown): ImageV2PresetSettings {
steps: turbo ? 8 : clampImageSteps(rec.steps, IMAGE_V2_STEPS_DEFAULT),
cfg: turbo ? 1 : clampImageCfg(rec.cfg, IMAGE_V2_CFG_DEFAULT),
megapixels: clampImageScaleMegapixels(rec.megapixels ?? rec.scaleMegapixels, 1),
- turbo
+ turbo,
+ strength: mode === 'refine' ? clampImageV2Denoise(rec.strength, IMAGE_V2_DENOISE_DEFAULT) : undefined
}
}
diff --git a/server/utils/imageChainV2.ts b/server/utils/imageChainV2.ts
index e288ee8..e56b62f 100644
--- a/server/utils/imageChainV2.ts
+++ b/server/utils/imageChainV2.ts
@@ -15,6 +15,7 @@ export type EditV2RunParams = {
task: ImageV2Task
image: EditImageFile
reference: EditImageFile | null
+ mask?: EditImageFile | null
prompt: string
negative: string
steps: number
@@ -25,11 +26,15 @@ export type EditV2RunParams = {
consistencyModel: number
consistencyClip: number
megapixels: number
+ strength?: number
}
export async function runEditV2(job: Job, params: EditV2RunParams) {
const library = job.library
if (!library) throw new Error('Edit job is missing library metadata')
+ if (params.mode === 'refine' && !params.mask) {
+ throw new Error('Refine requires a mask. Refusing to fall back to Edit.')
+ }
if (params.mode === 'compose' && !params.reference) {
throw new Error('Compose requires image B. Refusing to fall back to one-image generation.')
}
@@ -52,23 +57,35 @@ export async function runEditV2(job: Job, params: EditV2RunParams) {
await withImageComfyHost(job.imageComfyHost, async () => {
emitChainJob(job, {
type: 'status',
- message: params.mode === 'compose' ? 'Uploading stills A and B to Beast...' : 'Uploading still A to Beast...',
+ message: params.mode === 'refine'
+ ? 'Uploading canvas and mask to Beast...'
+ : params.mode === 'compose' ? 'Uploading stills A and B to Beast...' : 'Uploading still A to Beast...',
progress: 8
})
const uploaded = await uploadImage(params.image, job.id)
- const uploadedRef = params.reference
+ const uploadedRef = params.mode === 'refine'
+ ? null
+ : params.reference
+ ? await uploadImage({
+ ...params.reference,
+ filename: `ref_${params.reference.filename || 'image_b.png'}`
+ }, job.id)
+ : null
+ const uploadedMask = params.mode === 'refine' && params.mask
? await uploadImage({
- ...params.reference,
- filename: `ref_${params.reference.filename || 'image_b.png'}`
+ ...params.mask,
+ filename: `mask_${params.mask.filename || 'refine-mask.png'}`
}, job.id)
: null
if (job.status === 'cancelled') throw new Error('Job interrupted.')
emitChainJob(job, {
type: 'status',
- message: params.mode === 'compose'
- ? `Queueing Klein v2 compose (${params.task}) on Beast...`
- : 'Queueing Klein v2 edit on Beast...',
+ message: params.mode === 'refine'
+ ? 'Queueing Klein v2 refine on Beast...'
+ : params.mode === 'compose'
+ ? `Queueing Klein v2 compose (${params.task}) on Beast...`
+ : 'Queueing Klein v2 edit on Beast...',
progress: 12
})
await ensureComfyLoraNames('image')
@@ -80,6 +97,8 @@ export async function runEditV2(job: Job, params: EditV2RunParams) {
negative: params.negative,
imageAName: uploaded.name,
imageBName: uploadedRef?.name,
+ maskName: uploadedMask?.name,
+ strength: params.strength,
snofsModel: params.snofsModel,
snofsClip: params.snofsClip,
consistencyModel: params.consistencyModel,
@@ -137,7 +156,7 @@ export async function runEditV2(job: Job, params: EditV2RunParams) {
video: { filename: output.filename, subfolder: output.subfolder, type: output.type },
imageName: uploaded.name,
imageSubfolder: uploaded.subfolder,
- extraImageNames: uploadedRef?.name ? [uploadedRef.name] : [],
+ extraImageNames: [uploadedRef?.name, uploadedMask?.name].filter((name): name is string => Boolean(name)),
promptId: job.promptId
})
diff --git a/server/utils/imageWorkflowV2.ts b/server/utils/imageWorkflowV2.ts
index 36d91b2..4059341 100644
--- a/server/utils/imageWorkflowV2.ts
+++ b/server/utils/imageWorkflowV2.ts
@@ -1,5 +1,6 @@
import editTemplate from '../assets/klein_v2_edit.json'
import composeTemplate from '../assets/klein_v2_compose.json'
+import refineTemplate from '../assets/klein_v2_refine.json'
import { IMAGE_SCALE_TO_TOTAL_PIXELS } from '~/server/utils/comfy'
import { cachedComfyLoraNames } from '~/server/utils/loras'
import { resolveComfyLoraName, loraIdentityKey } from '~/utils/loras'
@@ -11,6 +12,8 @@ import {
IMAGE_V2_SNOFS_CLIP,
IMAGE_V2_SNOFS_LORA,
IMAGE_V2_SNOFS_MODEL,
+ IMAGE_V2_DENOISE_DEFAULT,
+ clampImageV2Denoise,
clampImageV2Strength,
composeImageV2Prompt,
type ImageV2Mode,
@@ -22,6 +25,8 @@ type WorkflowGraph = Record
const LOAD_A = '1'
const LOAD_B = '22'
+const LOAD_MASK = '30'
+const SCHEDULER_DENOISE = '17'
const SCALE_A = '2'
const SCALE_B = '23'
const PROMPT = '9'
@@ -35,6 +40,7 @@ const CONSISTENCY = '8'
export const IMAGE_V2_EDIT_WORKFLOW = 'klein_v2_edit.json'
export const IMAGE_V2_COMPOSE_WORKFLOW = 'klein_v2_compose.json'
+export const IMAGE_V2_REFINE_WORKFLOW = 'klein_v2_refine.json'
export interface ImageV2BuildParams {
mode: ImageV2Mode
@@ -43,6 +49,8 @@ export interface ImageV2BuildParams {
negative?: string
imageAName: string
imageBName?: string
+ maskName?: string
+ strength?: number
snofsModel?: number
snofsClip?: number
consistencyModel?: number
@@ -87,8 +95,36 @@ function patchScaleMegapixels(graph: WorkflowGraph, megapixels: number) {
}
}
+function graphHasMaskInput(graph: WorkflowGraph) {
+ return Object.values(graph).some((node) => {
+ const mask = node.inputs?.mask
+ return mask !== undefined && mask !== null && mask !== ''
+ })
+}
+
export function assertImageV2Graph(graph: WorkflowGraph, mode: ImageV2Mode, imageBName?: string) {
const loaders = loadImageNames(graph)
+ if (mode === 'refine') {
+ const mask = graph[LOAD_MASK]
+ if (!mask || mask.class_type !== 'LoadImage' || !String(mask.inputs.image || '').trim()) {
+ throw createError({
+ statusCode: 500,
+ statusMessage: 'Refine job is missing the mask image. Refusing to run.'
+ })
+ }
+ if (!graphHasMaskInput(graph)) {
+ throw createError({
+ statusCode: 500,
+ statusMessage: 'Refine graph has no mask input. Refusing to run.'
+ })
+ }
+ if (!('denoise' in (graph[SCHEDULER]?.inputs || {}))) {
+ throw createError({
+ statusCode: 500,
+ statusMessage: 'Refine graph has no denoise on the sampler. Refusing to run.'
+ })
+ }
+ }
if (mode === 'compose') {
if (loaders.length < 2) {
throw createError({
@@ -104,7 +140,7 @@ export function assertImageV2Graph(graph: WorkflowGraph, mode: ImageV2Mode, imag
})
}
}
- if (imageBName && loaders.length < 2) {
+ if (mode !== 'refine' && imageBName && loaders.length < 2) {
throw createError({
statusCode: 500,
statusMessage: 'image_b was sent but the executed graph has no second image input.'
@@ -124,7 +160,11 @@ export function assertImageV2Graph(graph: WorkflowGraph, mode: ImageV2Mode, imag
export function buildImageV2Workflow(params: ImageV2BuildParams) {
const compose = params.mode === 'compose'
- const graph = structuredClone(compose ? composeTemplate : editTemplate) as WorkflowGraph
+ const refine = params.mode === 'refine'
+ if (refine && !String(params.maskName || '').trim()) {
+ throw createError({ statusCode: 400, statusMessage: 'Refine requires a mask. Refusing to fall back to Edit.' })
+ }
+ const graph = structuredClone(refine ? refineTemplate : compose ? composeTemplate : editTemplate) as WorkflowGraph
const prompt = composeImageV2Prompt(params.mode, params.task, params.prompt)
const negative = String(params.negative || '')
const snofsModel = clampImageV2Strength(params.snofsModel, IMAGE_V2_SNOFS_MODEL)
@@ -133,10 +173,15 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) {
const consistencyClip = clampImageV2Strength(params.consistencyClip, IMAGE_V2_CONSISTENCY_CLIP)
const steps = clampImageSteps(params.steps, 24)
const cfg = clampImageCfg(params.cfg, 4)
- const workflowFile = compose ? IMAGE_V2_COMPOSE_WORKFLOW : IMAGE_V2_EDIT_WORKFLOW
+ const strength = refine ? clampImageV2Denoise(params.strength, IMAGE_V2_DENOISE_DEFAULT) : undefined
+ const workflowFile = refine ? IMAGE_V2_REFINE_WORKFLOW : compose ? IMAGE_V2_COMPOSE_WORKFLOW : IMAGE_V2_EDIT_WORKFLOW
setInput(graph, LOAD_A, 'image', params.imageAName)
if (compose) setInput(graph, LOAD_B, 'image', params.imageBName || '')
+ if (refine) {
+ setInput(graph, LOAD_MASK, 'image', params.maskName || '')
+ setInput(graph, SCHEDULER_DENOISE, 'denoise', strength)
+ }
setInput(graph, PROMPT, 'text', prompt)
setInput(graph, NEGATIVE, 'text', negative)
setInput(graph, NOISE, 'noise_seed', params.seed)
@@ -152,7 +197,7 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) {
setInput(graph, CONSISTENCY, 'strength_model', consistencyModel)
setInput(graph, CONSISTENCY, 'strength_clip', consistencyClip)
- assertImageV2Graph(graph, params.mode, params.imageBName)
+ assertImageV2Graph(graph, params.mode, refine ? undefined : params.imageBName)
const loaders = loadImageNames(graph)
console.log(JSON.stringify({
@@ -160,6 +205,9 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) {
workflow: workflowFile,
mode: params.mode,
task: params.task,
+ canvas: { id: LOAD_A, file: graph[LOAD_A]?.inputs.image },
+ mask: refine ? { id: LOAD_MASK, file: graph[LOAD_MASK]?.inputs.image } : undefined,
+ strength,
loadImage: Object.fromEntries(loaders.map(item => [item.id, { title: item.title, file: item.image }])),
loras: {
snofs: { name: graph[SNOFS]?.inputs.lora_name, model: snofsModel, clip: snofsClip },
@@ -171,14 +219,16 @@ export function buildImageV2Workflow(params: ImageV2BuildParams) {
megapixels: clampImageScaleMegapixels(params.megapixels ?? 1)
}))
- return { graph, workflowFile, loaders, prompt }
+ return { graph, workflowFile, loaders, prompt, strength }
}
export const IMAGE_V2_NODE_LABELS: Record = {
'1': 'Loading image A',
'22': 'Loading image B',
+ '30': 'Loading mask',
'2': 'Scaling image A',
'23': 'Scaling image B',
+ '31': 'Resizing mask',
'4': 'Loading Flux.2 Klein 9B',
'5': 'Loading CLIP',
'6': 'Loading VAE',
@@ -187,6 +237,7 @@ export const IMAGE_V2_NODE_LABELS: Record = {
'9': 'Encoding prompt',
'11': 'Encoding image A',
'24': 'Encoding image B',
+ '33': 'Applying mask',
'19': 'Sampling Klein v2',
'20': 'Decoding still',
'21': 'Saving still'
diff --git a/server/utils/studioQueue.ts b/server/utils/studioQueue.ts
index 3e78574..265fd5b 100644
--- a/server/utils/studioQueue.ts
+++ b/server/utils/studioQueue.ts
@@ -52,12 +52,15 @@ export interface StudioJobPayload {
scaleToTotalPixels?: boolean
scaleMegapixels?: number
imagePipeline?: 'v1' | 'v2'
- v2Mode?: 'edit' | 'compose'
- v2Task?: 'scene' | 'identity' | 'outfit' | 'face_lock'
+ v2Mode?: 'edit' | 'compose' | 'refine'
+ v2Task?: 'scene' | 'identity' | 'outfit' | 'face_lock' | 'refine'
snofsModel?: number
snofsClip?: number
consistencyModel?: number
consistencyClip?: number
+ maskStillId?: string
+ maskStillFilename?: string
+ refineStrength?: number
}
export interface StudioJob {
@@ -694,7 +697,18 @@ async function startStudioEditJob(item: StudioJob) {
: null
if (payload.imagePipeline === 'v2') {
- const mode = payload.v2Mode === 'compose' ? 'compose' : 'edit'
+ const mode = payload.v2Mode === 'compose' ? 'compose' : payload.v2Mode === 'refine' ? 'refine' : 'edit'
+ const maskId = payload.maskStillId
+ const mask = mode === 'refine' && maskId && existsSync(stillPath(item.ownerKey, maskId))
+ ? {
+ filename: payload.maskStillFilename || 'refine-mask.png',
+ data: readFileSync(stillPath(item.ownerKey, maskId)),
+ type: 'image/png'
+ }
+ : null
+ if (mode === 'refine' && !mask) {
+ throw new Error('Refine requires a mask. Refusing to fall back to Edit.')
+ }
if (mode === 'compose' && !reference) {
throw new Error('Compose requires image B. Refusing to fall back to one-image generation.')
}
@@ -732,7 +746,8 @@ async function startStudioEditJob(item: StudioJob) {
mode,
task: payload.v2Task || 'scene',
image,
- reference,
+ reference: mode === 'refine' ? null : reference,
+ mask,
prompt: payload.prompt,
negative: payload.negative || '',
steps: payload.steps,
@@ -742,7 +757,8 @@ async function startStudioEditJob(item: StudioJob) {
snofsClip: payload.snofsClip ?? 0.35,
consistencyModel: payload.consistencyModel ?? 0.7,
consistencyClip: payload.consistencyClip ?? 0.7,
- megapixels: payload.scaleMegapixels ?? 1
+ megapixels: payload.scaleMegapixels ?? 1,
+ strength: payload.refineStrength
}).catch((error) => {
const message = error instanceof Error ? error.message : String(error)
if (live && live.status !== 'error' && live.status !== 'cancelled' && live.status !== 'deferred') {
diff --git a/utils/imageV2.ts b/utils/imageV2.ts
index b91cda8..63fa535 100644
--- a/utils/imageV2.ts
+++ b/utils/imageV2.ts
@@ -1,5 +1,5 @@
-export const IMAGE_V2_MODES = ['edit', 'compose'] as const
-export const IMAGE_V2_TASKS = ['scene', 'identity', 'outfit', 'face_lock'] as const
+export const IMAGE_V2_MODES = ['edit', 'compose', 'refine'] as const
+export const IMAGE_V2_TASKS = ['scene', 'identity', 'outfit', 'face_lock', 'refine'] as const
export type ImageV2Mode = (typeof IMAGE_V2_MODES)[number]
export type ImageV2Task = (typeof IMAGE_V2_TASKS)[number]
@@ -18,8 +18,28 @@ export const IMAGE_V2_STRENGTH_STEP = 0.05
export const IMAGE_V2_SNOFS_LORA = 'klein_snofs_v1_4.safetensors'
export const IMAGE_V2_CONSISTENCY_LORA = 'Flux2-Klein-9B-consistency-V2.safetensors'
+export const IMAGE_V2_DENOISE_DEFAULT = 0.35
+export const IMAGE_V2_DENOISE_MIN = 0.15
+export const IMAGE_V2_DENOISE_MAX = 0.75
+export const IMAGE_V2_DENOISE_STEP = 0.01
+export const IMAGE_V2_REFINE_FACE = {
+ strength: 0.28,
+ snofsModel: 0.55,
+ snofsClip: 0.25,
+ consistencyModel: 0.75,
+ consistencyClip: 0.8
+}
+export const IMAGE_V2_REFINE_HAND = {
+ strength: 0.4,
+ snofsModel: 0.65,
+ snofsClip: 0.3,
+ consistencyModel: 0.7,
+ consistencyClip: 0.7
+}
+export const IMAGE_V2_REFINE_HEAVY = { strength: 0.55 }
+export const IMAGE_V2_REFINE_FACE_PROMPT = 'same face as the canvas, same glasses, same cheeks and jaw. Change only the face in the masked area.'
-export const IMAGE_V2_ROLE_HEADERS: Record, string> = {
+export const IMAGE_V2_ROLE_HEADERS: Record, string> = {
outfit: 'Person, face, body, pose, and background from image 1. Clothing only from image 2. Fit the outfit from image 2 to the body in image 1. Do not copy image 2’s face, body shape, or pose.',
face_lock: 'Body, pose, and scene from image 1. Exact face from image 2.',
identity: 'Same person as image 1. Use image 2 only to reinforce the face. Follow the user’s pose/scene prompt.'
@@ -42,10 +62,17 @@ export function clampImageV2Strength(raw: unknown, fallback: number) {
return Math.min(IMAGE_V2_STRENGTH_MAX, Math.max(IMAGE_V2_STRENGTH_MIN, Math.round(snapped * 100) / 100))
}
+export function clampImageV2Denoise(raw: unknown, fallback = IMAGE_V2_DENOISE_DEFAULT) {
+ const value = Number(raw)
+ if (!Number.isFinite(value)) return fallback
+ const snapped = Math.round(value / IMAGE_V2_DENOISE_STEP) * IMAGE_V2_DENOISE_STEP
+ return Math.min(IMAGE_V2_DENOISE_MAX, Math.max(IMAGE_V2_DENOISE_MIN, Math.round(snapped * 100) / 100))
+}
+
export function composeImageV2Prompt(mode: ImageV2Mode, task: ImageV2Task, prompt: string) {
const body = String(prompt || '').trim()
- if (mode !== 'compose' || task === 'scene') return body
- const header = IMAGE_V2_ROLE_HEADERS[task]
+ if (mode === 'refine' || mode !== 'compose' || task === 'scene') return body
+ const header = IMAGE_V2_ROLE_HEADERS[task as Exclude]
return header ? `${header}\n\n${body}` : body
}
@@ -61,4 +88,5 @@ export type ImageV2PresetSettings = {
cfg?: number
megapixels?: number
turbo?: boolean
+ strength?: number
}