From 608f7d8d47ed0c4d10508ad3a89991fd415a8d4f Mon Sep 17 00:00:00 2001 From: Towsty Date: Sun, 20 Sep 2026 16:22:55 -0500 Subject: [PATCH] Fix Qwen 2.1 GGUF load: promote Q8 norms to F32 and wire TextEncode latent. abenzerps Q8_0 ships 1D RMSNorms as packed Q8 (136 vs 128), which breaks Comfy rms_rope; tagger now dequantizes small tensors and the graph uses TextEncodeQwenImage21's 64-ch latent plus AuraFlow shift. Co-authored-by: Cursor --- docs/qwen21.md | 10 +- scripts/setup-qwen21.ps1 | 38 +++++++- scripts/tag-qwen21-gguf.py | 132 ++++++++++++++++++++++++++ server/assets/studio2_qwen21_t2i.json | 49 ++++------ server/utils/studio2/runner.ts | 10 +- 5 files changed, 200 insertions(+), 39 deletions(-) create mode 100644 scripts/tag-qwen21-gguf.py diff --git a/docs/qwen21.md b/docs/qwen21.md index 1c71dd9..70220f5 100644 --- a/docs/qwen21.md +++ b/docs/qwen21.md @@ -21,6 +21,14 @@ Custom node: `custom_nodes\ComfyUI-GGUF` (`UnetLoaderGGUF`). Do not install a se powershell -ExecutionPolicy Bypass -File scripts\setup-qwen21.ps1 ``` +DiT download uses the exact host invocation: + +```powershell +hf download hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/qwen-image-2.1-Q8_0.gguf +``` + +That lands in the Hugging Face hub cache; the setup script copies it into Shared `diffusion_models`. The abenzerps GGUF ships with `kv_count=0` (no `general.architecture`) **and** Q8_0-quantized 1D RMSNorm weights (logical 128 → packed 136), which breaks Comfy’s fused `rms_rope`. `scripts/tag-qwen21-gguf.py` rewrites the file with `general.architecture=qwen_image` and promotes small/1D tensors to F32 so city96 `UnetLoaderGGUF` can load it. + Restart Comfy **only when idle** (`COMFY_CONTROL_URL/status` → `gpu.busy=false`). Confirm `object_info` lists `UnetLoaderGGUF` and the three filenames. ## App @@ -28,7 +36,7 @@ Restart Comfy **only when idle** (`COMFY_CONTROL_URL/status` → `gpu.busy=false - Engine key: `qwen21` · UI label: **Qwen 2.1** - Generate (T2I) only in this build. Edit / Compose / Iterate / Video disable with: “Qwen 2.1 is T2I in this build”. - Sampler defaults: euler / simple / cfg **1** / steps **25** · `ModelSamplingAuraFlow` shift **3.1** -- Default canvas **1024×1024** (aspect 16:9 / 9:16 / 1:1 honored, multiples of 32). Drop to 768 if VRAM errors. +- Default canvas **1024×1024** (aspect 16:9 / 9:16 / 1:1 → long-edge square via `TextEncodeQwenImage21` resolution; multiples of 32). Drop to 768 if VRAM errors. - No hero / locks / Klein LoRA stack on this engine. Graph: `server/assets/studio2_qwen21_t2i.json`. diff --git a/scripts/setup-qwen21.ps1 b/scripts/setup-qwen21.ps1 index b64e0f7..aaa6292 100644 --- a/scripts/setup-qwen21.ps1 +++ b/scripts/setup-qwen21.ps1 @@ -146,7 +146,43 @@ if (-not $SkipDownload) { Write-Ok "HF CLI: $cli" # DiT GGUF only (no Q6/Q5/Q4, no full bf16 DiT) - Invoke-HfFile -Cli $cli -Repo 'abenzerps/Qwen-Image-2.1-Uncensored-GGUF' -File 'qwen-image-2.1-Q8_0.gguf' -LocalDir $Diff + # Exact invocation required on this host: + # hf download hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/qwen-image-2.1-Q8_0.gguf + # That form writes into the Hugging Face hub cache (not --local-dir). + Write-Step "hf download hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/qwen-image-2.1-Q8_0.gguf" + $ditName = 'qwen-image-2.1-Q8_0.gguf' + $ditDest = Join-Path $Diff $ditName + if ((Test-Path $ditDest) -and -not $Force -and (Get-Item $ditDest).Length -gt 1MB) { + Write-Ok "Already present: $ditDest ($((Get-Item $ditDest).Length) bytes)" + } else { + & $cli download "hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/$ditName" + if ($LASTEXITCODE -ne 0) { throw "hf download failed for DiT GGUF (exit $LASTEXITCODE)" } + $cacheRoot = Join-Path $env:USERPROFILE '.cache\huggingface\hub\models--abenzerps--Qwen-Image-2.1-Uncensored-GGUF' + $found = Get-ChildItem $cacheRoot -Recurse -File -Filter $ditName -ErrorAction SilentlyContinue | + Sort-Object LastWriteTime -Descending | + Select-Object -First 1 + if (-not $found) { throw "hf download finished but $ditName was not found under $cacheRoot" } + Ensure-Dir $Diff + Copy-Item -Force $found.FullName $ditDest + Write-Ok "Copied $($found.FullName) -> $ditDest ($((Get-Item $ditDest).Length) bytes)" + } + + # abenzerps ships kv_count=0 + Q8_0 1D RMSNorms (packed 136 vs logical 128). + # Tag architecture=qwen_image and promote small/1D tensors to F32 for ComfyUI-GGUF. + Write-Step "Tag DiT GGUF (architecture + F32 norms)" + $tagger = Join-Path $PSScriptRoot 'tag-qwen21-gguf.py' + $pyCandidates = @( + (Join-Path $root '.venv\Scripts\python.exe'), + (Join-Path (Split-Path $root) 'standalone-env\python.exe'), + (Get-Command python -ErrorAction SilentlyContinue | Select-Object -ExpandProperty Source) + ) | Where-Object { $_ -and (Test-Path $_) } + $py = $pyCandidates | Select-Object -First 1 + if (-not $py) { throw "No Python found to run $tagger" } + $fixed = Join-Path $Diff 'qwen-image-2.1-Q8_0.fixed.gguf' + & $py $tagger $ditDest $fixed + if ($LASTEXITCODE -ne 0) { throw "tag-qwen21-gguf.py failed (exit $LASTEXITCODE)" } + Copy-Item -Force $fixed $ditDest + Write-Ok "Installed tagged GGUF at $ditDest" # TE + VAE companions from Comfy-Org (INT8 TE for 16 GB VRAM) Invoke-HfFile -Cli $cli -Repo 'Comfy-Org/Qwen-Image-2.1' -File 'text_encoders/qwen3vl_8b_int8_convrot.safetensors' -LocalDir $TE diff --git a/scripts/tag-qwen21-gguf.py b/scripts/tag-qwen21-gguf.py new file mode 100644 index 0000000..7f79d1c --- /dev/null +++ b/scripts/tag-qwen21-gguf.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Make abenzerps Qwen Image 2.1 DiT GGUF loadable in city96 ComfyUI-GGUF. + +abenzerps/Qwen-Image-2.1-Uncensored-GGUF ships Q8_0 with: + - kv_count=0 (no general.architecture) + - 1D RMSNorm weights quantized to Q8_0 (logical 128 -> packed 136), + which breaks Comfy's fused rms_rope path + +This rewrite: + 1. Adds general.architecture=qwen_image + 2. Dequantizes small / 1D tensors to F32 (city96 convert keeps them hiprec) +""" +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +# Match city96 ComfyUI-GGUF/tools/convert.py QUANTIZATION_THRESHOLD +QUANTIZATION_THRESHOLD = 1024 + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("src", type=Path) + ap.add_argument("dst", type=Path, nargs="?", default=None) + ap.add_argument("--arch", default="qwen_image") + ap.add_argument("--name", default="qwen-image-2.1") + ap.add_argument("--inplace", action="store_true", help="Replace src after a successful tag") + args = ap.parse_args() + + import gguf + import numpy as np + + src = args.src.resolve() + if not src.is_file(): + print(f"missing source: {src}", file=sys.stderr) + return 1 + + dst = (args.dst.resolve() if args.dst else src.with_name(src.stem + ".tagged.gguf")) + + reader = gguf.GGUFReader(str(src)) + + def get_field(name: str): + field = reader.fields.get(name) + if field is None: + return None + try: + return field.contents() + except Exception: + return None + + if dst == src: + dst = src.with_name(src.stem + ".retag.gguf") + + f32 = gguf.GGMLQuantizationType.F32 + compat = {gguf.GGMLQuantizationType.F32, gguf.GGMLQuantizationType.F16, gguf.GGMLQuantizationType.BF16} + + print(f"rewriting {src} -> {dst} (arch={args.arch}, tensors={len(reader.tensors)})") + writer = gguf.GGUFWriter(str(dst), arch=args.arch, use_temp_file=True) + writer.add_name(args.name) + try: + writer.add_type("model") + except Exception: + pass + + promoted = 0 + for tensor in reader.tensors: + logical = tuple(int(x) for x in tensor.shape) + # GGUF stores dims reversed vs torch; logical numel is what matters + numel = 1 + for d in logical: + numel *= d + qtype = tensor.tensor_type + data = tensor.data + if hasattr(data, "copy"): + data = data.copy() + + needs_f32 = qtype not in compat and (len(logical) <= 1 or numel <= QUANTIZATION_THRESHOLD) + if needs_f32: + # Dequantize packed blocks -> float32 with logical shape (GGUF order) + dequant = gguf.quants.dequantize(np.asarray(data), qtype).astype(np.float32, copy=False) + expected = numel + if dequant.size != expected: + print( + f"dequant size mismatch {tensor.name}: got {dequant.size} want {expected} " + f"shape={logical} qtype={qtype}", + file=sys.stderr, + ) + return 3 + data = dequant.reshape(logical) + qtype = f32 + promoted += 1 + + writer.add_tensor(tensor.name, data, raw_dtype=qtype) + + writer.write_header_to_file() + writer.write_kv_data_to_file() + writer.write_tensors_to_file(progress=True) + writer.close() + + check = gguf.GGUFReader(str(dst)) + field = check.fields.get("general.architecture") + got = field.contents() if field is not None else None + print( + f"verified architecture={got!r} tensors={len(check.tensors)} " + f"promoted_f32={promoted} size={dst.stat().st_size}" + ) + if got != args.arch or len(check.tensors) != len(reader.tensors): + print("tag failed", file=sys.stderr) + return 2 + + # Spot-check a known norm weight is F32 @ 128 + sample = next((t for t in check.tensors if t.name.endswith("attn.norm_q.weight")), None) + if sample is not None: + print( + f"sample {sample.name}: shape={tuple(int(x) for x in sample.shape)} " + f"type={sample.tensor_type.name} data_shape={tuple(sample.data.shape)}" + ) + + if args.inplace: + bak = src.with_suffix(src.suffix + ".untagged.bak") + if bak.exists(): + bak.unlink() + src.replace(bak) + dst.replace(src) + print(f"inplace: {src} (backup {bak})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/server/assets/studio2_qwen21_t2i.json b/server/assets/studio2_qwen21_t2i.json index 1fc32fb..6b731b0 100644 --- a/server/assets/studio2_qwen21_t2i.json +++ b/server/assets/studio2_qwen21_t2i.json @@ -6,6 +6,14 @@ "class_type": "UnetLoaderGGUF", "_meta": { "title": "Load Qwen 2.1 GGUF" } }, + "3": { + "inputs": { + "shift": 3.1, + "model": ["4", 0] + }, + "class_type": "ModelSamplingAuraFlow", + "_meta": { "title": "Qwen 2.1 shift" } + }, "5": { "inputs": { "clip_name": "qwen3vl_8b_int8_convrot.safetensors", @@ -22,38 +30,15 @@ "class_type": "VAELoader", "_meta": { "title": "Load Qwen 2.1 VAE" } }, - "66": { - "inputs": { - "shift": 3.1, - "model": ["4", 0] - }, - "class_type": "ModelSamplingAuraFlow", - "_meta": { "title": "ModelSamplingAuraFlow" } - }, "9": { "inputs": { - "text": "", - "clip": ["5", 0] + "clip": ["5", 0], + "prompt": "", + "negative_prompt": "", + "resolution": 1024 }, - "class_type": "CLIPTextEncode", - "_meta": { "title": "Positive Prompt" } - }, - "10": { - "inputs": { - "text": "", - "clip": ["5", 0] - }, - "class_type": "CLIPTextEncode", - "_meta": { "title": "Negative Prompt" } - }, - "14": { - "inputs": { - "width": 1024, - "height": 1024, - "batch_size": 1 - }, - "class_type": "EmptySD3LatentImage", - "_meta": { "title": "Empty Latent" } + "class_type": "TextEncodeQwenImage21", + "_meta": { "title": "Text Encode Qwen Image 2.1" } }, "15": { "inputs": { @@ -63,10 +48,10 @@ "sampler_name": "euler", "scheduler": "simple", "denoise": 1, - "model": ["66", 0], + "model": ["3", 0], "positive": ["9", 0], - "negative": ["10", 0], - "latent_image": ["14", 0] + "negative": ["9", 1], + "latent_image": ["9", 2] }, "class_type": "KSampler", "_meta": { "title": "KSampler" } diff --git a/server/utils/studio2/runner.ts b/server/utils/studio2/runner.ts index fda8892..e1d05fa 100644 --- a/server/utils/studio2/runner.ts +++ b/server/utils/studio2/runner.ts @@ -135,17 +135,17 @@ async function prepareGraph(r: any) { s.width = size.width; s.height = size.height; graph = structuredClone(qwen21Template); - graph['9'].inputs.text = q.compiledPrompt; - graph['10'].inputs.text = stylePrompt(q.imageStyles, true); - graph['14'].inputs.width = size.width; - graph['14'].inputs.height = size.height; + graph['9'].inputs.prompt = q.compiledPrompt; + graph['9'].inputs.negative_prompt = stylePrompt(q.imageStyles, true); + // TextEncodeQwenImage21 builds the 64-ch empty latent from resolution (square T2I). + graph['9'].inputs.resolution = Math.max(size.width, size.height); graph['15'].inputs.seed = s.seed; graph['15'].inputs.steps = s.steps || 25; graph['15'].inputs.cfg = s.cfg ?? 1; graph['15'].inputs.sampler_name = 'euler'; graph['15'].inputs.scheduler = 'simple'; graph['21'].inputs.filename_prefix = prefix + '/image'; - r.sampleLatent = 'empty latent'; + r.sampleLatent = 'qwen21 textencode latent'; r.sampleDenoise = null; r.heroReferenceAttached = false; r.graphId = 'studio2_qwen21_t2i.json';