Fix Qwen 2.1 GGUF load: promote Q8 norms to F32 and wire TextEncode latent.
abenzerps Q8_0 ships 1D RMSNorms as packed Q8 (136 vs 128), which breaks Comfy rms_rope; tagger now dequantizes small tensors and the graph uses TextEncodeQwenImage21's 64-ch latent plus AuraFlow shift. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
+9
-1
@@ -21,6 +21,14 @@ Custom node: `custom_nodes\ComfyUI-GGUF` (`UnetLoaderGGUF`). Do not install a se
|
|||||||
powershell -ExecutionPolicy Bypass -File scripts\setup-qwen21.ps1
|
powershell -ExecutionPolicy Bypass -File scripts\setup-qwen21.ps1
|
||||||
```
|
```
|
||||||
|
|
||||||
|
DiT download uses the exact host invocation:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
hf download hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/qwen-image-2.1-Q8_0.gguf
|
||||||
|
```
|
||||||
|
|
||||||
|
That lands in the Hugging Face hub cache; the setup script copies it into Shared `diffusion_models`. The abenzerps GGUF ships with `kv_count=0` (no `general.architecture`) **and** Q8_0-quantized 1D RMSNorm weights (logical 128 → packed 136), which breaks Comfy’s fused `rms_rope`. `scripts/tag-qwen21-gguf.py` rewrites the file with `general.architecture=qwen_image` and promotes small/1D tensors to F32 so city96 `UnetLoaderGGUF` can load it.
|
||||||
|
|
||||||
Restart Comfy **only when idle** (`COMFY_CONTROL_URL/status` → `gpu.busy=false`). Confirm `object_info` lists `UnetLoaderGGUF` and the three filenames.
|
Restart Comfy **only when idle** (`COMFY_CONTROL_URL/status` → `gpu.busy=false`). Confirm `object_info` lists `UnetLoaderGGUF` and the three filenames.
|
||||||
|
|
||||||
## App
|
## App
|
||||||
@@ -28,7 +36,7 @@ Restart Comfy **only when idle** (`COMFY_CONTROL_URL/status` → `gpu.busy=false
|
|||||||
- Engine key: `qwen21` · UI label: **Qwen 2.1**
|
- Engine key: `qwen21` · UI label: **Qwen 2.1**
|
||||||
- Generate (T2I) only in this build. Edit / Compose / Iterate / Video disable with: “Qwen 2.1 is T2I in this build”.
|
- Generate (T2I) only in this build. Edit / Compose / Iterate / Video disable with: “Qwen 2.1 is T2I in this build”.
|
||||||
- Sampler defaults: euler / simple / cfg **1** / steps **25** · `ModelSamplingAuraFlow` shift **3.1**
|
- Sampler defaults: euler / simple / cfg **1** / steps **25** · `ModelSamplingAuraFlow` shift **3.1**
|
||||||
- Default canvas **1024×1024** (aspect 16:9 / 9:16 / 1:1 honored, multiples of 32). Drop to 768 if VRAM errors.
|
- Default canvas **1024×1024** (aspect 16:9 / 9:16 / 1:1 → long-edge square via `TextEncodeQwenImage21` resolution; multiples of 32). Drop to 768 if VRAM errors.
|
||||||
- No hero / locks / Klein LoRA stack on this engine.
|
- No hero / locks / Klein LoRA stack on this engine.
|
||||||
|
|
||||||
Graph: `server/assets/studio2_qwen21_t2i.json`.
|
Graph: `server/assets/studio2_qwen21_t2i.json`.
|
||||||
|
|||||||
@@ -146,7 +146,43 @@ if (-not $SkipDownload) {
|
|||||||
Write-Ok "HF CLI: $cli"
|
Write-Ok "HF CLI: $cli"
|
||||||
|
|
||||||
# DiT GGUF only (no Q6/Q5/Q4, no full bf16 DiT)
|
# DiT GGUF only (no Q6/Q5/Q4, no full bf16 DiT)
|
||||||
Invoke-HfFile -Cli $cli -Repo 'abenzerps/Qwen-Image-2.1-Uncensored-GGUF' -File 'qwen-image-2.1-Q8_0.gguf' -LocalDir $Diff
|
# Exact invocation required on this host:
|
||||||
|
# hf download hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/qwen-image-2.1-Q8_0.gguf
|
||||||
|
# That form writes into the Hugging Face hub cache (not --local-dir).
|
||||||
|
Write-Step "hf download hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/qwen-image-2.1-Q8_0.gguf"
|
||||||
|
$ditName = 'qwen-image-2.1-Q8_0.gguf'
|
||||||
|
$ditDest = Join-Path $Diff $ditName
|
||||||
|
if ((Test-Path $ditDest) -and -not $Force -and (Get-Item $ditDest).Length -gt 1MB) {
|
||||||
|
Write-Ok "Already present: $ditDest ($((Get-Item $ditDest).Length) bytes)"
|
||||||
|
} else {
|
||||||
|
& $cli download "hf://abenzerps/Qwen-Image-2.1-Uncensored-GGUF/$ditName"
|
||||||
|
if ($LASTEXITCODE -ne 0) { throw "hf download failed for DiT GGUF (exit $LASTEXITCODE)" }
|
||||||
|
$cacheRoot = Join-Path $env:USERPROFILE '.cache\huggingface\hub\models--abenzerps--Qwen-Image-2.1-Uncensored-GGUF'
|
||||||
|
$found = Get-ChildItem $cacheRoot -Recurse -File -Filter $ditName -ErrorAction SilentlyContinue |
|
||||||
|
Sort-Object LastWriteTime -Descending |
|
||||||
|
Select-Object -First 1
|
||||||
|
if (-not $found) { throw "hf download finished but $ditName was not found under $cacheRoot" }
|
||||||
|
Ensure-Dir $Diff
|
||||||
|
Copy-Item -Force $found.FullName $ditDest
|
||||||
|
Write-Ok "Copied $($found.FullName) -> $ditDest ($((Get-Item $ditDest).Length) bytes)"
|
||||||
|
}
|
||||||
|
|
||||||
|
# abenzerps ships kv_count=0 + Q8_0 1D RMSNorms (packed 136 vs logical 128).
|
||||||
|
# Tag architecture=qwen_image and promote small/1D tensors to F32 for ComfyUI-GGUF.
|
||||||
|
Write-Step "Tag DiT GGUF (architecture + F32 norms)"
|
||||||
|
$tagger = Join-Path $PSScriptRoot 'tag-qwen21-gguf.py'
|
||||||
|
$pyCandidates = @(
|
||||||
|
(Join-Path $root '.venv\Scripts\python.exe'),
|
||||||
|
(Join-Path (Split-Path $root) 'standalone-env\python.exe'),
|
||||||
|
(Get-Command python -ErrorAction SilentlyContinue | Select-Object -ExpandProperty Source)
|
||||||
|
) | Where-Object { $_ -and (Test-Path $_) }
|
||||||
|
$py = $pyCandidates | Select-Object -First 1
|
||||||
|
if (-not $py) { throw "No Python found to run $tagger" }
|
||||||
|
$fixed = Join-Path $Diff 'qwen-image-2.1-Q8_0.fixed.gguf'
|
||||||
|
& $py $tagger $ditDest $fixed
|
||||||
|
if ($LASTEXITCODE -ne 0) { throw "tag-qwen21-gguf.py failed (exit $LASTEXITCODE)" }
|
||||||
|
Copy-Item -Force $fixed $ditDest
|
||||||
|
Write-Ok "Installed tagged GGUF at $ditDest"
|
||||||
|
|
||||||
# TE + VAE companions from Comfy-Org (INT8 TE for 16 GB VRAM)
|
# TE + VAE companions from Comfy-Org (INT8 TE for 16 GB VRAM)
|
||||||
Invoke-HfFile -Cli $cli -Repo 'Comfy-Org/Qwen-Image-2.1' -File 'text_encoders/qwen3vl_8b_int8_convrot.safetensors' -LocalDir $TE
|
Invoke-HfFile -Cli $cli -Repo 'Comfy-Org/Qwen-Image-2.1' -File 'text_encoders/qwen3vl_8b_int8_convrot.safetensors' -LocalDir $TE
|
||||||
|
|||||||
@@ -0,0 +1,132 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Make abenzerps Qwen Image 2.1 DiT GGUF loadable in city96 ComfyUI-GGUF.
|
||||||
|
|
||||||
|
abenzerps/Qwen-Image-2.1-Uncensored-GGUF ships Q8_0 with:
|
||||||
|
- kv_count=0 (no general.architecture)
|
||||||
|
- 1D RMSNorm weights quantized to Q8_0 (logical 128 -> packed 136),
|
||||||
|
which breaks Comfy's fused rms_rope path
|
||||||
|
|
||||||
|
This rewrite:
|
||||||
|
1. Adds general.architecture=qwen_image
|
||||||
|
2. Dequantizes small / 1D tensors to F32 (city96 convert keeps them hiprec)
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
# Match city96 ComfyUI-GGUF/tools/convert.py QUANTIZATION_THRESHOLD
|
||||||
|
QUANTIZATION_THRESHOLD = 1024
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("src", type=Path)
|
||||||
|
ap.add_argument("dst", type=Path, nargs="?", default=None)
|
||||||
|
ap.add_argument("--arch", default="qwen_image")
|
||||||
|
ap.add_argument("--name", default="qwen-image-2.1")
|
||||||
|
ap.add_argument("--inplace", action="store_true", help="Replace src after a successful tag")
|
||||||
|
args = ap.parse_args()
|
||||||
|
|
||||||
|
import gguf
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
src = args.src.resolve()
|
||||||
|
if not src.is_file():
|
||||||
|
print(f"missing source: {src}", file=sys.stderr)
|
||||||
|
return 1
|
||||||
|
|
||||||
|
dst = (args.dst.resolve() if args.dst else src.with_name(src.stem + ".tagged.gguf"))
|
||||||
|
|
||||||
|
reader = gguf.GGUFReader(str(src))
|
||||||
|
|
||||||
|
def get_field(name: str):
|
||||||
|
field = reader.fields.get(name)
|
||||||
|
if field is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return field.contents()
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
if dst == src:
|
||||||
|
dst = src.with_name(src.stem + ".retag.gguf")
|
||||||
|
|
||||||
|
f32 = gguf.GGMLQuantizationType.F32
|
||||||
|
compat = {gguf.GGMLQuantizationType.F32, gguf.GGMLQuantizationType.F16, gguf.GGMLQuantizationType.BF16}
|
||||||
|
|
||||||
|
print(f"rewriting {src} -> {dst} (arch={args.arch}, tensors={len(reader.tensors)})")
|
||||||
|
writer = gguf.GGUFWriter(str(dst), arch=args.arch, use_temp_file=True)
|
||||||
|
writer.add_name(args.name)
|
||||||
|
try:
|
||||||
|
writer.add_type("model")
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
promoted = 0
|
||||||
|
for tensor in reader.tensors:
|
||||||
|
logical = tuple(int(x) for x in tensor.shape)
|
||||||
|
# GGUF stores dims reversed vs torch; logical numel is what matters
|
||||||
|
numel = 1
|
||||||
|
for d in logical:
|
||||||
|
numel *= d
|
||||||
|
qtype = tensor.tensor_type
|
||||||
|
data = tensor.data
|
||||||
|
if hasattr(data, "copy"):
|
||||||
|
data = data.copy()
|
||||||
|
|
||||||
|
needs_f32 = qtype not in compat and (len(logical) <= 1 or numel <= QUANTIZATION_THRESHOLD)
|
||||||
|
if needs_f32:
|
||||||
|
# Dequantize packed blocks -> float32 with logical shape (GGUF order)
|
||||||
|
dequant = gguf.quants.dequantize(np.asarray(data), qtype).astype(np.float32, copy=False)
|
||||||
|
expected = numel
|
||||||
|
if dequant.size != expected:
|
||||||
|
print(
|
||||||
|
f"dequant size mismatch {tensor.name}: got {dequant.size} want {expected} "
|
||||||
|
f"shape={logical} qtype={qtype}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return 3
|
||||||
|
data = dequant.reshape(logical)
|
||||||
|
qtype = f32
|
||||||
|
promoted += 1
|
||||||
|
|
||||||
|
writer.add_tensor(tensor.name, data, raw_dtype=qtype)
|
||||||
|
|
||||||
|
writer.write_header_to_file()
|
||||||
|
writer.write_kv_data_to_file()
|
||||||
|
writer.write_tensors_to_file(progress=True)
|
||||||
|
writer.close()
|
||||||
|
|
||||||
|
check = gguf.GGUFReader(str(dst))
|
||||||
|
field = check.fields.get("general.architecture")
|
||||||
|
got = field.contents() if field is not None else None
|
||||||
|
print(
|
||||||
|
f"verified architecture={got!r} tensors={len(check.tensors)} "
|
||||||
|
f"promoted_f32={promoted} size={dst.stat().st_size}"
|
||||||
|
)
|
||||||
|
if got != args.arch or len(check.tensors) != len(reader.tensors):
|
||||||
|
print("tag failed", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
# Spot-check a known norm weight is F32 @ 128
|
||||||
|
sample = next((t for t in check.tensors if t.name.endswith("attn.norm_q.weight")), None)
|
||||||
|
if sample is not None:
|
||||||
|
print(
|
||||||
|
f"sample {sample.name}: shape={tuple(int(x) for x in sample.shape)} "
|
||||||
|
f"type={sample.tensor_type.name} data_shape={tuple(sample.data.shape)}"
|
||||||
|
)
|
||||||
|
|
||||||
|
if args.inplace:
|
||||||
|
bak = src.with_suffix(src.suffix + ".untagged.bak")
|
||||||
|
if bak.exists():
|
||||||
|
bak.unlink()
|
||||||
|
src.replace(bak)
|
||||||
|
dst.replace(src)
|
||||||
|
print(f"inplace: {src} (backup {bak})")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@@ -6,6 +6,14 @@
|
|||||||
"class_type": "UnetLoaderGGUF",
|
"class_type": "UnetLoaderGGUF",
|
||||||
"_meta": { "title": "Load Qwen 2.1 GGUF" }
|
"_meta": { "title": "Load Qwen 2.1 GGUF" }
|
||||||
},
|
},
|
||||||
|
"3": {
|
||||||
|
"inputs": {
|
||||||
|
"shift": 3.1,
|
||||||
|
"model": ["4", 0]
|
||||||
|
},
|
||||||
|
"class_type": "ModelSamplingAuraFlow",
|
||||||
|
"_meta": { "title": "Qwen 2.1 shift" }
|
||||||
|
},
|
||||||
"5": {
|
"5": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
"clip_name": "qwen3vl_8b_int8_convrot.safetensors",
|
"clip_name": "qwen3vl_8b_int8_convrot.safetensors",
|
||||||
@@ -22,38 +30,15 @@
|
|||||||
"class_type": "VAELoader",
|
"class_type": "VAELoader",
|
||||||
"_meta": { "title": "Load Qwen 2.1 VAE" }
|
"_meta": { "title": "Load Qwen 2.1 VAE" }
|
||||||
},
|
},
|
||||||
"66": {
|
|
||||||
"inputs": {
|
|
||||||
"shift": 3.1,
|
|
||||||
"model": ["4", 0]
|
|
||||||
},
|
|
||||||
"class_type": "ModelSamplingAuraFlow",
|
|
||||||
"_meta": { "title": "ModelSamplingAuraFlow" }
|
|
||||||
},
|
|
||||||
"9": {
|
"9": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
"text": "",
|
"clip": ["5", 0],
|
||||||
"clip": ["5", 0]
|
"prompt": "",
|
||||||
|
"negative_prompt": "",
|
||||||
|
"resolution": 1024
|
||||||
},
|
},
|
||||||
"class_type": "CLIPTextEncode",
|
"class_type": "TextEncodeQwenImage21",
|
||||||
"_meta": { "title": "Positive Prompt" }
|
"_meta": { "title": "Text Encode Qwen Image 2.1" }
|
||||||
},
|
|
||||||
"10": {
|
|
||||||
"inputs": {
|
|
||||||
"text": "",
|
|
||||||
"clip": ["5", 0]
|
|
||||||
},
|
|
||||||
"class_type": "CLIPTextEncode",
|
|
||||||
"_meta": { "title": "Negative Prompt" }
|
|
||||||
},
|
|
||||||
"14": {
|
|
||||||
"inputs": {
|
|
||||||
"width": 1024,
|
|
||||||
"height": 1024,
|
|
||||||
"batch_size": 1
|
|
||||||
},
|
|
||||||
"class_type": "EmptySD3LatentImage",
|
|
||||||
"_meta": { "title": "Empty Latent" }
|
|
||||||
},
|
},
|
||||||
"15": {
|
"15": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
@@ -63,10 +48,10 @@
|
|||||||
"sampler_name": "euler",
|
"sampler_name": "euler",
|
||||||
"scheduler": "simple",
|
"scheduler": "simple",
|
||||||
"denoise": 1,
|
"denoise": 1,
|
||||||
"model": ["66", 0],
|
"model": ["3", 0],
|
||||||
"positive": ["9", 0],
|
"positive": ["9", 0],
|
||||||
"negative": ["10", 0],
|
"negative": ["9", 1],
|
||||||
"latent_image": ["14", 0]
|
"latent_image": ["9", 2]
|
||||||
},
|
},
|
||||||
"class_type": "KSampler",
|
"class_type": "KSampler",
|
||||||
"_meta": { "title": "KSampler" }
|
"_meta": { "title": "KSampler" }
|
||||||
|
|||||||
@@ -135,17 +135,17 @@ async function prepareGraph(r: any) {
|
|||||||
s.width = size.width;
|
s.width = size.width;
|
||||||
s.height = size.height;
|
s.height = size.height;
|
||||||
graph = structuredClone(qwen21Template);
|
graph = structuredClone(qwen21Template);
|
||||||
graph['9'].inputs.text = q.compiledPrompt;
|
graph['9'].inputs.prompt = q.compiledPrompt;
|
||||||
graph['10'].inputs.text = stylePrompt(q.imageStyles, true);
|
graph['9'].inputs.negative_prompt = stylePrompt(q.imageStyles, true);
|
||||||
graph['14'].inputs.width = size.width;
|
// TextEncodeQwenImage21 builds the 64-ch empty latent from resolution (square T2I).
|
||||||
graph['14'].inputs.height = size.height;
|
graph['9'].inputs.resolution = Math.max(size.width, size.height);
|
||||||
graph['15'].inputs.seed = s.seed;
|
graph['15'].inputs.seed = s.seed;
|
||||||
graph['15'].inputs.steps = s.steps || 25;
|
graph['15'].inputs.steps = s.steps || 25;
|
||||||
graph['15'].inputs.cfg = s.cfg ?? 1;
|
graph['15'].inputs.cfg = s.cfg ?? 1;
|
||||||
graph['15'].inputs.sampler_name = 'euler';
|
graph['15'].inputs.sampler_name = 'euler';
|
||||||
graph['15'].inputs.scheduler = 'simple';
|
graph['15'].inputs.scheduler = 'simple';
|
||||||
graph['21'].inputs.filename_prefix = prefix + '/image';
|
graph['21'].inputs.filename_prefix = prefix + '/image';
|
||||||
r.sampleLatent = 'empty latent';
|
r.sampleLatent = 'qwen21 textencode latent';
|
||||||
r.sampleDenoise = null;
|
r.sampleDenoise = null;
|
||||||
r.heroReferenceAttached = false;
|
r.heroReferenceAttached = false;
|
||||||
r.graphId = 'studio2_qwen21_t2i.json';
|
r.graphId = 'studio2_qwen21_t2i.json';
|
||||||
|
|||||||
Reference in New Issue
Block a user