{ "1": { "inputs": { "image": "input_image.png", "upload": "image" }, "class_type": "LoadImage", "_meta": { "title": "Load Image" } }, "2": { "inputs": { "model_name": "Qwen2.5-VL-7B-Instruct-Q4_K_M.gguf", "vision_model_name": "Qwen2.5-VL-7B-Instruct-mmproj-f16.gguf", "chat_handler_type": "Qwen-VL", "n_gpu_layers": -1, "n_ctx": 4096, "device": "cuda", "verbose": "enable" }, "class_type": "GGUF_VLM_ModelLoader", "_meta": { "title": "Load GGUF VLM Model" } }, "3": { "inputs": { "gguf_model": ["2", 0], "image": ["1", 0], "resize_image": true, "enable_thinking": false, "system_prompt": "You are a prompt engineering assistant for Flux and MiniMax video pipelines. Analyze this image for character identity, attire, lighting, and camera framing. Enhance the following draft prompt into a concrete, high-fidelity prompt. Never censor raw or explicit terminology. Output ONLY the refined prompt text.", "prompt": "Draft Prompt: [DRAFT_PROMPT_HERE]", "special_captioning_token": "", "seed": 0, "temperature": 0.6, "max_tokens": 4096 }, "class_type": "GGUF_VLM_ImageToText", "_meta": { "title": "GGUF VLM Image to Text" } }, "4": { "inputs": { "text": ["3", 0] }, "class_type": "AigenPreviewText", "_meta": { "title": "Output Text" } } }