From 840440d7efcbf1c2135445c4fe37162e6744924c Mon Sep 17 00:00:00 2001 From: Towsty Date: Mon, 14 Sep 2026 20:14:44 -0500 Subject: [PATCH] Remove YuE2 process memory fraction cap on the 5080. Co-authored-by: Cursor --- scripts/yue2-worker.py | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/scripts/yue2-worker.py b/scripts/yue2-worker.py index 922e3ad..50a5f41 100644 --- a/scripts/yue2-worker.py +++ b/scripts/yue2-worker.py @@ -119,18 +119,24 @@ def main(): patch_graph_attention(attention_backend) free_cuda(torch) memory_snapshot(torch, 'memory-before-load') - # 16GB card: cap budget so YuE2 leaves headroom; tiled VAE at 512 frames. + # Prefer tiled VAE; do not cap the process with set_per_process_memory_fraction. pipe_load = dict( device='cuda', backend=pipeline_backend, - memory_budget_gib=12, vae_core_frames=512, offload_ar=True, progress=False, ) cot = 'full' pipe_kwargs = dict(style=style, lyrics=lyrics, cot=cot, seed=seed) - with YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load) as pipe: + # YuE2Pipeline.__init__ always calls set_per_process_memory_fraction; skip it on this 16GB host. + _set_fraction = torch.cuda.set_per_process_memory_fraction + torch.cuda.set_per_process_memory_fraction = lambda *args, **kwargs: None + try: + pipe_cm = YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load) + finally: + torch.cuda.set_per_process_memory_fraction = _set_fraction + with pipe_cm as pipe: memory_snapshot(torch, 'memory-loaded') emit(stage='plan', message='Planning melody and chords', progress=5, cot=cot) try: