Remove YuE2 process memory fraction cap on the 5080.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-09-14 20:14:44 -05:00
co-authored by Cursor
parent ee84a99035
commit 840440d7ef
+9 -3
View File
@@ -119,18 +119,24 @@ def main():
patch_graph_attention(attention_backend)
free_cuda(torch)
memory_snapshot(torch, 'memory-before-load')
# 16GB card: cap budget so YuE2 leaves headroom; tiled VAE at 512 frames.
# Prefer tiled VAE; do not cap the process with set_per_process_memory_fraction.
pipe_load = dict(
device='cuda',
backend=pipeline_backend,
memory_budget_gib=12,
vae_core_frames=512,
offload_ar=True,
progress=False,
)
cot = 'full'
pipe_kwargs = dict(style=style, lyrics=lyrics, cot=cot, seed=seed)
with YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load) as pipe:
# YuE2Pipeline.__init__ always calls set_per_process_memory_fraction; skip it on this 16GB host.
_set_fraction = torch.cuda.set_per_process_memory_fraction
torch.cuda.set_per_process_memory_fraction = lambda *args, **kwargs: None
try:
pipe_cm = YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load)
finally:
torch.cuda.set_per_process_memory_fraction = _set_fraction
with pipe_cm as pipe:
memory_snapshot(torch, 'memory-loaded')
emit(stage='plan', message='Planning melody and chords', progress=5, cot=cot)
try: