Remove YuE2 process memory fraction cap on the 5080.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -119,18 +119,24 @@ def main():
|
||||
patch_graph_attention(attention_backend)
|
||||
free_cuda(torch)
|
||||
memory_snapshot(torch, 'memory-before-load')
|
||||
# 16GB card: cap budget so YuE2 leaves headroom; tiled VAE at 512 frames.
|
||||
# Prefer tiled VAE; do not cap the process with set_per_process_memory_fraction.
|
||||
pipe_load = dict(
|
||||
device='cuda',
|
||||
backend=pipeline_backend,
|
||||
memory_budget_gib=12,
|
||||
vae_core_frames=512,
|
||||
offload_ar=True,
|
||||
progress=False,
|
||||
)
|
||||
cot = 'full'
|
||||
pipe_kwargs = dict(style=style, lyrics=lyrics, cot=cot, seed=seed)
|
||||
with YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load) as pipe:
|
||||
# YuE2Pipeline.__init__ always calls set_per_process_memory_fraction; skip it on this 16GB host.
|
||||
_set_fraction = torch.cuda.set_per_process_memory_fraction
|
||||
torch.cuda.set_per_process_memory_fraction = lambda *args, **kwargs: None
|
||||
try:
|
||||
pipe_cm = YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load)
|
||||
finally:
|
||||
torch.cuda.set_per_process_memory_fraction = _set_fraction
|
||||
with pipe_cm as pipe:
|
||||
memory_snapshot(torch, 'memory-loaded')
|
||||
emit(stage='plan', message='Planning melody and chords', progress=5, cot=cot)
|
||||
try:
|
||||
|
||||
Reference in New Issue
Block a user