Remove YuE2 process memory fraction cap on the 5080.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -119,18 +119,24 @@ def main():
|
|||||||
patch_graph_attention(attention_backend)
|
patch_graph_attention(attention_backend)
|
||||||
free_cuda(torch)
|
free_cuda(torch)
|
||||||
memory_snapshot(torch, 'memory-before-load')
|
memory_snapshot(torch, 'memory-before-load')
|
||||||
# 16GB card: cap budget so YuE2 leaves headroom; tiled VAE at 512 frames.
|
# Prefer tiled VAE; do not cap the process with set_per_process_memory_fraction.
|
||||||
pipe_load = dict(
|
pipe_load = dict(
|
||||||
device='cuda',
|
device='cuda',
|
||||||
backend=pipeline_backend,
|
backend=pipeline_backend,
|
||||||
memory_budget_gib=12,
|
|
||||||
vae_core_frames=512,
|
vae_core_frames=512,
|
||||||
offload_ar=True,
|
offload_ar=True,
|
||||||
progress=False,
|
progress=False,
|
||||||
)
|
)
|
||||||
cot = 'full'
|
cot = 'full'
|
||||||
pipe_kwargs = dict(style=style, lyrics=lyrics, cot=cot, seed=seed)
|
pipe_kwargs = dict(style=style, lyrics=lyrics, cot=cot, seed=seed)
|
||||||
with YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load) as pipe:
|
# YuE2Pipeline.__init__ always calls set_per_process_memory_fraction; skip it on this 16GB host.
|
||||||
|
_set_fraction = torch.cuda.set_per_process_memory_fraction
|
||||||
|
torch.cuda.set_per_process_memory_fraction = lambda *args, **kwargs: None
|
||||||
|
try:
|
||||||
|
pipe_cm = YuE2Pipeline.from_pretrained(model, vae=vae, **pipe_load)
|
||||||
|
finally:
|
||||||
|
torch.cuda.set_per_process_memory_fraction = _set_fraction
|
||||||
|
with pipe_cm as pipe:
|
||||||
memory_snapshot(torch, 'memory-loaded')
|
memory_snapshot(torch, 'memory-loaded')
|
||||||
emit(stage='plan', message='Planning melody and chords', progress=5, cot=cot)
|
emit(stage='plan', message='Planning melody and chords', progress=5, cot=cot)
|
||||||
try:
|
try:
|
||||||
|
|||||||
Reference in New Issue
Block a user