Avoid YuEGP pinned-memory exhaustion on 32GB Windows hosts

This commit is contained in:
Towsty
2026-09-08 07:04:13 -05:00
parent 7665285a0d
commit 4c97980646
3 changed files with 35 additions and 2 deletions
+4
View File
@@ -101,3 +101,7 @@ Upstream reference: [deepbeepmeep/YuEGP](https://github.com/deepbeepmeep/YuEGP/t
- The earlier test failed during decoder construction: MMGP sets the default PyTorch device to CUDA. The adapter now defers decoder creation until the transformer stages finish, releases their VRAM, and constructs the decoder in an explicit CPU context. It also writes WAV directly instead of depending on a Windows MP3 backend. - The earlier test failed during decoder construction: MMGP sets the default PyTorch device to CUDA. The adapter now defers decoder creation until the transformer stages finish, releases their VRAM, and constructs the decoder in an explicit CPU context. It also writes WAV directly instead of depending on a Windows MP3 backend.
- No generation test was run after that fix, at the user's request. Successful song generation, peak VRAM, and wall time remain unverified. - No generation test was run after that fix, at the user's request. Successful song generation, peak VRAM, and wall time remain unverified.
- This isolated environment currently has no FlashAttention or Triton. It uses upstream SDPA with compilation off. The optimized FlashAttention cache path in YuEGP is therefore not active; profile 1 remains selected and unquantized, but advertised four-minute timings must not be assumed for this installation. - This isolated environment currently has no FlashAttention or Triton. It uses upstream SDPA with compilation off. The optimized FlashAttention cache path in YuEGP is therefore not active; profile 1 remains selected and unquantized, but advertised four-minute timings must not be assumed for this installation.
## Windows startup memory correction
The later profile-1 run reached Stage 1 but failed before its first forward pass. MMGP logged a failed pinned-host-memory allocation after reserving about 13GB, then CUDA reported OOM in special-token preparation. On Windows hosts with less than 48GB system RAM, the adapter now passes `pinnedMemory=False` to `offload.profile`. Profile 1, unquantized weights, default VRAM budgets, 60s, and compile-off remain unchanged. This disables transfer-staging RAM pinning, not profile 1, and does not switch to profile 2 or 3. CUDA synchronization and free/allocated GPU and available RAM diagnostics now bracket initialization and Stage 1 completion. No generation was run to validate this correction; subsequent inference may still encounter a separate VRAM limit.
+21 -2
View File
@@ -61,6 +61,16 @@ def load_codec_on_cpu(ns, torch):
return codec return codec
def memory_profile_options(profile, compile_enabled, platform, total_ram):
options = dict(profile_no=profile, quantizeTransformer=profile == 3,
compile=compile_enabled, verboseLevel=1)
# Windows CUDA pinned allocations can fail before inference on 32GB hosts.
# This affects transfer staging only, not weight precision or VRAM budgets.
if platform == 'win32' and total_ram < 48 * 1024 ** 3:
options['pinnedMemory'] = False
return options
def main(): def main():
import psutil import psutil
parent = psutil.Process(os.getppid()) parent = psutil.Process(os.getppid())
@@ -126,8 +136,16 @@ def main():
model2.generation_config.cache_implementation = None model2.generation_config.cache_implementation = None
model._validate_model_kwargs = lambda _: None model._validate_model_kwargs = lambda _: None
model2._validate_model_kwargs = lambda _: None model2._validate_model_kwargs = lambda _: None
offloader = ns['offload'].profile({'transformer': model, 'stage2': model2}, profile_no=cli.profile, profile_options = memory_profile_options(cli.profile, compile_enabled, sys.platform, psutil.virtual_memory().total)
quantizeTransformer=cli.profile == 3, compile=compile_enabled, verboseLevel=1) emit(stage='loading', message='Preparing YuEGP memory', profileOptions=profile_options)
offloader = ns['offload'].profile({'transformer': model, 'stage2': model2}, **profile_options)
def memory_checkpoint(stage):
torch.cuda.synchronize()
free, total = torch.cuda.mem_get_info()
emit(stage=stage, message=f'YuEGP {stage}', gpuFreeMiB=round(free / 1024 ** 2),
gpuTotalMiB=round(total / 1024 ** 2), gpuAllocatedMiB=round(torch.cuda.memory_allocated() / 1024 ** 2),
ramAvailableMiB=round(psutil.virtual_memory().available / 1024 ** 2))
memory_checkpoint('memory-ready')
args = SimpleNamespace(use_audio_prompt=False, use_dual_tracks_prompt=False, rescale=True, args = SimpleNamespace(use_audio_prompt=False, use_dual_tracks_prompt=False, rescale=True,
output_dir=str(output), cuda_idx=0) output_dir=str(output), cuda_idx=0)
ns.update(model=model, model_stage2=model2, device=device, codec_model=None, args=args, ns.update(model=model, model_stage2=model2, device=device, codec_model=None, args=args,
@@ -150,6 +168,7 @@ def main():
progress=round(100 * done / max(1, total), 1)) progress=round(100 * done / max(1, total), 1))
emit(stage='stage1', message='Generating song tokens', progress=0) emit(stage='stage1', message='Generating song tokens', progress=0)
stems = ns['stage1_inference'](tags, lyrics, 1, max_tokens, seed, state, callback) stems = ns['stage1_inference'](tags, lyrics, 1, max_tokens, seed, state, callback)
memory_checkpoint('stage1-finished')
emit(stage='stage2', message='Generating audio detail', progress=0) emit(stage='stage2', message='Generating audio detail', progress=0)
results = ns['stage2_inference'](model2, stems, str(output / 'stage2'), results = ns['stage2_inference'](model2, stems, str(output / 'stage2'),
batch_size=20 if cli.profile == 1 else 4, state=state, callback=callback) batch_size=20 if cli.profile == 1 else 4, state=state, callback=callback)
+10
View File
@@ -11,6 +11,16 @@ spec.loader.exec_module(worker)
class WorkerTests(unittest.TestCase): class WorkerTests(unittest.TestCase):
def test_windows_memory_fix_preserves_profile_and_precision(self):
options = worker.memory_profile_options(1, False, 'win32', 32 * 1024 ** 3)
self.assertFalse(options['pinnedMemory'])
self.assertEqual(options['profile_no'], 1)
self.assertFalse(options['quantizeTransformer'])
self.assertFalse(options['compile'])
self.assertNotIn('budgets', options)
self.assertNotIn('pinnedMemory', worker.memory_profile_options(1, False, 'linux', 32 * 1024 ** 3))
self.assertTrue(worker.memory_profile_options(3, False, 'win32', 32 * 1024 ** 3)['quantizeTransformer'])
def test_lyrics_preserve_words_and_normalize_ui_headings(self): def test_lyrics_preserve_words_and_normalize_ui_headings(self):
self.assertEqual(worker.normalize_lyrics('[Pre-Chorus]\nEvery word stays\n[Outro]\n'), '[prechorus]\nEvery word stays\n\n') self.assertEqual(worker.normalize_lyrics('[Pre-Chorus]\nEvery word stays\n[Outro]\n'), '[prechorus]\nEvery word stays\n\n')
self.assertEqual(worker.normalize_lyrics('[Verse 1]\nHello'), '[verse1]\nHello\n\n') self.assertEqual(worker.normalize_lyrics('[Verse 1]\nHello'), '[verse1]\nHello\n\n')