From 4c97980646ebce49239b4b4ebce93ccd79a9d701 Mon Sep 17 00:00:00 2001 From: Towsty Date: Tue, 8 Sep 2026 07:04:13 -0500 Subject: [PATCH] Avoid YuEGP pinned-memory exhaustion on 32GB Windows hosts --- docs/yuegp-replacement.md | 4 ++++ scripts/yuegp-worker.py | 23 +++++++++++++++++++++-- tests/test_yuegp_worker.py | 10 ++++++++++ 3 files changed, 35 insertions(+), 2 deletions(-) diff --git a/docs/yuegp-replacement.md b/docs/yuegp-replacement.md index c4dc516..2625bdd 100644 --- a/docs/yuegp-replacement.md +++ b/docs/yuegp-replacement.md @@ -101,3 +101,7 @@ Upstream reference: [deepbeepmeep/YuEGP](https://github.com/deepbeepmeep/YuEGP/t - The earlier test failed during decoder construction: MMGP sets the default PyTorch device to CUDA. The adapter now defers decoder creation until the transformer stages finish, releases their VRAM, and constructs the decoder in an explicit CPU context. It also writes WAV directly instead of depending on a Windows MP3 backend. - No generation test was run after that fix, at the user's request. Successful song generation, peak VRAM, and wall time remain unverified. - This isolated environment currently has no FlashAttention or Triton. It uses upstream SDPA with compilation off. The optimized FlashAttention cache path in YuEGP is therefore not active; profile 1 remains selected and unquantized, but advertised four-minute timings must not be assumed for this installation. + +## Windows startup memory correction + +The later profile-1 run reached Stage 1 but failed before its first forward pass. MMGP logged a failed pinned-host-memory allocation after reserving about 13GB, then CUDA reported OOM in special-token preparation. On Windows hosts with less than 48GB system RAM, the adapter now passes `pinnedMemory=False` to `offload.profile`. Profile 1, unquantized weights, default VRAM budgets, 60s, and compile-off remain unchanged. This disables transfer-staging RAM pinning, not profile 1, and does not switch to profile 2 or 3. CUDA synchronization and free/allocated GPU and available RAM diagnostics now bracket initialization and Stage 1 completion. No generation was run to validate this correction; subsequent inference may still encounter a separate VRAM limit. diff --git a/scripts/yuegp-worker.py b/scripts/yuegp-worker.py index feedaf9..e7e0b76 100644 --- a/scripts/yuegp-worker.py +++ b/scripts/yuegp-worker.py @@ -61,6 +61,16 @@ def load_codec_on_cpu(ns, torch): return codec +def memory_profile_options(profile, compile_enabled, platform, total_ram): + options = dict(profile_no=profile, quantizeTransformer=profile == 3, + compile=compile_enabled, verboseLevel=1) + # Windows CUDA pinned allocations can fail before inference on 32GB hosts. + # This affects transfer staging only, not weight precision or VRAM budgets. + if platform == 'win32' and total_ram < 48 * 1024 ** 3: + options['pinnedMemory'] = False + return options + + def main(): import psutil parent = psutil.Process(os.getppid()) @@ -126,8 +136,16 @@ def main(): model2.generation_config.cache_implementation = None model._validate_model_kwargs = lambda _: None model2._validate_model_kwargs = lambda _: None - offloader = ns['offload'].profile({'transformer': model, 'stage2': model2}, profile_no=cli.profile, - quantizeTransformer=cli.profile == 3, compile=compile_enabled, verboseLevel=1) + profile_options = memory_profile_options(cli.profile, compile_enabled, sys.platform, psutil.virtual_memory().total) + emit(stage='loading', message='Preparing YuEGP memory', profileOptions=profile_options) + offloader = ns['offload'].profile({'transformer': model, 'stage2': model2}, **profile_options) + def memory_checkpoint(stage): + torch.cuda.synchronize() + free, total = torch.cuda.mem_get_info() + emit(stage=stage, message=f'YuEGP {stage}', gpuFreeMiB=round(free / 1024 ** 2), + gpuTotalMiB=round(total / 1024 ** 2), gpuAllocatedMiB=round(torch.cuda.memory_allocated() / 1024 ** 2), + ramAvailableMiB=round(psutil.virtual_memory().available / 1024 ** 2)) + memory_checkpoint('memory-ready') args = SimpleNamespace(use_audio_prompt=False, use_dual_tracks_prompt=False, rescale=True, output_dir=str(output), cuda_idx=0) ns.update(model=model, model_stage2=model2, device=device, codec_model=None, args=args, @@ -150,6 +168,7 @@ def main(): progress=round(100 * done / max(1, total), 1)) emit(stage='stage1', message='Generating song tokens', progress=0) stems = ns['stage1_inference'](tags, lyrics, 1, max_tokens, seed, state, callback) + memory_checkpoint('stage1-finished') emit(stage='stage2', message='Generating audio detail', progress=0) results = ns['stage2_inference'](model2, stems, str(output / 'stage2'), batch_size=20 if cli.profile == 1 else 4, state=state, callback=callback) diff --git a/tests/test_yuegp_worker.py b/tests/test_yuegp_worker.py index 58f0637..04a77b8 100644 --- a/tests/test_yuegp_worker.py +++ b/tests/test_yuegp_worker.py @@ -11,6 +11,16 @@ spec.loader.exec_module(worker) class WorkerTests(unittest.TestCase): + def test_windows_memory_fix_preserves_profile_and_precision(self): + options = worker.memory_profile_options(1, False, 'win32', 32 * 1024 ** 3) + self.assertFalse(options['pinnedMemory']) + self.assertEqual(options['profile_no'], 1) + self.assertFalse(options['quantizeTransformer']) + self.assertFalse(options['compile']) + self.assertNotIn('budgets', options) + self.assertNotIn('pinnedMemory', worker.memory_profile_options(1, False, 'linux', 32 * 1024 ** 3)) + self.assertTrue(worker.memory_profile_options(3, False, 'win32', 32 * 1024 ** 3)['quantizeTransformer']) + def test_lyrics_preserve_words_and_normalize_ui_headings(self): self.assertEqual(worker.normalize_lyrics('[Pre-Chorus]\nEvery word stays\n[Outro]\n'), '[prechorus]\nEvery word stays\n\n') self.assertEqual(worker.normalize_lyrics('[Verse 1]\nHello'), '[verse1]\nHello\n\n')