Tighten YuE2 VRAM use on the 5080 and empty GPU before launch.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Towsty
2026-09-14 20:01:06 -05:00
co-authored by Cursor
parent 62671d05da
commit ee84a99035
3 changed files with 89 additions and 37 deletions
+7 -5
View File
@@ -1,5 +1,6 @@
"""CPU-only YuE2 worker regression tests; never import torch or run inference."""
import importlib.util
import os
from pathlib import Path
import unittest
from types import SimpleNamespace
@@ -26,13 +27,14 @@ class WorkerTests(unittest.TestCase):
backends=SimpleNamespace(cudnn=SimpleNamespace(is_available=lambda: True)),
)
attention, pipeline = worker.resolve_attention_backend(torch_mod)
self.assertEqual(attention, 'cudnn')
self.assertEqual(pipeline, 'torch')
self.assertNotEqual(attention, 'flash')
torch_mod.backends.cudnn.is_available = lambda: False
attention, pipeline = worker.resolve_attention_backend(torch_mod)
self.assertEqual(attention, 'sdpa')
self.assertEqual(pipeline, 'torch-eager')
self.assertNotEqual(attention, 'flash')
self.assertTrue(os.environ.get('PYTORCH_CUDA_ALLOC_CONF', '').startswith('expandable_segments'))
def test_cuda_oom_detection(self):
self.assertTrue(worker.is_cuda_oom(RuntimeError('CUDA out of memory. Tried to allocate 2.49 GiB')))
self.assertFalse(worker.is_cuda_oom(RuntimeError('bad lyrics')))
if __name__ == '__main__':