fix(diffusers): auto-detect CUDA instead of defaulting to CPU (#11891)

The device fell back to CPU unless the model config set cuda: true,
while MPS right below was auto-detected — GPU hosts silently rendered
on CPU for any gallery entry missing the flag. Use CUDA whenever torch
reports it available (ROCm builds included), keep cuda: true as an
explicit force, and allow pinning with the device: model option (e.g.
options: ["device:cpu"]). Gallery entries stay untouched.

Assisted-by: Claude:claude-fable-5

Signed-off-by: Plamen K. Kosseff <p.kosseff@gmail.com>
This commit is contained in:
Plamen K. Kosseff authored and GitHub committed 2026-09-08 08:44:35 +02:00
1 parent e0179a1d3b
commit ba88fb13ce
3 files changed
+43 -7

No files matched your search

+22 -6
View File
@@ -122,6 +122,21 @@ from diffusers.schedulers import (
UniPCMultistepScheduler,
)
def select_device(request_cuda, device_option, cuda_available, xpu, mps_available):
"""Pick the pipeline device. An explicit `device:` model option wins;
otherwise CUDA is used whenever torch reports it available (ROCm
builds included) or the model config forces it with `cuda: true`,
keeping the pre-existing XPU/MPS overrides. CPU is the fallback, not
the default."""
if device_option:
return device_option
device = "cuda" if (request_cuda or cuda_available) else "cpu"
if xpu:
device = "xpu"
if mps_available:
device = "mps"
return device
def is_float(s):
"""Check if a string can be converted to float."""
try:
@@ -627,12 +642,13 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
# modify LoraAdapter to be relative to modelFileBase
request.LoraAdapter = os.path.join(request.ModelPath, request.LoraAdapter)
device = "cpu" if not request.CUDA else "cuda"
if XPU:
device = "xpu"
mps_available = hasattr(torch.backends, "mps") and torch.backends.mps.is_available()
if mps_available:
device = "mps"
device = select_device(
request.CUDA,
self.options.pop("device", None),
torch.cuda.is_available(),
XPU,
hasattr(torch.backends, "mps") and torch.backends.mps.is_available(),
)
self.device = device
if request.LoraAdapter:
# Check if its a local file and not a directory ( we load lora differently for a safetensor file )
+20
View File
@@ -7,6 +7,7 @@ import time
from unittest.mock import patch, MagicMock
# Import dynamic loader for testing (these don't need gRPC)
import backend
import diffusers_dynamic_loader as loader
from diffusers import DiffusionPipeline, StableDiffusionPipeline
@@ -425,3 +426,22 @@ class TestGenerateImageOptionsKwargsMerge(unittest.TestCase):
self.assertEqual(pipeline.kwargs["num_inference_steps"], 4)
finally:
os.unlink(dst_path)
class TestDeviceSelection(unittest.TestCase):
"""Unit tests for backend.select_device (no GPU required)."""
def test_autodetect_cuda(self):
self.assertEqual(backend.select_device(False, None, True, False, False), "cuda")
def test_cpu_fallback(self):
self.assertEqual(backend.select_device(False, None, False, False, False), "cpu")
def test_forced_cuda(self):
self.assertEqual(backend.select_device(True, None, False, False, False), "cuda")
def test_device_option_wins(self):
self.assertEqual(backend.select_device(True, "cpu", True, True, True), "cpu")
def test_mps_overrides(self):
self.assertEqual(backend.select_device(False, None, True, False, True), "mps")