fix(sidecar): fall back to CPU when the CUDA runtime is unusable
This commit is contained in:
parent
96b24e279c
commit
4588b65dfa
3 changed files with 605 additions and 37 deletions
110
apps/desktop/sidecar/device_policy.py
Normal file
110
apps/desktop/sidecar/device_policy.py
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
"""
|
||||
STT 추론 장치 선택 정책 (순수 로직, 외부 IO 없음).
|
||||
|
||||
main.py(FastAPI 어댑터)가 ctranslate2/faster-whisper 호출을 주입하고, 이 모듈은
|
||||
"어떤 장치·정밀도로 올릴지"와 "언제 CPU로 내려갈지"만 결정한다.
|
||||
|
||||
배경: ctranslate2 는 CUDA 런타임을 정적 링크하므로 NVIDIA 드라이버만 있어도
|
||||
get_cuda_device_count() 가 1 이상을 돌려준다. 그런데 cuBLAS(cublas64_12.dll)가
|
||||
없으면 모델 생성은 성공하고 첫 전사에서야 RuntimeError 가 난다. Pascal(CC 6.x)
|
||||
GPU 는 float16 을 지원하지 않아 모델 생성부터 ValueError 가 난다. 그래서
|
||||
1) 지원 compute type 중 쓸 수 있는 것을 고르고,
|
||||
2) GPU 로 만든 직후 짧은 무음 전사로 실제 동작을 검증하고,
|
||||
3) 실패하면 CPU(int8)로 다시 만든다.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Iterable, Optional, TypeVar
|
||||
|
||||
M = TypeVar("M")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DeviceChoice:
|
||||
"""모델을 올릴 장치와 연산 정밀도."""
|
||||
|
||||
device: str
|
||||
compute_type: str
|
||||
|
||||
@property
|
||||
def is_gpu(self) -> bool:
|
||||
return self.device == "cuda"
|
||||
|
||||
|
||||
CPU_CHOICE = DeviceChoice(device="cpu", compute_type="int8")
|
||||
|
||||
# 선호 순서. float16 을 못 쓰는 GPU(Pascal 등)는 int8_float32 → float32 로 내려간다.
|
||||
_CUDA_COMPUTE_PREFERENCE: tuple[str, ...] = ("float16", "int8_float32", "float32")
|
||||
|
||||
# GPU 런타임/라이브러리 문제로 판단할 오류 메시지 조각 (소문자 비교).
|
||||
_GPU_ERROR_MARKERS: tuple[str, ...] = (
|
||||
"cublas",
|
||||
"cudnn",
|
||||
"cuda",
|
||||
"cufft",
|
||||
"nvrtc",
|
||||
"float16 compute",
|
||||
"efficient float16",
|
||||
)
|
||||
|
||||
|
||||
def choose_cuda_compute_type(supported: Iterable[str]) -> Optional[str]:
|
||||
"""GPU 가 지원하는 compute type 중 선호 순서대로 첫 번째를 고른다. 없으면 None."""
|
||||
supported_set = {s.lower() for s in supported}
|
||||
for candidate in _CUDA_COMPUTE_PREFERENCE:
|
||||
if candidate in supported_set:
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def choose_gpu_device(cuda_count: int, supported: Iterable[str]) -> Optional[DeviceChoice]:
|
||||
"""CUDA 장치 수와 지원 compute type 으로 GPU 후보를 정한다. 쓸 수 없으면 None."""
|
||||
if cuda_count <= 0:
|
||||
return None
|
||||
compute_type = choose_cuda_compute_type(supported)
|
||||
if compute_type is None:
|
||||
return None
|
||||
return DeviceChoice(device="cuda", compute_type=compute_type)
|
||||
|
||||
|
||||
def is_gpu_runtime_error(exc: BaseException) -> bool:
|
||||
"""CUDA 라이브러리 로드 실패·미지원 정밀도 등 CPU 로 내려가면 풀리는 오류인가."""
|
||||
if not isinstance(exc, (RuntimeError, ValueError, OSError)):
|
||||
return False
|
||||
message = str(exc).lower()
|
||||
return any(marker in message for marker in _GPU_ERROR_MARKERS)
|
||||
|
||||
|
||||
def load_with_fallback(
|
||||
build: Callable[[DeviceChoice], M],
|
||||
verify: Callable[[M], None],
|
||||
preferred: Optional[DeviceChoice],
|
||||
on_fallback: Optional[Callable[[DeviceChoice, BaseException], None]] = None,
|
||||
) -> tuple[M, DeviceChoice]:
|
||||
"""preferred(GPU)로 만들어 검증하고, 실패하면 CPU 로 다시 만든다.
|
||||
|
||||
- GPU 생성 실패: GPU 관련 오류일 때만 CPU 로 내려간다. 그 밖의 오류(모델 파일
|
||||
다운로드 실패 등)는 CPU 로 다시 해도 똑같이 실패하므로 그대로 올린다.
|
||||
- GPU 검증 실패: 모델은 만들어졌으므로 장치 문제다. 오류 종류와 무관하게 CPU 로 간다.
|
||||
- CPU 후보이거나 후보가 없으면 검증 없이 CPU 로 만든다.
|
||||
"""
|
||||
if preferred is not None and preferred.is_gpu:
|
||||
try:
|
||||
model = build(preferred)
|
||||
except Exception as exc:
|
||||
if not is_gpu_runtime_error(exc):
|
||||
raise
|
||||
if on_fallback is not None:
|
||||
on_fallback(preferred, exc)
|
||||
else:
|
||||
try:
|
||||
verify(model)
|
||||
return model, preferred
|
||||
except Exception as exc:
|
||||
# 검증에 실패한 GPU 모델의 참조를 먼저 끊어 VRAM 을 돌려받는다.
|
||||
del model
|
||||
if on_fallback is not None:
|
||||
on_fallback(preferred, exc)
|
||||
return build(CPU_CHOICE), CPU_CHOICE
|
||||
Loading…
Add table
Add a link
Reference in a new issue