fix(sidecar): fall back to CPU when the CUDA runtime is unusable

This commit is contained in:
Yun Chan 2026-09-28 02:16:15 +09:00
parent 96b24e279c
commit 4588b65dfa
3 changed files with 605 additions and 37 deletions

View file

@ -0,0 +1,110 @@
"""
STT 추론 장치 선택 정책 (순수 로직, 외부 IO 없음).
main.py(FastAPI 어댑터)가 ctranslate2/faster-whisper 호출을 주입하고, 이 모듈은
"어떤 장치·정밀도로 올릴지"와 "언제 CPU로 내려갈지"만 결정한다.
배경: ctranslate2 는 CUDA 런타임을 정적 링크하므로 NVIDIA 드라이버만 있어도
get_cuda_device_count() 가 1 이상을 돌려준다. 그런데 cuBLAS(cublas64_12.dll)가
없으면 모델 생성은 성공하고 첫 전사에서야 RuntimeError 가 난다. Pascal(CC 6.x)
GPU 는 float16 을 지원하지 않아 모델 생성부터 ValueError 가 난다. 그래서
1) 지원 compute type 중 쓸 수 있는 것을 고르고,
2) GPU 로 만든 직후 짧은 무음 전사로 실제 동작을 검증하고,
3) 실패하면 CPU(int8)로 다시 만든다.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Callable, Iterable, Optional, TypeVar
M = TypeVar("M")
@dataclass(frozen=True)
class DeviceChoice:
"""모델을 올릴 장치와 연산 정밀도."""
device: str
compute_type: str
@property
def is_gpu(self) -> bool:
return self.device == "cuda"
CPU_CHOICE = DeviceChoice(device="cpu", compute_type="int8")
# 선호 순서. float16 을 못 쓰는 GPU(Pascal 등)는 int8_float32 → float32 로 내려간다.
_CUDA_COMPUTE_PREFERENCE: tuple[str, ...] = ("float16", "int8_float32", "float32")
# GPU 런타임/라이브러리 문제로 판단할 오류 메시지 조각 (소문자 비교).
_GPU_ERROR_MARKERS: tuple[str, ...] = (
"cublas",
"cudnn",
"cuda",
"cufft",
"nvrtc",
"float16 compute",
"efficient float16",
)
def choose_cuda_compute_type(supported: Iterable[str]) -> Optional[str]:
"""GPU 가 지원하는 compute type 중 선호 순서대로 첫 번째를 고른다. 없으면 None."""
supported_set = {s.lower() for s in supported}
for candidate in _CUDA_COMPUTE_PREFERENCE:
if candidate in supported_set:
return candidate
return None
def choose_gpu_device(cuda_count: int, supported: Iterable[str]) -> Optional[DeviceChoice]:
"""CUDA 장치 수와 지원 compute type 으로 GPU 후보를 정한다. 쓸 수 없으면 None."""
if cuda_count <= 0:
return None
compute_type = choose_cuda_compute_type(supported)
if compute_type is None:
return None
return DeviceChoice(device="cuda", compute_type=compute_type)
def is_gpu_runtime_error(exc: BaseException) -> bool:
"""CUDA 라이브러리 로드 실패·미지원 정밀도 등 CPU 로 내려가면 풀리는 오류인가."""
if not isinstance(exc, (RuntimeError, ValueError, OSError)):
return False
message = str(exc).lower()
return any(marker in message for marker in _GPU_ERROR_MARKERS)
def load_with_fallback(
build: Callable[[DeviceChoice], M],
verify: Callable[[M], None],
preferred: Optional[DeviceChoice],
on_fallback: Optional[Callable[[DeviceChoice, BaseException], None]] = None,
) -> tuple[M, DeviceChoice]:
"""preferred(GPU)로 만들어 검증하고, 실패하면 CPU 로 다시 만든다.
- GPU 생성 실패: GPU 관련 오류일 때만 CPU 로 내려간다. 그 밖의 오류(모델 파일
다운로드 실패 등)는 CPU 로 다시 해도 똑같이 실패하므로 그대로 올린다.
- GPU 검증 실패: 모델은 만들어졌으므로 장치 문제다. 오류 종류와 무관하게 CPU 로 간다.
- CPU 후보이거나 후보가 없으면 검증 없이 CPU 로 만든다.
"""
if preferred is not None and preferred.is_gpu:
try:
model = build(preferred)
except Exception as exc:
if not is_gpu_runtime_error(exc):
raise
if on_fallback is not None:
on_fallback(preferred, exc)
else:
try:
verify(model)
return model, preferred
except Exception as exc:
# 검증에 실패한 GPU 모델의 참조를 먼저 끊어 VRAM 을 돌려받는다.
del model
if on_fallback is not None:
on_fallback(preferred, exc)
return build(CPU_CHOICE), CPU_CHOICE