""" STT 추론 장치 선택 정책 (순수 로직, 외부 IO 없음). main.py(FastAPI 어댑터)가 ctranslate2/faster-whisper 호출을 주입하고, 이 모듈은 "어떤 장치·정밀도로 올릴지"와 "언제 CPU로 내려갈지"만 결정한다. 배경: ctranslate2 는 CUDA 런타임을 정적 링크하므로 NVIDIA 드라이버만 있어도 get_cuda_device_count() 가 1 이상을 돌려준다. 그런데 cuBLAS(cublas64_12.dll)가 없으면 모델 생성은 성공하고 첫 전사에서야 RuntimeError 가 난다. Pascal(CC 6.x) GPU 는 float16 을 지원하지 않아 모델 생성부터 ValueError 가 난다. 그래서 1) 지원 compute type 중 쓸 수 있는 것을 고르고, 2) GPU 로 만든 직후 짧은 무음 전사로 실제 동작을 검증하고, 3) 실패하면 CPU(int8)로 다시 만든다. """ from __future__ import annotations from dataclasses import dataclass from typing import Callable, Iterable, Optional, TypeVar M = TypeVar("M") @dataclass(frozen=True) class DeviceChoice: """모델을 올릴 장치와 연산 정밀도.""" device: str compute_type: str @property def is_gpu(self) -> bool: return self.device == "cuda" CPU_CHOICE = DeviceChoice(device="cpu", compute_type="int8") # 선호 순서. float16 을 못 쓰는 GPU(Pascal 등)는 int8_float32 → float32 로 내려간다. _CUDA_COMPUTE_PREFERENCE: tuple[str, ...] = ("float16", "int8_float32", "float32") # GPU 런타임/라이브러리 문제로 판단할 오류 메시지 조각 (소문자 비교). _GPU_ERROR_MARKERS: tuple[str, ...] = ( "cublas", "cudnn", "cuda", "cufft", "nvrtc", "float16 compute", "efficient float16", ) def choose_cuda_compute_type(supported: Iterable[str]) -> Optional[str]: """GPU 가 지원하는 compute type 중 선호 순서대로 첫 번째를 고른다. 없으면 None.""" supported_set = {s.lower() for s in supported} for candidate in _CUDA_COMPUTE_PREFERENCE: if candidate in supported_set: return candidate return None def choose_gpu_device(cuda_count: int, supported: Iterable[str]) -> Optional[DeviceChoice]: """CUDA 장치 수와 지원 compute type 으로 GPU 후보를 정한다. 쓸 수 없으면 None.""" if cuda_count <= 0: return None compute_type = choose_cuda_compute_type(supported) if compute_type is None: return None return DeviceChoice(device="cuda", compute_type=compute_type) def is_gpu_runtime_error(exc: BaseException) -> bool: """CUDA 라이브러리 로드 실패·미지원 정밀도 등 CPU 로 내려가면 풀리는 오류인가.""" if not isinstance(exc, (RuntimeError, ValueError, OSError)): return False message = str(exc).lower() return any(marker in message for marker in _GPU_ERROR_MARKERS) def load_with_fallback( build: Callable[[DeviceChoice], M], verify: Callable[[M], None], preferred: Optional[DeviceChoice], on_fallback: Optional[Callable[[DeviceChoice, BaseException], None]] = None, ) -> tuple[M, DeviceChoice]: """preferred(GPU)로 만들어 검증하고, 실패하면 CPU 로 다시 만든다. - GPU 생성 실패: GPU 관련 오류일 때만 CPU 로 내려간다. 그 밖의 오류(모델 파일 다운로드 실패 등)는 CPU 로 다시 해도 똑같이 실패하므로 그대로 올린다. - GPU 검증 실패: 모델은 만들어졌으므로 장치 문제다. 오류 종류와 무관하게 CPU 로 간다. - CPU 후보이거나 후보가 없으면 검증 없이 CPU 로 만든다. """ if preferred is not None and preferred.is_gpu: try: model = build(preferred) except Exception as exc: if not is_gpu_runtime_error(exc): raise if on_fallback is not None: on_fallback(preferred, exc) else: try: verify(model) return model, preferred except Exception as exc: # 검증에 실패한 GPU 모델의 참조를 먼저 끊어 VRAM 을 돌려받는다. del model if on_fallback is not None: on_fallback(preferred, exc) return build(CPU_CHOICE), CPU_CHOICE