110 lines
4.2 KiB
Python
110 lines
4.2 KiB
Python
"""
|
|
STT 추론 장치 선택 정책 (순수 로직, 외부 IO 없음).
|
|
|
|
main.py(FastAPI 어댑터)가 ctranslate2/faster-whisper 호출을 주입하고, 이 모듈은
|
|
"어떤 장치·정밀도로 올릴지"와 "언제 CPU로 내려갈지"만 결정한다.
|
|
|
|
배경: ctranslate2 는 CUDA 런타임을 정적 링크하므로 NVIDIA 드라이버만 있어도
|
|
get_cuda_device_count() 가 1 이상을 돌려준다. 그런데 cuBLAS(cublas64_12.dll)가
|
|
없으면 모델 생성은 성공하고 첫 전사에서야 RuntimeError 가 난다. Pascal(CC 6.x)
|
|
GPU 는 float16 을 지원하지 않아 모델 생성부터 ValueError 가 난다. 그래서
|
|
1) 지원 compute type 중 쓸 수 있는 것을 고르고,
|
|
2) GPU 로 만든 직후 짧은 무음 전사로 실제 동작을 검증하고,
|
|
3) 실패하면 CPU(int8)로 다시 만든다.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from typing import Callable, Iterable, Optional, TypeVar
|
|
|
|
M = TypeVar("M")
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class DeviceChoice:
|
|
"""모델을 올릴 장치와 연산 정밀도."""
|
|
|
|
device: str
|
|
compute_type: str
|
|
|
|
@property
|
|
def is_gpu(self) -> bool:
|
|
return self.device == "cuda"
|
|
|
|
|
|
CPU_CHOICE = DeviceChoice(device="cpu", compute_type="int8")
|
|
|
|
# 선호 순서. float16 을 못 쓰는 GPU(Pascal 등)는 int8_float32 → float32 로 내려간다.
|
|
_CUDA_COMPUTE_PREFERENCE: tuple[str, ...] = ("float16", "int8_float32", "float32")
|
|
|
|
# GPU 런타임/라이브러리 문제로 판단할 오류 메시지 조각 (소문자 비교).
|
|
_GPU_ERROR_MARKERS: tuple[str, ...] = (
|
|
"cublas",
|
|
"cudnn",
|
|
"cuda",
|
|
"cufft",
|
|
"nvrtc",
|
|
"float16 compute",
|
|
"efficient float16",
|
|
)
|
|
|
|
|
|
def choose_cuda_compute_type(supported: Iterable[str]) -> Optional[str]:
|
|
"""GPU 가 지원하는 compute type 중 선호 순서대로 첫 번째를 고른다. 없으면 None."""
|
|
supported_set = {s.lower() for s in supported}
|
|
for candidate in _CUDA_COMPUTE_PREFERENCE:
|
|
if candidate in supported_set:
|
|
return candidate
|
|
return None
|
|
|
|
|
|
def choose_gpu_device(cuda_count: int, supported: Iterable[str]) -> Optional[DeviceChoice]:
|
|
"""CUDA 장치 수와 지원 compute type 으로 GPU 후보를 정한다. 쓸 수 없으면 None."""
|
|
if cuda_count <= 0:
|
|
return None
|
|
compute_type = choose_cuda_compute_type(supported)
|
|
if compute_type is None:
|
|
return None
|
|
return DeviceChoice(device="cuda", compute_type=compute_type)
|
|
|
|
|
|
def is_gpu_runtime_error(exc: BaseException) -> bool:
|
|
"""CUDA 라이브러리 로드 실패·미지원 정밀도 등 CPU 로 내려가면 풀리는 오류인가."""
|
|
if not isinstance(exc, (RuntimeError, ValueError, OSError)):
|
|
return False
|
|
message = str(exc).lower()
|
|
return any(marker in message for marker in _GPU_ERROR_MARKERS)
|
|
|
|
|
|
def load_with_fallback(
|
|
build: Callable[[DeviceChoice], M],
|
|
verify: Callable[[M], None],
|
|
preferred: Optional[DeviceChoice],
|
|
on_fallback: Optional[Callable[[DeviceChoice, BaseException], None]] = None,
|
|
) -> tuple[M, DeviceChoice]:
|
|
"""preferred(GPU)로 만들어 검증하고, 실패하면 CPU 로 다시 만든다.
|
|
|
|
- GPU 생성 실패: GPU 관련 오류일 때만 CPU 로 내려간다. 그 밖의 오류(모델 파일
|
|
다운로드 실패 등)는 CPU 로 다시 해도 똑같이 실패하므로 그대로 올린다.
|
|
- GPU 검증 실패: 모델은 만들어졌으므로 장치 문제다. 오류 종류와 무관하게 CPU 로 간다.
|
|
- CPU 후보이거나 후보가 없으면 검증 없이 CPU 로 만든다.
|
|
"""
|
|
if preferred is not None and preferred.is_gpu:
|
|
try:
|
|
model = build(preferred)
|
|
except Exception as exc:
|
|
if not is_gpu_runtime_error(exc):
|
|
raise
|
|
if on_fallback is not None:
|
|
on_fallback(preferred, exc)
|
|
else:
|
|
try:
|
|
verify(model)
|
|
return model, preferred
|
|
except Exception as exc:
|
|
# 검증에 실패한 GPU 모델의 참조를 먼저 끊어 VRAM 을 돌려받는다.
|
|
del model
|
|
if on_fallback is not None:
|
|
on_fallback(preferred, exc)
|
|
return build(CPU_CHOICE), CPU_CHOICE
|