fix: red-team round 3 hardening across desktop, mobile, core and server
Batch of red-team r3 fixes that were in the working tree before the 2026-09-28 design overhaul, committed as one unit with their tests. - desktop main: STT timeouts and sidecar, voice recording store, sync (credentials, audio, knowledge reindex, push gates), runtime provisioner, update policy, AltGr keybindings, voice-command policy, dictionary file codec/limits, meeting transcript condensing and a local recording ledger so interrupted-session recovery only closes meetings this device recorded (a phone's live meeting is left alone). - mobile: login CSRF via implicit token callbacks rejected, account deletion/retention, durable queue retention, knowledge realtime without unfiltered DELETE, meeting re-record failure paths, cloud STT client, preferences store/resync. - core: text chunking splits long unbroken transcripts to fit, template field policy, dictionary limits, meeting markdown inline handling. - server: payple webhook policy and cancellation order scope, meeting document generation quota, team RPC null-role guard, unified LLM quota in-flight accounting, knowledge chunk vector index, meeting re-record failure paths (migrations 20260929*). - ci: portable/runtime feed gates, update-policy schema, Forgejo file delete and alias planning. Four older tests are updated to the new contracts rather than the old behavior: token-pair auth callbacks are rejected, knowledge realtime no longer subscribes to DELETE, long transcript lines are split, and meeting recovery requires the local recording ledger for empty rows.
This commit is contained in:
parent
2428ede03d
commit
ba9ef9741e
161 changed files with 17056 additions and 2379 deletions
|
|
@ -36,7 +36,7 @@ from typing import AsyncGenerator
|
|||
|
||||
import numpy as np
|
||||
import uvicorn
|
||||
from fastapi import FastAPI, File, Form, UploadFile
|
||||
from fastapi import FastAPI, File, Form, Request, UploadFile
|
||||
from fastapi.responses import JSONResponse
|
||||
|
||||
from device_policy import (
|
||||
|
|
@ -79,6 +79,21 @@ _model_devices: dict[str, DeviceChoice] = {}
|
|||
_server: uvicorn.Server | None = None
|
||||
_models_dir: Path | None = None
|
||||
|
||||
# 모델 추론·로딩은 한 번에 하나만 돈다(모델 교체와 전사가 겹치지 않게). 무거운 작업은
|
||||
# asyncio.to_thread 로 돌려, 긴 전사·로딩 중에도 /health · /uia/focus · /download/status 가 응답한다.
|
||||
# (예전엔 async 핸들러 안에서 동기로 돌아 이벤트 루프 전체가 막혔다.)
|
||||
_inference_lock = asyncio.Lock()
|
||||
# 클라이언트가 끊긴 전사를 확인하는 주기(초)
|
||||
_DISCONNECT_POLL_SECONDS = 0.5
|
||||
|
||||
|
||||
class TranscriptionCancelled(Exception):
|
||||
"""클라이언트가 연결을 끊어 전사를 중단했다 (타임아웃·취소)."""
|
||||
|
||||
|
||||
class ModelNotInstalled(Exception):
|
||||
"""models-dir 에도 HF 캐시에도 없는 모델 — 암묵적으로 내려받지 않는다."""
|
||||
|
||||
# ── 다운로드 상태 (스레드 공유) ────────────────────────────
|
||||
|
||||
_download_lock = threading.Lock()
|
||||
|
|
@ -352,6 +367,7 @@ def _run_transcription(
|
|||
model: "WhisperModel",
|
||||
audio_array: np.ndarray,
|
||||
transcribe_kwargs: dict,
|
||||
cancel: threading.Event | None = None,
|
||||
) -> tuple[list[dict], str, str]:
|
||||
"""모델로 전사하고 세그먼트를 끝까지 소비한다.
|
||||
|
||||
|
|
@ -374,6 +390,9 @@ def _run_transcription(
|
|||
full_text_parts: list[str] = []
|
||||
|
||||
for segment in segments_iter:
|
||||
# 클라이언트가 떠났으면(타임아웃·취소) 남은 디코딩을 버린다 — 끝까지 돌면 다음 받아쓰기가 막힌다.
|
||||
if cancel is not None and cancel.is_set():
|
||||
raise TranscriptionCancelled("client disconnected")
|
||||
seg_dict = {
|
||||
"text": segment.text.strip(),
|
||||
"start": round(segment.start, 3),
|
||||
|
|
@ -486,6 +505,22 @@ def _reload_on_cpu(model_id: str, is_primary: bool) -> "WhisperModel":
|
|||
return model
|
||||
|
||||
|
||||
def _ensure_model_available(model_id: str) -> None:
|
||||
"""models-dir 을 쓰는 설치본에서, 받지 않은 모델을 /load 가 HF 에서 몰래 내려받지 않게 한다.
|
||||
|
||||
models-dir 에 있거나 HF 캐시에 이미 있으면 통과한다. models-dir 을 지정하지 않은 실행(개발·테스트)은
|
||||
예전처럼 faster-whisper 에 맡긴다(HF 캐시만 사용).
|
||||
"""
|
||||
if _models_dir is None or _local_model_dir(model_id) is not None:
|
||||
return
|
||||
try:
|
||||
from faster_whisper.utils import download_model
|
||||
|
||||
download_model(model_id, local_files_only=True)
|
||||
except Exception as exc: # noqa: BLE001 — 캐시에 없으면 여러 종류의 예외가 난다
|
||||
raise ModelNotInstalled(f"모델이 설치되어 있지 않습니다: {model_id}") from exc
|
||||
|
||||
|
||||
@app.post("/load")
|
||||
async def load_model(body: dict) -> JSONResponse: # noqa: ANN001
|
||||
"""Whisper 모델을 로딩한다.
|
||||
|
|
@ -518,17 +553,28 @@ async def load_model(body: dict) -> JSONResponse: # noqa: ANN001
|
|||
)
|
||||
|
||||
try:
|
||||
if slot == "aux":
|
||||
# 보조 자리는 하나만 둔다 — 다른 보조 모델은 내려 VRAM 을 돌려받는다.
|
||||
_aux_models.clear()
|
||||
_aux_models[model_id] = _load_on_best_device(model_id)
|
||||
else:
|
||||
# 모델 교체 시 이전 모델을 먼저 해제해 VRAM/RAM을 회수한다.
|
||||
_model = None
|
||||
_model = _load_on_best_device(model_id)
|
||||
_model_id = model_id
|
||||
# 기본 모델이 된 모델은 보조 자리에 중복으로 들고 있지 않는다.
|
||||
_aux_models.pop(model_id, None)
|
||||
# 설치하지 않은 모델은 올리지 않는다 — 쓰던 모델을 내리기 전에 확인한다.
|
||||
await asyncio.to_thread(_ensure_model_available, model_id)
|
||||
except ModelNotInstalled as exc:
|
||||
logger.warning("모델 로딩 거부 (미설치): %s", exc)
|
||||
return JSONResponse(
|
||||
status_code=404,
|
||||
content={"status": "error", "code": "model_not_installed", "message": str(exc)},
|
||||
)
|
||||
|
||||
try:
|
||||
async with _inference_lock:
|
||||
if slot == "aux":
|
||||
# 보조 자리는 하나만 둔다 — 다른 보조 모델은 내려 VRAM 을 돌려받는다.
|
||||
_aux_models.clear()
|
||||
_aux_models[model_id] = await asyncio.to_thread(_load_on_best_device, model_id)
|
||||
else:
|
||||
# 모델 교체 시 이전 모델을 먼저 해제해 VRAM/RAM을 회수한다.
|
||||
_model = None
|
||||
_model = await asyncio.to_thread(_load_on_best_device, model_id)
|
||||
_model_id = model_id
|
||||
# 기본 모델이 된 모델은 보조 자리에 중복으로 들고 있지 않는다.
|
||||
_aux_models.pop(model_id, None)
|
||||
|
||||
load_time_ms = int((time.monotonic() - start_time) * 1000)
|
||||
logger.info("모델 로딩 완료: %s (slot=%s, %dms)", model_id, slot, load_time_ms)
|
||||
|
|
@ -549,8 +595,52 @@ async def load_model(body: dict) -> JSONResponse: # noqa: ANN001
|
|||
)
|
||||
|
||||
|
||||
async def _watch_disconnect(request: Request, cancel: threading.Event) -> None:
|
||||
"""클라이언트가 연결을 끊으면(타임아웃 abort 등) cancel 을 세운다."""
|
||||
while not cancel.is_set():
|
||||
if await request.is_disconnected():
|
||||
cancel.set()
|
||||
return
|
||||
await asyncio.sleep(_DISCONNECT_POLL_SECONDS)
|
||||
|
||||
|
||||
def _transcribe_blocking(
|
||||
model: "WhisperModel",
|
||||
audio_array: np.ndarray,
|
||||
transcribe_kwargs: dict,
|
||||
resolved_model_id: str | None,
|
||||
is_primary: bool,
|
||||
cancel: threading.Event,
|
||||
) -> tuple[list[dict], str, str]:
|
||||
"""워커 스레드에서 전사한다 (GPU 런타임 오류면 CPU 로 다시 올려 한 번만 재시도)."""
|
||||
try:
|
||||
return _run_transcription(model, audio_array, dict(transcribe_kwargs), cancel)
|
||||
except TranscriptionCancelled:
|
||||
raise
|
||||
except Exception as exc:
|
||||
# cuBLAS 미설치 PC 등: 검증을 통과했더라도 실제 전사에서 GPU 런타임 오류가
|
||||
# 나면 이 모델을 CPU 로 다시 올려 한 번만 재시도한다.
|
||||
device = _model_devices.get(resolved_model_id or "")
|
||||
if not (
|
||||
resolved_model_id
|
||||
and device is not None
|
||||
and device.is_gpu
|
||||
and is_gpu_runtime_error(exc)
|
||||
):
|
||||
raise
|
||||
logger.warning(
|
||||
"전사 중 GPU 런타임 오류 → CPU 로 재로딩 후 재시도 (%s): %s",
|
||||
resolved_model_id,
|
||||
exc,
|
||||
)
|
||||
_disable_gpu(exc)
|
||||
cpu_model = _reload_on_cpu(resolved_model_id, is_primary)
|
||||
return _run_transcription(cpu_model, audio_array, dict(transcribe_kwargs), cancel)
|
||||
|
||||
|
||||
@app.post("/transcribe")
|
||||
async def transcribe(
|
||||
request: Request,
|
||||
audio: UploadFile = File(...),
|
||||
language: str = Form("auto"),
|
||||
vad_filter: str = Form("true"),
|
||||
|
|
@ -619,31 +709,24 @@ async def transcribe(
|
|||
is_partial=is_partial,
|
||||
)
|
||||
|
||||
cancel = threading.Event()
|
||||
watcher = asyncio.create_task(_watch_disconnect(request, cancel))
|
||||
try:
|
||||
segments_list, full_text, detected_language = _run_transcription(
|
||||
model, audio_array, dict(transcribe_kwargs)
|
||||
)
|
||||
except Exception as exc:
|
||||
# cuBLAS 미설치 PC 등: 검증을 통과했더라도 실제 전사에서 GPU 런타임 오류가
|
||||
# 나면 이 모델을 CPU 로 다시 올려 한 번만 재시도한다.
|
||||
device = _model_devices.get(resolved_model_id or "")
|
||||
if not (
|
||||
resolved_model_id
|
||||
and device is not None
|
||||
and device.is_gpu
|
||||
and is_gpu_runtime_error(exc)
|
||||
):
|
||||
raise
|
||||
logger.warning(
|
||||
"전사 중 GPU 런타임 오류 → CPU 로 재로딩 후 재시도 (%s): %s",
|
||||
resolved_model_id,
|
||||
exc,
|
||||
)
|
||||
_disable_gpu(exc)
|
||||
model = _reload_on_cpu(resolved_model_id, is_primary)
|
||||
segments_list, full_text, detected_language = _run_transcription(
|
||||
model, audio_array, dict(transcribe_kwargs)
|
||||
)
|
||||
async with _inference_lock:
|
||||
if cancel.is_set():
|
||||
raise TranscriptionCancelled("client disconnected while queued")
|
||||
segments_list, full_text, detected_language = await asyncio.to_thread(
|
||||
_transcribe_blocking,
|
||||
model,
|
||||
audio_array,
|
||||
transcribe_kwargs,
|
||||
resolved_model_id,
|
||||
is_primary,
|
||||
cancel,
|
||||
)
|
||||
finally:
|
||||
cancel.set()
|
||||
watcher.cancel()
|
||||
|
||||
processing_time = int((time.monotonic() - start_time) * 1000)
|
||||
|
||||
|
|
@ -665,6 +748,12 @@ async def transcribe(
|
|||
}
|
||||
)
|
||||
|
||||
except TranscriptionCancelled as exc:
|
||||
logger.info("전사 중단: %s", exc)
|
||||
return JSONResponse(
|
||||
status_code=499,
|
||||
content={"status": "error", "code": "cancelled", "message": str(exc)},
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.error("전사 실패: %s", exc, exc_info=True)
|
||||
return JSONResponse(
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue