fix: red-team round 3 hardening across desktop, mobile, core and server

Batch of red-team r3 fixes that were in the working tree before the
2026-09-28 design overhaul, committed as one unit with their tests.

- desktop main: STT timeouts and sidecar, voice recording store, sync
  (credentials, audio, knowledge reindex, push gates), runtime
  provisioner, update policy, AltGr keybindings, voice-command policy,
  dictionary file codec/limits, meeting transcript condensing and a
  local recording ledger so interrupted-session recovery only closes
  meetings this device recorded (a phone's live meeting is left alone).
- mobile: login CSRF via implicit token callbacks rejected, account
  deletion/retention, durable queue retention, knowledge realtime
  without unfiltered DELETE, meeting re-record failure paths, cloud STT
  client, preferences store/resync.
- core: text chunking splits long unbroken transcripts to fit, template
  field policy, dictionary limits, meeting markdown inline handling.
- server: payple webhook policy and cancellation order scope, meeting
  document generation quota, team RPC null-role guard, unified LLM
  quota in-flight accounting, knowledge chunk vector index, meeting
  re-record failure paths (migrations 20260929*).
- ci: portable/runtime feed gates, update-policy schema, Forgejo file
  delete and alias planning.

Four older tests are updated to the new contracts rather than the old
behavior: token-pair auth callbacks are rejected, knowledge realtime no
longer subscribes to DELETE, long transcript lines are split, and
meeting recovery requires the local recording ledger for empty rows.
This commit is contained in:
Yun Chan 2026-09-28 20:45:52 +09:00
parent 2428ede03d
commit ba9ef9741e
161 changed files with 17056 additions and 2379 deletions

View file

@ -36,7 +36,7 @@ from typing import AsyncGenerator
import numpy as np
import uvicorn
from fastapi import FastAPI, File, Form, UploadFile
from fastapi import FastAPI, File, Form, Request, UploadFile
from fastapi.responses import JSONResponse
from device_policy import (
@ -79,6 +79,21 @@ _model_devices: dict[str, DeviceChoice] = {}
_server: uvicorn.Server | None = None
_models_dir: Path | None = None
# 모델 추론·로딩은 한 번에 하나만 돈다(모델 교체와 전사가 겹치지 않게). 무거운 작업은
# asyncio.to_thread 로 돌려, 긴 전사·로딩 중에도 /health · /uia/focus · /download/status 가 응답한다.
# (예전엔 async 핸들러 안에서 동기로 돌아 이벤트 루프 전체가 막혔다.)
_inference_lock = asyncio.Lock()
# 클라이언트가 끊긴 전사를 확인하는 주기(초)
_DISCONNECT_POLL_SECONDS = 0.5
class TranscriptionCancelled(Exception):
"""클라이언트가 연결을 끊어 전사를 중단했다 (타임아웃·취소)."""
class ModelNotInstalled(Exception):
"""models-dir 에도 HF 캐시에도 없는 모델 — 암묵적으로 내려받지 않는다."""
# ── 다운로드 상태 (스레드 공유) ────────────────────────────
_download_lock = threading.Lock()
@ -352,6 +367,7 @@ def _run_transcription(
model: "WhisperModel",
audio_array: np.ndarray,
transcribe_kwargs: dict,
cancel: threading.Event | None = None,
) -> tuple[list[dict], str, str]:
"""모델로 전사하고 세그먼트를 끝까지 소비한다.
@ -374,6 +390,9 @@ def _run_transcription(
full_text_parts: list[str] = []
for segment in segments_iter:
# 클라이언트가 떠났으면(타임아웃·취소) 남은 디코딩을 버린다 — 끝까지 돌면 다음 받아쓰기가 막힌다.
if cancel is not None and cancel.is_set():
raise TranscriptionCancelled("client disconnected")
seg_dict = {
"text": segment.text.strip(),
"start": round(segment.start, 3),
@ -486,6 +505,22 @@ def _reload_on_cpu(model_id: str, is_primary: bool) -> "WhisperModel":
return model
def _ensure_model_available(model_id: str) -> None:
"""models-dir 을 쓰는 설치본에서, 받지 않은 모델을 /load 가 HF 에서 몰래 내려받지 않게 한다.
models-dir 에 있거나 HF 캐시에 이미 있으면 통과한다. models-dir 을 지정하지 않은 실행(개발·테스트)은
예전처럼 faster-whisper 에 맡긴다(HF 캐시만 사용).
"""
if _models_dir is None or _local_model_dir(model_id) is not None:
return
try:
from faster_whisper.utils import download_model
download_model(model_id, local_files_only=True)
except Exception as exc: # noqa: BLE001 — 캐시에 없으면 여러 종류의 예외가 난다
raise ModelNotInstalled(f"모델이 설치되어 있지 않습니다: {model_id}") from exc
@app.post("/load")
async def load_model(body: dict) -> JSONResponse: # noqa: ANN001
"""Whisper 모델을 로딩한다.
@ -518,17 +553,28 @@ async def load_model(body: dict) -> JSONResponse: # noqa: ANN001
)
try:
if slot == "aux":
# 보조 자리는 하나만 둔다 — 다른 보조 모델은 내려 VRAM 을 돌려받는다.
_aux_models.clear()
_aux_models[model_id] = _load_on_best_device(model_id)
else:
# 모델 교체 시 이전 모델을 먼저 해제해 VRAM/RAM을 회수한다.
_model = None
_model = _load_on_best_device(model_id)
_model_id = model_id
# 기본 모델이 된 모델은 보조 자리에 중복으로 들고 있지 않는다.
_aux_models.pop(model_id, None)
# 설치하지 않은 모델은 올리지 않는다 — 쓰던 모델을 내리기 전에 확인한다.
await asyncio.to_thread(_ensure_model_available, model_id)
except ModelNotInstalled as exc:
logger.warning("모델 로딩 거부 (미설치): %s", exc)
return JSONResponse(
status_code=404,
content={"status": "error", "code": "model_not_installed", "message": str(exc)},
)
try:
async with _inference_lock:
if slot == "aux":
# 보조 자리는 하나만 둔다 — 다른 보조 모델은 내려 VRAM 을 돌려받는다.
_aux_models.clear()
_aux_models[model_id] = await asyncio.to_thread(_load_on_best_device, model_id)
else:
# 모델 교체 시 이전 모델을 먼저 해제해 VRAM/RAM을 회수한다.
_model = None
_model = await asyncio.to_thread(_load_on_best_device, model_id)
_model_id = model_id
# 기본 모델이 된 모델은 보조 자리에 중복으로 들고 있지 않는다.
_aux_models.pop(model_id, None)
load_time_ms = int((time.monotonic() - start_time) * 1000)
logger.info("모델 로딩 완료: %s (slot=%s, %dms)", model_id, slot, load_time_ms)
@ -549,8 +595,52 @@ async def load_model(body: dict) -> JSONResponse: # noqa: ANN001
)
async def _watch_disconnect(request: Request, cancel: threading.Event) -> None:
"""클라이언트가 연결을 끊으면(타임아웃 abort 등) cancel 을 세운다."""
while not cancel.is_set():
if await request.is_disconnected():
cancel.set()
return
await asyncio.sleep(_DISCONNECT_POLL_SECONDS)
def _transcribe_blocking(
model: "WhisperModel",
audio_array: np.ndarray,
transcribe_kwargs: dict,
resolved_model_id: str | None,
is_primary: bool,
cancel: threading.Event,
) -> tuple[list[dict], str, str]:
"""워커 스레드에서 전사한다 (GPU 런타임 오류면 CPU 로 다시 올려 한 번만 재시도)."""
try:
return _run_transcription(model, audio_array, dict(transcribe_kwargs), cancel)
except TranscriptionCancelled:
raise
except Exception as exc:
# cuBLAS 미설치 PC 등: 검증을 통과했더라도 실제 전사에서 GPU 런타임 오류가
# 나면 이 모델을 CPU 로 다시 올려 한 번만 재시도한다.
device = _model_devices.get(resolved_model_id or "")
if not (
resolved_model_id
and device is not None
and device.is_gpu
and is_gpu_runtime_error(exc)
):
raise
logger.warning(
"전사 중 GPU 런타임 오류 → CPU 로 재로딩 후 재시도 (%s): %s",
resolved_model_id,
exc,
)
_disable_gpu(exc)
cpu_model = _reload_on_cpu(resolved_model_id, is_primary)
return _run_transcription(cpu_model, audio_array, dict(transcribe_kwargs), cancel)
@app.post("/transcribe")
async def transcribe(
request: Request,
audio: UploadFile = File(...),
language: str = Form("auto"),
vad_filter: str = Form("true"),
@ -619,31 +709,24 @@ async def transcribe(
is_partial=is_partial,
)
cancel = threading.Event()
watcher = asyncio.create_task(_watch_disconnect(request, cancel))
try:
segments_list, full_text, detected_language = _run_transcription(
model, audio_array, dict(transcribe_kwargs)
)
except Exception as exc:
# cuBLAS 미설치 PC 등: 검증을 통과했더라도 실제 전사에서 GPU 런타임 오류가
# 나면 이 모델을 CPU 로 다시 올려 한 번만 재시도한다.
device = _model_devices.get(resolved_model_id or "")
if not (
resolved_model_id
and device is not None
and device.is_gpu
and is_gpu_runtime_error(exc)
):
raise
logger.warning(
"전사 중 GPU 런타임 오류 → CPU 로 재로딩 후 재시도 (%s): %s",
resolved_model_id,
exc,
)
_disable_gpu(exc)
model = _reload_on_cpu(resolved_model_id, is_primary)
segments_list, full_text, detected_language = _run_transcription(
model, audio_array, dict(transcribe_kwargs)
)
async with _inference_lock:
if cancel.is_set():
raise TranscriptionCancelled("client disconnected while queued")
segments_list, full_text, detected_language = await asyncio.to_thread(
_transcribe_blocking,
model,
audio_array,
transcribe_kwargs,
resolved_model_id,
is_primary,
cancel,
)
finally:
cancel.set()
watcher.cancel()
processing_time = int((time.monotonic() - start_time) * 1000)
@ -665,6 +748,12 @@ async def transcribe(
}
)
except TranscriptionCancelled as exc:
logger.info("전사 중단: %s", exc)
return JSONResponse(
status_code=499,
content={"status": "error", "code": "cancelled", "message": str(exc)},
)
except Exception as exc:
logger.error("전사 실패: %s", exc, exc_info=True)
return JSONResponse(

View file

@ -0,0 +1,134 @@
"""
사이드카 회귀 테스트 (r3-1).
- models-dir 을 쓰는 설치본에서 받지 않은 모델을 /load 하면 HF 에서 몰래 내려받지 않고 404 로 거부하며,
쓰던 기본 모델을 내리지 않는다.
- 클라이언트가 떠난 전사(cancel)는 세그먼트 반복을 멈춘다 — 끝까지 디코딩하며 다음 요청을 막지 않는다.
실행 (apps/desktop/sidecar 에서):
.venv/Scripts/python.exe -m unittest discover -s tests -v
"""
from __future__ import annotations
import sys
import tempfile
import threading
import types
import unittest
from pathlib import Path
from typing import Iterator
from unittest import mock
SIDECAR_DIR = Path(__file__).resolve().parent.parent
if str(SIDECAR_DIR) not in sys.path:
sys.path.insert(0, str(SIDECAR_DIR))
import numpy as np # noqa: E402
from fastapi.testclient import TestClient # noqa: E402
import main # noqa: E402
class _Segment:
def __init__(self, text: str) -> None:
self.text = text
self.start = 0.0
self.end = 1.0
self.avg_logprob = -0.1
class _Info:
language = "ko"
class _FakeModel:
created: list["_FakeModel"] = []
def __init__(self, source: str, **_kwargs: object) -> None:
self.source = source
_FakeModel.created.append(self)
def transcribe(self, _audio: np.ndarray, **_kwargs: object) -> tuple[Iterator[_Segment], _Info]:
def gen() -> Iterator[_Segment]:
for index in range(5):
yield _Segment(f"segment {index}")
return gen(), _Info()
class ImplicitDownloadGuardTest(unittest.TestCase):
def setUp(self) -> None:
_FakeModel.created = []
self._tmp = tempfile.TemporaryDirectory()
main._models_dir = Path(self._tmp.name)
main._gpu_available = False
main._gpu_choice = None
main._aux_models.clear()
main._model_devices.clear()
self.previous = _FakeModel("turbo")
main._model = self.previous # type: ignore[assignment]
main._model_id = "large-v3-turbo"
fake_utils = types.ModuleType("faster_whisper.utils")
def download_model(model_id: str, local_files_only: bool = False, **_kwargs: object) -> str:
if local_files_only:
raise FileNotFoundError(f"{model_id} not in cache")
raise AssertionError("must never download implicitly")
fake_utils.download_model = download_model # type: ignore[attr-defined]
fake_module = types.ModuleType("faster_whisper")
fake_module.WhisperModel = _FakeModel # type: ignore[attr-defined]
fake_module.utils = fake_utils # type: ignore[attr-defined]
self._patch = mock.patch.dict(
sys.modules, {"faster_whisper": fake_module, "faster_whisper.utils": fake_utils}
)
self._patch.start()
self.client = TestClient(main.app)
def tearDown(self) -> None:
self._patch.stop()
self._tmp.cleanup()
main._models_dir = None
main._model = None
main._model_id = None
main._aux_models.clear()
main._model_devices.clear()
def test_uninstalled_model_is_rejected_without_unloading_primary(self) -> None:
res = self.client.post("/load", json={"model_id": "large-v3"})
self.assertEqual(res.status_code, 404)
self.assertEqual(res.json()["code"], "model_not_installed")
self.assertIs(main._model, self.previous)
self.assertEqual(main._model_id, "large-v3-turbo")
self.assertEqual(len(_FakeModel.created), 1)
def test_installed_model_in_models_dir_loads(self) -> None:
model_dir = Path(self._tmp.name) / "small"
model_dir.mkdir()
(model_dir / "model.bin").write_bytes(b"x")
with mock.patch.object(main, "_probe_model"):
res = self.client.post("/load", json={"model_id": "small"})
self.assertEqual(res.status_code, 200)
self.assertEqual(main._model_id, "small")
class TranscriptionCancelTest(unittest.TestCase):
def test_cancelled_transcription_stops_iterating_segments(self) -> None:
cancel = threading.Event()
cancel.set()
with self.assertRaises(main.TranscriptionCancelled):
main._run_transcription(_FakeModel("x"), np.zeros(16000, dtype=np.float32), {}, cancel)
def test_uncancelled_transcription_consumes_all_segments(self) -> None:
segments, text, language = main._run_transcription(
_FakeModel("x"), np.zeros(16000, dtype=np.float32), {}, threading.Event()
)
self.assertEqual(len(segments), 5)
self.assertEqual(language, "ko")
self.assertTrue(text.startswith("segment 0"))
if __name__ == "__main__":
unittest.main()