// tests/main/services/voice-mode-redteam-r1-3.test.ts // VoiceModeService 세션 런(SessionRun) 회귀 테스트: // - 취소된 세션의 늦은 STT 결과가 다음 세션에 삽입되지 않는다 // - 지연 전사 중 두 번째 release 가 받아쓰기를 취소하지 않는다 // - 오디오 캡처 참조는 세션당 정확히 한 번 놓는다 (start 대기 중 취소 포함) // - 녹음 중 캡처가 죽으면(장치 분리) 에러로 알린다 // - 회의 모드가 소유한 자막은 자막 핫키로 끌 수 없다 // - 표시 계층 이벤트(phase / insert-failed) import { describe, it, expect, beforeEach, vi } from 'vitest' import { EventEmitter } from 'events' import { ErrorCode } from '@d3ro/core/errors' import type { KeyBindingTriggerPayload } from '../../../src/main/services/KeyBindingService' vi.mock('../../../src/main/services/LoggerService', () => ({ getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }), })) interface Deferred { promise: Promise resolve: (value: T) => void reject: (err: unknown) => void } function deferred(): Deferred { let resolve!: (value: T) => void let reject!: (err: unknown) => void const promise = new Promise((res, rej) => { resolve = res reject = rej }) return { promise, resolve, reject } } const result = (text: string): { text: string; segments: never[]; language: string; duration: number; processingTime: number } => ({ text, segments: [], language: 'ko', duration: 1, processingTime: 1, }) const mockSTT = vi.hoisted(() => ({ initialize: vi.fn(), transcribe: vi.fn(), transcribePartial: vi.fn(async () => ''), getModels: vi.fn(() => [ { id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 }, ]), getStatus: vi.fn(() => ({ engineState: 'ready', activeModel: 'base', engineVersion: null, gpuAccelerated: false })), })) vi.mock('../../../src/main/services/LocalSTTService', () => ({ getLocalSTTService: () => mockSTT, resetLocalSTTServiceForTests: () => undefined, })) const audioBus = new EventEmitter() const mockAudio = vi.hoisted(() => ({ start: vi.fn(), stop: vi.fn(), on: vi.fn(), off: vi.fn(), })) vi.mock('../../../src/main/services/AudioCaptureService', () => ({ getAudioCaptureService: () => mockAudio, })) const keyBinding = vi.hoisted(() => ({ handler: null as ((payload: KeyBindingTriggerPayload) => void) | null, })) vi.mock('../../../src/main/services/KeyBindingService', () => ({ getKeyBindingService: () => ({ on: (_ev: string, fn: (payload: KeyBindingTriggerPayload) => void) => { keyBinding.handler = fn }, off: vi.fn(), }), })) const config = vi.hoisted(() => ({ values: {} as Record })) vi.mock('../../../src/main/services/ConfigService', () => ({ configGet: vi.fn((key: string) => config.values[key]), })) const mockInsert = vi.hoisted(() => ({ insertText: vi.fn() })) vi.mock('../../../src/main/services/TextInsertService', () => ({ getTextInsertService: () => mockInsert, })) vi.mock('../../../src/main/services/LocalLLMService', () => ({ getLocalLLMService: () => ({ isAvailable: () => true, processText: vi.fn(async (text: string) => text), generate: vi.fn(), chatStream: vi.fn(), cancelGeneration: vi.fn(), }), })) const caption = vi.hoisted(() => ({ state: 'inactive' as string, stop: vi.fn(), start: vi.fn(async () => undefined), })) vi.mock('../../../src/main/services/CaptionService', () => ({ getCaptionService: () => ({ getState: () => caption.state, stop: caption.stop, start: caption.start, }), })) vi.mock('../../../src/main/services/MeetingModeService', () => ({ getMeetingModeService: () => ({ getState: () => 'recording' }), })) type VoiceModeModule = typeof import('../../../src/main/services/VoiceModeService') let getVoiceModeService: VoiceModeModule['getVoiceModeService'] const SPEECH = Buffer.alloc(16000 * 2) // 1초 async function flush(times = 5): Promise { for (let i = 0; i < times; i++) await new Promise((r) => setImmediate(r)) } /** Date.now 를 앞으로 당겨 accidental-press(700ms) 판정을 건너뛴다 */ function advanceClock(ms: number): void { const base = Date.now() vi.spyOn(Date, 'now').mockReturnValue(base + ms) } beforeEach(async () => { vi.restoreAllMocks() vi.resetModules() vi.clearAllMocks() audioBus.removeAllListeners() config.values = { sttModelId: 'base', defaultLLMAction: 'none', autoInsert: true, llmBackend: 'local' } caption.state = 'inactive' keyBinding.handler = null // once 큐가 테스트 사이로 새지 않게 완전히 초기화한다 for (const fn of [mockSTT.initialize, mockSTT.transcribe, mockSTT.transcribePartial, mockSTT.getModels, mockAudio.start, mockAudio.stop, mockInsert.insertText, caption.stop]) { fn.mockReset() } mockSTT.initialize.mockResolvedValue(undefined) mockSTT.transcribe.mockResolvedValue(result('기본 전사')) mockSTT.transcribePartial.mockResolvedValue('') mockSTT.getModels.mockReturnValue([ { id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 }, ]) mockAudio.start.mockResolvedValue(undefined) mockAudio.stop.mockResolvedValue(undefined) mockAudio.on.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => { audioBus.on(ev, fn) }) mockAudio.off.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => { audioBus.off(ev, fn) }) mockInsert.insertText.mockResolvedValue({ success: true, method: 'clipboard', textLength: 1, durationMs: 1 }) const mod = await import('../../../src/main/services/VoiceModeService') mod.resetVoiceModeServiceForTests() getVoiceModeService = mod.getVoiceModeService }) describe('세션 런 — 늦게 도착한 결과', () => { it('취소된 세션 A 의 STT 결과는 새 세션 B 에 삽입되지 않는다', async () => { const svc = getVoiceModeService() const sttA = deferred>() mockSTT.transcribe.mockReturnValueOnce(sttA.promise) const completed: string[] = [] svc.on('session-completed', ({ finalText }) => completed.push(finalText)) // 세션 A: 녹음 → 정지 → 전사 진행 중 await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) const stopA = svc.stopSession() await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)) // A 취소 → B 시작 svc.cancelSession() await svc.startSession('dictation') const sessionB = svc.currentSession expect(sessionB).not.toBeNull() // A 의 결과가 늦게 도착 sttA.resolve(result('세션 A 텍스트')) await stopA await flush() expect(completed).toEqual([]) expect(mockInsert.insertText).not.toHaveBeenCalled() expect(svc.currentSession?.id).toBe(sessionB?.id) expect(svc.isActive).toBe(true) }) it('취소된 세션의 늦은 STT 실패는 새 세션을 에러로 끝내지 않는다', async () => { const svc = getVoiceModeService() const sttA = deferred>() mockSTT.transcribe.mockReturnValueOnce(sttA.promise) const errors: number[] = [] svc.on('error', ({ error }) => errors.push(error.code)) await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) const stopA = svc.stopSession() await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)) svc.cancelSession() await svc.startSession('dictation') sttA.reject(new Error('sidecar exploded')) await stopA await flush() expect(errors).toEqual([]) expect(svc.isActive).toBe(true) }) }) describe('세션 런 — 정지 멱등성', () => { it('지연 전사 중 두 번째 정지는 받아쓰기를 취소하지 않는다', async () => { const init = deferred() mockSTT.initialize.mockReturnValueOnce(init.promise) const stt = deferred>() mockSTT.transcribe.mockReturnValueOnce(stt.promise) const svc = getVoiceModeService() const cancelled: string[] = [] const completed: string[] = [] svc.on('session-cancelled', ({ reason }) => cancelled.push(reason)) svc.on('session-completed', ({ finalText }) => completed.push(finalText)) await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() // STT 미준비 → 대기 init.resolve() // 모델 로딩 완료 → 지연 전사 시작 (버퍼 비움) await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)) await svc.stopSession() // 두 번째 release expect(cancelled).toEqual([]) stt.resolve(result('살아남은 받아쓰기')) await vi.waitFor(() => expect(completed).toEqual(['살아남은 받아쓰기'])) expect(mockInsert.insertText).toHaveBeenCalledWith('살아남은 받아쓰기', undefined) }) }) describe('세션 런 — 오디오 참조 소유', () => { it('정지 후 "No speech" 에러가 나도 audio.stop 은 세션당 한 번이다', async () => { mockSTT.transcribe.mockResolvedValueOnce(result(' ')) const svc = getVoiceModeService() const errors: number[] = [] svc.on('error', ({ error }) => errors.push(error.code)) await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() await vi.waitFor(() => expect(errors).toEqual([ErrorCode.STTNoAudioData])) expect(mockAudio.start).toHaveBeenCalledTimes(1) expect(mockAudio.stop).toHaveBeenCalledTimes(1) }) it('audio.start 대기 중 취소돼도 늦게 잡힌 참조를 정확히 한 번 놓는다', async () => { const start = deferred() mockAudio.start.mockReturnValueOnce(start.promise) const svc = getVoiceModeService() const starting = svc.startSession('dictation') await vi.waitFor(() => expect(mockAudio.start).toHaveBeenCalledTimes(1)) svc.cancelSession() expect(mockAudio.stop).not.toHaveBeenCalled() start.resolve() await starting await flush() expect(mockAudio.stop).toHaveBeenCalledTimes(1) expect(audioBus.listenerCount('audio-data')).toBe(0) }) it('세션이 끝나면 오디오 핸들러가 모두 해제된다 (다음 세션 프레임 중복 없음)', async () => { const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() await vi.waitFor(() => expect(svc.isActive).toBe(false)) await svc.startSession('dictation') expect(audioBus.listenerCount('audio-data')).toBe(1) expect(audioBus.listenerCount('stopped')).toBe(1) }) }) describe('캡처 손실', () => { it('오디오 없이 캡처가 죽으면 AudioCaptureFailed 에러를 낸다', async () => { const svc = getVoiceModeService() const errors: number[] = [] svc.on('error', ({ error }) => errors.push(error.code)) await svc.startSession('dictation') audioBus.emit('stopped', { reason: 'device-lost' }) await flush() expect(errors).toEqual([ErrorCode.AudioCaptureFailed]) expect(svc.isActive).toBe(false) // 캡처 쪽이 이미 정리했으므로 다른 소비자의 참조를 깎지 않는다 expect(mockAudio.stop).not.toHaveBeenCalled() }) it("수동 정지('manual')는 캡처 손실로 보지 않는다", async () => { const svc = getVoiceModeService() const errors: number[] = [] svc.on('error', ({ error }) => errors.push(error.code)) await svc.startSession('dictation') audioBus.emit('stopped', { reason: 'manual' }) await flush() expect(errors).toEqual([]) expect(svc.isActive).toBe(true) }) }) describe('표시 이벤트', () => { it("모델 확인을 통과한 뒤에만 'recording' phase 를 내보낸다", async () => { mockSTT.getModels.mockReturnValueOnce([ { id: 'base', name: 'Base', sizeBytes: 0, downloaded: false, languages: [], accuracy: 2, speed: 4 }, ]) const svc = getVoiceModeService() const phases: string[] = [] const errors: number[] = [] svc.on('phase', ({ phase }) => phases.push(phase)) svc.on('error', ({ error }) => errors.push(error.code)) await svc.startSession('dictation') expect(phases).toEqual([]) expect(errors).toEqual([ErrorCode.STTModelNotFound]) }) it("녹음 → 정지에서 'recording' 다음 'thinking' phase 가 나간다", async () => { const svc = getVoiceModeService() const phases: string[] = [] svc.on('phase', ({ phase }) => phases.push(phase)) await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(phases).toEqual(['recording', 'thinking']) }) it('자동 삽입이 실패하면 insert-failed 를 내보낸다', async () => { mockInsert.insertText.mockRejectedValueOnce(new Error('clipboard busy')) const svc = getVoiceModeService() const failed: string[] = [] svc.on('insert-failed', ({ text }) => failed.push(text)) await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() await vi.waitFor(() => expect(failed).toEqual(['기본 전사'])) }) }) describe('자막 핫키', () => { it('회의 모드가 소유한 자막 세션은 자막 핫키로 멈추지 않는다', async () => { caption.state = 'active' caption.stop.mockResolvedValue(false) const svc = getVoiceModeService() svc.connectKeyBindings() expect(keyBinding.handler).not.toBeNull() keyBinding.handler?.({ actionId: 'caption', type: 'pressed', timestamp: Date.now(), isDoublePress: false, holdMode: false, } as KeyBindingTriggerPayload) // 소유자('user')를 밝혀 요청하므로 CaptionService 가 거부할 수 있다 await vi.waitFor(() => expect(caption.stop).toHaveBeenCalledWith('user')) expect(caption.start).not.toHaveBeenCalled() }) })