// tests/main/services/voice-mode-redteam-r2-7.test.ts // VoiceModeService 회귀 테스트 (round 2 #7): // - 더블탭으로 핸즈프리를 끄면 두 번째 탭이 처리 뒤 새 녹음을 시작하지 않는다 (큐가 막히지 않음) // - 스크린 컨텍스트의 창 조회가 세션·마이크 시작을 막지 않는다 // - 원문 삽입 경로(none)에선 선택 텍스트용 복사 단축키를 보내지 않는다 // - '체인' 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다 // - 정지 시 캡처가 흘려보내는 꼬리 오디오를 받는다 import { describe, it, expect, beforeEach, vi } from 'vitest' import { EventEmitter } from 'events' import type { KeyBindingTriggerPayload } from '../../../src/main/services/KeyBindingService' vi.mock('../../../src/main/services/LoggerService', () => ({ getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }), })) interface Deferred { promise: Promise resolve: (value: T) => void } function deferred(): Deferred { let resolve!: (value: T) => void const promise = new Promise((res) => { resolve = res }) return { promise, resolve } } type SttResult = { text: string; segments: never[]; language: string; duration: number; processingTime: number } const result = (text: string): SttResult => ({ text, segments: [], language: 'ko', duration: 1, processingTime: 1 }) const mockSTT = vi.hoisted(() => ({ initialize: vi.fn(), transcribe: vi.fn(), transcribePartial: vi.fn(async () => ''), getModels: vi.fn(), getStatus: vi.fn(() => ({ engineState: 'ready', activeModel: 'base', engineVersion: null, gpuAccelerated: false })), })) vi.mock('../../../src/main/services/LocalSTTService', () => ({ getLocalSTTService: () => mockSTT, resetLocalSTTServiceForTests: () => undefined, })) const audioBus = new EventEmitter() const mockAudio = vi.hoisted(() => ({ start: vi.fn(), stop: vi.fn(), on: vi.fn(), off: vi.fn() })) vi.mock('../../../src/main/services/AudioCaptureService', () => ({ getAudioCaptureService: () => mockAudio, })) const keyBinding = vi.hoisted(() => ({ handler: null as ((payload: KeyBindingTriggerPayload) => void) | null, })) vi.mock('../../../src/main/services/KeyBindingService', () => ({ getKeyBindingService: () => ({ on: (_ev: string, fn: (payload: KeyBindingTriggerPayload) => void) => { keyBinding.handler = fn }, off: vi.fn(), }), })) const config = vi.hoisted(() => ({ values: {} as Record })) vi.mock('../../../src/main/services/ConfigService', () => ({ configGet: vi.fn((key: string) => config.values[key]), })) const mockInsert = vi.hoisted(() => ({ insertText: vi.fn() })) vi.mock('../../../src/main/services/TextInsertService', () => ({ getTextInsertService: () => mockInsert, })) vi.mock('../../../src/main/services/LocalLLMService', () => ({ getLocalLLMService: () => ({ isAvailable: () => true }), })) const gateway = vi.hoisted(() => ({ processText: vi.fn() })) vi.mock('../../../src/main/services/llm/LlmGateway', () => ({ getLlmGateway: () => gateway, })) vi.mock('../../../src/main/services/CaptionService', () => ({ getCaptionService: () => ({ getState: () => 'inactive', stop: vi.fn(), start: vi.fn() }), })) const chain = vi.hoisted(() => ({ execute: vi.fn() })) vi.mock('../../../src/main/services/ChainService', () => ({ getChainService: () => chain, })) const instructions = vi.hoisted(() => ({ byId: {} as Record, })) vi.mock('../../../src/main/services/CustomInstructionService', () => ({ getCustomInstructionService: () => ({ getById: (id: string) => instructions.byId[id] ?? null }), })) const voiceCommand = vi.hoisted(() => ({ instructionId: null as string | null })) vi.mock('../../../src/main/services/VoiceCommandService', () => ({ getVoiceCommandService: () => ({ isEnabled: () => voiceCommand.instructionId !== null, match: (text: string) => voiceCommand.instructionId ? { matched: true, ruleId: 'r', instructionId: voiceCommand.instructionId, cleanedText: text, matchedKeyword: 'k' } : { matched: false, ruleId: null, instructionId: null, cleanedText: text, matchedKeyword: null }, }), })) const screen = vi.hoisted(() => ({ enabled: false, captureContext: vi.fn(), captureSelectedText: vi.fn(), })) vi.mock('../../../src/main/services/ScreenContextService', () => ({ getScreenContextService: () => ({ isEnabled: () => screen.enabled, captureContext: screen.captureContext, captureSelectedText: screen.captureSelectedText, buildContextPrompt: (ctx: { appName: string | null; selectedText: string | null }) => `[ctx app=${ctx.appName ?? '-'} sel=${ctx.selectedText ?? '-'}]\n`, }), })) type VoiceModeModule = typeof import('../../../src/main/services/VoiceModeService') let getVoiceModeService: VoiceModeModule['getVoiceModeService'] const SPEECH = Buffer.alloc(16000 * 2) // 1초 async function flush(times = 8): Promise { for (let i = 0; i < times; i++) await new Promise((r) => setImmediate(r)) } function advanceClock(ms: number): void { const base = Date.now() vi.spyOn(Date, 'now').mockReturnValue(base + ms) } function key( actionId: 'dictation' | 'hands-free', type: 'pressed' | 'released', isDoublePress: boolean, ): KeyBindingTriggerPayload { return { actionId, type, isDoublePress, holdMode: actionId === 'dictation', timestamp: Date.now(), durationMs: 0, } as unknown as KeyBindingTriggerPayload } function screenContext(appName: string): { context: { appName: string; windowTitle: string; selectedText: null; capturedAt: number }; selectedTextAttempted: boolean } { return { context: { appName, windowTitle: 'w', selectedText: null, capturedAt: 0 }, selectedTextAttempted: false } } beforeEach(async () => { vi.restoreAllMocks() vi.resetModules() audioBus.removeAllListeners() config.values = { sttModelId: 'base', defaultLLMAction: 'none', autoInsert: true, llmBackend: 'local' } keyBinding.handler = null screen.enabled = false voiceCommand.instructionId = null instructions.byId = {} for (const fn of [ mockSTT.initialize, mockSTT.transcribe, mockSTT.transcribePartial, mockSTT.getModels, mockAudio.start, mockAudio.stop, mockAudio.on, mockAudio.off, mockInsert.insertText, gateway.processText, chain.execute, screen.captureContext, screen.captureSelectedText, ]) { fn.mockReset() } mockSTT.initialize.mockResolvedValue(undefined) mockSTT.transcribe.mockResolvedValue(result('기본 전사')) mockSTT.transcribePartial.mockResolvedValue('') mockSTT.getModels.mockReturnValue([ { id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 }, ]) mockAudio.start.mockResolvedValue(undefined) mockAudio.stop.mockResolvedValue(undefined) mockAudio.on.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => { audioBus.on(ev, fn) }) mockAudio.off.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => { audioBus.off(ev, fn) }) mockInsert.insertText.mockResolvedValue({ success: true, method: 'clipboard', textLength: 1, durationMs: 1 }) gateway.processText.mockImplementation(async (text: string) => `LLM(${text})`) chain.execute.mockResolvedValue({ finalText: '체인 결과' }) screen.captureContext.mockResolvedValue(screenContext('Code')) screen.captureSelectedText.mockResolvedValue('선택 문장') const mod = await import('../../../src/main/services/VoiceModeService') mod.resetVoiceModeServiceForTests() getVoiceModeService = mod.getVoiceModeService }) describe('더블탭 핸즈프리 정지', () => { it('정지 더블탭의 두 번째 탭은 전사가 끝난 뒤 새 핸즈프리 녹음을 시작하지 않는다', async () => { const stt = deferred() mockSTT.transcribe.mockReturnValueOnce(stt.promise) const svc = getVoiceModeService() svc.connectKeyBindings() const completed: string[] = [] svc.on('session-completed', ({ finalText }) => completed.push(finalText)) const emit = (p: KeyBindingTriggerPayload): void => keyBinding.handler?.(p) // 더블탭으로 핸즈프리 켜기 emit(key('hands-free', 'pressed', true)) emit(key('hands-free', 'released', true)) await vi.waitFor(() => expect(mockAudio.start).toHaveBeenCalledTimes(1)) await vi.waitFor(() => expect(svc.isActive).toBe(true)) audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(3000) // 더블탭으로 끄기: 첫 탭(dictation press/release) → 정지, 두 번째 탭(hands-free press) emit(key('dictation', 'pressed', false)) emit(key('dictation', 'released', false)) await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)) emit(key('hands-free', 'pressed', true)) emit(key('hands-free', 'released', true)) await flush() // 전사 완료 stt.resolve(result('핸즈프리 받아쓰기')) await vi.waitFor(() => expect(completed).toEqual(['핸즈프리 받아쓰기'])) await flush() expect(mockAudio.start).toHaveBeenCalledTimes(1) expect(svc.isActive).toBe(false) }) it('공개 stopSession() 은 여전히 전사·삽입이 끝날 때까지 기다린다', async () => { const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(mockSTT.transcribe).toHaveBeenCalledTimes(1) expect(svc.isActive).toBe(false) }) }) describe('스크린 컨텍스트 — 창 조회가 세션을 막지 않는다', () => { it('창 조회가 끝나기 전에 세션과 마이크가 시작되고, 결과는 LLM 입력에 합쳐진다', async () => { screen.enabled = true config.values.defaultLLMAction = 'refine' const probe = deferred>() screen.captureContext.mockReturnValueOnce(probe.promise) const svc = getVoiceModeService() const started: string[] = [] svc.on('session-started', ({ session }) => started.push(session.id)) const starting = svc.startSession('dictation') await Promise.race([starting, new Promise((r) => setTimeout(r, 300))]) expect(started).toHaveLength(1) expect(mockAudio.start).toHaveBeenCalledTimes(1) expect(screen.captureContext).toHaveBeenCalledWith(false) audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) const stopping = svc.stopSession() probe.resolve(screenContext('Code')) await stopping await starting expect(gateway.processText).toHaveBeenCalledTimes(1) const [text, action] = gateway.processText.mock.calls[0] as [string, string] expect(action).toBe('refine') expect(text).toBe('[ctx app=Code sel=선택 문장]\n기본 전사') }) }) describe('스크린 컨텍스트 — 선택 텍스트 캡처', () => { it("원문 삽입(none) 경로에선 복사 단축키(선택 텍스트 캡처)를 보내지 않는다", async () => { screen.enabled = true config.values.defaultLLMAction = 'none' const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() await flush() expect(mockInsert.insertText).toHaveBeenCalledWith('기본 전사', undefined) expect(screen.captureSelectedText).not.toHaveBeenCalled() }) it('LLM 경로면 정지 시점에 선택 텍스트를 캡처한다', async () => { screen.enabled = true config.values.defaultLLMAction = 'refine' const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(screen.captureSelectedText).toHaveBeenCalledTimes(1) }) it("none 이어도 음성 단축키로 LLM 경로가 되면 그때 선택 텍스트를 캡처해 쓴다", async () => { screen.enabled = true config.values.defaultLLMAction = 'none' voiceCommand.instructionId = 'i1' instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' } const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(screen.captureSelectedText).toHaveBeenCalledTimes(1) expect(gateway.processText).toHaveBeenCalledTimes(1) const [text, action] = gateway.processText.mock.calls[0] as [string, string] expect(action).toBe('custom') expect(text).toContain('[ctx app=Code sel=선택 문장]') }) }) describe('LLM 경로 — 체인 + 음성 단축키', () => { it('체인 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다', async () => { config.values.defaultLLMAction = 'chain' config.values.activeChainId = 'c1' voiceCommand.instructionId = 'i1' instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' } const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(chain.execute).not.toHaveBeenCalled() expect(gateway.processText).toHaveBeenCalledTimes(1) const [, action, , systemPrompt] = gateway.processText.mock.calls[0] as [string, string, unknown, string] expect(action).toBe('custom') expect(systemPrompt).toContain('요약해줘') }) it('음성 단축키가 없으면 활성 체인을 실행한다', async () => { config.values.defaultLLMAction = 'chain' config.values.activeChainId = 'c1' const svc = getVoiceModeService() const completed: string[] = [] svc.on('session-completed', ({ finalText }) => completed.push(finalText)) await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(chain.execute).toHaveBeenCalledWith('c1', '기본 전사') expect(gateway.processText).not.toHaveBeenCalled() expect(completed).toEqual(['체인 결과']) }) }) describe('정지 시 꼬리 오디오', () => { it('캡처가 정지 중에 흘려보낸 꼬리 오디오까지 전사에 넣는다', async () => { const TAIL = Buffer.alloc(3200) // 100ms mockAudio.stop.mockImplementation(async () => { // AudioCaptureService.stop() 은 SoX 를 죽이기 전에 남은 오디오를 emit 한다 audioBus.emit('audio-data', { buffer: TAIL }) }) const svc = getVoiceModeService() await svc.startSession('dictation') audioBus.emit('audio-data', { buffer: SPEECH }) advanceClock(1000) await svc.stopSession() expect(mockSTT.transcribe).toHaveBeenCalledTimes(1) const [audio] = mockSTT.transcribe.mock.calls[0] as [Buffer] expect(audio.length).toBe(SPEECH.length + TAIL.length) expect(audioBus.listenerCount('audio-data')).toBe(0) }) })