fix(voice): stop mic-test ref stealing, unblock action queue, keep tail audio, and move context/LLM routing behind ports and pure policies

This commit is contained in:
Yun Chan 2026-09-28 02:16:16 +09:00
parent 2cd462333f
commit 2f94d24c99
13 changed files with 1672 additions and 353 deletions

View file

@ -0,0 +1,230 @@
// tests/main/services/audio-capture-redteam-r2-7.test.ts
// AudioCaptureService / 마이크 테스트 IPC 회귀 테스트:
// - kill 된 옛 SoX 의 늦은 'close' 가 새 캡처를 오류로 끝내지 않는다
// - 마이크 테스트는 자기가 잡은 참조만 놓는다 (회의·자막의 캡처를 빼앗지 않는다)
// - 정지 시 SoX 를 곧바로 죽이지 않고 꼬리 오디오와 잔여 프레임을 흘려보낸다
// - SoX 버퍼를 줄여(--buffer 1024) 청크 지연·손실을 줄인다
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'
import { EventEmitter } from 'events'
import { ipcMain } from 'electron'
import { IPC_CHANNELS } from '@d3ro/core/ipc-channels'
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn(() => 'default'),
configSet: vi.fn(),
}))
vi.mock('../../../src/main/utils/paths', () => ({
getSoxPath: () => 'sox',
}))
vi.mock('../../../src/main/windows/WindowManager', () => ({
getMainWindow: () => null,
}))
class FakeProcess extends EventEmitter {
stdout = new EventEmitter()
stderr = new EventEmitter()
kill = vi.fn()
}
const spawned = vi.hoisted(() => ({ list: [] as unknown[], args: [] as string[][] }))
vi.mock('child_process', async (importOriginal) => {
const actual = await importOriginal<typeof import('child_process')>()
return {
...actual,
spawn: vi.fn((_cmd: string, args: string[]) => {
const proc = new FakeProcess()
spawned.list.push(proc)
spawned.args.push(args)
return proc
}),
}
})
type Module = typeof import('../../../src/main/services/AudioCaptureService')
let mod: Module
const FRAME = 1920 // 60ms
function proc(i: number): FakeProcess {
return spawned.list[i] as FakeProcess
}
beforeEach(async () => {
vi.resetModules()
vi.useRealTimers()
spawned.list = []
spawned.args = []
mod = await import('../../../src/main/services/AudioCaptureService')
mod.resetAudioCaptureServiceForTests()
})
afterEach(() => {
vi.useRealTimers()
})
describe('SoX 프로세스 수명', () => {
it("정지 후 곧바로 재시작하면 옛 SoX 의 늦은 'close' 가 새 캡처를 죽이지 않는다", async () => {
const audio = mod.getAudioCaptureService()
const stopped: string[] = []
audio.on('stopped', ({ reason }) => stopped.push(reason))
await audio.start()
await audio.stop()
await audio.start()
expect(spawned.list).toHaveLength(2)
// 옛 프로세스의 'close' 가 kill 몇 ms 뒤 도착
proc(0).emit('close', null)
proc(0).emit('error', new Error('late'))
expect(audio.state).toBe('capturing')
expect(proc(1).kill).not.toHaveBeenCalled()
expect(stopped).toEqual(['manual'])
})
it('SoX 인자에 --buffer 1024 가 들어간다', async () => {
await mod.getAudioCaptureService().start()
const args = spawned.args[0]
const i = args.indexOf('--buffer')
expect(i).toBeGreaterThanOrEqual(0)
expect(args[i + 1]).toBe(String(mod.SOX_BUFFER_BYTES))
expect(mod.SOX_BUFFER_BYTES).toBe(1024)
})
})
describe('정지 시 꼬리 오디오', () => {
it('정지 중 도착한 청크와 60ms 미만 잔여 바이트를 흘려보낸 뒤 SoX 를 끝낸다', async () => {
const audio = mod.getAudioCaptureService()
const received: number[] = []
audio.on('audio-data', ({ buffer }) => received.push(buffer.length))
await audio.start()
proc(0).stdout.emit('data', Buffer.alloc(FRAME + 100)) // 1 프레임 + 잔여 100
const stopping = audio.stop()
expect(proc(0).kill).not.toHaveBeenCalled()
// SoX 버퍼에 남아 있던 끝부분이 정지 직후 도착
proc(0).stdout.emit('data', Buffer.alloc(FRAME))
await stopping
expect(proc(0).kill).toHaveBeenCalledTimes(1)
// 1920 (첫 프레임) + 1920 (잔여 100 + 꼬리 1820) + 잔여 100 (마지막 부분 프레임)
expect(received).toEqual([FRAME, FRAME, 100])
expect(received.reduce((a, b) => a + b, 0)).toBe(FRAME * 2 + 100)
expect(audio.state).toBe('idle')
})
it('꼬리 수집 중 새 소비자가 오면 정지를 취소하고 같은 SoX 로 이어 간다', async () => {
const audio = mod.getAudioCaptureService()
const stopped: string[] = []
audio.on('stopped', ({ reason }) => stopped.push(reason))
await audio.start()
const stopping = audio.stop()
await audio.start()
await stopping
expect(spawned.list).toHaveLength(1)
expect(proc(0).kill).not.toHaveBeenCalled()
expect(audio.state).toBe('capturing')
expect(stopped).toEqual([])
await audio.stop()
expect(proc(0).kill).toHaveBeenCalledTimes(1)
expect(stopped).toEqual(['manual'])
})
})
describe('캡처 임대(lease)', () => {
it('release 는 멱등이고 다른 소비자의 참조를 깎지 않는다', async () => {
const audio = mod.getAudioCaptureService()
await audio.start() // 회의·자막이 잡은 참조
const lease = await audio.acquire()
await lease.release()
await lease.release()
expect(audio.state).toBe('capturing')
expect(proc(0).kill).not.toHaveBeenCalled()
})
it('캡처가 죽었다 다시 시작됐으면 옛 임대의 release 는 새 캡처를 멈추지 않는다', async () => {
const audio = mod.getAudioCaptureService()
const lease = await audio.acquire()
proc(0).emit('close', 1) // 장치 분리
expect(audio.state).toBe('error')
await audio.start() // 다른 소비자가 새로 시작
await lease.release()
expect(audio.state).toBe('capturing')
expect(proc(1).kill).not.toHaveBeenCalled()
})
})
describe('마이크 테스트 IPC', () => {
function handler(channel: string): () => Promise<unknown> {
const call = vi.mocked(ipcMain.handle).mock.calls.find(([ch]) => ch === channel)
if (!call) throw new Error(`handler not registered: ${channel}`)
return call[1] as unknown as () => Promise<unknown>
}
it('회의 캡처 중에 마이크 테스트를 해도 회의 캡처는 끊기지 않는다', async () => {
vi.mocked(ipcMain.handle).mockClear()
const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers')
registerAudioHandlers()
const audio = mod.getAudioCaptureService()
// 회의(자막) 가 캡처를 잡고 있다
await audio.start()
const meetingFrames: number[] = []
audio.on('audio-data', ({ buffer }) => meetingFrames.push(buffer.length))
// 설정 창에서 TEST → 곧바로 STOP
await handler(IPC_CHANNELS.AUDIO.TEST_DEVICE)()
await handler(IPC_CHANNELS.AUDIO.STOP_TEST)()
await new Promise((r) => setTimeout(r, mod.TAIL_FLUSH_MS + 20))
// 옛 SoX 를 죽이거나 새로 띄우지 않았고 회의 오디오가 계속 들어온다
expect(spawned.list).toHaveLength(1)
expect(proc(0).kill).not.toHaveBeenCalled()
expect(audio.state).toBe('capturing')
proc(0).stdout.emit('data', Buffer.alloc(FRAME))
expect(meetingFrames).toEqual([FRAME])
})
it('테스트가 돌고 있지 않을 때의 STOP 은 다른 소비자의 참조를 놓지 않는다', async () => {
vi.mocked(ipcMain.handle).mockClear()
const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers')
registerAudioHandlers()
const audio = mod.getAudioCaptureService()
await audio.start()
await handler(IPC_CHANNELS.AUDIO.STOP_TEST)()
await new Promise((r) => setTimeout(r, mod.TAIL_FLUSH_MS + 20))
expect(audio.state).toBe('capturing')
expect(proc(0).kill).not.toHaveBeenCalled()
})
it('테스트 단독이면 STOP 으로 캡처가 끝난다', async () => {
vi.mocked(ipcMain.handle).mockClear()
const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers')
registerAudioHandlers()
const audio = mod.getAudioCaptureService()
await handler(IPC_CHANNELS.AUDIO.TEST_DEVICE)()
expect(audio.state).toBe('capturing')
await handler(IPC_CHANNELS.AUDIO.STOP_TEST)()
await vi.waitFor(() => expect(audio.state).toBe('idle'))
expect(proc(0).kill).toHaveBeenCalledTimes(1)
})
})

View file

@ -0,0 +1,147 @@
// tests/main/services/screen-context-redteam-r2-7.test.ts
// ScreenContextService + ForegroundWindowPort 회귀 테스트:
// - 활성 창 조회는 포트(koffi/osascript 어댑터)에 맡기고 PowerShell 을 띄우지 않는다
// - Windows 어댑터는 실제 앱 이름을 돌려준다 (예전 PowerShell 경로는 $PID 버그로 늘 'powershell')
// - 콘솔/터미널이 포그라운드면 복사 단축키를 보내지 않는다 (SIGINT·입력 줄 지우기 방지)
// - macOS 에선 ⌘+C 를 보낸다
import { describe, it, expect, beforeEach, vi } from 'vitest'
import { clipboard } from 'electron'
import type { ForegroundWindowInfo } from '../../../src/main/utils/win32-foreground'
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn(() => true),
configSet: vi.fn(),
}))
vi.mock('../../../src/main/services/modifier-state', () => ({
waitForModifiersReleased: vi.fn(async () => true),
}))
const nut = vi.hoisted(() => ({ pressKey: vi.fn(async () => undefined), releaseKey: vi.fn(async () => undefined) }))
vi.mock('@nut-tree-fork/nut-js', () => ({
keyboard: nut,
Key: { LeftControl: 'LeftControl', LeftSuper: 'LeftSuper', C: 'C' },
}))
const childProcess = vi.hoisted(() => ({ execFile: vi.fn(), exec: vi.fn(), spawn: vi.fn() }))
vi.mock('child_process', async (importOriginal) => {
const actual = await importOriginal<typeof import('child_process')>()
return { ...actual, ...childProcess }
})
type ScreenModule = typeof import('../../../src/main/services/ScreenContextService')
type PortModule = typeof import('../../../src/main/services/foreground-window-port')
let screenMod: ScreenModule
let portMod: PortModule
Object.assign(clipboard, { clear: vi.fn() })
function fakePort(appName: string | null, windowTitle: string | null = 'Title'): { current: ReturnType<typeof vi.fn> } {
return { current: vi.fn(async () => ({ appName, windowTitle })) }
}
function info(patch: Partial<ForegroundWindowInfo>): ForegroundWindowInfo {
return { hwnd: 1, title: '', processId: 1, appName: '', appPath: null, bounds: null, ...patch }
}
beforeEach(async () => {
vi.resetModules()
vi.clearAllMocks()
screenMod = await import('../../../src/main/services/ScreenContextService')
portMod = await import('../../../src/main/services/foreground-window-port')
})
describe('captureContext — ForegroundWindowPort', () => {
it('포트가 준 앱·창 정보를 쓰고 자식 프로세스를 띄우지 않는다', async () => {
const port = fakePort('Code', 'main.ts - D3ROVoice')
const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' })
const { context, selectedTextAttempted } = await svc.captureContext(false)
expect(context.appName).toBe('Code')
expect(context.windowTitle).toBe('main.ts - D3ROVoice')
expect(selectedTextAttempted).toBe(false)
expect(childProcess.execFile).not.toHaveBeenCalled()
expect(childProcess.spawn).not.toHaveBeenCalled()
expect(svc.buildContextPrompt(context)).toContain('활성 앱: Code')
})
it('포트가 실패해도 컨텍스트 캡처는 계속된다', async () => {
const port = { current: vi.fn(async () => { throw new Error('ffi') }) }
const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' })
const { context } = await svc.captureContext(false)
expect(context.appName).toBeNull()
expect(svc.buildContextPrompt(context)).toBe('')
})
})
describe('Windows 포그라운드 어댑터 (koffi)', () => {
it('실행 파일 이름에서 .exe 를 떼어 실제 앱 이름을 돌려준다', async () => {
const port = portMod.createWin32ForegroundWindowPort(() =>
info({ appName: 'Code.exe', title: 'main.ts', appPath: 'C:\\Code\\Code.exe' }),
)
await expect(port.current()).resolves.toEqual({ appName: 'Code', windowTitle: 'main.ts' })
})
it('빈 값과 조회 실패는 null 이다', async () => {
await expect(portMod.createWin32ForegroundWindowPort(() => info({})).current()).resolves.toEqual({
appName: null,
windowTitle: null,
})
await expect(portMod.createWin32ForegroundWindowPort(() => null).current()).resolves.toEqual({
appName: null,
windowTitle: null,
})
})
})
describe('선택 텍스트 캡처 — 복사 단축키 안전성', () => {
it.each(['WindowsTerminal', 'conhost', 'pwsh', 'Terminal', 'iTerm2', 'WindowsTerminal.exe'])(
'포그라운드가 터미널(%s)이면 복사 단축키를 보내지 않는다',
async (appName) => {
const svc = new screenMod.ScreenContextService({ foreground: fakePort(appName), platform: 'win32' })
await expect(svc.captureSelectedText()).resolves.toBeNull()
expect(nut.pressKey).not.toHaveBeenCalled()
expect(clipboard.clear).not.toHaveBeenCalled()
},
)
it('Windows 에선 Ctrl+C 를 보낸다', async () => {
vi.mocked(clipboard.readText).mockReturnValueOnce('').mockReturnValueOnce('선택')
const svc = new screenMod.ScreenContextService({ foreground: fakePort('Code'), platform: 'win32' })
await expect(svc.captureSelectedText()).resolves.toBe('선택')
expect(nut.pressKey).toHaveBeenCalledWith('LeftControl', 'C')
})
it('macOS 에선 ⌘+C 를 보낸다', async () => {
vi.mocked(clipboard.readText).mockReturnValueOnce('').mockReturnValueOnce('선택')
const svc = new screenMod.ScreenContextService({ foreground: fakePort('Notes'), platform: 'darwin' })
await expect(svc.captureSelectedText()).resolves.toBe('선택')
expect(nut.pressKey).toHaveBeenCalledWith('LeftSuper', 'C')
expect(nut.pressKey).not.toHaveBeenCalledWith('LeftControl', 'C')
})
it('captureContext(true) 도 터미널이면 선택 텍스트를 건너뛴다', async () => {
const port = fakePort('WindowsTerminal')
const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' })
const { context, selectedTextAttempted } = await svc.captureContext(true)
expect(selectedTextAttempted).toBe(true)
expect(context.selectedText).toBeNull()
expect(nut.pressKey).not.toHaveBeenCalled()
// 포그라운드는 한 번만 조회한다
expect(port.current).toHaveBeenCalledTimes(1)
})
})
describe('isTerminalHost', () => {
it('대소문자·확장자와 무관하게 판별하고, 일반 앱은 아니다', () => {
expect(portMod.isTerminalHost('WINDOWSTERMINAL.EXE')).toBe(true)
expect(portMod.isTerminalHost('Code')).toBe(false)
expect(portMod.isTerminalHost(null)).toBe(false)
})
})

View file

@ -0,0 +1,84 @@
// tests/main/services/voice-intent.test.ts
// 키 제스처 → 세션 의도 표 테스트 (더블탭 on/off, hold, 처리 중 토글).
import { describe, it, expect } from 'vitest'
import {
deriveVoiceIntentPhase,
resolveGestureHoldMode,
resolveGestureMode,
resolveVoiceIntent,
type VoiceGesture,
type VoiceIntent,
type VoiceIntentPhase,
} from '../../../src/main/services/voice-intent'
const dictationPress: VoiceGesture = { type: 'press', mode: 'dictation', holdMode: true }
const dictationRelease: VoiceGesture = { type: 'release', mode: 'dictation', holdMode: true }
const handsFreePress: VoiceGesture = { type: 'press', mode: 'hands-free', holdMode: false }
const handsFreeRelease: VoiceGesture = { type: 'release', mode: 'hands-free', holdMode: false }
const START_DICTATION: VoiceIntent = { kind: 'start', mode: 'dictation' }
const START_HANDS_FREE: VoiceIntent = { kind: 'start', mode: 'hands-free' }
const STOP: VoiceIntent = { kind: 'stop' }
const IGNORE: VoiceIntent = { kind: 'ignore' }
describe('resolveVoiceIntent', () => {
it.each<[string, VoiceGesture, VoiceIntentPhase, VoiceIntent]>([
// hold-to-talk
['idle 에서 누르면 받아쓰기를 시작한다', dictationPress, 'idle', START_DICTATION],
['녹음 중 누름은 무시한다', dictationPress, 'recording', IGNORE],
['녹음 중 떼면 정지한다', dictationRelease, 'recording', STOP],
['idle 에서 뗌은 무시한다', dictationRelease, 'idle', IGNORE],
['처리 중 뗌은 무시한다', dictationRelease, 'processing', IGNORE],
// 핸즈프리 토글 (더블탭)
['idle 에서 더블탭은 핸즈프리를 켠다', handsFreePress, 'idle', START_HANDS_FREE],
['녹음 중 더블탭은 끈다', handsFreePress, 'recording', STOP],
['처리 중 더블탭은 새 녹음을 시작하지 않는다', handsFreePress, 'processing', IGNORE],
['토글의 release 는 늘 무시한다', handsFreeRelease, 'recording', IGNORE],
['토글의 release 는 idle 에서도 무시한다', handsFreeRelease, 'idle', IGNORE],
// 더블탭 정지의 첫 탭(dictation press/release)이 핸즈프리 세션을 멈춘다 (기존 동작 보존)
['핸즈프리 녹음 중 첫 탭 press 는 무시', dictationPress, 'recording', IGNORE],
['핸즈프리 녹음 중 첫 탭 release 가 정지', dictationRelease, 'recording', STOP],
['처리 중 받아쓰기 누름은 무시', dictationPress, 'processing', IGNORE],
])('%s', (_label, gesture, phase, expected) => {
expect(resolveVoiceIntent(gesture, phase)).toEqual(expected)
})
it('holdMode 가 아닌 dictation release 는 정지하지 않는다', () => {
expect(resolveVoiceIntent({ type: 'release', mode: 'dictation', holdMode: false }, 'recording')).toEqual(IGNORE)
})
it('더블탭 on → 더블탭 off 시퀀스: 두 번째 탭은 처리 중에 무시된다', () => {
// on
expect(resolveVoiceIntent(handsFreePress, 'idle')).toEqual(START_HANDS_FREE)
// off: 첫 탭 press/release, 정지 후 처리 중에 두 번째 탭
expect(resolveVoiceIntent(dictationPress, 'recording')).toEqual(IGNORE)
expect(resolveVoiceIntent(dictationRelease, 'recording')).toEqual(STOP)
expect(resolveVoiceIntent(handsFreePress, 'processing')).toEqual(IGNORE)
expect(resolveVoiceIntent(handsFreeRelease, 'processing')).toEqual(IGNORE)
})
})
describe('deriveVoiceIntentPhase', () => {
it('세션 상태에서 단계를 도출한다', () => {
expect(deriveVoiceIntentPhase(false, false)).toBe('idle')
expect(deriveVoiceIntentPhase(false, true)).toBe('idle')
expect(deriveVoiceIntentPhase(true, false)).toBe('recording')
expect(deriveVoiceIntentPhase(true, true)).toBe('processing')
})
})
describe('제스처 입력 정규화', () => {
it('더블탭이거나 hands-free 액션이면 핸즈프리다', () => {
expect(resolveGestureMode('dictation', false)).toBe('dictation')
expect(resolveGestureMode('dictation', true)).toBe('hands-free')
expect(resolveGestureMode('hands-free', false)).toBe('hands-free')
expect(resolveGestureMode('command', false)).toBe('dictation')
})
it("'command' 는 dictation 파이프라인 fallback 이라 hold-to-talk 이다", () => {
expect(resolveGestureHoldMode('command', false)).toBe(true)
expect(resolveGestureHoldMode('dictation', true)).toBe(true)
expect(resolveGestureHoldMode('hands-free', false)).toBe(false)
})
})

View file

@ -0,0 +1,373 @@
// tests/main/services/voice-mode-redteam-r2-7.test.ts
// VoiceModeService 회귀 테스트 (round 2 #7):
// - 더블탭으로 핸즈프리를 끄면 두 번째 탭이 처리 뒤 새 녹음을 시작하지 않는다 (큐가 막히지 않음)
// - 스크린 컨텍스트의 창 조회가 세션·마이크 시작을 막지 않는다
// - 원문 삽입 경로(none)에선 선택 텍스트용 복사 단축키를 보내지 않는다
// - '체인' 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다
// - 정지 시 캡처가 흘려보내는 꼬리 오디오를 받는다
import { describe, it, expect, beforeEach, vi } from 'vitest'
import { EventEmitter } from 'events'
import type { KeyBindingTriggerPayload } from '../../../src/main/services/KeyBindingService'
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
interface Deferred<T> {
promise: Promise<T>
resolve: (value: T) => void
}
function deferred<T>(): Deferred<T> {
let resolve!: (value: T) => void
const promise = new Promise<T>((res) => {
resolve = res
})
return { promise, resolve }
}
type SttResult = { text: string; segments: never[]; language: string; duration: number; processingTime: number }
const result = (text: string): SttResult => ({ text, segments: [], language: 'ko', duration: 1, processingTime: 1 })
const mockSTT = vi.hoisted(() => ({
initialize: vi.fn(),
transcribe: vi.fn(),
transcribePartial: vi.fn(async () => ''),
getModels: vi.fn(),
getStatus: vi.fn(() => ({ engineState: 'ready', activeModel: 'base', engineVersion: null, gpuAccelerated: false })),
}))
vi.mock('../../../src/main/services/LocalSTTService', () => ({
getLocalSTTService: () => mockSTT,
resetLocalSTTServiceForTests: () => undefined,
}))
const audioBus = new EventEmitter()
const mockAudio = vi.hoisted(() => ({ start: vi.fn(), stop: vi.fn(), on: vi.fn(), off: vi.fn() }))
vi.mock('../../../src/main/services/AudioCaptureService', () => ({
getAudioCaptureService: () => mockAudio,
}))
const keyBinding = vi.hoisted(() => ({
handler: null as ((payload: KeyBindingTriggerPayload) => void) | null,
}))
vi.mock('../../../src/main/services/KeyBindingService', () => ({
getKeyBindingService: () => ({
on: (_ev: string, fn: (payload: KeyBindingTriggerPayload) => void) => {
keyBinding.handler = fn
},
off: vi.fn(),
}),
}))
const config = vi.hoisted(() => ({ values: {} as Record<string, unknown> }))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn((key: string) => config.values[key]),
}))
const mockInsert = vi.hoisted(() => ({ insertText: vi.fn() }))
vi.mock('../../../src/main/services/TextInsertService', () => ({
getTextInsertService: () => mockInsert,
}))
vi.mock('../../../src/main/services/LocalLLMService', () => ({
getLocalLLMService: () => ({ isAvailable: () => true }),
}))
const gateway = vi.hoisted(() => ({ processText: vi.fn() }))
vi.mock('../../../src/main/services/llm/LlmGateway', () => ({
getLlmGateway: () => gateway,
}))
vi.mock('../../../src/main/services/CaptionService', () => ({
getCaptionService: () => ({ getState: () => 'inactive', stop: vi.fn(), start: vi.fn() }),
}))
const chain = vi.hoisted(() => ({ execute: vi.fn() }))
vi.mock('../../../src/main/services/ChainService', () => ({
getChainService: () => chain,
}))
const instructions = vi.hoisted(() => ({
byId: {} as Record<string, { id: string; name: string; prompt: string }>,
}))
vi.mock('../../../src/main/services/CustomInstructionService', () => ({
getCustomInstructionService: () => ({ getById: (id: string) => instructions.byId[id] ?? null }),
}))
const voiceCommand = vi.hoisted(() => ({ instructionId: null as string | null }))
vi.mock('../../../src/main/services/VoiceCommandService', () => ({
getVoiceCommandService: () => ({
isEnabled: () => voiceCommand.instructionId !== null,
match: (text: string) =>
voiceCommand.instructionId
? { matched: true, ruleId: 'r', instructionId: voiceCommand.instructionId, cleanedText: text, matchedKeyword: 'k' }
: { matched: false, ruleId: null, instructionId: null, cleanedText: text, matchedKeyword: null },
}),
}))
const screen = vi.hoisted(() => ({
enabled: false,
captureContext: vi.fn(),
captureSelectedText: vi.fn(),
}))
vi.mock('../../../src/main/services/ScreenContextService', () => ({
getScreenContextService: () => ({
isEnabled: () => screen.enabled,
captureContext: screen.captureContext,
captureSelectedText: screen.captureSelectedText,
buildContextPrompt: (ctx: { appName: string | null; selectedText: string | null }) =>
`[ctx app=${ctx.appName ?? '-'} sel=${ctx.selectedText ?? '-'}]\n`,
}),
}))
type VoiceModeModule = typeof import('../../../src/main/services/VoiceModeService')
let getVoiceModeService: VoiceModeModule['getVoiceModeService']
const SPEECH = Buffer.alloc(16000 * 2) // 1초
async function flush(times = 8): Promise<void> {
for (let i = 0; i < times; i++) await new Promise((r) => setImmediate(r))
}
function advanceClock(ms: number): void {
const base = Date.now()
vi.spyOn(Date, 'now').mockReturnValue(base + ms)
}
function key(
actionId: 'dictation' | 'hands-free',
type: 'pressed' | 'released',
isDoublePress: boolean,
): KeyBindingTriggerPayload {
return {
actionId,
type,
isDoublePress,
holdMode: actionId === 'dictation',
timestamp: Date.now(),
durationMs: 0,
} as unknown as KeyBindingTriggerPayload
}
function screenContext(appName: string): { context: { appName: string; windowTitle: string; selectedText: null; capturedAt: number }; selectedTextAttempted: boolean } {
return { context: { appName, windowTitle: 'w', selectedText: null, capturedAt: 0 }, selectedTextAttempted: false }
}
beforeEach(async () => {
vi.restoreAllMocks()
vi.resetModules()
audioBus.removeAllListeners()
config.values = { sttModelId: 'base', defaultLLMAction: 'none', autoInsert: true, llmBackend: 'local' }
keyBinding.handler = null
screen.enabled = false
voiceCommand.instructionId = null
instructions.byId = {}
for (const fn of [
mockSTT.initialize, mockSTT.transcribe, mockSTT.transcribePartial, mockSTT.getModels,
mockAudio.start, mockAudio.stop, mockAudio.on, mockAudio.off, mockInsert.insertText,
gateway.processText, chain.execute, screen.captureContext, screen.captureSelectedText,
]) {
fn.mockReset()
}
mockSTT.initialize.mockResolvedValue(undefined)
mockSTT.transcribe.mockResolvedValue(result('기본 전사'))
mockSTT.transcribePartial.mockResolvedValue('')
mockSTT.getModels.mockReturnValue([
{ id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 },
])
mockAudio.start.mockResolvedValue(undefined)
mockAudio.stop.mockResolvedValue(undefined)
mockAudio.on.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => {
audioBus.on(ev, fn)
})
mockAudio.off.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => {
audioBus.off(ev, fn)
})
mockInsert.insertText.mockResolvedValue({ success: true, method: 'clipboard', textLength: 1, durationMs: 1 })
gateway.processText.mockImplementation(async (text: string) => `LLM(${text})`)
chain.execute.mockResolvedValue({ finalText: '체인 결과' })
screen.captureContext.mockResolvedValue(screenContext('Code'))
screen.captureSelectedText.mockResolvedValue('선택 문장')
const mod = await import('../../../src/main/services/VoiceModeService')
mod.resetVoiceModeServiceForTests()
getVoiceModeService = mod.getVoiceModeService
})
describe('더블탭 핸즈프리 정지', () => {
it('정지 더블탭의 두 번째 탭은 전사가 끝난 뒤 새 핸즈프리 녹음을 시작하지 않는다', async () => {
const stt = deferred<SttResult>()
mockSTT.transcribe.mockReturnValueOnce(stt.promise)
const svc = getVoiceModeService()
svc.connectKeyBindings()
const completed: string[] = []
svc.on('session-completed', ({ finalText }) => completed.push(finalText))
const emit = (p: KeyBindingTriggerPayload): void => keyBinding.handler?.(p)
// 더블탭으로 핸즈프리 켜기
emit(key('hands-free', 'pressed', true))
emit(key('hands-free', 'released', true))
await vi.waitFor(() => expect(mockAudio.start).toHaveBeenCalledTimes(1))
await vi.waitFor(() => expect(svc.isActive).toBe(true))
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(3000)
// 더블탭으로 끄기: 첫 탭(dictation press/release) → 정지, 두 번째 탭(hands-free press)
emit(key('dictation', 'pressed', false))
emit(key('dictation', 'released', false))
await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1))
emit(key('hands-free', 'pressed', true))
emit(key('hands-free', 'released', true))
await flush()
// 전사 완료
stt.resolve(result('핸즈프리 받아쓰기'))
await vi.waitFor(() => expect(completed).toEqual(['핸즈프리 받아쓰기']))
await flush()
expect(mockAudio.start).toHaveBeenCalledTimes(1)
expect(svc.isActive).toBe(false)
})
it('공개 stopSession() 은 여전히 전사·삽입이 끝날 때까지 기다린다', async () => {
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)
expect(svc.isActive).toBe(false)
})
})
describe('스크린 컨텍스트 — 창 조회가 세션을 막지 않는다', () => {
it('창 조회가 끝나기 전에 세션과 마이크가 시작되고, 결과는 LLM 입력에 합쳐진다', async () => {
screen.enabled = true
config.values.defaultLLMAction = 'refine'
const probe = deferred<ReturnType<typeof screenContext>>()
screen.captureContext.mockReturnValueOnce(probe.promise)
const svc = getVoiceModeService()
const started: string[] = []
svc.on('session-started', ({ session }) => started.push(session.id))
const starting = svc.startSession('dictation')
await Promise.race([starting, new Promise((r) => setTimeout(r, 300))])
expect(started).toHaveLength(1)
expect(mockAudio.start).toHaveBeenCalledTimes(1)
expect(screen.captureContext).toHaveBeenCalledWith(false)
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
const stopping = svc.stopSession()
probe.resolve(screenContext('Code'))
await stopping
await starting
expect(gateway.processText).toHaveBeenCalledTimes(1)
const [text, action] = gateway.processText.mock.calls[0] as [string, string]
expect(action).toBe('refine')
expect(text).toBe('[ctx app=Code sel=선택 문장]\n기본 전사')
})
})
describe('스크린 컨텍스트 — 선택 텍스트 캡처', () => {
it("원문 삽입(none) 경로에선 복사 단축키(선택 텍스트 캡처)를 보내지 않는다", async () => {
screen.enabled = true
config.values.defaultLLMAction = 'none'
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
await flush()
expect(mockInsert.insertText).toHaveBeenCalledWith('기본 전사', undefined)
expect(screen.captureSelectedText).not.toHaveBeenCalled()
})
it('LLM 경로면 정지 시점에 선택 텍스트를 캡처한다', async () => {
screen.enabled = true
config.values.defaultLLMAction = 'refine'
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(screen.captureSelectedText).toHaveBeenCalledTimes(1)
})
it("none 이어도 음성 단축키로 LLM 경로가 되면 그때 선택 텍스트를 캡처해 쓴다", async () => {
screen.enabled = true
config.values.defaultLLMAction = 'none'
voiceCommand.instructionId = 'i1'
instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' }
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(screen.captureSelectedText).toHaveBeenCalledTimes(1)
expect(gateway.processText).toHaveBeenCalledTimes(1)
const [text, action] = gateway.processText.mock.calls[0] as [string, string]
expect(action).toBe('custom')
expect(text).toContain('[ctx app=Code sel=선택 문장]')
})
})
describe('LLM 경로 — 체인 + 음성 단축키', () => {
it('체인 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다', async () => {
config.values.defaultLLMAction = 'chain'
config.values.activeChainId = 'c1'
voiceCommand.instructionId = 'i1'
instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' }
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(chain.execute).not.toHaveBeenCalled()
expect(gateway.processText).toHaveBeenCalledTimes(1)
const [, action, , systemPrompt] = gateway.processText.mock.calls[0] as [string, string, unknown, string]
expect(action).toBe('custom')
expect(systemPrompt).toContain('요약해줘')
})
it('음성 단축키가 없으면 활성 체인을 실행한다', async () => {
config.values.defaultLLMAction = 'chain'
config.values.activeChainId = 'c1'
const svc = getVoiceModeService()
const completed: string[] = []
svc.on('session-completed', ({ finalText }) => completed.push(finalText))
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(chain.execute).toHaveBeenCalledWith('c1', '기본 전사')
expect(gateway.processText).not.toHaveBeenCalled()
expect(completed).toEqual(['체인 결과'])
})
})
describe('정지 시 꼬리 오디오', () => {
it('캡처가 정지 중에 흘려보낸 꼬리 오디오까지 전사에 넣는다', async () => {
const TAIL = Buffer.alloc(3200) // 100ms
mockAudio.stop.mockImplementation(async () => {
// AudioCaptureService.stop() 은 SoX 를 죽이기 전에 남은 오디오를 emit 한다
audioBus.emit('audio-data', { buffer: TAIL })
})
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)
const [audio] = mockSTT.transcribe.mock.calls[0] as [Buffer]
expect(audio.length).toBe(SPEECH.length + TAIL.length)
expect(audioBus.listenerCount('audio-data')).toBe(0)
})
})

View file

@ -38,6 +38,19 @@ vi.mock('../../../src/main/services/PremiumLLMService', () => ({
getPremiumLLMService: vi.fn(),
}))
// 활성 창 조회는 koffi FFI 리더(ForegroundWindowPort)로 한다 — 자식 프로세스를 띄우지 않는다.
const foregroundReader = vi.hoisted(() => ({
getForegroundWindowInfo: vi.fn(() => ({
hwnd: 1,
title: 'Test window',
processId: 1,
appName: 'TestApp.exe',
appPath: null,
})),
}))
vi.mock('../../../src/main/utils/win32-foreground', () => foregroundReader)
function createProcess(args: unknown[]): EventEmitter & {
stdout: EventEmitter
stderr: EventEmitter
@ -128,8 +141,8 @@ describe('Windows child process visibility', () => {
expectWindowsHideOnAllCalls(childProcess.exec)
})
// 장치 목록·활성 창 조회는 win32 분기에서만 PowerShell 을 부른다.
it.skipIf(process.platform !== 'win32')('hides Windows device discovery and active-window PowerShell calls', async () => {
// 장치 목록 조회는 win32 분기에서만 PowerShell 을 부른다. 활성 창 조회는 FFI 로 하므로 프로세스가 없다.
it.skipIf(process.platform !== 'win32')('hides Windows device discovery and spawns no process for the active window', async () => {
const audioModule = await import('../../../src/main/services/AudioCaptureService')
const audioService = audioModule.getAudioCaptureService() as unknown as {
_getDevicesWindows(): Promise<unknown>
@ -145,10 +158,9 @@ describe('Windows child process visibility', () => {
timeout: 3000,
windowsHide: true,
})
expect(childProcess.execFileAsync.mock.calls[0]?.[2]).toMatchObject({
timeout: 3000,
windowsHide: true,
})
expect(foregroundReader.getForegroundWindowInfo).toHaveBeenCalled()
expect(childProcess.execFileAsync).not.toHaveBeenCalled()
expect(childProcess.execFile).not.toHaveBeenCalled()
})
it('declares hidden ffmpeg processes for conversion, duration, and chunk extraction', () => {