fix(voice): stop mic-test ref stealing, unblock action queue, keep tail audio, and move context/LLM routing behind ports and pure policies
This commit is contained in:
parent
2cd462333f
commit
2f94d24c99
13 changed files with 1672 additions and 353 deletions
373
apps/desktop/tests/main/services/voice-mode-redteam-r2-7.test.ts
Normal file
373
apps/desktop/tests/main/services/voice-mode-redteam-r2-7.test.ts
Normal file
|
|
@ -0,0 +1,373 @@
|
|||
// tests/main/services/voice-mode-redteam-r2-7.test.ts
|
||||
// VoiceModeService 회귀 테스트 (round 2 #7):
|
||||
// - 더블탭으로 핸즈프리를 끄면 두 번째 탭이 처리 뒤 새 녹음을 시작하지 않는다 (큐가 막히지 않음)
|
||||
// - 스크린 컨텍스트의 창 조회가 세션·마이크 시작을 막지 않는다
|
||||
// - 원문 삽입 경로(none)에선 선택 텍스트용 복사 단축키를 보내지 않는다
|
||||
// - '체인' 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다
|
||||
// - 정지 시 캡처가 흘려보내는 꼬리 오디오를 받는다
|
||||
|
||||
import { describe, it, expect, beforeEach, vi } from 'vitest'
|
||||
import { EventEmitter } from 'events'
|
||||
import type { KeyBindingTriggerPayload } from '../../../src/main/services/KeyBindingService'
|
||||
|
||||
vi.mock('../../../src/main/services/LoggerService', () => ({
|
||||
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
|
||||
}))
|
||||
|
||||
interface Deferred<T> {
|
||||
promise: Promise<T>
|
||||
resolve: (value: T) => void
|
||||
}
|
||||
|
||||
function deferred<T>(): Deferred<T> {
|
||||
let resolve!: (value: T) => void
|
||||
const promise = new Promise<T>((res) => {
|
||||
resolve = res
|
||||
})
|
||||
return { promise, resolve }
|
||||
}
|
||||
|
||||
type SttResult = { text: string; segments: never[]; language: string; duration: number; processingTime: number }
|
||||
const result = (text: string): SttResult => ({ text, segments: [], language: 'ko', duration: 1, processingTime: 1 })
|
||||
|
||||
const mockSTT = vi.hoisted(() => ({
|
||||
initialize: vi.fn(),
|
||||
transcribe: vi.fn(),
|
||||
transcribePartial: vi.fn(async () => ''),
|
||||
getModels: vi.fn(),
|
||||
getStatus: vi.fn(() => ({ engineState: 'ready', activeModel: 'base', engineVersion: null, gpuAccelerated: false })),
|
||||
}))
|
||||
|
||||
vi.mock('../../../src/main/services/LocalSTTService', () => ({
|
||||
getLocalSTTService: () => mockSTT,
|
||||
resetLocalSTTServiceForTests: () => undefined,
|
||||
}))
|
||||
|
||||
const audioBus = new EventEmitter()
|
||||
const mockAudio = vi.hoisted(() => ({ start: vi.fn(), stop: vi.fn(), on: vi.fn(), off: vi.fn() }))
|
||||
vi.mock('../../../src/main/services/AudioCaptureService', () => ({
|
||||
getAudioCaptureService: () => mockAudio,
|
||||
}))
|
||||
|
||||
const keyBinding = vi.hoisted(() => ({
|
||||
handler: null as ((payload: KeyBindingTriggerPayload) => void) | null,
|
||||
}))
|
||||
vi.mock('../../../src/main/services/KeyBindingService', () => ({
|
||||
getKeyBindingService: () => ({
|
||||
on: (_ev: string, fn: (payload: KeyBindingTriggerPayload) => void) => {
|
||||
keyBinding.handler = fn
|
||||
},
|
||||
off: vi.fn(),
|
||||
}),
|
||||
}))
|
||||
|
||||
const config = vi.hoisted(() => ({ values: {} as Record<string, unknown> }))
|
||||
vi.mock('../../../src/main/services/ConfigService', () => ({
|
||||
configGet: vi.fn((key: string) => config.values[key]),
|
||||
}))
|
||||
|
||||
const mockInsert = vi.hoisted(() => ({ insertText: vi.fn() }))
|
||||
vi.mock('../../../src/main/services/TextInsertService', () => ({
|
||||
getTextInsertService: () => mockInsert,
|
||||
}))
|
||||
|
||||
vi.mock('../../../src/main/services/LocalLLMService', () => ({
|
||||
getLocalLLMService: () => ({ isAvailable: () => true }),
|
||||
}))
|
||||
|
||||
const gateway = vi.hoisted(() => ({ processText: vi.fn() }))
|
||||
vi.mock('../../../src/main/services/llm/LlmGateway', () => ({
|
||||
getLlmGateway: () => gateway,
|
||||
}))
|
||||
|
||||
vi.mock('../../../src/main/services/CaptionService', () => ({
|
||||
getCaptionService: () => ({ getState: () => 'inactive', stop: vi.fn(), start: vi.fn() }),
|
||||
}))
|
||||
|
||||
const chain = vi.hoisted(() => ({ execute: vi.fn() }))
|
||||
vi.mock('../../../src/main/services/ChainService', () => ({
|
||||
getChainService: () => chain,
|
||||
}))
|
||||
|
||||
const instructions = vi.hoisted(() => ({
|
||||
byId: {} as Record<string, { id: string; name: string; prompt: string }>,
|
||||
}))
|
||||
vi.mock('../../../src/main/services/CustomInstructionService', () => ({
|
||||
getCustomInstructionService: () => ({ getById: (id: string) => instructions.byId[id] ?? null }),
|
||||
}))
|
||||
|
||||
const voiceCommand = vi.hoisted(() => ({ instructionId: null as string | null }))
|
||||
vi.mock('../../../src/main/services/VoiceCommandService', () => ({
|
||||
getVoiceCommandService: () => ({
|
||||
isEnabled: () => voiceCommand.instructionId !== null,
|
||||
match: (text: string) =>
|
||||
voiceCommand.instructionId
|
||||
? { matched: true, ruleId: 'r', instructionId: voiceCommand.instructionId, cleanedText: text, matchedKeyword: 'k' }
|
||||
: { matched: false, ruleId: null, instructionId: null, cleanedText: text, matchedKeyword: null },
|
||||
}),
|
||||
}))
|
||||
|
||||
const screen = vi.hoisted(() => ({
|
||||
enabled: false,
|
||||
captureContext: vi.fn(),
|
||||
captureSelectedText: vi.fn(),
|
||||
}))
|
||||
vi.mock('../../../src/main/services/ScreenContextService', () => ({
|
||||
getScreenContextService: () => ({
|
||||
isEnabled: () => screen.enabled,
|
||||
captureContext: screen.captureContext,
|
||||
captureSelectedText: screen.captureSelectedText,
|
||||
buildContextPrompt: (ctx: { appName: string | null; selectedText: string | null }) =>
|
||||
`[ctx app=${ctx.appName ?? '-'} sel=${ctx.selectedText ?? '-'}]\n`,
|
||||
}),
|
||||
}))
|
||||
|
||||
type VoiceModeModule = typeof import('../../../src/main/services/VoiceModeService')
|
||||
let getVoiceModeService: VoiceModeModule['getVoiceModeService']
|
||||
|
||||
const SPEECH = Buffer.alloc(16000 * 2) // 1초
|
||||
|
||||
async function flush(times = 8): Promise<void> {
|
||||
for (let i = 0; i < times; i++) await new Promise((r) => setImmediate(r))
|
||||
}
|
||||
|
||||
function advanceClock(ms: number): void {
|
||||
const base = Date.now()
|
||||
vi.spyOn(Date, 'now').mockReturnValue(base + ms)
|
||||
}
|
||||
|
||||
function key(
|
||||
actionId: 'dictation' | 'hands-free',
|
||||
type: 'pressed' | 'released',
|
||||
isDoublePress: boolean,
|
||||
): KeyBindingTriggerPayload {
|
||||
return {
|
||||
actionId,
|
||||
type,
|
||||
isDoublePress,
|
||||
holdMode: actionId === 'dictation',
|
||||
timestamp: Date.now(),
|
||||
durationMs: 0,
|
||||
} as unknown as KeyBindingTriggerPayload
|
||||
}
|
||||
|
||||
function screenContext(appName: string): { context: { appName: string; windowTitle: string; selectedText: null; capturedAt: number }; selectedTextAttempted: boolean } {
|
||||
return { context: { appName, windowTitle: 'w', selectedText: null, capturedAt: 0 }, selectedTextAttempted: false }
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
vi.restoreAllMocks()
|
||||
vi.resetModules()
|
||||
audioBus.removeAllListeners()
|
||||
config.values = { sttModelId: 'base', defaultLLMAction: 'none', autoInsert: true, llmBackend: 'local' }
|
||||
keyBinding.handler = null
|
||||
screen.enabled = false
|
||||
voiceCommand.instructionId = null
|
||||
instructions.byId = {}
|
||||
for (const fn of [
|
||||
mockSTT.initialize, mockSTT.transcribe, mockSTT.transcribePartial, mockSTT.getModels,
|
||||
mockAudio.start, mockAudio.stop, mockAudio.on, mockAudio.off, mockInsert.insertText,
|
||||
gateway.processText, chain.execute, screen.captureContext, screen.captureSelectedText,
|
||||
]) {
|
||||
fn.mockReset()
|
||||
}
|
||||
mockSTT.initialize.mockResolvedValue(undefined)
|
||||
mockSTT.transcribe.mockResolvedValue(result('기본 전사'))
|
||||
mockSTT.transcribePartial.mockResolvedValue('')
|
||||
mockSTT.getModels.mockReturnValue([
|
||||
{ id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 },
|
||||
])
|
||||
mockAudio.start.mockResolvedValue(undefined)
|
||||
mockAudio.stop.mockResolvedValue(undefined)
|
||||
mockAudio.on.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => {
|
||||
audioBus.on(ev, fn)
|
||||
})
|
||||
mockAudio.off.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => {
|
||||
audioBus.off(ev, fn)
|
||||
})
|
||||
mockInsert.insertText.mockResolvedValue({ success: true, method: 'clipboard', textLength: 1, durationMs: 1 })
|
||||
gateway.processText.mockImplementation(async (text: string) => `LLM(${text})`)
|
||||
chain.execute.mockResolvedValue({ finalText: '체인 결과' })
|
||||
screen.captureContext.mockResolvedValue(screenContext('Code'))
|
||||
screen.captureSelectedText.mockResolvedValue('선택 문장')
|
||||
const mod = await import('../../../src/main/services/VoiceModeService')
|
||||
mod.resetVoiceModeServiceForTests()
|
||||
getVoiceModeService = mod.getVoiceModeService
|
||||
})
|
||||
|
||||
describe('더블탭 핸즈프리 정지', () => {
|
||||
it('정지 더블탭의 두 번째 탭은 전사가 끝난 뒤 새 핸즈프리 녹음을 시작하지 않는다', async () => {
|
||||
const stt = deferred<SttResult>()
|
||||
mockSTT.transcribe.mockReturnValueOnce(stt.promise)
|
||||
const svc = getVoiceModeService()
|
||||
svc.connectKeyBindings()
|
||||
const completed: string[] = []
|
||||
svc.on('session-completed', ({ finalText }) => completed.push(finalText))
|
||||
const emit = (p: KeyBindingTriggerPayload): void => keyBinding.handler?.(p)
|
||||
|
||||
// 더블탭으로 핸즈프리 켜기
|
||||
emit(key('hands-free', 'pressed', true))
|
||||
emit(key('hands-free', 'released', true))
|
||||
await vi.waitFor(() => expect(mockAudio.start).toHaveBeenCalledTimes(1))
|
||||
await vi.waitFor(() => expect(svc.isActive).toBe(true))
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(3000)
|
||||
|
||||
// 더블탭으로 끄기: 첫 탭(dictation press/release) → 정지, 두 번째 탭(hands-free press)
|
||||
emit(key('dictation', 'pressed', false))
|
||||
emit(key('dictation', 'released', false))
|
||||
await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1))
|
||||
emit(key('hands-free', 'pressed', true))
|
||||
emit(key('hands-free', 'released', true))
|
||||
await flush()
|
||||
|
||||
// 전사 완료
|
||||
stt.resolve(result('핸즈프리 받아쓰기'))
|
||||
await vi.waitFor(() => expect(completed).toEqual(['핸즈프리 받아쓰기']))
|
||||
await flush()
|
||||
|
||||
expect(mockAudio.start).toHaveBeenCalledTimes(1)
|
||||
expect(svc.isActive).toBe(false)
|
||||
})
|
||||
|
||||
it('공개 stopSession() 은 여전히 전사·삽입이 끝날 때까지 기다린다', async () => {
|
||||
const svc = getVoiceModeService()
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)
|
||||
expect(svc.isActive).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('스크린 컨텍스트 — 창 조회가 세션을 막지 않는다', () => {
|
||||
it('창 조회가 끝나기 전에 세션과 마이크가 시작되고, 결과는 LLM 입력에 합쳐진다', async () => {
|
||||
screen.enabled = true
|
||||
config.values.defaultLLMAction = 'refine'
|
||||
const probe = deferred<ReturnType<typeof screenContext>>()
|
||||
screen.captureContext.mockReturnValueOnce(probe.promise)
|
||||
const svc = getVoiceModeService()
|
||||
const started: string[] = []
|
||||
svc.on('session-started', ({ session }) => started.push(session.id))
|
||||
|
||||
const starting = svc.startSession('dictation')
|
||||
await Promise.race([starting, new Promise((r) => setTimeout(r, 300))])
|
||||
expect(started).toHaveLength(1)
|
||||
expect(mockAudio.start).toHaveBeenCalledTimes(1)
|
||||
expect(screen.captureContext).toHaveBeenCalledWith(false)
|
||||
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
const stopping = svc.stopSession()
|
||||
probe.resolve(screenContext('Code'))
|
||||
await stopping
|
||||
await starting
|
||||
|
||||
expect(gateway.processText).toHaveBeenCalledTimes(1)
|
||||
const [text, action] = gateway.processText.mock.calls[0] as [string, string]
|
||||
expect(action).toBe('refine')
|
||||
expect(text).toBe('[ctx app=Code sel=선택 문장]\n기본 전사')
|
||||
})
|
||||
})
|
||||
|
||||
describe('스크린 컨텍스트 — 선택 텍스트 캡처', () => {
|
||||
it("원문 삽입(none) 경로에선 복사 단축키(선택 텍스트 캡처)를 보내지 않는다", async () => {
|
||||
screen.enabled = true
|
||||
config.values.defaultLLMAction = 'none'
|
||||
const svc = getVoiceModeService()
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
await flush()
|
||||
|
||||
expect(mockInsert.insertText).toHaveBeenCalledWith('기본 전사', undefined)
|
||||
expect(screen.captureSelectedText).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('LLM 경로면 정지 시점에 선택 텍스트를 캡처한다', async () => {
|
||||
screen.enabled = true
|
||||
config.values.defaultLLMAction = 'refine'
|
||||
const svc = getVoiceModeService()
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
expect(screen.captureSelectedText).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it("none 이어도 음성 단축키로 LLM 경로가 되면 그때 선택 텍스트를 캡처해 쓴다", async () => {
|
||||
screen.enabled = true
|
||||
config.values.defaultLLMAction = 'none'
|
||||
voiceCommand.instructionId = 'i1'
|
||||
instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' }
|
||||
const svc = getVoiceModeService()
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
|
||||
expect(screen.captureSelectedText).toHaveBeenCalledTimes(1)
|
||||
expect(gateway.processText).toHaveBeenCalledTimes(1)
|
||||
const [text, action] = gateway.processText.mock.calls[0] as [string, string]
|
||||
expect(action).toBe('custom')
|
||||
expect(text).toContain('[ctx app=Code sel=선택 문장]')
|
||||
})
|
||||
})
|
||||
|
||||
describe('LLM 경로 — 체인 + 음성 단축키', () => {
|
||||
it('체인 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다', async () => {
|
||||
config.values.defaultLLMAction = 'chain'
|
||||
config.values.activeChainId = 'c1'
|
||||
voiceCommand.instructionId = 'i1'
|
||||
instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' }
|
||||
const svc = getVoiceModeService()
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
|
||||
expect(chain.execute).not.toHaveBeenCalled()
|
||||
expect(gateway.processText).toHaveBeenCalledTimes(1)
|
||||
const [, action, , systemPrompt] = gateway.processText.mock.calls[0] as [string, string, unknown, string]
|
||||
expect(action).toBe('custom')
|
||||
expect(systemPrompt).toContain('요약해줘')
|
||||
})
|
||||
|
||||
it('음성 단축키가 없으면 활성 체인을 실행한다', async () => {
|
||||
config.values.defaultLLMAction = 'chain'
|
||||
config.values.activeChainId = 'c1'
|
||||
const svc = getVoiceModeService()
|
||||
const completed: string[] = []
|
||||
svc.on('session-completed', ({ finalText }) => completed.push(finalText))
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
|
||||
expect(chain.execute).toHaveBeenCalledWith('c1', '기본 전사')
|
||||
expect(gateway.processText).not.toHaveBeenCalled()
|
||||
expect(completed).toEqual(['체인 결과'])
|
||||
})
|
||||
})
|
||||
|
||||
describe('정지 시 꼬리 오디오', () => {
|
||||
it('캡처가 정지 중에 흘려보낸 꼬리 오디오까지 전사에 넣는다', async () => {
|
||||
const TAIL = Buffer.alloc(3200) // 100ms
|
||||
mockAudio.stop.mockImplementation(async () => {
|
||||
// AudioCaptureService.stop() 은 SoX 를 죽이기 전에 남은 오디오를 emit 한다
|
||||
audioBus.emit('audio-data', { buffer: TAIL })
|
||||
})
|
||||
const svc = getVoiceModeService()
|
||||
await svc.startSession('dictation')
|
||||
audioBus.emit('audio-data', { buffer: SPEECH })
|
||||
advanceClock(1000)
|
||||
await svc.stopSession()
|
||||
|
||||
expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)
|
||||
const [audio] = mockSTT.transcribe.mock.calls[0] as [Buffer]
|
||||
expect(audio.length).toBe(SPEECH.length + TAIL.length)
|
||||
expect(audioBus.listenerCount('audio-data')).toBe(0)
|
||||
})
|
||||
})
|
||||
Loading…
Add table
Add a link
Reference in a new issue