From 2f94d24c99ad9d792b4d43f7596e383de513aa67 Mon Sep 17 00:00:00 2001 From: Yun Chan Date: Mon, 28 Sep 2026 02:16:16 +0900 Subject: [PATCH] fix(voice): stop mic-test ref stealing, unblock action queue, keep tail audio, and move context/LLM routing behind ports and pure policies --- apps/desktop/src/main/ipc/audio-handlers.ts | 13 +- .../src/main/services/AudioCaptureService.ts | 202 +++++++--- .../src/main/services/ScreenContextService.ts | 180 +++------ .../src/main/services/VoiceModeService.ts | 374 ++++++++++-------- .../main/services/foreground-window-port.ts | 156 ++++++++ .../desktop/src/main/services/voice-intent.ts | 85 ++++ .../audio-capture-redteam-r2-7.test.ts | 230 +++++++++++ .../screen-context-redteam-r2-7.test.ts | 147 +++++++ .../tests/main/services/voice-intent.test.ts | 84 ++++ .../services/voice-mode-redteam-r2-7.test.ts | 373 +++++++++++++++++ .../windows-child-process-hide.test.ts | 24 +- .../core/__tests__/voice-llm-route.test.ts | 68 ++++ packages/core/src/voice-llm-route.ts | 89 +++++ 13 files changed, 1672 insertions(+), 353 deletions(-) create mode 100644 apps/desktop/src/main/services/foreground-window-port.ts create mode 100644 apps/desktop/src/main/services/voice-intent.ts create mode 100644 apps/desktop/tests/main/services/audio-capture-redteam-r2-7.test.ts create mode 100644 apps/desktop/tests/main/services/screen-context-redteam-r2-7.test.ts create mode 100644 apps/desktop/tests/main/services/voice-intent.test.ts create mode 100644 apps/desktop/tests/main/services/voice-mode-redteam-r2-7.test.ts create mode 100644 packages/core/__tests__/voice-llm-route.test.ts create mode 100644 packages/core/src/voice-llm-route.ts diff --git a/apps/desktop/src/main/ipc/audio-handlers.ts b/apps/desktop/src/main/ipc/audio-handlers.ts index 9e99af6..6aac281 100644 --- a/apps/desktop/src/main/ipc/audio-handlers.ts +++ b/apps/desktop/src/main/ipc/audio-handlers.ts @@ -3,7 +3,7 @@ import { ipcMain } from 'electron' import { IPC_CHANNELS } from '@d3ro/core/ipc-channels' import { ipcSuccess, ipcError, ErrorCode } from '@d3ro/core/errors' -import { getAudioCaptureService, calculateRMS } from '../services/AudioCaptureService' +import { getAudioCaptureService, calculateRMS, type CaptureLease } from '../services/AudioCaptureService' import { getMainWindow } from '../windows/WindowManager' import { configGet, configSet } from '../services/ConfigService' import type { SetDeviceParams } from '@d3ro/core/types' @@ -12,6 +12,11 @@ let testTimer: ReturnType | null = null let testAudioHandler: ((payload: { buffer: Buffer; timestamp: number }) => void) | null = null let testLevelInterval: ReturnType | null = null let testLastRms = 0 +/** + * 테스트가 실제로 잡은 캡처 참조. 테스트가 돌고 있지 않을 때 stop() 을 부르면 회의·자막·받아쓰기가 + * 잡고 있는 참조를 빼앗아 그쪽 마이크가 조용히 끊긴다 — 테스트는 자기 임대만 놓는다. + */ +let testLease: CaptureLease | null = null function stopAudioTest(): void { const service = getAudioCaptureService() @@ -28,7 +33,9 @@ function stopAudioTest(): void { testTimer = null } testLastRms = 0 - service.stop().catch(() => { /* ignore */ }) + const lease = testLease + testLease = null + lease?.release().catch(() => { /* ignore */ }) } export function registerAudioHandlers(): void { @@ -62,7 +69,7 @@ export function registerAudioHandlers(): void { stopAudioTest() const service = getAudioCaptureService() - await service.start() + testLease = await service.acquire() // 오디오 데이터에서 RMS 추적 testAudioHandler = (payload: { buffer: Buffer; timestamp: number }) => { diff --git a/apps/desktop/src/main/services/AudioCaptureService.ts b/apps/desktop/src/main/services/AudioCaptureService.ts index a2debe8..828ff93 100644 --- a/apps/desktop/src/main/services/AudioCaptureService.ts +++ b/apps/desktop/src/main/services/AudioCaptureService.ts @@ -18,8 +18,44 @@ const logger = getLogger('AudioCaptureService') /** 60ms 프레임 크기 (바이트): 16000 * 2 * 0.06 = 1920 */ const FRAME_SIZE_BYTES = (AUDIO_FORMAT.SAMPLE_RATE * AUDIO_FORMAT.BYTES_PER_SAMPLE * 60) / 1000 +/** + * SoX stdout 버퍼 크기(바이트). SoX 기본값 8192 는 16kHz 16bit mono 에서 256ms 라, 청크가 늦게 오고 + * 정지 시 버퍼에 남은 최대 256ms(끝 음절)가 버려졌다. 1024 = 32ms. + */ +export const SOX_BUFFER_BYTES = 1024 + +/** + * 마지막 참조를 놓을 때 SoX 를 죽이기 전에 버퍼·파이프에 남은 꼬리 오디오를 받아 내는 시간. + * SoX 버퍼(32ms)의 두 배 이상이어야 정지 직전까지의 소리가 한 번은 쓰여 나온다. + */ +export const TAIL_FLUSH_MS = 80 + export type CaptureState = 'idle' | 'starting' | 'capturing' | 'stopping' | 'error' +/** + * 캡처 참조 한 개. release() 는 멱등이고, 그사이 캡처가 죽었다 다시 시작됐으면(세대가 바뀜) + * 다른 소비자의 참조를 깎지 않는다. + */ +export interface CaptureLease { + release(): Promise +} + +/** SoX 녹음 인자 (Windows: -t waveaudio 필수, 그 외: -d) */ +export function buildSoxRecordArgs(deviceArg: string, platform: NodeJS.Platform = process.platform): string[] { + const input = platform === 'win32' ? ['-t', 'waveaudio', deviceArg] : ['-d'] + return [ + '--buffer', String(SOX_BUFFER_BYTES), + ...input, + '--no-show-progress', + '-r', String(AUDIO_FORMAT.SAMPLE_RATE), + '-c', String(AUDIO_FORMAT.CHANNELS), + '-b', '16', + '-e', 'signed-integer', + '-t', 'raw', + '-', + ] +} + interface AudioCaptureEvents { 'audio-data': (payload: { buffer: Buffer; timestamp: number }) => void 'audio-level': (payload: { level: number; timestamp: number }) => void @@ -59,6 +95,12 @@ class AudioCaptureService extends EventEmitter { private _soxProcess: ChildProcess | null = null private _stream: Readable | null = null private _levelInterval: ReturnType | null = null + /** 새 SoX 를 띄울 때마다 증가 — 임대(lease)가 자기 캡처인지 판단한다 */ + private _generation = 0 + /** 진행 중인 꼬리 수집 정지의 토큰 (0 = 없음). start() 가 0 으로 만들어 정지를 취소한다 */ + private _pendingStopToken = 0 + private _stopSeq = 0 + private _cancelTailWait: (() => void) | null = null /** 프레임 조립용 잔여 바이트 버퍼 */ private _residualBuffer: Buffer = Buffer.alloc(0) @@ -83,6 +125,28 @@ class AudioCaptureService extends EventEmitter { return this._currentDevice } + /** + * 캡처 참조를 하나 잡고 임대(lease)로 돌려준다. 참조를 잡지 않은 쪽이 stop() 을 불러 + * 다른 소비자(회의·자막·받아쓰기)의 참조를 빼앗는 일을 막는다. + */ + async acquire(deviceId?: string): Promise { + await this.start(deviceId) + if (this._state !== 'capturing') { + throw new D3ROError(ErrorCode.AudioCaptureStartFailed, `Audio capture not running (state: ${this._state})`) + } + const generation = this._generation + let released = false + return { + release: async () => { + if (released) return + released = true + // 캡처가 죽었다가(장치 분리 등) 다시 시작됐으면 이 임대는 이미 무효다 + if (this._generation !== generation) return + await this.stop() + }, + } + } + async start(deviceId?: string): Promise { if (this._state === 'capturing') { this._refCount++ @@ -90,6 +154,17 @@ class AudioCaptureService extends EventEmitter { return } + // 마지막 참조가 꼬리를 받는 중에 새 소비자가 오면 정지를 취소하고 같은 SoX 로 이어 간다 + if (this._state === 'stopping' && this._pendingStopToken !== 0) { + this._pendingStopToken = 0 + this._cancelTailWait?.() + this._state = 'capturing' + this._refCount = 1 + this._startLevelTimer() + logger.debug('Pending stop cancelled by a new consumer — capture continues') + return + } + if (this._state !== 'idle' && this._state !== 'error') { logger.warn(`Cannot start capture in state: ${this._state}`) return @@ -119,33 +194,15 @@ class AudioCaptureService extends EventEmitter { ? selectedDeviceId : 'default' - // Windows: sox -t waveaudio --no-show-progress -r 16000 -c 1 -b 16 -e signed-integer -t raw - - // Linux/Mac: sox -d --no-show-progress ... - const soxArgs = process.platform === 'win32' - ? [ - '-t', 'waveaudio', deviceArg, - '--no-show-progress', - '-r', String(AUDIO_FORMAT.SAMPLE_RATE), - '-c', String(AUDIO_FORMAT.CHANNELS), - '-b', '16', - '-e', 'signed-integer', - '-t', 'raw', - '-', - ] - : [ - '-d', - '--no-show-progress', - '-r', String(AUDIO_FORMAT.SAMPLE_RATE), - '-c', String(AUDIO_FORMAT.CHANNELS), - '-b', '16', - '-e', 'signed-integer', - '-t', 'raw', - '-', - ] + // Windows: sox --buffer 1024 -t waveaudio --no-show-progress -r 16000 -c 1 -b 16 -e signed-integer -t raw - + // Linux/Mac: sox --buffer 1024 -d --no-show-progress ... + const soxArgs = buildSoxRecordArgs(deviceArg) logger.info(`SoX args: ${soxArgs.join(' ')}`) - this._soxProcess = spawn(soxExe, soxArgs, { stdio: ['pipe', 'pipe', 'pipe'], windowsHide: true }) - this._stream = this._soxProcess.stdout + const child = spawn(soxExe, soxArgs, { stdio: ['pipe', 'pipe', 'pipe'], windowsHide: true }) + this._soxProcess = child + this._generation++ + this._stream = child.stdout this._residualBuffer = Buffer.alloc(0) this._levelAccumulator = [] @@ -159,13 +216,16 @@ class AudioCaptureService extends EventEmitter { }) // SoX stderr → 로그 - this._soxProcess.stderr?.on('data', (chunk: Buffer) => { + child.stderr?.on('data', (chunk: Buffer) => { const msg = chunk.toString().trim() if (msg) logger.warn(`SoX stderr: ${msg}`) }) - // SoX 프로세스 종료 감지 - this._soxProcess.on('close', (code) => { + // SoX 프로세스 종료 감지. + // 자기 프로세스인지 먼저 확인한다 — 예전엔 kill 한 옛 SoX 의 'close' 가 몇 ms 뒤에 와서 + // 그사이 시작된 새 캡처('capturing')를 오류로 끝냈다(회의 중 마이크 테스트로 회의 마이크가 끊김). + child.on('close', (code) => { + if (this._soxProcess !== child) return if (this._state === 'capturing') { logger.warn(`SoX process exited unexpectedly (code: ${code})`) this._handleError( @@ -175,7 +235,8 @@ class AudioCaptureService extends EventEmitter { } }) - this._soxProcess.on('error', (err: Error) => { + child.on('error', (err: Error) => { + if (this._soxProcess !== child) return logger.error(`SoX process spawn error: ${err.message}`) const hint = soxExe === 'sox' && (err as NodeJS.ErrnoException).code === 'ENOENT' @@ -197,9 +258,7 @@ class AudioCaptureService extends EventEmitter { this._refCount = 1 // 100ms 간격으로 audio-level 이벤트 emit - this._levelInterval = setInterval(() => { - this._emitAudioLevel() - }, TIMING.AUDIO_LEVEL_INTERVAL) + this._startLevelTimer() this.emit('started', { deviceId: this._currentDevice.deviceId }) logger.info('Audio capture started') @@ -225,7 +284,17 @@ class AudioCaptureService extends EventEmitter { return } + // 곧바로 kill 하면 SoX 버퍼·파이프에 남은 끝부분(끝 음절)이 버려진다. + // 잠깐 더 읽어 소비자에게 흘려보낸 뒤 정리한다. 그사이 start() 가 오면 정지를 취소한다. this._state = 'stopping' + this._stopLevelTimer() + const token = ++this._stopSeq + this._pendingStopToken = token + await this._waitForTail() + if (this._pendingStopToken !== token || this._state !== 'stopping') return + this._pendingStopToken = 0 + + this._emitResidual() this._cleanup() this._state = 'idle' this._currentDevice = null @@ -233,6 +302,48 @@ class AudioCaptureService extends EventEmitter { logger.info('Audio capture stopped') } + /** TAIL_FLUSH_MS 만큼(또는 SoX 가 먼저 끝날 때까지, 또는 정지가 취소될 때까지) 기다린다 */ + private _waitForTail(): Promise { + const child = this._soxProcess + return new Promise((resolve) => { + let done = false + const finish = (): void => { + if (done) return + done = true + clearTimeout(timer) + child?.off('close', finish) + if (this._cancelTailWait === finish) this._cancelTailWait = null + resolve() + } + const timer = setTimeout(finish, TAIL_FLUSH_MS) + child?.once('close', finish) + this._cancelTailWait = finish + }) + } + + /** 60ms 프레임에 못 미친 잔여 바이트를 마지막 프레임으로 내보낸다 (샘플 경계로 자름) */ + private _emitResidual(): void { + const usable = this._residualBuffer.length - (this._residualBuffer.length % AUDIO_FORMAT.BYTES_PER_SAMPLE) + if (usable <= 0) return + const frame = Buffer.from(this._residualBuffer.subarray(0, usable)) + this._residualBuffer = Buffer.alloc(0) + this.emit('audio-data', { buffer: frame, timestamp: Date.now() }) + } + + private _startLevelTimer(): void { + this._stopLevelTimer() + this._levelInterval = setInterval(() => { + this._emitAudioLevel() + }, TIMING.AUDIO_LEVEL_INTERVAL) + } + + private _stopLevelTimer(): void { + if (this._levelInterval) { + clearInterval(this._levelInterval) + this._levelInterval = null + } + } + async getDevices(): Promise { // 캐싱: 네이티브 프로세스 호출이 느리므로 한번만 실행 if (this._cachedDevices) return this._cachedDevices @@ -378,6 +489,8 @@ class AudioCaptureService extends EventEmitter { } dispose(): void { + this._pendingStopToken = 0 + this._cancelTailWait?.() this._cleanup() this._state = 'idle' this._currentDevice = null @@ -396,13 +509,7 @@ class AudioCaptureService extends EventEmitter { ? selectedDeviceId : 'default' - const soxArgs = process.platform === 'win32' - ? ['-t', 'waveaudio', deviceArg, '--no-show-progress', - '-r', String(AUDIO_FORMAT.SAMPLE_RATE), '-c', String(AUDIO_FORMAT.CHANNELS), - '-b', '16', '-e', 'signed-integer', '-t', 'raw', '-'] - : ['-d', '--no-show-progress', - '-r', String(AUDIO_FORMAT.SAMPLE_RATE), '-c', String(AUDIO_FORMAT.CHANNELS), - '-b', '16', '-e', 'signed-integer', '-t', 'raw', '-'] + const soxArgs = buildSoxRecordArgs(deviceArg) return new Promise((resolve, reject) => { const proc = spawn(soxExe, soxArgs, { stdio: ['ignore', 'pipe', 'pipe'], windowsHide: true }) @@ -505,6 +612,7 @@ class AudioCaptureService extends EventEmitter { this._cleanup() this._state = 'error' + this._refCount = 0 this._currentDevice = null this.emit('error', { error }) this.emit('stopped', { reason: stopReason }) @@ -514,10 +622,7 @@ class AudioCaptureService extends EventEmitter { * 녹음 프로세스와 타이머를 정리한다. */ private _cleanup(): void { - if (this._levelInterval) { - clearInterval(this._levelInterval) - this._levelInterval = null - } + this._stopLevelTimer() if (this._stream) { this._stream.removeAllListeners() @@ -525,12 +630,19 @@ class AudioCaptureService extends EventEmitter { } if (this._soxProcess) { + const child = this._soxProcess + this._soxProcess = null + // 이 프로세스의 늦은 'close'/'error' 가 다음 캡처를 건드리지 못하게 떼어 낸다. + // 'error' 는 리스너가 없으면 throw 하므로 빈 sink 를 남긴다. + child.removeAllListeners('close') + child.removeAllListeners('error') + child.on('error', () => { /* 정리된 프로세스의 늦은 에러 무시 */ }) + child.stderr?.removeAllListeners('data') try { - this._soxProcess.kill() + child.kill() } catch { // 이미 종료된 프로세스 kill 시 에러 무시 } - this._soxProcess = null } this._residualBuffer = Buffer.alloc(0) diff --git a/apps/desktop/src/main/services/ScreenContextService.ts b/apps/desktop/src/main/services/ScreenContextService.ts index dc12651..dab6093 100644 --- a/apps/desktop/src/main/services/ScreenContextService.ts +++ b/apps/desktop/src/main/services/ScreenContextService.ts @@ -1,6 +1,8 @@ // src/main/services/ScreenContextService.ts // 활성 윈도우 정보 + 선택된 텍스트를 캡처하여 LLM 프롬프트에 컨텍스트로 제공한다. // Phase 10.2 스크린 컨텍스트. Speakly ContextService 참조. +// 활성 창 조회는 ForegroundWindowPort(foreground-window-port.ts)에 맡긴다 — 이 서비스는 컨텍스트 +// 정책(무엇을 언제 캡처해 어떻게 프롬프트로 만드는가)만 소유한다. import { clipboard } from 'electron' import { getLogger } from './LoggerService' @@ -8,6 +10,11 @@ import { configGet, configSet } from './ConfigService' import { D3ROError, ErrorCode } from '@d3ro/core/errors' import type { ScreenContext, CaptureContextResult } from '@d3ro/core/types' import { waitForModifiersReleased } from './modifier-state' +import { + createDefaultForegroundWindowPort, + isTerminalHost, + type ForegroundWindowPort, +} from './foreground-window-port' const logger = getLogger('screen-context') @@ -30,6 +37,13 @@ interface NutKeyboard { Key: NutModule['Key'] } +export interface ScreenContextServiceDeps { + /** 포그라운드 창 조회 (기본: 플랫폼 어댑터) */ + foreground?: ForegroundWindowPort + /** 복사 단축키 수정자 결정용 플랫폼 (기본: process.platform) */ + platform?: NodeJS.Platform +} + // ============================================================ // 클립보드 스냅샷 (TextInsertService와 동일 패턴) // ============================================================ @@ -46,9 +60,16 @@ interface ClipboardSnapshot { // ScreenContextService // ============================================================ -class ScreenContextService { +export class ScreenContextService { private _nutKeyboard: NutKeyboard | null = null private _nutLoadPromise: Promise | null = null + private readonly _foreground: ForegroundWindowPort + private readonly _platform: NodeJS.Platform + + constructor(deps: ScreenContextServiceDeps = {}) { + this._platform = deps.platform ?? process.platform + this._foreground = deps.foreground ?? createDefaultForegroundWindowPort(this._platform) + } /** * nut-js lazy dynamic import (TextInsertService 패턴). @@ -90,7 +111,7 @@ class ScreenContextService { /** * 활성 윈도우 정보 + 선택된 텍스트를 캡처한다. * - * @param captureSelectedText true이면 Ctrl+C로 선택 텍스트 캡처 시도 + * @param captureSelectedText true이면 복사 단축키로 선택 텍스트 캡처 시도 */ async captureContext(captureSelectedText = true): Promise { const capturedAt = Date.now() @@ -99,9 +120,9 @@ class ScreenContextService { let selectedText: string | null = null let selectedTextAttempted = false - // 1. 활성 윈도우 정보 (PowerShell) + // 1. 활성 윈도우 정보 (ForegroundWindowPort) try { - const appInfo = await this._getActiveWindowInfo() + const appInfo = await this._foreground.current() appName = appInfo.appName windowTitle = appInfo.windowTitle } catch (error) { @@ -115,7 +136,7 @@ class ScreenContextService { if (captureSelectedText) { selectedTextAttempted = true try { - selectedText = await this.captureSelectedText() + selectedText = await this._captureSelectionIn(appName) } catch (error) { logger.warn( `Failed to capture selected text: ${error instanceof Error ? error.message : String(error)}` @@ -142,13 +163,30 @@ class ScreenContextService { * 선택된 텍스트만 캡처한다 (클립보드 방식). * 수정자 키가 눌려 있으면 뗄 때까지 잠깐 기다리고, 그래도 쥐고 있으면 캡처하지 않고 null 을 * 돌려준다(클립보드도 건드리지 않는다). + * 포그라운드가 콘솔/터미널이면 복사 단축키를 보내지 않는다 — 선택이 없을 때 Ctrl+C 는 + * 실행 중 프로세스를 중단(SIGINT)하거나 입력 줄을 지운다. */ async captureSelectedText(): Promise { + let foregroundApp: string | null = null + try { + foregroundApp = (await this._foreground.current()).appName + } catch { + // 조회 실패 시엔 앱을 모른 채 진행한다 + } + return this._captureSelectionIn(foregroundApp) + } + + /** 포그라운드 앱을 이미 알고 있을 때의 선택 텍스트 캡처 */ + private async _captureSelectionIn(foregroundApp: string | null): Promise { const released = await waitForModifiersReleased(MODIFIER_RELEASE_TIMEOUT_MS) if (!released) { logger.info('Selected text capture skipped: modifier keys still held') return null } + if (isTerminalHost(foregroundApp)) { + logger.info(`Selected text capture skipped: terminal host (${foregroundApp ?? 'unknown'})`) + return null + } return this._captureSelectedText() } @@ -202,132 +240,9 @@ class ScreenContextService { // ── 내부 구현 ───────────────────────────────────────── - /** - * 활성 윈도우의 프로세스명과 타이틀을 가져온다. - * - Windows: PowerShell + user32.dll - * - macOS: osascript (System Events) - * - Linux: 미지원 (null 반환) - * - * 참고: macOS는 Accessibility 권한이 필요하다. 첫 호출 시 시스템이 - * 권한 요청 다이얼로그를 띄운다. 사용자가 거부하면 { null, null } 반환. - */ - private async _getActiveWindowInfo(): Promise<{ - appName: string | null - windowTitle: string | null - }> { - if (process.platform === 'win32') { - return this._getActiveWindowInfoWin32() - } - if (process.platform === 'darwin') { - return this._getActiveWindowInfoDarwin() - } - return { appName: null, windowTitle: null } - } - - /** - * Windows: PowerShell + user32.dll로 활성 윈도우 조회. - */ - private async _getActiveWindowInfoWin32(): Promise<{ - appName: string | null - windowTitle: string | null - }> { - const { execFile } = await import('child_process') - const { promisify } = await import('util') - const execFileAsync = promisify(execFile) - - const psScript = ` -Add-Type @" -using System; -using System.Runtime.InteropServices; -using System.Text; -public class Win32 { - [DllImport("user32.dll")] public static extern IntPtr GetForegroundWindow(); - [DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr hWnd, out uint lpdwProcessId); - [DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr hWnd, StringBuilder lpString, int nMaxCount); -} -"@ -$hwnd = [Win32]::GetForegroundWindow() -$pid = 0 -[void][Win32]::GetWindowThreadProcessId($hwnd, [ref]$pid) -$proc = Get-Process -Id $pid -ErrorAction SilentlyContinue -$sb = New-Object System.Text.StringBuilder 512 -[void][Win32]::GetWindowText($hwnd, $sb, 512) -$procName = if ($proc) { $proc.ProcessName } else { '' } -$title = $sb.ToString() -"$procName$([char]10)$title" -`.trim() - - try { - const { stdout } = await execFileAsync('powershell.exe', [ - '-NoProfile', - '-NonInteractive', - '-ExecutionPolicy', 'Bypass', - '-Command', psScript, - ], { timeout: 3000, windowsHide: true }) - - const lines = stdout.trim().split('\n') - const appName = lines[0]?.trim() || null - const windowTitle = lines[1]?.trim() || null - - return { appName, windowTitle } - } catch (error) { - logger.warn( - `PowerShell active window query failed: ${error instanceof Error ? error.message : String(error)}` - ) - return { appName: null, windowTitle: null } - } - } - - /** - * macOS: osascript (AppleScript)로 frontmost process와 윈도우 타이틀 조회. - * Accessibility 권한 필요. - */ - private async _getActiveWindowInfoDarwin(): Promise<{ - appName: string | null - windowTitle: string | null - }> { - const { execFile } = await import('child_process') - const { promisify } = await import('util') - const execFileAsync = promisify(execFile) - - // AppleScript: 프로세스명과 앞 윈도우 타이틀을 2줄로 반환. - // 윈도우가 없는 앱도 있으므로 try-fallback. - const script = ` -try - tell application "System Events" - set frontApp to first process whose frontmost is true - set appName to name of frontApp - try - set winTitle to name of front window of frontApp - on error - set winTitle to "" - end try - return appName & linefeed & winTitle - end tell -on error errMsg - return "" & linefeed & "" -end try -`.trim() - - try { - const { stdout } = await execFileAsync('/usr/bin/osascript', ['-e', script], { - timeout: 3000 - }) - const lines = stdout.split('\n') - const appName = lines[0]?.trim() || null - const windowTitle = lines[1]?.trim() || null - return { appName, windowTitle } - } catch (error) { - logger.warn( - `osascript active window query failed: ${error instanceof Error ? error.message : String(error)}` - ) - return { appName: null, windowTitle: null } - } - } - /** * 선택된 텍스트를 클립보드 방식으로 캡처한다. - * TextInsertService의 역방향: clipboard save → Ctrl+C simulate → clipboard read → clipboard restore + * TextInsertService의 역방향: clipboard save → 복사 단축키 simulate → clipboard read → clipboard restore */ private async _captureSelectedText(): Promise { const nut = await this._ensureNut() @@ -339,9 +254,10 @@ end try // 2. 클립보드 비우기 (이전 내용이 남아있으면 "선택 없음"을 감지할 수 없으므로) clipboard.clear() - // 3. Ctrl+C 시뮬레이션 - await nut.pressKey(nut.Key.LeftControl, nut.Key.C) - await nut.releaseKey(nut.Key.LeftControl, nut.Key.C) + // 3. 복사 단축키 시뮬레이션 (macOS ⌘+C, 그 외 Ctrl+C) + const modifier = this._platform === 'darwin' ? nut.Key.LeftSuper : nut.Key.LeftControl + await nut.pressKey(modifier, nut.Key.C) + await nut.releaseKey(modifier, nut.Key.C) // 4. 클립보드에 텍스트가 복사될 때까지 대기 await this._sleep(150) diff --git a/apps/desktop/src/main/services/VoiceModeService.ts b/apps/desktop/src/main/services/VoiceModeService.ts index be10be6..cd4d695 100644 --- a/apps/desktop/src/main/services/VoiceModeService.ts +++ b/apps/desktop/src/main/services/VoiceModeService.ts @@ -35,6 +35,13 @@ import { TIMING } from '@d3ro/core/constants' import { RecognitionState, AudioState } from '@d3ro/core/types' import type { KeyBindingActionId, VoiceMode, VoiceState, LLMAction } from '@d3ro/core/types' import type { ScreenContext } from '@d3ro/core/types' +import { resolveVoiceLlmRoute, routeUsesLlm, type VoiceLlmRoute } from '@d3ro/core/voice-llm-route' +import { + deriveVoiceIntentPhase, + resolveGestureHoldMode, + resolveGestureMode, + resolveVoiceIntent, +} from './voice-intent' const logger = getLogger('VoiceModeService') @@ -126,6 +133,8 @@ class SessionRun { /** 스크린 컨텍스트: 선택 텍스트 캡처 여부와 진행 중 작업 */ captureSelection = false selectionCapture: Promise | null = null + /** 스크린 컨텍스트: 녹음 시작 시점 활성 창 조회 (기다리지 않고 LLM 단계에서 합친다) */ + windowContext: Promise | null = null dataHandler: ((payload: { buffer: Buffer }) => void) | null = null levelHandler: ((payload: { level: number }) => void) | null = null @@ -169,6 +178,19 @@ function describe(err: unknown): string { return err instanceof Error ? err.message : String(err) } +/** LLM 을 실제로 부르는 경로 */ +type LlmProcessingRoute = Exclude + +/** + * 정지 요청의 결과. 오디오 해제까지는 요청 안에서 끝나고, 전사·LLM 후처리는 `finished` 로 따로 돈다. + * (async 함수가 Promise 를 반환하면 펼쳐지므로 객체로 감싼다) + */ +interface StopOutcome { + finished: Promise +} + +const NOTHING_PENDING: StopOutcome = { finished: Promise.resolve() } + // ============================================================ // VoiceModeService // ============================================================ @@ -229,11 +251,11 @@ class VoiceModeService extends EventEmitter { return } - const holdMode = this._resolveHoldMode(payload.actionId, payload.holdMode) + const holdMode = resolveGestureHoldMode(payload.actionId, payload.holdMode) const action: VoiceAction = { type: payload.type === 'pressed' ? 'press' : 'release', timestamp: payload.timestamp, - mode: this._resolveMode(payload.actionId, payload.isDoublePress), + mode: resolveGestureMode(payload.actionId, payload.isDoublePress), actionId: payload.actionId, holdMode } @@ -300,16 +322,23 @@ class VoiceModeService extends EventEmitter { // 선택 텍스트(Ctrl+C)는 여기서 보내지 않는다: 이 시점엔 사용자가 아직 핫키(기본 Right Alt)를 // 누르고 있어 대상 앱에 Ctrl+Alt+C 가 들어가고(복사 안 됨, AltGr 배열에선 문자 입력), // 클립보드만 비워진다. 선택 텍스트는 녹음 종료(키를 뗀 뒤)에 캡처한다. - let screenContext: ScreenContext | null = null - let captureSelection = false + // 활성 창 조회는 기다리지 않는다 — 예전엔 조회(PowerShell, 0.7~3초)가 세션·마이크 시작 앞을 + // 막아 첫 말이 잘리고 짧은 hold-to-talk 가 'too-short' 로 취소됐다. 결과는 LLM 단계에서 합친다. + let windowContext: Promise | null = null try { const { getScreenContextService } = await import('./ScreenContextService') const ctx = getScreenContextService() if (ctx.isEnabled()) { - captureSelection = true - const result = await ctx.captureContext(false) - screenContext = result.context - logger.info(`Screen context captured: ${screenContext.appName ?? 'unknown'}`) + windowContext = ctx.captureContext(false).then( + (result) => { + logger.info(`Screen context captured: ${result.context.appName ?? 'unknown'}`) + return result.context + }, + (err: unknown) => { + logger.debug(`Screen context capture skipped: ${describe(err)}`) + return null + }, + ) } } catch { // 컨텍스트 캡처 실패 시 무시 — 핵심 기능 아님 @@ -320,7 +349,8 @@ class VoiceModeService extends EventEmitter { // 세션 생성 const run = new SessionRun(randomUUID()) - run.captureSelection = captureSelection + run.captureSelection = windowContext !== null + run.windowContext = windowContext this._run = run this._session = { id: run.id, @@ -332,7 +362,7 @@ class VoiceModeService extends EventEmitter { transcription: '', processedText: null, accidentalPress: false, - screenContext, + screenContext: null, } this._setRecognitionState(RecognitionState.PREPARING) @@ -360,16 +390,30 @@ class VoiceModeService extends EventEmitter { await this._startAudio(run) } + /** + * 세션을 정지하고 전사·후처리가 끝날 때까지 기다린다 (IPC 등 공개 API). + * 키 입력 경로는 _requestStop 만 기다려 액션 큐를 막지 않는다. + */ async stopSession(): Promise { + const { finished } = await this._requestStop() + await finished + } + + /** + * 정지 요청과 오디오 해제까지만 수행한다. 전사·LLM 은 run 에 묶어 `finished` 로 따로 돈다. + * 예전엔 정지가 STT·LLM 전체(수 초)를 await 해 액션 큐를 막았고, 그사이 들어온 입력이 + * 세션이 끝난 뒤 다른 의미로 재생됐다(더블탭 정지의 두 번째 탭이 새 핸즈프리 녹음을 시작). + */ + private async _requestStop(): Promise { const run = this._run const session = this._session - if (!run || !session || !this._isCurrent(run) || this._isInTerminalState()) return + if (!run || !session || !this._isCurrent(run) || this._isInTerminalState()) return NOTHING_PENDING // 정지는 세션당 한 번이다. 지연 전사가 도는 중에 두 번째 release/토글이 오면 // 예전엔 비어 있는 버퍼를 보고 'too-short' 로 조용히 취소해 받아쓰기를 잃었다. if (run.stopRequested) { logger.info('Stop already requested for this session — ignoring') - return + return NOTHING_PENDING } run.stopRequested = true @@ -380,23 +424,29 @@ class VoiceModeService extends EventEmitter { session.accidentalPress = true logger.info(`Accidental press detected (${duration}ms < ${TIMING.MIN_AUDIO_DURATION}ms)`) this._cancelSession('too-short') - return + return NOTHING_PENDING } // 오디오 캡처 중지 this._setAudioState(AudioState.STOPPED) await this._releaseAudio(run) - if (!this._isCurrent(run)) return + if (!this._isCurrent(run)) return NOTHING_PENDING // Speakly 패턴: 녹음 종료 후 thinking 상태로 전환 this.emit('phase', { phase: 'thinking', sessionId: run.id }) - // 핫키를 뗀 뒤라 Ctrl+C 가 안전하다 — 전사와 병렬로 선택 텍스트를 캡처한다 - this._startSelectionCapture(run) + // 핫키를 뗀 뒤라 복사 단축키가 안전하다 — LLM 이 쓸 것이 확실하면 전사와 병렬로 선택 텍스트를 + // 캡처한다. 원문 삽입 경로(none·LLM 불가)에선 보내지 않는다: 쓰지도 않을 Ctrl+C 가 터미널에선 + // 프로세스를 중단시킨다. 음성 단축키로 LLM 경로가 될 수도 있으면 LLM 단계에서 늦게 캡처한다. + if (routeUsesLlm(this._resolveLlmRoute(null))) this._startSelectionCapture(run) // 버퍼가 있으면 STT에 전달 if (run.audioBuffer.length > 0 && run.sttReady) { - await this._transcribe(run) + return { + finished: this._transcribe(run).catch((err: unknown) => { + logger.error(`Unhandled error in _transcribe after stop: ${describe(err)}`) + }), + } } else if (run.audioBuffer.length > 0 && !run.sttReady) { // STT 아직 준비 안 됨 → _initSTT 완료 후 _tryFlushAll이 전사 logger.info('Waiting for STT to be ready before transcribing') @@ -415,6 +465,7 @@ class VoiceModeService extends EventEmitter { logger.warn('No audio buffer, cancelling session') this._cancelSession('too-short') } + return NOTHING_PENDING } cancelSession(): void { @@ -601,17 +652,28 @@ class VoiceModeService extends EventEmitter { */ private async _releaseAudio(run: SessionRun): Promise { run.audioReleaseRequested = true - this._detachAudioHandlers(run) - run.audioStarted = false this._stopPartialLoop(run) - if (!run.audioAcquired) return - run.audioAcquired = false - try { - await getAudioCaptureService().stop() - } catch (error) { - logger.warn(`Audio stop error: ${describe(error)}`) + // 현재 세션의 정상 정지면 캡처가 꼬리 오디오(SoX 버퍼에 남은 끝 음절)를 흘려보낼 때까지 + // 핸들러를 붙여 둔다 — 먼저 떼면 그 끝부분을 잃는다. 취소·교체된 run 은 꼬리가 필요 없다. + // (그동안 audioStarted 를 유지해 _tryFlushAll 이 전사를 앞당기지 않게 한다) + const keepForTail = run.audioAcquired && this._isCurrent(run) + if (!keepForTail) { + this._detachAudioHandlers(run) + run.audioStarted = false } + + if (run.audioAcquired) { + run.audioAcquired = false + try { + await getAudioCaptureService().stop() + } catch (error) { + logger.warn(`Audio stop error: ${describe(error)}`) + } + } + + this._detachAudioHandlers(run) + run.audioStarted = false } // ── 스크린 컨텍스트(선택 텍스트) ───────────────────────── @@ -629,18 +691,37 @@ class VoiceModeService extends EventEmitter { })() } - /** 선택 텍스트 캡처가 끝나기를 기다렸다가 세션 컨텍스트에 합친다. */ - private async _applySelectionCapture(run: SessionRun): Promise { - if (!run.selectionCapture) return - const selectedText = await run.selectionCapture - if (!selectedText || !this._isCurrent(run) || !this._session) return - const base: ScreenContext = this._session.screenContext ?? { + /** + * 활성 창 조회(녹음 시작 시점)와 선택 텍스트 캡처(정지 시점)가 끝나기를 기다렸다가 + * 세션 컨텍스트로 합친다. + */ + private async _applyScreenContext(run: SessionRun): Promise { + if (!run.windowContext && !run.selectionCapture) return + const [windowContext, selectedText] = await Promise.all([ + run.windowContext ?? Promise.resolve(null), + run.selectionCapture ?? Promise.resolve(null), + ]) + if (!this._isCurrent(run) || !this._session) return + if (!windowContext && !selectedText) return + const base: ScreenContext = windowContext ?? this._session.screenContext ?? { appName: null, windowTitle: null, selectedText: null, capturedAt: Date.now(), } - this._session.screenContext = { ...base, selectedText } + this._session.screenContext = selectedText ? { ...base, selectedText } : base + } + + /** 세션 컨텍스트를 LLM 입력 접두사로 만든다 (없으면 빈 문자열) */ + private async _buildContextPrefix(): Promise { + const screenContext = this._session?.screenContext + if (!screenContext) return '' + try { + const { getScreenContextService } = await import('./ScreenContextService') + return getScreenContextService().buildContextPrompt(screenContext) + } catch { + return '' + } } // ── 실시간 부분 전사(미리보기) ───────────────────────────── @@ -806,7 +887,6 @@ class VoiceModeService extends EventEmitter { // Phase 10.5: 음성 단축키 — 키워드 매칭으로 LLM 명령어 자동 선택 let effectiveText = result.text - let overrideAction: string | null = null let overrideInstructionId: string | null = null try { const { getVoiceCommandService } = await import('./VoiceCommandService') @@ -815,7 +895,6 @@ class VoiceModeService extends EventEmitter { const match = vcSvc.match(result.text) if (match.matched && match.instructionId) { effectiveText = match.cleanedText - overrideAction = 'custom' overrideInstructionId = match.instructionId logger.info(`Voice command matched: keyword="${match.matchedKeyword}", instruction=${match.instructionId}`) } @@ -825,12 +904,9 @@ class VoiceModeService extends EventEmitter { } if (!this._isCurrent(run)) return - const llmAction = overrideAction ?? configGet('defaultLLMAction') - const backend = configGet('llmBackend') - const ollamaUnavailable = backend === 'local' && !getLocalLLMService().isAvailable() - const skipLLM = llmAction === 'none' || ollamaUnavailable - if (skipLLM) { - if (llmAction !== 'none' && ollamaUnavailable) { + const route = this._resolveLlmRoute(overrideInstructionId) + if (route.kind === 'insert-raw') { + if (route.reason === 'llm-unavailable') { this.emit('error', { error: new D3ROError( ErrorCode.LLMServerUnreachable, @@ -842,7 +918,7 @@ class VoiceModeService extends EventEmitter { } void this._completeSession(run, effectiveText) } else { - await this._processWithLLM(run, effectiveText, overrideInstructionId) + await this._processWithLLM(run, effectiveText, route) } } catch (error) { if (!this._isCurrent(run)) return @@ -874,50 +950,42 @@ class VoiceModeService extends EventEmitter { }) } + /** 현재 설정으로 LLM 후처리 경로를 결정한다 (정책은 @d3ro/core/voice-llm-route 가 소유) */ + private _resolveLlmRoute(overrideInstructionId: string | null): VoiceLlmRoute { + const backend = configGet('llmBackend') + return resolveVoiceLlmRoute({ + defaultAction: configGet('defaultLLMAction'), + overrideInstructionId, + activeChainId: configGet('activeChainId') ?? null, + activeInstructionId: configGet('activeInstructionId') ?? null, + localLlmAvailable: backend === 'local' ? getLocalLLMService().isAvailable() : true, + backend, + }) + } + private async _processWithLLM( run: SessionRun, transcribedText: string, - overrideInstructionId?: string | null, + route: LlmProcessingRoute, ): Promise { if (!this._isCurrent(run)) return try { - const configuredAction = configGet('defaultLLMAction') - // 음성 단축키가 특정 명령을 지목해 들어왔다면 기본 액션이 'none'이어도 처리한다. - // 명시적 요청을 기본 설정이 무효화하면 안 된다. - if (configuredAction === 'none' && !overrideInstructionId) { - void this._completeSession(run, transcribedText) - return - } - - // 여기서 'none'이 남아 있다면 overrideInstructionId가 반드시 있다 → custom 경로. - const action: LLMAction = configuredAction === 'none' ? 'custom' : configuredAction - - // Phase 10.2: 스크린 컨텍스트를 LLM 프롬프트에 주입 (선택 텍스트는 정지 시점 캡처) - await this._applySelectionCapture(run) + // Phase 10.2: 스크린 컨텍스트를 LLM 프롬프트에 주입 (선택 텍스트는 정지 이후 캡처). + // 정지 시점에 LLM 경로가 확실하지 않아 캡처를 미뤘다면 여기서 시작한다(멱등). + this._startSelectionCapture(run) + await this._applyScreenContext(run) if (!this._isCurrent(run)) return - let contextPrefix = '' - if (this._session?.screenContext) { - try { - const { getScreenContextService } = await import('./ScreenContextService') - contextPrefix = getScreenContextService().buildContextPrompt(this._session.screenContext) - } catch { - // 무시 - } - } + const userText = (await this._buildContextPrefix()) + transcribedText // Phase 10.4: 체인 모드 처리 - if (action === 'chain') { + if (route.kind === 'chain') { try { const { getChainService } = await import('./ChainService') - const activeChainId = configGet('activeChainId') - if (activeChainId) { - const chainResult = await getChainService().execute(activeChainId, contextPrefix + transcribedText) - if (!this._isCurrent(run)) return - if (this._session) this._session.processedText = chainResult.finalText - void this._completeSession(run, chainResult.finalText) - return - } + const chainResult = await getChainService().execute(route.chainId, userText) + if (!this._isCurrent(run)) return + if (this._session) this._session.processedText = chainResult.finalText + void this._completeSession(run, chainResult.finalText) } catch (error) { if (!this._isCurrent(run)) return logger.warn(`Chain execution failed: ${describe(error)}`) @@ -927,60 +995,14 @@ class VoiceModeService extends EventEmitter { ? error : new D3ROError(ErrorCode.ChainExecutionFailed, `Chain execution failed: ${describe(error)}`), ) - return } + return } - let processedText: string - - // 음성 단축키 오버라이드 또는 활성 명령어 - const effectiveInstructionId = overrideInstructionId - ?? configGet('activeInstructionId') - - const targetLanguage = resolveTargetLanguage() - - if (action === 'custom' || overrideInstructionId) { - const userText = contextPrefix + transcribedText - let invocationText = userText - let instructionPrompt: string | undefined - - if (effectiveInstructionId) { - const { getCustomInstructionService } = await import('./CustomInstructionService') - const instruction = getCustomInstructionService().getById(effectiveInstructionId) - if (instruction) { - const invocation = buildInstructionInvocation( - instruction.prompt, - userText, - targetLanguage, - ) - invocationText = invocation.text - instructionPrompt = invocation.systemPrompt - logger.info(`Using custom instruction: "${instruction.name}"`) - } else { - logger.warn( - `Custom instruction not found: ${effectiveInstructionId} — processing without an instruction`, - ) - } - } - - processedText = await this._runProcessorWithFallback( - invocationText, - 'custom', - undefined, - instructionPrompt, - ) - } else { - logger.info( - action === 'translate' - ? `Processing with LLM (action: ${action}, targetLanguage: ${targetLanguage})` - : `Processing with LLM (action: ${action})`, - ) - processedText = await this._runProcessorWithFallback( - contextPrefix + transcribedText, - action, - action === 'translate' ? targetLanguage : undefined, - ) - } + const processedText = + route.kind === 'instruction' + ? await this._processWithInstruction(userText, route.instructionId) + : await this._processWithAction(userText, route.action) if (!this._isCurrent(run)) return @@ -1003,6 +1025,45 @@ class VoiceModeService extends EventEmitter { } } + /** 지시문(음성 단축키 또는 활성 지시문)으로 처리한다. 지시문이 없으면 지시문 없이 custom 처리 */ + private async _processWithInstruction(userText: string, instructionId: string | null): Promise { + let invocationText = userText + let instructionPrompt: string | undefined + + if (instructionId) { + const { getCustomInstructionService } = await import('./CustomInstructionService') + const instruction = getCustomInstructionService().getById(instructionId) + if (instruction) { + const invocation = buildInstructionInvocation(instruction.prompt, userText, resolveTargetLanguage()) + invocationText = invocation.text + instructionPrompt = invocation.systemPrompt + logger.info(`Using custom instruction: "${instruction.name}"`) + } else { + logger.warn(`Custom instruction not found: ${instructionId} — processing without an instruction`) + } + } + + return this._runProcessorWithFallback(invocationText, 'custom', undefined, instructionPrompt) + } + + /** 일반 액션(refine/translate/…)으로 처리한다 */ + private async _processWithAction( + userText: string, + action: Extract['action'], + ): Promise { + const targetLanguage = resolveTargetLanguage() + logger.info( + action === 'translate' + ? `Processing with LLM (action: ${action}, targetLanguage: ${targetLanguage})` + : `Processing with LLM (action: ${action})`, + ) + return this._runProcessorWithFallback( + userText, + action, + action === 'translate' ? targetLanguage : undefined, + ) + } + // ── 세션 완료/취소 ───────────────────────────────────── private async _completeSession(run: SessionRun, finalText: string): Promise { @@ -1161,10 +1222,8 @@ class VoiceModeService extends EventEmitter { try { switch (action.type) { case 'press': - await this._handlePress(action) - break case 'release': - await this._handleRelease(action) + await this._handleGesture(action) break case 'escape': this.cancelSession() @@ -1175,48 +1234,29 @@ class VoiceModeService extends EventEmitter { } } - private async _handlePress(action: VoiceAction): Promise { - if (action.mode === 'dictation') { - // Dictation: hold-to-talk — press로 시작 - if (!this.isActive) { - await this.startSession('dictation') - } - } else if (action.mode === 'hands-free') { - // HandsFree: toggle - if (this.isActive) { - await this.stopSession() - } else { - await this.startSession('hands-free') - } - } - } - - private async _handleRelease(action: VoiceAction): Promise { - if (action.holdMode && action.mode === 'dictation' && this.isActive) { - // Dictation: hold-to-talk — release로 즉시 종료 - // Speakly 패턴: 딜레이 없이 즉시 stop (딜레이가 race condition 유발) - await this.stopSession() - } - // HandsFree(토글): release 무시 - } - - // ── 유틸리티 ─────────────────────────────────────────── - - private _resolveMode(actionId: KeyBindingActionId, isDoublePress: boolean): VoiceMode { - if (actionId === 'hands-free' || isDoublePress) { - return 'hands-free' - } - return 'dictation' - } - /** - * 'command' 액션에는 아직 전용 핸들러가 없다. - * 현행 동작대로 dictation 파이프라인으로 fallback 하며, 그 경로는 hold-to-talk 이므로 - * KEYBINDING_ACTIONS 의 holdMode(false)가 아니라 dictation 과 같은 값을 쓴다. + * 키 제스처를 세션 의도로 바꿔(voice-intent) 실행한다. + * 정지는 오디오 해제까지만 기다린다 — 전사·LLM 동안 큐가 풀려 있어야 처리 중에 들어온 토글이 + * 'processing' 단계에서 무시되고, 세션이 끝난 뒤 새 녹음으로 재생되지 않는다. */ - private _resolveHoldMode(actionId: KeyBindingActionId, specHoldMode: boolean): boolean { - if (actionId === 'command') return true - return specHoldMode + private async _handleGesture(action: VoiceAction): Promise { + if (action.type === 'escape') return + const phase = deriveVoiceIntentPhase(this.isActive, this._run?.stopRequested ?? false) + const intent = resolveVoiceIntent( + { type: action.type, mode: action.mode, holdMode: action.holdMode }, + phase, + ) + switch (intent.kind) { + case 'start': + await this.startSession(intent.mode) + break + case 'stop': + // Speakly 패턴: 딜레이 없이 즉시 stop (딜레이가 race condition 유발) + await this._requestStop() + break + case 'ignore': + break + } } // ── 자막 모드 토글 (Phase 10.1) ───────────────────────── diff --git a/apps/desktop/src/main/services/foreground-window-port.ts b/apps/desktop/src/main/services/foreground-window-port.ts new file mode 100644 index 0000000..ee16e26 --- /dev/null +++ b/apps/desktop/src/main/services/foreground-window-port.ts @@ -0,0 +1,156 @@ +// src/main/services/foreground-window-port.ts +// 포그라운드(활성) 창 조회 포트와 플랫폼 어댑터. +// +// ScreenContextService(컨텍스트 정책)는 "지금 앞에 있는 앱이 무엇인가"만 알면 되고, 그것을 어떻게 +// 읽는지(FFI, AppleScript)는 알 필요가 없다. 예전엔 서비스가 PowerShell + Add-Type 스크립트를 직접 +// 들고 있었다 — koffi 리더(utils/win32-foreground.ts)와 중복이었고, 그 사본은 읽기 전용 $PID 에 +// 대입하는 버그로 앱 이름이 늘 'powershell' 이었으며, 호출마다 0.7~3초가 걸렸다. + +import { getLogger } from './LoggerService' +import { getForegroundWindowInfo, type ForegroundWindowInfo } from '../utils/win32-foreground' + +const logger = getLogger('foreground-window') + +export interface ActiveWindowInfo { + /** 프로세스 이름 (확장자 없음, 예: 'chrome'). 알 수 없으면 null */ + appName: string | null + /** 창 제목. 알 수 없으면 null */ + windowTitle: string | null +} + +/** 포그라운드 창 조회 포트. macOS(osascript)는 비동기라 Promise 로 통일한다. */ +export interface ForegroundWindowPort { + current(): Promise +} + +const UNKNOWN: ActiveWindowInfo = { appName: null, windowTitle: null } + +function blankToNull(value: string | null | undefined): string | null { + const trimmed = value?.trim() + return trimmed ? trimmed : null +} + +/** 'Code.exe' → 'Code'. 예전 PowerShell 경로(Process.ProcessName)와 같은 모양을 유지한다. */ +export function toProcessName(exeName: string): string { + return exeName.replace(/\.exe$/iu, '') +} + +/** Windows: koffi FFI 리더(동기, 수 ms)를 감싼다. */ +export function createWin32ForegroundWindowPort( + read: () => ForegroundWindowInfo | null = getForegroundWindowInfo, +): ForegroundWindowPort { + return { + async current() { + const info = read() + if (!info) return UNKNOWN + return { + appName: blankToNull(toProcessName(info.appName)), + windowTitle: blankToNull(info.title), + } + }, + } +} + +type ExecFileText = ( + file: string, + args: string[], + options: { timeout: number }, +) => Promise<{ stdout: string }> + +async function defaultExecFileText( + file: string, + args: string[], + options: { timeout: number }, +): Promise<{ stdout: string }> { + const { execFile } = await import('child_process') + const { promisify } = await import('util') + const { stdout } = await promisify(execFile)(file, args, { ...options, encoding: 'utf8' }) + return { stdout: String(stdout) } +} + +// AppleScript: 프로세스명과 앞 윈도우 타이틀을 2줄로 반환. 윈도우가 없는 앱도 있으므로 try-fallback. +const DARWIN_SCRIPT = ` +try + tell application "System Events" + set frontApp to first process whose frontmost is true + set appName to name of frontApp + try + set winTitle to name of front window of frontApp + on error + set winTitle to "" + end try + return appName & linefeed & winTitle + end tell +on error errMsg + return "" & linefeed & "" +end try +`.trim() + +/** + * macOS: osascript (System Events). Accessibility 권한이 필요하다 — 거부되면 { null, null }. + */ +export function createDarwinForegroundWindowPort( + execFileText: ExecFileText = defaultExecFileText, +): ForegroundWindowPort { + return { + async current() { + try { + const { stdout } = await execFileText('/usr/bin/osascript', ['-e', DARWIN_SCRIPT], { timeout: 3000 }) + const lines = stdout.split('\n') + return { appName: blankToNull(lines[0]), windowTitle: blankToNull(lines[1]) } + } catch (error) { + logger.warn( + `osascript active window query failed: ${error instanceof Error ? error.message : String(error)}`, + ) + return UNKNOWN + } + }, + } +} + +/** 미지원 플랫폼(Linux 등) */ +export function createNullForegroundWindowPort(): ForegroundWindowPort { + return { current: async () => UNKNOWN } +} + +export function createDefaultForegroundWindowPort( + platform: NodeJS.Platform = process.platform, +): ForegroundWindowPort { + if (platform === 'win32') return createWin32ForegroundWindowPort() + if (platform === 'darwin') return createDarwinForegroundWindowPort() + return createNullForegroundWindowPort() +} + +// ── 선택 텍스트 캡처 안전성 ───────────────────────────────── + +/** + * 콘솔/터미널 호스트(소문자 프로세스 이름). 이런 창에서 선택 없이 Ctrl+C 를 보내면 복사가 아니라 + * SIGINT(실행 중 프로세스 중단)나 입력 줄 지우기가 된다 — 선택 텍스트 캡처를 건너뛴다. + */ +const TERMINAL_HOSTS = new Set([ + 'windowsterminal', + 'openconsole', + 'conhost', + 'cmd', + 'powershell', + 'pwsh', + 'powershell_ise', + 'wezterm-gui', + 'alacritty', + 'mintty', + 'conemu', + 'conemu64', + 'kitty', + 'hyper', + 'tabby', + 'warp', + 'terminal', + 'iterm2', + 'iterm', + 'ghostty', +]) + +export function isTerminalHost(appName: string | null): boolean { + if (!appName) return false + return TERMINAL_HOSTS.has(toProcessName(appName).trim().toLowerCase()) +} diff --git a/apps/desktop/src/main/services/voice-intent.ts b/apps/desktop/src/main/services/voice-intent.ts new file mode 100644 index 0000000..c2b213f --- /dev/null +++ b/apps/desktop/src/main/services/voice-intent.ts @@ -0,0 +1,85 @@ +// src/main/services/voice-intent.ts +// 키 제스처 → 음성 세션 의도(start/stop/ignore) 매핑 정책. 순수 함수만 둔다 (서비스·IO 의존 없음). +// +// 예전엔 이 규칙이 VoiceModeService 의 _handlePress / _handleRelease / _resolveMode / +// _resolveHoldMode 에 흩어져 있었고, '처리 중(processing)' 이라는 단계가 코드에 드러나지 않아 +// 액션 큐의 직렬화 순서에 암묵적으로 기댔다. 정지가 전사·LLM 전체를 await 하는 동안 큐에 쌓인 +// 입력이 세션이 끝난 뒤 다른 의미로 재생됐다(더블탭 정지의 두 번째 탭이 새 핸즈프리를 시작). +// 이제 단계를 명시적으로 받아 의도를 돌려준다. 처리 중의 토글·누름은 'ignore' 다. +// +// 남는 한계: 처리가 더블탭 간격(TIMING.DOUBLE_PRESS_DURATION)보다 빨리 끝나 세션이 이미 idle 이면 +// 두 번째 탭은 여전히 새 핸즈프리 세션을 시작한다. 의도 판단에 시간 정보가 필요한 경우로, 이 모듈의 +// 범위 밖이다. + +import type { KeyBindingActionId, VoiceMode } from '@d3ro/core/types' + +/** 의도 판단에 쓰는 세션 단계 */ +export type VoiceIntentPhase = 'idle' | 'recording' | 'processing' + +export interface VoiceGesture { + type: 'press' | 'release' + mode: VoiceMode + /** hold-to-talk 여부 — release 에서 세션을 끊을지 결정한다 */ + holdMode: boolean +} + +export type VoiceIntent = + | { kind: 'start'; mode: VoiceMode } + | { kind: 'stop' } + | { kind: 'ignore' } + +const IGNORE: VoiceIntent = { kind: 'ignore' } +const STOP: VoiceIntent = { kind: 'stop' } + +/** + * 세션 상태에서 의도 판단용 단계를 도출한다. 별도 상태 필드를 두지 않는다. + * @param active 세션이 살아 있다 (비터미널) + * @param stopRequested 이 세션에 정지가 이미 요청됐다 (오디오 해제 이후 = 전사/LLM 처리 중) + */ +export function deriveVoiceIntentPhase(active: boolean, stopRequested: boolean): VoiceIntentPhase { + if (!active) return 'idle' + return stopRequested ? 'processing' : 'recording' +} + +/** + * 키 제스처를 세션 의도로 바꾼다. + * + * | 제스처 | idle | recording | processing | + * |---------------------------------|------------------|-----------|------------| + * | press dictation | start dictation | ignore | ignore | + * | press hands-free (토글) | start hands-free | stop | ignore | + * | release dictation + holdMode | ignore | stop | ignore | + * | release 그 외 (토글의 release) | ignore | ignore | ignore | + * + * 핸즈프리 세션도 dictation release(holdMode) 한 번으로 멈춘다 — 기본 바인딩에서 더블탭 정지의 + * 첫 탭이 이 경로다(두 번째 탭은 processing 에서 ignore). + */ +export function resolveVoiceIntent(gesture: VoiceGesture, phase: VoiceIntentPhase): VoiceIntent { + if (phase === 'processing') return IGNORE + + if (gesture.type === 'press') { + if (gesture.mode === 'dictation') { + return phase === 'idle' ? { kind: 'start', mode: 'dictation' } : IGNORE + } + return phase === 'idle' ? { kind: 'start', mode: 'hands-free' } : STOP + } + + if (gesture.holdMode && gesture.mode === 'dictation' && phase === 'recording') return STOP + return IGNORE +} + +/** 키바인딩 액션 → 세션 모드. 더블탭은 항상 핸즈프리다. */ +export function resolveGestureMode(actionId: KeyBindingActionId, isDoublePress: boolean): VoiceMode { + if (actionId === 'hands-free' || isDoublePress) return 'hands-free' + return 'dictation' +} + +/** + * 'command' 액션에는 아직 전용 핸들러가 없다. + * 현행 동작대로 dictation 파이프라인으로 fallback 하며, 그 경로는 hold-to-talk 이므로 + * KEYBINDING_ACTIONS 의 holdMode(false)가 아니라 dictation 과 같은 값을 쓴다. + */ +export function resolveGestureHoldMode(actionId: KeyBindingActionId, specHoldMode: boolean): boolean { + if (actionId === 'command') return true + return specHoldMode +} diff --git a/apps/desktop/tests/main/services/audio-capture-redteam-r2-7.test.ts b/apps/desktop/tests/main/services/audio-capture-redteam-r2-7.test.ts new file mode 100644 index 0000000..c096513 --- /dev/null +++ b/apps/desktop/tests/main/services/audio-capture-redteam-r2-7.test.ts @@ -0,0 +1,230 @@ +// tests/main/services/audio-capture-redteam-r2-7.test.ts +// AudioCaptureService / 마이크 테스트 IPC 회귀 테스트: +// - kill 된 옛 SoX 의 늦은 'close' 가 새 캡처를 오류로 끝내지 않는다 +// - 마이크 테스트는 자기가 잡은 참조만 놓는다 (회의·자막의 캡처를 빼앗지 않는다) +// - 정지 시 SoX 를 곧바로 죽이지 않고 꼬리 오디오와 잔여 프레임을 흘려보낸다 +// - SoX 버퍼를 줄여(--buffer 1024) 청크 지연·손실을 줄인다 + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { EventEmitter } from 'events' +import { ipcMain } from 'electron' +import { IPC_CHANNELS } from '@d3ro/core/ipc-channels' + +vi.mock('../../../src/main/services/LoggerService', () => ({ + getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }), +})) + +vi.mock('../../../src/main/services/ConfigService', () => ({ + configGet: vi.fn(() => 'default'), + configSet: vi.fn(), +})) + +vi.mock('../../../src/main/utils/paths', () => ({ + getSoxPath: () => 'sox', +})) + +vi.mock('../../../src/main/windows/WindowManager', () => ({ + getMainWindow: () => null, +})) + +class FakeProcess extends EventEmitter { + stdout = new EventEmitter() + stderr = new EventEmitter() + kill = vi.fn() +} + +const spawned = vi.hoisted(() => ({ list: [] as unknown[], args: [] as string[][] })) + +vi.mock('child_process', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + spawn: vi.fn((_cmd: string, args: string[]) => { + const proc = new FakeProcess() + spawned.list.push(proc) + spawned.args.push(args) + return proc + }), + } +}) + +type Module = typeof import('../../../src/main/services/AudioCaptureService') +let mod: Module + +const FRAME = 1920 // 60ms + +function proc(i: number): FakeProcess { + return spawned.list[i] as FakeProcess +} + +beforeEach(async () => { + vi.resetModules() + vi.useRealTimers() + spawned.list = [] + spawned.args = [] + mod = await import('../../../src/main/services/AudioCaptureService') + mod.resetAudioCaptureServiceForTests() +}) + +afterEach(() => { + vi.useRealTimers() +}) + +describe('SoX 프로세스 수명', () => { + it("정지 후 곧바로 재시작하면 옛 SoX 의 늦은 'close' 가 새 캡처를 죽이지 않는다", async () => { + const audio = mod.getAudioCaptureService() + const stopped: string[] = [] + audio.on('stopped', ({ reason }) => stopped.push(reason)) + + await audio.start() + await audio.stop() + await audio.start() + expect(spawned.list).toHaveLength(2) + + // 옛 프로세스의 'close' 가 kill 몇 ms 뒤 도착 + proc(0).emit('close', null) + proc(0).emit('error', new Error('late')) + + expect(audio.state).toBe('capturing') + expect(proc(1).kill).not.toHaveBeenCalled() + expect(stopped).toEqual(['manual']) + }) + + it('SoX 인자에 --buffer 1024 가 들어간다', async () => { + await mod.getAudioCaptureService().start() + const args = spawned.args[0] + const i = args.indexOf('--buffer') + expect(i).toBeGreaterThanOrEqual(0) + expect(args[i + 1]).toBe(String(mod.SOX_BUFFER_BYTES)) + expect(mod.SOX_BUFFER_BYTES).toBe(1024) + }) +}) + +describe('정지 시 꼬리 오디오', () => { + it('정지 중 도착한 청크와 60ms 미만 잔여 바이트를 흘려보낸 뒤 SoX 를 끝낸다', async () => { + const audio = mod.getAudioCaptureService() + const received: number[] = [] + audio.on('audio-data', ({ buffer }) => received.push(buffer.length)) + + await audio.start() + proc(0).stdout.emit('data', Buffer.alloc(FRAME + 100)) // 1 프레임 + 잔여 100 + + const stopping = audio.stop() + expect(proc(0).kill).not.toHaveBeenCalled() + // SoX 버퍼에 남아 있던 끝부분이 정지 직후 도착 + proc(0).stdout.emit('data', Buffer.alloc(FRAME)) + await stopping + + expect(proc(0).kill).toHaveBeenCalledTimes(1) + // 1920 (첫 프레임) + 1920 (잔여 100 + 꼬리 1820) + 잔여 100 (마지막 부분 프레임) + expect(received).toEqual([FRAME, FRAME, 100]) + expect(received.reduce((a, b) => a + b, 0)).toBe(FRAME * 2 + 100) + expect(audio.state).toBe('idle') + }) + + it('꼬리 수집 중 새 소비자가 오면 정지를 취소하고 같은 SoX 로 이어 간다', async () => { + const audio = mod.getAudioCaptureService() + const stopped: string[] = [] + audio.on('stopped', ({ reason }) => stopped.push(reason)) + + await audio.start() + const stopping = audio.stop() + await audio.start() + await stopping + + expect(spawned.list).toHaveLength(1) + expect(proc(0).kill).not.toHaveBeenCalled() + expect(audio.state).toBe('capturing') + expect(stopped).toEqual([]) + + await audio.stop() + expect(proc(0).kill).toHaveBeenCalledTimes(1) + expect(stopped).toEqual(['manual']) + }) +}) + +describe('캡처 임대(lease)', () => { + it('release 는 멱등이고 다른 소비자의 참조를 깎지 않는다', async () => { + const audio = mod.getAudioCaptureService() + await audio.start() // 회의·자막이 잡은 참조 + const lease = await audio.acquire() + + await lease.release() + await lease.release() + + expect(audio.state).toBe('capturing') + expect(proc(0).kill).not.toHaveBeenCalled() + }) + + it('캡처가 죽었다 다시 시작됐으면 옛 임대의 release 는 새 캡처를 멈추지 않는다', async () => { + const audio = mod.getAudioCaptureService() + const lease = await audio.acquire() + proc(0).emit('close', 1) // 장치 분리 + expect(audio.state).toBe('error') + + await audio.start() // 다른 소비자가 새로 시작 + await lease.release() + + expect(audio.state).toBe('capturing') + expect(proc(1).kill).not.toHaveBeenCalled() + }) +}) + +describe('마이크 테스트 IPC', () => { + function handler(channel: string): () => Promise { + const call = vi.mocked(ipcMain.handle).mock.calls.find(([ch]) => ch === channel) + if (!call) throw new Error(`handler not registered: ${channel}`) + return call[1] as unknown as () => Promise + } + + it('회의 캡처 중에 마이크 테스트를 해도 회의 캡처는 끊기지 않는다', async () => { + vi.mocked(ipcMain.handle).mockClear() + const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers') + registerAudioHandlers() + const audio = mod.getAudioCaptureService() + + // 회의(자막) 가 캡처를 잡고 있다 + await audio.start() + const meetingFrames: number[] = [] + audio.on('audio-data', ({ buffer }) => meetingFrames.push(buffer.length)) + + // 설정 창에서 TEST → 곧바로 STOP + await handler(IPC_CHANNELS.AUDIO.TEST_DEVICE)() + await handler(IPC_CHANNELS.AUDIO.STOP_TEST)() + await new Promise((r) => setTimeout(r, mod.TAIL_FLUSH_MS + 20)) + + // 옛 SoX 를 죽이거나 새로 띄우지 않았고 회의 오디오가 계속 들어온다 + expect(spawned.list).toHaveLength(1) + expect(proc(0).kill).not.toHaveBeenCalled() + expect(audio.state).toBe('capturing') + proc(0).stdout.emit('data', Buffer.alloc(FRAME)) + expect(meetingFrames).toEqual([FRAME]) + }) + + it('테스트가 돌고 있지 않을 때의 STOP 은 다른 소비자의 참조를 놓지 않는다', async () => { + vi.mocked(ipcMain.handle).mockClear() + const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers') + registerAudioHandlers() + const audio = mod.getAudioCaptureService() + await audio.start() + + await handler(IPC_CHANNELS.AUDIO.STOP_TEST)() + await new Promise((r) => setTimeout(r, mod.TAIL_FLUSH_MS + 20)) + + expect(audio.state).toBe('capturing') + expect(proc(0).kill).not.toHaveBeenCalled() + }) + + it('테스트 단독이면 STOP 으로 캡처가 끝난다', async () => { + vi.mocked(ipcMain.handle).mockClear() + const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers') + registerAudioHandlers() + const audio = mod.getAudioCaptureService() + + await handler(IPC_CHANNELS.AUDIO.TEST_DEVICE)() + expect(audio.state).toBe('capturing') + await handler(IPC_CHANNELS.AUDIO.STOP_TEST)() + await vi.waitFor(() => expect(audio.state).toBe('idle')) + expect(proc(0).kill).toHaveBeenCalledTimes(1) + }) +}) diff --git a/apps/desktop/tests/main/services/screen-context-redteam-r2-7.test.ts b/apps/desktop/tests/main/services/screen-context-redteam-r2-7.test.ts new file mode 100644 index 0000000..4c162e0 --- /dev/null +++ b/apps/desktop/tests/main/services/screen-context-redteam-r2-7.test.ts @@ -0,0 +1,147 @@ +// tests/main/services/screen-context-redteam-r2-7.test.ts +// ScreenContextService + ForegroundWindowPort 회귀 테스트: +// - 활성 창 조회는 포트(koffi/osascript 어댑터)에 맡기고 PowerShell 을 띄우지 않는다 +// - Windows 어댑터는 실제 앱 이름을 돌려준다 (예전 PowerShell 경로는 $PID 버그로 늘 'powershell') +// - 콘솔/터미널이 포그라운드면 복사 단축키를 보내지 않는다 (SIGINT·입력 줄 지우기 방지) +// - macOS 에선 ⌘+C 를 보낸다 + +import { describe, it, expect, beforeEach, vi } from 'vitest' +import { clipboard } from 'electron' +import type { ForegroundWindowInfo } from '../../../src/main/utils/win32-foreground' + +vi.mock('../../../src/main/services/LoggerService', () => ({ + getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }), +})) + +vi.mock('../../../src/main/services/ConfigService', () => ({ + configGet: vi.fn(() => true), + configSet: vi.fn(), +})) + +vi.mock('../../../src/main/services/modifier-state', () => ({ + waitForModifiersReleased: vi.fn(async () => true), +})) + +const nut = vi.hoisted(() => ({ pressKey: vi.fn(async () => undefined), releaseKey: vi.fn(async () => undefined) })) +vi.mock('@nut-tree-fork/nut-js', () => ({ + keyboard: nut, + Key: { LeftControl: 'LeftControl', LeftSuper: 'LeftSuper', C: 'C' }, +})) + +const childProcess = vi.hoisted(() => ({ execFile: vi.fn(), exec: vi.fn(), spawn: vi.fn() })) +vi.mock('child_process', async (importOriginal) => { + const actual = await importOriginal() + return { ...actual, ...childProcess } +}) + +type ScreenModule = typeof import('../../../src/main/services/ScreenContextService') +type PortModule = typeof import('../../../src/main/services/foreground-window-port') +let screenMod: ScreenModule +let portMod: PortModule + +Object.assign(clipboard, { clear: vi.fn() }) + +function fakePort(appName: string | null, windowTitle: string | null = 'Title'): { current: ReturnType } { + return { current: vi.fn(async () => ({ appName, windowTitle })) } +} + +function info(patch: Partial): ForegroundWindowInfo { + return { hwnd: 1, title: '', processId: 1, appName: '', appPath: null, bounds: null, ...patch } +} + +beforeEach(async () => { + vi.resetModules() + vi.clearAllMocks() + screenMod = await import('../../../src/main/services/ScreenContextService') + portMod = await import('../../../src/main/services/foreground-window-port') +}) + +describe('captureContext — ForegroundWindowPort', () => { + it('포트가 준 앱·창 정보를 쓰고 자식 프로세스를 띄우지 않는다', async () => { + const port = fakePort('Code', 'main.ts - D3ROVoice') + const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' }) + + const { context, selectedTextAttempted } = await svc.captureContext(false) + + expect(context.appName).toBe('Code') + expect(context.windowTitle).toBe('main.ts - D3ROVoice') + expect(selectedTextAttempted).toBe(false) + expect(childProcess.execFile).not.toHaveBeenCalled() + expect(childProcess.spawn).not.toHaveBeenCalled() + expect(svc.buildContextPrompt(context)).toContain('활성 앱: Code') + }) + + it('포트가 실패해도 컨텍스트 캡처는 계속된다', async () => { + const port = { current: vi.fn(async () => { throw new Error('ffi') }) } + const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' }) + const { context } = await svc.captureContext(false) + expect(context.appName).toBeNull() + expect(svc.buildContextPrompt(context)).toBe('') + }) +}) + +describe('Windows 포그라운드 어댑터 (koffi)', () => { + it('실행 파일 이름에서 .exe 를 떼어 실제 앱 이름을 돌려준다', async () => { + const port = portMod.createWin32ForegroundWindowPort(() => + info({ appName: 'Code.exe', title: 'main.ts', appPath: 'C:\\Code\\Code.exe' }), + ) + await expect(port.current()).resolves.toEqual({ appName: 'Code', windowTitle: 'main.ts' }) + }) + + it('빈 값과 조회 실패는 null 이다', async () => { + await expect(portMod.createWin32ForegroundWindowPort(() => info({})).current()).resolves.toEqual({ + appName: null, + windowTitle: null, + }) + await expect(portMod.createWin32ForegroundWindowPort(() => null).current()).resolves.toEqual({ + appName: null, + windowTitle: null, + }) + }) +}) + +describe('선택 텍스트 캡처 — 복사 단축키 안전성', () => { + it.each(['WindowsTerminal', 'conhost', 'pwsh', 'Terminal', 'iTerm2', 'WindowsTerminal.exe'])( + '포그라운드가 터미널(%s)이면 복사 단축키를 보내지 않는다', + async (appName) => { + const svc = new screenMod.ScreenContextService({ foreground: fakePort(appName), platform: 'win32' }) + await expect(svc.captureSelectedText()).resolves.toBeNull() + expect(nut.pressKey).not.toHaveBeenCalled() + expect(clipboard.clear).not.toHaveBeenCalled() + }, + ) + + it('Windows 에선 Ctrl+C 를 보낸다', async () => { + vi.mocked(clipboard.readText).mockReturnValueOnce('').mockReturnValueOnce('선택') + const svc = new screenMod.ScreenContextService({ foreground: fakePort('Code'), platform: 'win32' }) + await expect(svc.captureSelectedText()).resolves.toBe('선택') + expect(nut.pressKey).toHaveBeenCalledWith('LeftControl', 'C') + }) + + it('macOS 에선 ⌘+C 를 보낸다', async () => { + vi.mocked(clipboard.readText).mockReturnValueOnce('').mockReturnValueOnce('선택') + const svc = new screenMod.ScreenContextService({ foreground: fakePort('Notes'), platform: 'darwin' }) + await expect(svc.captureSelectedText()).resolves.toBe('선택') + expect(nut.pressKey).toHaveBeenCalledWith('LeftSuper', 'C') + expect(nut.pressKey).not.toHaveBeenCalledWith('LeftControl', 'C') + }) + + it('captureContext(true) 도 터미널이면 선택 텍스트를 건너뛴다', async () => { + const port = fakePort('WindowsTerminal') + const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' }) + const { context, selectedTextAttempted } = await svc.captureContext(true) + expect(selectedTextAttempted).toBe(true) + expect(context.selectedText).toBeNull() + expect(nut.pressKey).not.toHaveBeenCalled() + // 포그라운드는 한 번만 조회한다 + expect(port.current).toHaveBeenCalledTimes(1) + }) +}) + +describe('isTerminalHost', () => { + it('대소문자·확장자와 무관하게 판별하고, 일반 앱은 아니다', () => { + expect(portMod.isTerminalHost('WINDOWSTERMINAL.EXE')).toBe(true) + expect(portMod.isTerminalHost('Code')).toBe(false) + expect(portMod.isTerminalHost(null)).toBe(false) + }) +}) diff --git a/apps/desktop/tests/main/services/voice-intent.test.ts b/apps/desktop/tests/main/services/voice-intent.test.ts new file mode 100644 index 0000000..0f7ec20 --- /dev/null +++ b/apps/desktop/tests/main/services/voice-intent.test.ts @@ -0,0 +1,84 @@ +// tests/main/services/voice-intent.test.ts +// 키 제스처 → 세션 의도 표 테스트 (더블탭 on/off, hold, 처리 중 토글). + +import { describe, it, expect } from 'vitest' +import { + deriveVoiceIntentPhase, + resolveGestureHoldMode, + resolveGestureMode, + resolveVoiceIntent, + type VoiceGesture, + type VoiceIntent, + type VoiceIntentPhase, +} from '../../../src/main/services/voice-intent' + +const dictationPress: VoiceGesture = { type: 'press', mode: 'dictation', holdMode: true } +const dictationRelease: VoiceGesture = { type: 'release', mode: 'dictation', holdMode: true } +const handsFreePress: VoiceGesture = { type: 'press', mode: 'hands-free', holdMode: false } +const handsFreeRelease: VoiceGesture = { type: 'release', mode: 'hands-free', holdMode: false } + +const START_DICTATION: VoiceIntent = { kind: 'start', mode: 'dictation' } +const START_HANDS_FREE: VoiceIntent = { kind: 'start', mode: 'hands-free' } +const STOP: VoiceIntent = { kind: 'stop' } +const IGNORE: VoiceIntent = { kind: 'ignore' } + +describe('resolveVoiceIntent', () => { + it.each<[string, VoiceGesture, VoiceIntentPhase, VoiceIntent]>([ + // hold-to-talk + ['idle 에서 누르면 받아쓰기를 시작한다', dictationPress, 'idle', START_DICTATION], + ['녹음 중 누름은 무시한다', dictationPress, 'recording', IGNORE], + ['녹음 중 떼면 정지한다', dictationRelease, 'recording', STOP], + ['idle 에서 뗌은 무시한다', dictationRelease, 'idle', IGNORE], + ['처리 중 뗌은 무시한다', dictationRelease, 'processing', IGNORE], + // 핸즈프리 토글 (더블탭) + ['idle 에서 더블탭은 핸즈프리를 켠다', handsFreePress, 'idle', START_HANDS_FREE], + ['녹음 중 더블탭은 끈다', handsFreePress, 'recording', STOP], + ['처리 중 더블탭은 새 녹음을 시작하지 않는다', handsFreePress, 'processing', IGNORE], + ['토글의 release 는 늘 무시한다', handsFreeRelease, 'recording', IGNORE], + ['토글의 release 는 idle 에서도 무시한다', handsFreeRelease, 'idle', IGNORE], + // 더블탭 정지의 첫 탭(dictation press/release)이 핸즈프리 세션을 멈춘다 (기존 동작 보존) + ['핸즈프리 녹음 중 첫 탭 press 는 무시', dictationPress, 'recording', IGNORE], + ['핸즈프리 녹음 중 첫 탭 release 가 정지', dictationRelease, 'recording', STOP], + ['처리 중 받아쓰기 누름은 무시', dictationPress, 'processing', IGNORE], + ])('%s', (_label, gesture, phase, expected) => { + expect(resolveVoiceIntent(gesture, phase)).toEqual(expected) + }) + + it('holdMode 가 아닌 dictation release 는 정지하지 않는다', () => { + expect(resolveVoiceIntent({ type: 'release', mode: 'dictation', holdMode: false }, 'recording')).toEqual(IGNORE) + }) + + it('더블탭 on → 더블탭 off 시퀀스: 두 번째 탭은 처리 중에 무시된다', () => { + // on + expect(resolveVoiceIntent(handsFreePress, 'idle')).toEqual(START_HANDS_FREE) + // off: 첫 탭 press/release, 정지 후 처리 중에 두 번째 탭 + expect(resolveVoiceIntent(dictationPress, 'recording')).toEqual(IGNORE) + expect(resolveVoiceIntent(dictationRelease, 'recording')).toEqual(STOP) + expect(resolveVoiceIntent(handsFreePress, 'processing')).toEqual(IGNORE) + expect(resolveVoiceIntent(handsFreeRelease, 'processing')).toEqual(IGNORE) + }) +}) + +describe('deriveVoiceIntentPhase', () => { + it('세션 상태에서 단계를 도출한다', () => { + expect(deriveVoiceIntentPhase(false, false)).toBe('idle') + expect(deriveVoiceIntentPhase(false, true)).toBe('idle') + expect(deriveVoiceIntentPhase(true, false)).toBe('recording') + expect(deriveVoiceIntentPhase(true, true)).toBe('processing') + }) +}) + +describe('제스처 입력 정규화', () => { + it('더블탭이거나 hands-free 액션이면 핸즈프리다', () => { + expect(resolveGestureMode('dictation', false)).toBe('dictation') + expect(resolveGestureMode('dictation', true)).toBe('hands-free') + expect(resolveGestureMode('hands-free', false)).toBe('hands-free') + expect(resolveGestureMode('command', false)).toBe('dictation') + }) + + it("'command' 는 dictation 파이프라인 fallback 이라 hold-to-talk 이다", () => { + expect(resolveGestureHoldMode('command', false)).toBe(true) + expect(resolveGestureHoldMode('dictation', true)).toBe(true) + expect(resolveGestureHoldMode('hands-free', false)).toBe(false) + }) +}) diff --git a/apps/desktop/tests/main/services/voice-mode-redteam-r2-7.test.ts b/apps/desktop/tests/main/services/voice-mode-redteam-r2-7.test.ts new file mode 100644 index 0000000..e5971cd --- /dev/null +++ b/apps/desktop/tests/main/services/voice-mode-redteam-r2-7.test.ts @@ -0,0 +1,373 @@ +// tests/main/services/voice-mode-redteam-r2-7.test.ts +// VoiceModeService 회귀 테스트 (round 2 #7): +// - 더블탭으로 핸즈프리를 끄면 두 번째 탭이 처리 뒤 새 녹음을 시작하지 않는다 (큐가 막히지 않음) +// - 스크린 컨텍스트의 창 조회가 세션·마이크 시작을 막지 않는다 +// - 원문 삽입 경로(none)에선 선택 텍스트용 복사 단축키를 보내지 않는다 +// - '체인' 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다 +// - 정지 시 캡처가 흘려보내는 꼬리 오디오를 받는다 + +import { describe, it, expect, beforeEach, vi } from 'vitest' +import { EventEmitter } from 'events' +import type { KeyBindingTriggerPayload } from '../../../src/main/services/KeyBindingService' + +vi.mock('../../../src/main/services/LoggerService', () => ({ + getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }), +})) + +interface Deferred { + promise: Promise + resolve: (value: T) => void +} + +function deferred(): Deferred { + let resolve!: (value: T) => void + const promise = new Promise((res) => { + resolve = res + }) + return { promise, resolve } +} + +type SttResult = { text: string; segments: never[]; language: string; duration: number; processingTime: number } +const result = (text: string): SttResult => ({ text, segments: [], language: 'ko', duration: 1, processingTime: 1 }) + +const mockSTT = vi.hoisted(() => ({ + initialize: vi.fn(), + transcribe: vi.fn(), + transcribePartial: vi.fn(async () => ''), + getModels: vi.fn(), + getStatus: vi.fn(() => ({ engineState: 'ready', activeModel: 'base', engineVersion: null, gpuAccelerated: false })), +})) + +vi.mock('../../../src/main/services/LocalSTTService', () => ({ + getLocalSTTService: () => mockSTT, + resetLocalSTTServiceForTests: () => undefined, +})) + +const audioBus = new EventEmitter() +const mockAudio = vi.hoisted(() => ({ start: vi.fn(), stop: vi.fn(), on: vi.fn(), off: vi.fn() })) +vi.mock('../../../src/main/services/AudioCaptureService', () => ({ + getAudioCaptureService: () => mockAudio, +})) + +const keyBinding = vi.hoisted(() => ({ + handler: null as ((payload: KeyBindingTriggerPayload) => void) | null, +})) +vi.mock('../../../src/main/services/KeyBindingService', () => ({ + getKeyBindingService: () => ({ + on: (_ev: string, fn: (payload: KeyBindingTriggerPayload) => void) => { + keyBinding.handler = fn + }, + off: vi.fn(), + }), +})) + +const config = vi.hoisted(() => ({ values: {} as Record })) +vi.mock('../../../src/main/services/ConfigService', () => ({ + configGet: vi.fn((key: string) => config.values[key]), +})) + +const mockInsert = vi.hoisted(() => ({ insertText: vi.fn() })) +vi.mock('../../../src/main/services/TextInsertService', () => ({ + getTextInsertService: () => mockInsert, +})) + +vi.mock('../../../src/main/services/LocalLLMService', () => ({ + getLocalLLMService: () => ({ isAvailable: () => true }), +})) + +const gateway = vi.hoisted(() => ({ processText: vi.fn() })) +vi.mock('../../../src/main/services/llm/LlmGateway', () => ({ + getLlmGateway: () => gateway, +})) + +vi.mock('../../../src/main/services/CaptionService', () => ({ + getCaptionService: () => ({ getState: () => 'inactive', stop: vi.fn(), start: vi.fn() }), +})) + +const chain = vi.hoisted(() => ({ execute: vi.fn() })) +vi.mock('../../../src/main/services/ChainService', () => ({ + getChainService: () => chain, +})) + +const instructions = vi.hoisted(() => ({ + byId: {} as Record, +})) +vi.mock('../../../src/main/services/CustomInstructionService', () => ({ + getCustomInstructionService: () => ({ getById: (id: string) => instructions.byId[id] ?? null }), +})) + +const voiceCommand = vi.hoisted(() => ({ instructionId: null as string | null })) +vi.mock('../../../src/main/services/VoiceCommandService', () => ({ + getVoiceCommandService: () => ({ + isEnabled: () => voiceCommand.instructionId !== null, + match: (text: string) => + voiceCommand.instructionId + ? { matched: true, ruleId: 'r', instructionId: voiceCommand.instructionId, cleanedText: text, matchedKeyword: 'k' } + : { matched: false, ruleId: null, instructionId: null, cleanedText: text, matchedKeyword: null }, + }), +})) + +const screen = vi.hoisted(() => ({ + enabled: false, + captureContext: vi.fn(), + captureSelectedText: vi.fn(), +})) +vi.mock('../../../src/main/services/ScreenContextService', () => ({ + getScreenContextService: () => ({ + isEnabled: () => screen.enabled, + captureContext: screen.captureContext, + captureSelectedText: screen.captureSelectedText, + buildContextPrompt: (ctx: { appName: string | null; selectedText: string | null }) => + `[ctx app=${ctx.appName ?? '-'} sel=${ctx.selectedText ?? '-'}]\n`, + }), +})) + +type VoiceModeModule = typeof import('../../../src/main/services/VoiceModeService') +let getVoiceModeService: VoiceModeModule['getVoiceModeService'] + +const SPEECH = Buffer.alloc(16000 * 2) // 1초 + +async function flush(times = 8): Promise { + for (let i = 0; i < times; i++) await new Promise((r) => setImmediate(r)) +} + +function advanceClock(ms: number): void { + const base = Date.now() + vi.spyOn(Date, 'now').mockReturnValue(base + ms) +} + +function key( + actionId: 'dictation' | 'hands-free', + type: 'pressed' | 'released', + isDoublePress: boolean, +): KeyBindingTriggerPayload { + return { + actionId, + type, + isDoublePress, + holdMode: actionId === 'dictation', + timestamp: Date.now(), + durationMs: 0, + } as unknown as KeyBindingTriggerPayload +} + +function screenContext(appName: string): { context: { appName: string; windowTitle: string; selectedText: null; capturedAt: number }; selectedTextAttempted: boolean } { + return { context: { appName, windowTitle: 'w', selectedText: null, capturedAt: 0 }, selectedTextAttempted: false } +} + +beforeEach(async () => { + vi.restoreAllMocks() + vi.resetModules() + audioBus.removeAllListeners() + config.values = { sttModelId: 'base', defaultLLMAction: 'none', autoInsert: true, llmBackend: 'local' } + keyBinding.handler = null + screen.enabled = false + voiceCommand.instructionId = null + instructions.byId = {} + for (const fn of [ + mockSTT.initialize, mockSTT.transcribe, mockSTT.transcribePartial, mockSTT.getModels, + mockAudio.start, mockAudio.stop, mockAudio.on, mockAudio.off, mockInsert.insertText, + gateway.processText, chain.execute, screen.captureContext, screen.captureSelectedText, + ]) { + fn.mockReset() + } + mockSTT.initialize.mockResolvedValue(undefined) + mockSTT.transcribe.mockResolvedValue(result('기본 전사')) + mockSTT.transcribePartial.mockResolvedValue('') + mockSTT.getModels.mockReturnValue([ + { id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 }, + ]) + mockAudio.start.mockResolvedValue(undefined) + mockAudio.stop.mockResolvedValue(undefined) + mockAudio.on.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => { + audioBus.on(ev, fn) + }) + mockAudio.off.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => { + audioBus.off(ev, fn) + }) + mockInsert.insertText.mockResolvedValue({ success: true, method: 'clipboard', textLength: 1, durationMs: 1 }) + gateway.processText.mockImplementation(async (text: string) => `LLM(${text})`) + chain.execute.mockResolvedValue({ finalText: '체인 결과' }) + screen.captureContext.mockResolvedValue(screenContext('Code')) + screen.captureSelectedText.mockResolvedValue('선택 문장') + const mod = await import('../../../src/main/services/VoiceModeService') + mod.resetVoiceModeServiceForTests() + getVoiceModeService = mod.getVoiceModeService +}) + +describe('더블탭 핸즈프리 정지', () => { + it('정지 더블탭의 두 번째 탭은 전사가 끝난 뒤 새 핸즈프리 녹음을 시작하지 않는다', async () => { + const stt = deferred() + mockSTT.transcribe.mockReturnValueOnce(stt.promise) + const svc = getVoiceModeService() + svc.connectKeyBindings() + const completed: string[] = [] + svc.on('session-completed', ({ finalText }) => completed.push(finalText)) + const emit = (p: KeyBindingTriggerPayload): void => keyBinding.handler?.(p) + + // 더블탭으로 핸즈프리 켜기 + emit(key('hands-free', 'pressed', true)) + emit(key('hands-free', 'released', true)) + await vi.waitFor(() => expect(mockAudio.start).toHaveBeenCalledTimes(1)) + await vi.waitFor(() => expect(svc.isActive).toBe(true)) + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(3000) + + // 더블탭으로 끄기: 첫 탭(dictation press/release) → 정지, 두 번째 탭(hands-free press) + emit(key('dictation', 'pressed', false)) + emit(key('dictation', 'released', false)) + await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)) + emit(key('hands-free', 'pressed', true)) + emit(key('hands-free', 'released', true)) + await flush() + + // 전사 완료 + stt.resolve(result('핸즈프리 받아쓰기')) + await vi.waitFor(() => expect(completed).toEqual(['핸즈프리 받아쓰기'])) + await flush() + + expect(mockAudio.start).toHaveBeenCalledTimes(1) + expect(svc.isActive).toBe(false) + }) + + it('공개 stopSession() 은 여전히 전사·삽입이 끝날 때까지 기다린다', async () => { + const svc = getVoiceModeService() + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + expect(mockSTT.transcribe).toHaveBeenCalledTimes(1) + expect(svc.isActive).toBe(false) + }) +}) + +describe('스크린 컨텍스트 — 창 조회가 세션을 막지 않는다', () => { + it('창 조회가 끝나기 전에 세션과 마이크가 시작되고, 결과는 LLM 입력에 합쳐진다', async () => { + screen.enabled = true + config.values.defaultLLMAction = 'refine' + const probe = deferred>() + screen.captureContext.mockReturnValueOnce(probe.promise) + const svc = getVoiceModeService() + const started: string[] = [] + svc.on('session-started', ({ session }) => started.push(session.id)) + + const starting = svc.startSession('dictation') + await Promise.race([starting, new Promise((r) => setTimeout(r, 300))]) + expect(started).toHaveLength(1) + expect(mockAudio.start).toHaveBeenCalledTimes(1) + expect(screen.captureContext).toHaveBeenCalledWith(false) + + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + const stopping = svc.stopSession() + probe.resolve(screenContext('Code')) + await stopping + await starting + + expect(gateway.processText).toHaveBeenCalledTimes(1) + const [text, action] = gateway.processText.mock.calls[0] as [string, string] + expect(action).toBe('refine') + expect(text).toBe('[ctx app=Code sel=선택 문장]\n기본 전사') + }) +}) + +describe('스크린 컨텍스트 — 선택 텍스트 캡처', () => { + it("원문 삽입(none) 경로에선 복사 단축키(선택 텍스트 캡처)를 보내지 않는다", async () => { + screen.enabled = true + config.values.defaultLLMAction = 'none' + const svc = getVoiceModeService() + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + await flush() + + expect(mockInsert.insertText).toHaveBeenCalledWith('기본 전사', undefined) + expect(screen.captureSelectedText).not.toHaveBeenCalled() + }) + + it('LLM 경로면 정지 시점에 선택 텍스트를 캡처한다', async () => { + screen.enabled = true + config.values.defaultLLMAction = 'refine' + const svc = getVoiceModeService() + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + expect(screen.captureSelectedText).toHaveBeenCalledTimes(1) + }) + + it("none 이어도 음성 단축키로 LLM 경로가 되면 그때 선택 텍스트를 캡처해 쓴다", async () => { + screen.enabled = true + config.values.defaultLLMAction = 'none' + voiceCommand.instructionId = 'i1' + instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' } + const svc = getVoiceModeService() + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + + expect(screen.captureSelectedText).toHaveBeenCalledTimes(1) + expect(gateway.processText).toHaveBeenCalledTimes(1) + const [text, action] = gateway.processText.mock.calls[0] as [string, string] + expect(action).toBe('custom') + expect(text).toContain('[ctx app=Code sel=선택 문장]') + }) +}) + +describe('LLM 경로 — 체인 + 음성 단축키', () => { + it('체인 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다', async () => { + config.values.defaultLLMAction = 'chain' + config.values.activeChainId = 'c1' + voiceCommand.instructionId = 'i1' + instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' } + const svc = getVoiceModeService() + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + + expect(chain.execute).not.toHaveBeenCalled() + expect(gateway.processText).toHaveBeenCalledTimes(1) + const [, action, , systemPrompt] = gateway.processText.mock.calls[0] as [string, string, unknown, string] + expect(action).toBe('custom') + expect(systemPrompt).toContain('요약해줘') + }) + + it('음성 단축키가 없으면 활성 체인을 실행한다', async () => { + config.values.defaultLLMAction = 'chain' + config.values.activeChainId = 'c1' + const svc = getVoiceModeService() + const completed: string[] = [] + svc.on('session-completed', ({ finalText }) => completed.push(finalText)) + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + + expect(chain.execute).toHaveBeenCalledWith('c1', '기본 전사') + expect(gateway.processText).not.toHaveBeenCalled() + expect(completed).toEqual(['체인 결과']) + }) +}) + +describe('정지 시 꼬리 오디오', () => { + it('캡처가 정지 중에 흘려보낸 꼬리 오디오까지 전사에 넣는다', async () => { + const TAIL = Buffer.alloc(3200) // 100ms + mockAudio.stop.mockImplementation(async () => { + // AudioCaptureService.stop() 은 SoX 를 죽이기 전에 남은 오디오를 emit 한다 + audioBus.emit('audio-data', { buffer: TAIL }) + }) + const svc = getVoiceModeService() + await svc.startSession('dictation') + audioBus.emit('audio-data', { buffer: SPEECH }) + advanceClock(1000) + await svc.stopSession() + + expect(mockSTT.transcribe).toHaveBeenCalledTimes(1) + const [audio] = mockSTT.transcribe.mock.calls[0] as [Buffer] + expect(audio.length).toBe(SPEECH.length + TAIL.length) + expect(audioBus.listenerCount('audio-data')).toBe(0) + }) +}) diff --git a/apps/desktop/tests/main/services/windows-child-process-hide.test.ts b/apps/desktop/tests/main/services/windows-child-process-hide.test.ts index c3af191..7ac40be 100644 --- a/apps/desktop/tests/main/services/windows-child-process-hide.test.ts +++ b/apps/desktop/tests/main/services/windows-child-process-hide.test.ts @@ -38,6 +38,19 @@ vi.mock('../../../src/main/services/PremiumLLMService', () => ({ getPremiumLLMService: vi.fn(), })) +// 활성 창 조회는 koffi FFI 리더(ForegroundWindowPort)로 한다 — 자식 프로세스를 띄우지 않는다. +const foregroundReader = vi.hoisted(() => ({ + getForegroundWindowInfo: vi.fn(() => ({ + hwnd: 1, + title: 'Test window', + processId: 1, + appName: 'TestApp.exe', + appPath: null, + })), +})) + +vi.mock('../../../src/main/utils/win32-foreground', () => foregroundReader) + function createProcess(args: unknown[]): EventEmitter & { stdout: EventEmitter stderr: EventEmitter @@ -128,8 +141,8 @@ describe('Windows child process visibility', () => { expectWindowsHideOnAllCalls(childProcess.exec) }) - // 장치 목록·활성 창 조회는 win32 분기에서만 PowerShell 을 부른다. - it.skipIf(process.platform !== 'win32')('hides Windows device discovery and active-window PowerShell calls', async () => { + // 장치 목록 조회는 win32 분기에서만 PowerShell 을 부른다. 활성 창 조회는 FFI 로 하므로 프로세스가 없다. + it.skipIf(process.platform !== 'win32')('hides Windows device discovery and spawns no process for the active window', async () => { const audioModule = await import('../../../src/main/services/AudioCaptureService') const audioService = audioModule.getAudioCaptureService() as unknown as { _getDevicesWindows(): Promise @@ -145,10 +158,9 @@ describe('Windows child process visibility', () => { timeout: 3000, windowsHide: true, }) - expect(childProcess.execFileAsync.mock.calls[0]?.[2]).toMatchObject({ - timeout: 3000, - windowsHide: true, - }) + expect(foregroundReader.getForegroundWindowInfo).toHaveBeenCalled() + expect(childProcess.execFileAsync).not.toHaveBeenCalled() + expect(childProcess.execFile).not.toHaveBeenCalled() }) it('declares hidden ffmpeg processes for conversion, duration, and chunk extraction', () => { diff --git a/packages/core/__tests__/voice-llm-route.test.ts b/packages/core/__tests__/voice-llm-route.test.ts new file mode 100644 index 0000000..fe1e57f --- /dev/null +++ b/packages/core/__tests__/voice-llm-route.test.ts @@ -0,0 +1,68 @@ +// packages/core/__tests__/voice-llm-route.test.ts +// 받아쓰기 LLM 경로 결정 우선순위 표 테스트 (override > chain > instruction > action > none). + +import { describe, expect, it } from 'vitest' +import { resolveVoiceLlmRoute, routeUsesLlm, type VoiceLlmRouteInput } from '../src/voice-llm-route' + +const base: VoiceLlmRouteInput = { + defaultAction: 'refine', + overrideInstructionId: null, + activeChainId: null, + activeInstructionId: null, + localLlmAvailable: true, + backend: 'local', +} + +function route(patch: Partial): ReturnType { + return resolveVoiceLlmRoute({ ...base, ...patch }) +} + +describe('resolveVoiceLlmRoute — 우선순위 표', () => { + it.each([ + // [설명, 입력, 기대 경로] + ['override 는 체인보다 앞선다', { defaultAction: 'chain', activeChainId: 'c1', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }], + ['override 는 none 을 이긴다', { defaultAction: 'none', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }], + ['override 는 일반 액션을 이긴다', { defaultAction: 'translate', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }], + ['override 는 활성 지시문보다 앞선다', { defaultAction: 'custom', activeInstructionId: 'i1', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }], + ['활성 체인이 있으면 체인', { defaultAction: 'chain', activeChainId: 'c1' }, { kind: 'chain', chainId: 'c1' }], + ['체인인데 활성 체인이 없으면 일반 액션 chain (기존 동작)', { defaultAction: 'chain', activeChainId: null }, { kind: 'action', action: 'chain' }], + ['custom 은 활성 지시문', { defaultAction: 'custom', activeInstructionId: 'i1' }, { kind: 'instruction', instructionId: 'i1', source: 'active' }], + ['custom 인데 활성 지시문이 없으면 지시문 없이 custom', { defaultAction: 'custom', activeInstructionId: '' }, { kind: 'instruction', instructionId: null, source: 'active' }], + ['활성 지시문은 일반 액션을 덮어쓰지 않는다', { defaultAction: 'refine', activeInstructionId: 'i1' }, { kind: 'action', action: 'refine' }], + ['translate 는 일반 액션', { defaultAction: 'translate' }, { kind: 'action', action: 'translate' }], + ['none 은 원문 삽입', { defaultAction: 'none' }, { kind: 'insert-raw', reason: 'none' }], + ] as const)('%s', (_label, patch, expected) => { + expect(route(patch as Partial)).toEqual(expected) + }) +}) + +describe('resolveVoiceLlmRoute — 로컬 LLM 가용성', () => { + it('로컬 백엔드에서 LLM 을 쓸 수 없으면 사유를 밝혀 원문 삽입', () => { + expect(route({ localLlmAvailable: false })).toEqual({ kind: 'insert-raw', reason: 'llm-unavailable' }) + expect(route({ localLlmAvailable: false, overrideInstructionId: 'i9' })).toEqual({ + kind: 'insert-raw', + reason: 'llm-unavailable', + }) + expect(route({ localLlmAvailable: false, defaultAction: 'chain', activeChainId: 'c1' })).toEqual({ + kind: 'insert-raw', + reason: 'llm-unavailable', + }) + }) + + it("none 은 LLM 가용성과 무관하게 'none' 사유다 (경고 없음)", () => { + expect(route({ defaultAction: 'none', localLlmAvailable: false })).toEqual({ kind: 'insert-raw', reason: 'none' }) + }) + + it('온라인 백엔드는 로컬 가용성을 보지 않는다', () => { + expect(route({ backend: 'online', localLlmAvailable: false })).toEqual({ kind: 'action', action: 'refine' }) + expect(route({ backend: undefined, localLlmAvailable: false })).toEqual({ kind: 'action', action: 'refine' }) + }) +}) + +describe('routeUsesLlm', () => { + it('원문 삽입만 LLM 을 쓰지 않는다', () => { + expect(routeUsesLlm({ kind: 'insert-raw', reason: 'none' })).toBe(false) + expect(routeUsesLlm({ kind: 'chain', chainId: 'c' })).toBe(true) + expect(routeUsesLlm({ kind: 'action', action: 'refine' })).toBe(true) + }) +}) diff --git a/packages/core/src/voice-llm-route.ts b/packages/core/src/voice-llm-route.ts new file mode 100644 index 0000000..d6d9bfb --- /dev/null +++ b/packages/core/src/voice-llm-route.ts @@ -0,0 +1,89 @@ +// packages/core/src/voice-llm-route.ts +// 받아쓰기 세션의 LLM 후처리 경로 결정 — 순수 함수만 둔다 (IO·설정 저장소 의존 없음). +// +// 예전엔 같은 결정을 VoiceModeService._transcribe(스킵 여부)와 _processWithLLM(무엇을 실행할지)이 +// 서로 다른 규칙으로 두 번 내렸다. 두 사본이 어긋나 '체인' 기본 액션에서 음성 단축키가 지목한 +// 지시문이 조용히 무시됐다. 이제 경로는 여기서 한 번만 정하고, 서비스는 결과에 따라 IO 만 실행한다. +// +// 우선순위: 음성 단축키(override) > 체인 > 지시문(custom) > 일반 액션 > none +// - 지시문 경로는 기본 액션이 'custom' 이거나 override 가 있을 때만 탄다. +// (활성 지시문이 있어도 refine/translate 등 일반 액션을 덮어쓰지 않는다) +// - 체인을 골랐는데 활성 체인이 없으면 일반 액션 'chain' 으로 LLM 에 넘긴다(기존 동작 유지). +// - 로컬 백엔드인데 로컬 LLM 을 쓸 수 없으면 원문 삽입으로 떨어지고, 그 이유를 밝힌다 +// (이 경우에만 호출자가 경고를 띄운다). + +import type { LLMAction, LLMActionSelection } from './types' + +export type VoiceLlmBackend = 'local' | 'online' + +export interface VoiceLlmRouteInput { + /** 설정의 기본 후처리 액션 */ + defaultAction: LLMActionSelection + /** 음성 단축키가 지목한 지시문 id (없으면 null) */ + overrideInstructionId: string | null + /** 활성 체인 id */ + activeChainId: string | null + /** 활성 지시문 id ('custom' 기본 액션에서만 쓴다) */ + activeInstructionId: string | null + /** 로컬 LLM 사용 가능 여부 */ + localLlmAvailable: boolean + /** LLM 백엔드 — 알 수 없으면 undefined (가용성 검사를 건너뛴다) */ + backend: VoiceLlmBackend | undefined +} + +export type VoiceLlmRoute = + /** LLM 없이 원문을 삽입한다 */ + | { kind: 'insert-raw'; reason: 'none' | 'llm-unavailable' } + /** 활성 체인을 실행한다 */ + | { kind: 'chain'; chainId: string } + /** 지시문('custom')으로 처리한다. instructionId 가 null 이면 지시문 없이 custom 처리 */ + | { kind: 'instruction'; instructionId: string | null; source: 'override' | 'active' } + /** 일반 액션으로 처리한다 */ + | { kind: 'action'; action: Exclude } + +function nonEmpty(value: string | null | undefined): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +/** LLM 이 필요한 경로를 고른다 (가용성 검사 전). */ +function resolveDesiredRoute(input: VoiceLlmRouteInput): VoiceLlmRoute { + const override = nonEmpty(input.overrideInstructionId) + if (override) { + // 명시적 음성 명령은 기본 설정(none·chain 포함)이 무효화하지 않는다 + return { kind: 'instruction', instructionId: override, source: 'override' } + } + + switch (input.defaultAction) { + case 'none': + return { kind: 'insert-raw', reason: 'none' } + case 'chain': { + const chainId = nonEmpty(input.activeChainId) + return chainId ? { kind: 'chain', chainId } : { kind: 'action', action: 'chain' } + } + case 'custom': + return { + kind: 'instruction', + instructionId: nonEmpty(input.activeInstructionId), + source: 'active', + } + default: + return { kind: 'action', action: input.defaultAction } + } +} + +/** 이 경로가 LLM 호출을 필요로 하는가 */ +export function routeUsesLlm(route: VoiceLlmRoute): boolean { + return route.kind !== 'insert-raw' +} + +/** + * 받아쓰기 한 건의 LLM 후처리 경로를 결정한다. + */ +export function resolveVoiceLlmRoute(input: VoiceLlmRouteInput): VoiceLlmRoute { + const desired = resolveDesiredRoute(input) + if (!routeUsesLlm(desired)) return desired + if (input.backend === 'local' && !input.localLlmAvailable) { + return { kind: 'insert-raw', reason: 'llm-unavailable' } + } + return desired +}