fix(voice): stop mic-test ref stealing, unblock action queue, keep tail audio, and move context/LLM routing behind ports and pure policies

This commit is contained in:
Yun Chan 2026-09-28 02:16:16 +09:00
parent 2cd462333f
commit 2f94d24c99
13 changed files with 1672 additions and 353 deletions

View file

@ -3,7 +3,7 @@
import { ipcMain } from 'electron'
import { IPC_CHANNELS } from '@d3ro/core/ipc-channels'
import { ipcSuccess, ipcError, ErrorCode } from '@d3ro/core/errors'
import { getAudioCaptureService, calculateRMS } from '../services/AudioCaptureService'
import { getAudioCaptureService, calculateRMS, type CaptureLease } from '../services/AudioCaptureService'
import { getMainWindow } from '../windows/WindowManager'
import { configGet, configSet } from '../services/ConfigService'
import type { SetDeviceParams } from '@d3ro/core/types'
@ -12,6 +12,11 @@ let testTimer: ReturnType<typeof setTimeout> | null = null
let testAudioHandler: ((payload: { buffer: Buffer; timestamp: number }) => void) | null = null
let testLevelInterval: ReturnType<typeof setInterval> | null = null
let testLastRms = 0
/**
* 테스트가 실제로 잡은 캡처 참조. 테스트가 돌고 있지 않을 때 stop() 을 부르면 회의·자막·받아쓰기가
* 잡고 있는 참조를 빼앗아 그쪽 마이크가 조용히 끊긴다 — 테스트는 자기 임대만 놓는다.
*/
let testLease: CaptureLease | null = null
function stopAudioTest(): void {
const service = getAudioCaptureService()
@ -28,7 +33,9 @@ function stopAudioTest(): void {
testTimer = null
}
testLastRms = 0
service.stop().catch(() => { /* ignore */ })
const lease = testLease
testLease = null
lease?.release().catch(() => { /* ignore */ })
}
export function registerAudioHandlers(): void {
@ -62,7 +69,7 @@ export function registerAudioHandlers(): void {
stopAudioTest()
const service = getAudioCaptureService()
await service.start()
testLease = await service.acquire()
// 오디오 데이터에서 RMS 추적
testAudioHandler = (payload: { buffer: Buffer; timestamp: number }) => {

View file

@ -18,8 +18,44 @@ const logger = getLogger('AudioCaptureService')
/** 60ms 프레임 크기 (바이트): 16000 * 2 * 0.06 = 1920 */
const FRAME_SIZE_BYTES = (AUDIO_FORMAT.SAMPLE_RATE * AUDIO_FORMAT.BYTES_PER_SAMPLE * 60) / 1000
/**
* SoX stdout 버퍼 크기(바이트). SoX 기본값 8192 는 16kHz 16bit mono 에서 256ms 라, 청크가 늦게 오고
* 정지 시 버퍼에 남은 최대 256ms(끝 음절)가 버려졌다. 1024 = 32ms.
*/
export const SOX_BUFFER_BYTES = 1024
/**
* 마지막 참조를 놓을 때 SoX 를 죽이기 전에 버퍼·파이프에 남은 꼬리 오디오를 받아 내는 시간.
* SoX 버퍼(32ms)의 두 배 이상이어야 정지 직전까지의 소리가 한 번은 쓰여 나온다.
*/
export const TAIL_FLUSH_MS = 80
export type CaptureState = 'idle' | 'starting' | 'capturing' | 'stopping' | 'error'
/**
* 캡처 참조 한 개. release() 는 멱등이고, 그사이 캡처가 죽었다 다시 시작됐으면(세대가 바뀜)
* 다른 소비자의 참조를 깎지 않는다.
*/
export interface CaptureLease {
release(): Promise<void>
}
/** SoX 녹음 인자 (Windows: -t waveaudio 필수, 그 외: -d) */
export function buildSoxRecordArgs(deviceArg: string, platform: NodeJS.Platform = process.platform): string[] {
const input = platform === 'win32' ? ['-t', 'waveaudio', deviceArg] : ['-d']
return [
'--buffer', String(SOX_BUFFER_BYTES),
...input,
'--no-show-progress',
'-r', String(AUDIO_FORMAT.SAMPLE_RATE),
'-c', String(AUDIO_FORMAT.CHANNELS),
'-b', '16',
'-e', 'signed-integer',
'-t', 'raw',
'-',
]
}
interface AudioCaptureEvents {
'audio-data': (payload: { buffer: Buffer; timestamp: number }) => void
'audio-level': (payload: { level: number; timestamp: number }) => void
@ -59,6 +95,12 @@ class AudioCaptureService extends EventEmitter {
private _soxProcess: ChildProcess | null = null
private _stream: Readable | null = null
private _levelInterval: ReturnType<typeof setInterval> | null = null
/** 새 SoX 를 띄울 때마다 증가 — 임대(lease)가 자기 캡처인지 판단한다 */
private _generation = 0
/** 진행 중인 꼬리 수집 정지의 토큰 (0 = 없음). start() 가 0 으로 만들어 정지를 취소한다 */
private _pendingStopToken = 0
private _stopSeq = 0
private _cancelTailWait: (() => void) | null = null
/** 프레임 조립용 잔여 바이트 버퍼 */
private _residualBuffer: Buffer = Buffer.alloc(0)
@ -83,6 +125,28 @@ class AudioCaptureService extends EventEmitter {
return this._currentDevice
}
/**
* 캡처 참조를 하나 잡고 임대(lease)로 돌려준다. 참조를 잡지 않은 쪽이 stop() 을 불러
* 다른 소비자(회의·자막·받아쓰기)의 참조를 빼앗는 일을 막는다.
*/
async acquire(deviceId?: string): Promise<CaptureLease> {
await this.start(deviceId)
if (this._state !== 'capturing') {
throw new D3ROError(ErrorCode.AudioCaptureStartFailed, `Audio capture not running (state: ${this._state})`)
}
const generation = this._generation
let released = false
return {
release: async () => {
if (released) return
released = true
// 캡처가 죽었다가(장치 분리 등) 다시 시작됐으면 이 임대는 이미 무효다
if (this._generation !== generation) return
await this.stop()
},
}
}
async start(deviceId?: string): Promise<void> {
if (this._state === 'capturing') {
this._refCount++
@ -90,6 +154,17 @@ class AudioCaptureService extends EventEmitter {
return
}
// 마지막 참조가 꼬리를 받는 중에 새 소비자가 오면 정지를 취소하고 같은 SoX 로 이어 간다
if (this._state === 'stopping' && this._pendingStopToken !== 0) {
this._pendingStopToken = 0
this._cancelTailWait?.()
this._state = 'capturing'
this._refCount = 1
this._startLevelTimer()
logger.debug('Pending stop cancelled by a new consumer — capture continues')
return
}
if (this._state !== 'idle' && this._state !== 'error') {
logger.warn(`Cannot start capture in state: ${this._state}`)
return
@ -119,33 +194,15 @@ class AudioCaptureService extends EventEmitter {
? selectedDeviceId
: 'default'
// Windows: sox -t waveaudio <device> --no-show-progress -r 16000 -c 1 -b 16 -e signed-integer -t raw -
// Linux/Mac: sox -d --no-show-progress ...
const soxArgs = process.platform === 'win32'
? [
'-t', 'waveaudio', deviceArg,
'--no-show-progress',
'-r', String(AUDIO_FORMAT.SAMPLE_RATE),
'-c', String(AUDIO_FORMAT.CHANNELS),
'-b', '16',
'-e', 'signed-integer',
'-t', 'raw',
'-',
]
: [
'-d',
'--no-show-progress',
'-r', String(AUDIO_FORMAT.SAMPLE_RATE),
'-c', String(AUDIO_FORMAT.CHANNELS),
'-b', '16',
'-e', 'signed-integer',
'-t', 'raw',
'-',
]
// Windows: sox --buffer 1024 -t waveaudio <device> --no-show-progress -r 16000 -c 1 -b 16 -e signed-integer -t raw -
// Linux/Mac: sox --buffer 1024 -d --no-show-progress ...
const soxArgs = buildSoxRecordArgs(deviceArg)
logger.info(`SoX args: ${soxArgs.join(' ')}`)
this._soxProcess = spawn(soxExe, soxArgs, { stdio: ['pipe', 'pipe', 'pipe'], windowsHide: true })
this._stream = this._soxProcess.stdout
const child = spawn(soxExe, soxArgs, { stdio: ['pipe', 'pipe', 'pipe'], windowsHide: true })
this._soxProcess = child
this._generation++
this._stream = child.stdout
this._residualBuffer = Buffer.alloc(0)
this._levelAccumulator = []
@ -159,13 +216,16 @@ class AudioCaptureService extends EventEmitter {
})
// SoX stderr → 로그
this._soxProcess.stderr?.on('data', (chunk: Buffer) => {
child.stderr?.on('data', (chunk: Buffer) => {
const msg = chunk.toString().trim()
if (msg) logger.warn(`SoX stderr: ${msg}`)
})
// SoX 프로세스 종료 감지
this._soxProcess.on('close', (code) => {
// SoX 프로세스 종료 감지.
// 자기 프로세스인지 먼저 확인한다 — 예전엔 kill 한 옛 SoX 의 'close' 가 몇 ms 뒤에 와서
// 그사이 시작된 새 캡처('capturing')를 오류로 끝냈다(회의 중 마이크 테스트로 회의 마이크가 끊김).
child.on('close', (code) => {
if (this._soxProcess !== child) return
if (this._state === 'capturing') {
logger.warn(`SoX process exited unexpectedly (code: ${code})`)
this._handleError(
@ -175,7 +235,8 @@ class AudioCaptureService extends EventEmitter {
}
})
this._soxProcess.on('error', (err: Error) => {
child.on('error', (err: Error) => {
if (this._soxProcess !== child) return
logger.error(`SoX process spawn error: ${err.message}`)
const hint =
soxExe === 'sox' && (err as NodeJS.ErrnoException).code === 'ENOENT'
@ -197,9 +258,7 @@ class AudioCaptureService extends EventEmitter {
this._refCount = 1
// 100ms 간격으로 audio-level 이벤트 emit
this._levelInterval = setInterval(() => {
this._emitAudioLevel()
}, TIMING.AUDIO_LEVEL_INTERVAL)
this._startLevelTimer()
this.emit('started', { deviceId: this._currentDevice.deviceId })
logger.info('Audio capture started')
@ -225,7 +284,17 @@ class AudioCaptureService extends EventEmitter {
return
}
// 곧바로 kill 하면 SoX 버퍼·파이프에 남은 끝부분(끝 음절)이 버려진다.
// 잠깐 더 읽어 소비자에게 흘려보낸 뒤 정리한다. 그사이 start() 가 오면 정지를 취소한다.
this._state = 'stopping'
this._stopLevelTimer()
const token = ++this._stopSeq
this._pendingStopToken = token
await this._waitForTail()
if (this._pendingStopToken !== token || this._state !== 'stopping') return
this._pendingStopToken = 0
this._emitResidual()
this._cleanup()
this._state = 'idle'
this._currentDevice = null
@ -233,6 +302,48 @@ class AudioCaptureService extends EventEmitter {
logger.info('Audio capture stopped')
}
/** TAIL_FLUSH_MS 만큼(또는 SoX 가 먼저 끝날 때까지, 또는 정지가 취소될 때까지) 기다린다 */
private _waitForTail(): Promise<void> {
const child = this._soxProcess
return new Promise<void>((resolve) => {
let done = false
const finish = (): void => {
if (done) return
done = true
clearTimeout(timer)
child?.off('close', finish)
if (this._cancelTailWait === finish) this._cancelTailWait = null
resolve()
}
const timer = setTimeout(finish, TAIL_FLUSH_MS)
child?.once('close', finish)
this._cancelTailWait = finish
})
}
/** 60ms 프레임에 못 미친 잔여 바이트를 마지막 프레임으로 내보낸다 (샘플 경계로 자름) */
private _emitResidual(): void {
const usable = this._residualBuffer.length - (this._residualBuffer.length % AUDIO_FORMAT.BYTES_PER_SAMPLE)
if (usable <= 0) return
const frame = Buffer.from(this._residualBuffer.subarray(0, usable))
this._residualBuffer = Buffer.alloc(0)
this.emit('audio-data', { buffer: frame, timestamp: Date.now() })
}
private _startLevelTimer(): void {
this._stopLevelTimer()
this._levelInterval = setInterval(() => {
this._emitAudioLevel()
}, TIMING.AUDIO_LEVEL_INTERVAL)
}
private _stopLevelTimer(): void {
if (this._levelInterval) {
clearInterval(this._levelInterval)
this._levelInterval = null
}
}
async getDevices(): Promise<AudioDevice[]> {
// 캐싱: 네이티브 프로세스 호출이 느리므로 한번만 실행
if (this._cachedDevices) return this._cachedDevices
@ -378,6 +489,8 @@ class AudioCaptureService extends EventEmitter {
}
dispose(): void {
this._pendingStopToken = 0
this._cancelTailWait?.()
this._cleanup()
this._state = 'idle'
this._currentDevice = null
@ -396,13 +509,7 @@ class AudioCaptureService extends EventEmitter {
? selectedDeviceId
: 'default'
const soxArgs = process.platform === 'win32'
? ['-t', 'waveaudio', deviceArg, '--no-show-progress',
'-r', String(AUDIO_FORMAT.SAMPLE_RATE), '-c', String(AUDIO_FORMAT.CHANNELS),
'-b', '16', '-e', 'signed-integer', '-t', 'raw', '-']
: ['-d', '--no-show-progress',
'-r', String(AUDIO_FORMAT.SAMPLE_RATE), '-c', String(AUDIO_FORMAT.CHANNELS),
'-b', '16', '-e', 'signed-integer', '-t', 'raw', '-']
const soxArgs = buildSoxRecordArgs(deviceArg)
return new Promise((resolve, reject) => {
const proc = spawn(soxExe, soxArgs, { stdio: ['ignore', 'pipe', 'pipe'], windowsHide: true })
@ -505,6 +612,7 @@ class AudioCaptureService extends EventEmitter {
this._cleanup()
this._state = 'error'
this._refCount = 0
this._currentDevice = null
this.emit('error', { error })
this.emit('stopped', { reason: stopReason })
@ -514,10 +622,7 @@ class AudioCaptureService extends EventEmitter {
* 녹음 프로세스와 타이머를 정리한다.
*/
private _cleanup(): void {
if (this._levelInterval) {
clearInterval(this._levelInterval)
this._levelInterval = null
}
this._stopLevelTimer()
if (this._stream) {
this._stream.removeAllListeners()
@ -525,12 +630,19 @@ class AudioCaptureService extends EventEmitter {
}
if (this._soxProcess) {
const child = this._soxProcess
this._soxProcess = null
// 이 프로세스의 늦은 'close'/'error' 가 다음 캡처를 건드리지 못하게 떼어 낸다.
// 'error' 는 리스너가 없으면 throw 하므로 빈 sink 를 남긴다.
child.removeAllListeners('close')
child.removeAllListeners('error')
child.on('error', () => { /* 정리된 프로세스의 늦은 에러 무시 */ })
child.stderr?.removeAllListeners('data')
try {
this._soxProcess.kill()
child.kill()
} catch {
// 이미 종료된 프로세스 kill 시 에러 무시
}
this._soxProcess = null
}
this._residualBuffer = Buffer.alloc(0)

View file

@ -1,6 +1,8 @@
// src/main/services/ScreenContextService.ts
// 활성 윈도우 정보 + 선택된 텍스트를 캡처하여 LLM 프롬프트에 컨텍스트로 제공한다.
// Phase 10.2 스크린 컨텍스트. Speakly ContextService 참조.
// 활성 창 조회는 ForegroundWindowPort(foreground-window-port.ts)에 맡긴다 — 이 서비스는 컨텍스트
// 정책(무엇을 언제 캡처해 어떻게 프롬프트로 만드는가)만 소유한다.
import { clipboard } from 'electron'
import { getLogger } from './LoggerService'
@ -8,6 +10,11 @@ import { configGet, configSet } from './ConfigService'
import { D3ROError, ErrorCode } from '@d3ro/core/errors'
import type { ScreenContext, CaptureContextResult } from '@d3ro/core/types'
import { waitForModifiersReleased } from './modifier-state'
import {
createDefaultForegroundWindowPort,
isTerminalHost,
type ForegroundWindowPort,
} from './foreground-window-port'
const logger = getLogger('screen-context')
@ -30,6 +37,13 @@ interface NutKeyboard {
Key: NutModule['Key']
}
export interface ScreenContextServiceDeps {
/** 포그라운드 창 조회 (기본: 플랫폼 어댑터) */
foreground?: ForegroundWindowPort
/** 복사 단축키 수정자 결정용 플랫폼 (기본: process.platform) */
platform?: NodeJS.Platform
}
// ============================================================
// 클립보드 스냅샷 (TextInsertService와 동일 패턴)
// ============================================================
@ -46,9 +60,16 @@ interface ClipboardSnapshot {
// ScreenContextService
// ============================================================
class ScreenContextService {
export class ScreenContextService {
private _nutKeyboard: NutKeyboard | null = null
private _nutLoadPromise: Promise<void> | null = null
private readonly _foreground: ForegroundWindowPort
private readonly _platform: NodeJS.Platform
constructor(deps: ScreenContextServiceDeps = {}) {
this._platform = deps.platform ?? process.platform
this._foreground = deps.foreground ?? createDefaultForegroundWindowPort(this._platform)
}
/**
* nut-js lazy dynamic import (TextInsertService 패턴).
@ -90,7 +111,7 @@ class ScreenContextService {
/**
* 활성 윈도우 정보 + 선택된 텍스트를 캡처한다.
*
* @param captureSelectedText true이면 Ctrl+C로 선택 텍스트 캡처 시도
* @param captureSelectedText true이면 복사 단축키로 선택 텍스트 캡처 시도
*/
async captureContext(captureSelectedText = true): Promise<CaptureContextResult> {
const capturedAt = Date.now()
@ -99,9 +120,9 @@ class ScreenContextService {
let selectedText: string | null = null
let selectedTextAttempted = false
// 1. 활성 윈도우 정보 (PowerShell)
// 1. 활성 윈도우 정보 (ForegroundWindowPort)
try {
const appInfo = await this._getActiveWindowInfo()
const appInfo = await this._foreground.current()
appName = appInfo.appName
windowTitle = appInfo.windowTitle
} catch (error) {
@ -115,7 +136,7 @@ class ScreenContextService {
if (captureSelectedText) {
selectedTextAttempted = true
try {
selectedText = await this.captureSelectedText()
selectedText = await this._captureSelectionIn(appName)
} catch (error) {
logger.warn(
`Failed to capture selected text: ${error instanceof Error ? error.message : String(error)}`
@ -142,13 +163,30 @@ class ScreenContextService {
* 선택된 텍스트만 캡처한다 (클립보드 방식).
* 수정자 키가 눌려 있으면 뗄 때까지 잠깐 기다리고, 그래도 쥐고 있으면 캡처하지 않고 null 을
* 돌려준다(클립보드도 건드리지 않는다).
* 포그라운드가 콘솔/터미널이면 복사 단축키를 보내지 않는다 — 선택이 없을 때 Ctrl+C 는
* 실행 중 프로세스를 중단(SIGINT)하거나 입력 줄을 지운다.
*/
async captureSelectedText(): Promise<string | null> {
let foregroundApp: string | null = null
try {
foregroundApp = (await this._foreground.current()).appName
} catch {
// 조회 실패 시엔 앱을 모른 채 진행한다
}
return this._captureSelectionIn(foregroundApp)
}
/** 포그라운드 앱을 이미 알고 있을 때의 선택 텍스트 캡처 */
private async _captureSelectionIn(foregroundApp: string | null): Promise<string | null> {
const released = await waitForModifiersReleased(MODIFIER_RELEASE_TIMEOUT_MS)
if (!released) {
logger.info('Selected text capture skipped: modifier keys still held')
return null
}
if (isTerminalHost(foregroundApp)) {
logger.info(`Selected text capture skipped: terminal host (${foregroundApp ?? 'unknown'})`)
return null
}
return this._captureSelectedText()
}
@ -202,132 +240,9 @@ class ScreenContextService {
// ── 내부 구현 ─────────────────────────────────────────
/**
* 활성 윈도우의 프로세스명과 타이틀을 가져온다.
* - Windows: PowerShell + user32.dll
* - macOS: osascript (System Events)
* - Linux: 미지원 (null 반환)
*
* 참고: macOS는 Accessibility 권한이 필요하다. 첫 호출 시 시스템이
* 권한 요청 다이얼로그를 띄운다. 사용자가 거부하면 { null, null } 반환.
*/
private async _getActiveWindowInfo(): Promise<{
appName: string | null
windowTitle: string | null
}> {
if (process.platform === 'win32') {
return this._getActiveWindowInfoWin32()
}
if (process.platform === 'darwin') {
return this._getActiveWindowInfoDarwin()
}
return { appName: null, windowTitle: null }
}
/**
* Windows: PowerShell + user32.dll로 활성 윈도우 조회.
*/
private async _getActiveWindowInfoWin32(): Promise<{
appName: string | null
windowTitle: string | null
}> {
const { execFile } = await import('child_process')
const { promisify } = await import('util')
const execFileAsync = promisify(execFile)
const psScript = `
Add-Type @"
using System;
using System.Runtime.InteropServices;
using System.Text;
public class Win32 {
[DllImport("user32.dll")] public static extern IntPtr GetForegroundWindow();
[DllImport("user32.dll")] public static extern uint GetWindowThreadProcessId(IntPtr hWnd, out uint lpdwProcessId);
[DllImport("user32.dll", CharSet = CharSet.Unicode)] public static extern int GetWindowText(IntPtr hWnd, StringBuilder lpString, int nMaxCount);
}
"@
$hwnd = [Win32]::GetForegroundWindow()
$pid = 0
[void][Win32]::GetWindowThreadProcessId($hwnd, [ref]$pid)
$proc = Get-Process -Id $pid -ErrorAction SilentlyContinue
$sb = New-Object System.Text.StringBuilder 512
[void][Win32]::GetWindowText($hwnd, $sb, 512)
$procName = if ($proc) { $proc.ProcessName } else { '' }
$title = $sb.ToString()
"$procName$([char]10)$title"
`.trim()
try {
const { stdout } = await execFileAsync('powershell.exe', [
'-NoProfile',
'-NonInteractive',
'-ExecutionPolicy', 'Bypass',
'-Command', psScript,
], { timeout: 3000, windowsHide: true })
const lines = stdout.trim().split('\n')
const appName = lines[0]?.trim() || null
const windowTitle = lines[1]?.trim() || null
return { appName, windowTitle }
} catch (error) {
logger.warn(
`PowerShell active window query failed: ${error instanceof Error ? error.message : String(error)}`
)
return { appName: null, windowTitle: null }
}
}
/**
* macOS: osascript (AppleScript)로 frontmost process와 윈도우 타이틀 조회.
* Accessibility 권한 필요.
*/
private async _getActiveWindowInfoDarwin(): Promise<{
appName: string | null
windowTitle: string | null
}> {
const { execFile } = await import('child_process')
const { promisify } = await import('util')
const execFileAsync = promisify(execFile)
// AppleScript: 프로세스명과 앞 윈도우 타이틀을 2줄로 반환.
// 윈도우가 없는 앱도 있으므로 try-fallback.
const script = `
try
tell application "System Events"
set frontApp to first process whose frontmost is true
set appName to name of frontApp
try
set winTitle to name of front window of frontApp
on error
set winTitle to ""
end try
return appName & linefeed & winTitle
end tell
on error errMsg
return "" & linefeed & ""
end try
`.trim()
try {
const { stdout } = await execFileAsync('/usr/bin/osascript', ['-e', script], {
timeout: 3000
})
const lines = stdout.split('\n')
const appName = lines[0]?.trim() || null
const windowTitle = lines[1]?.trim() || null
return { appName, windowTitle }
} catch (error) {
logger.warn(
`osascript active window query failed: ${error instanceof Error ? error.message : String(error)}`
)
return { appName: null, windowTitle: null }
}
}
/**
* 선택된 텍스트를 클립보드 방식으로 캡처한다.
* TextInsertService의 역방향: clipboard save → Ctrl+C simulate → clipboard read → clipboard restore
* TextInsertService의 역방향: clipboard save → 복사 단축키 simulate → clipboard read → clipboard restore
*/
private async _captureSelectedText(): Promise<string | null> {
const nut = await this._ensureNut()
@ -339,9 +254,10 @@ end try
// 2. 클립보드 비우기 (이전 내용이 남아있으면 "선택 없음"을 감지할 수 없으므로)
clipboard.clear()
// 3. Ctrl+C 시뮬레이션
await nut.pressKey(nut.Key.LeftControl, nut.Key.C)
await nut.releaseKey(nut.Key.LeftControl, nut.Key.C)
// 3. 복사 단축키 시뮬레이션 (macOS ⌘+C, 그 외 Ctrl+C)
const modifier = this._platform === 'darwin' ? nut.Key.LeftSuper : nut.Key.LeftControl
await nut.pressKey(modifier, nut.Key.C)
await nut.releaseKey(modifier, nut.Key.C)
// 4. 클립보드에 텍스트가 복사될 때까지 대기
await this._sleep(150)

View file

@ -35,6 +35,13 @@ import { TIMING } from '@d3ro/core/constants'
import { RecognitionState, AudioState } from '@d3ro/core/types'
import type { KeyBindingActionId, VoiceMode, VoiceState, LLMAction } from '@d3ro/core/types'
import type { ScreenContext } from '@d3ro/core/types'
import { resolveVoiceLlmRoute, routeUsesLlm, type VoiceLlmRoute } from '@d3ro/core/voice-llm-route'
import {
deriveVoiceIntentPhase,
resolveGestureHoldMode,
resolveGestureMode,
resolveVoiceIntent,
} from './voice-intent'
const logger = getLogger('VoiceModeService')
@ -126,6 +133,8 @@ class SessionRun {
/** 스크린 컨텍스트: 선택 텍스트 캡처 여부와 진행 중 작업 */
captureSelection = false
selectionCapture: Promise<string | null> | null = null
/** 스크린 컨텍스트: 녹음 시작 시점 활성 창 조회 (기다리지 않고 LLM 단계에서 합친다) */
windowContext: Promise<ScreenContext | null> | null = null
dataHandler: ((payload: { buffer: Buffer }) => void) | null = null
levelHandler: ((payload: { level: number }) => void) | null = null
@ -169,6 +178,19 @@ function describe(err: unknown): string {
return err instanceof Error ? err.message : String(err)
}
/** LLM 을 실제로 부르는 경로 */
type LlmProcessingRoute = Exclude<VoiceLlmRoute, { kind: 'insert-raw' }>
/**
* 정지 요청의 결과. 오디오 해제까지는 요청 안에서 끝나고, 전사·LLM 후처리는 `finished` 로 따로 돈다.
* (async 함수가 Promise 를 반환하면 펼쳐지므로 객체로 감싼다)
*/
interface StopOutcome {
finished: Promise<void>
}
const NOTHING_PENDING: StopOutcome = { finished: Promise.resolve() }
// ============================================================
// VoiceModeService
// ============================================================
@ -229,11 +251,11 @@ class VoiceModeService extends EventEmitter {
return
}
const holdMode = this._resolveHoldMode(payload.actionId, payload.holdMode)
const holdMode = resolveGestureHoldMode(payload.actionId, payload.holdMode)
const action: VoiceAction = {
type: payload.type === 'pressed' ? 'press' : 'release',
timestamp: payload.timestamp,
mode: this._resolveMode(payload.actionId, payload.isDoublePress),
mode: resolveGestureMode(payload.actionId, payload.isDoublePress),
actionId: payload.actionId,
holdMode
}
@ -300,16 +322,23 @@ class VoiceModeService extends EventEmitter {
// 선택 텍스트(Ctrl+C)는 여기서 보내지 않는다: 이 시점엔 사용자가 아직 핫키(기본 Right Alt)를
// 누르고 있어 대상 앱에 Ctrl+Alt+C 가 들어가고(복사 안 됨, AltGr 배열에선 문자 입력),
// 클립보드만 비워진다. 선택 텍스트는 녹음 종료(키를 뗀 뒤)에 캡처한다.
let screenContext: ScreenContext | null = null
let captureSelection = false
// 활성 창 조회는 기다리지 않는다 — 예전엔 조회(PowerShell, 0.7~3초)가 세션·마이크 시작 앞을
// 막아 첫 말이 잘리고 짧은 hold-to-talk 가 'too-short' 로 취소됐다. 결과는 LLM 단계에서 합친다.
let windowContext: Promise<ScreenContext | null> | null = null
try {
const { getScreenContextService } = await import('./ScreenContextService')
const ctx = getScreenContextService()
if (ctx.isEnabled()) {
captureSelection = true
const result = await ctx.captureContext(false)
screenContext = result.context
logger.info(`Screen context captured: ${screenContext.appName ?? 'unknown'}`)
windowContext = ctx.captureContext(false).then(
(result) => {
logger.info(`Screen context captured: ${result.context.appName ?? 'unknown'}`)
return result.context
},
(err: unknown) => {
logger.debug(`Screen context capture skipped: ${describe(err)}`)
return null
},
)
}
} catch {
// 컨텍스트 캡처 실패 시 무시 — 핵심 기능 아님
@ -320,7 +349,8 @@ class VoiceModeService extends EventEmitter {
// 세션 생성
const run = new SessionRun(randomUUID())
run.captureSelection = captureSelection
run.captureSelection = windowContext !== null
run.windowContext = windowContext
this._run = run
this._session = {
id: run.id,
@ -332,7 +362,7 @@ class VoiceModeService extends EventEmitter {
transcription: '',
processedText: null,
accidentalPress: false,
screenContext,
screenContext: null,
}
this._setRecognitionState(RecognitionState.PREPARING)
@ -360,16 +390,30 @@ class VoiceModeService extends EventEmitter {
await this._startAudio(run)
}
/**
* 세션을 정지하고 전사·후처리가 끝날 때까지 기다린다 (IPC 등 공개 API).
* 키 입력 경로는 _requestStop 만 기다려 액션 큐를 막지 않는다.
*/
async stopSession(): Promise<void> {
const { finished } = await this._requestStop()
await finished
}
/**
* 정지 요청과 오디오 해제까지만 수행한다. 전사·LLM 은 run 에 묶어 `finished` 로 따로 돈다.
* 예전엔 정지가 STT·LLM 전체(수 초)를 await 해 액션 큐를 막았고, 그사이 들어온 입력이
* 세션이 끝난 뒤 다른 의미로 재생됐다(더블탭 정지의 두 번째 탭이 새 핸즈프리 녹음을 시작).
*/
private async _requestStop(): Promise<StopOutcome> {
const run = this._run
const session = this._session
if (!run || !session || !this._isCurrent(run) || this._isInTerminalState()) return
if (!run || !session || !this._isCurrent(run) || this._isInTerminalState()) return NOTHING_PENDING
// 정지는 세션당 한 번이다. 지연 전사가 도는 중에 두 번째 release/토글이 오면
// 예전엔 비어 있는 버퍼를 보고 'too-short' 로 조용히 취소해 받아쓰기를 잃었다.
if (run.stopRequested) {
logger.info('Stop already requested for this session — ignoring')
return
return NOTHING_PENDING
}
run.stopRequested = true
@ -380,23 +424,29 @@ class VoiceModeService extends EventEmitter {
session.accidentalPress = true
logger.info(`Accidental press detected (${duration}ms < ${TIMING.MIN_AUDIO_DURATION}ms)`)
this._cancelSession('too-short')
return
return NOTHING_PENDING
}
// 오디오 캡처 중지
this._setAudioState(AudioState.STOPPED)
await this._releaseAudio(run)
if (!this._isCurrent(run)) return
if (!this._isCurrent(run)) return NOTHING_PENDING
// Speakly 패턴: 녹음 종료 후 thinking 상태로 전환
this.emit('phase', { phase: 'thinking', sessionId: run.id })
// 핫키를 뗀 뒤라 Ctrl+C 가 안전하다 — 전사와 병렬로 선택 텍스트를 캡처한다
this._startSelectionCapture(run)
// 핫키를 뗀 뒤라 복사 단축키가 안전하다 — LLM 이 쓸 것이 확실하면 전사와 병렬로 선택 텍스트를
// 캡처한다. 원문 삽입 경로(none·LLM 불가)에선 보내지 않는다: 쓰지도 않을 Ctrl+C 가 터미널에선
// 프로세스를 중단시킨다. 음성 단축키로 LLM 경로가 될 수도 있으면 LLM 단계에서 늦게 캡처한다.
if (routeUsesLlm(this._resolveLlmRoute(null))) this._startSelectionCapture(run)
// 버퍼가 있으면 STT에 전달
if (run.audioBuffer.length > 0 && run.sttReady) {
await this._transcribe(run)
return {
finished: this._transcribe(run).catch((err: unknown) => {
logger.error(`Unhandled error in _transcribe after stop: ${describe(err)}`)
}),
}
} else if (run.audioBuffer.length > 0 && !run.sttReady) {
// STT 아직 준비 안 됨 → _initSTT 완료 후 _tryFlushAll이 전사
logger.info('Waiting for STT to be ready before transcribing')
@ -415,6 +465,7 @@ class VoiceModeService extends EventEmitter {
logger.warn('No audio buffer, cancelling session')
this._cancelSession('too-short')
}
return NOTHING_PENDING
}
cancelSession(): void {
@ -601,11 +652,18 @@ class VoiceModeService extends EventEmitter {
*/
private async _releaseAudio(run: SessionRun): Promise<void> {
run.audioReleaseRequested = true
this._detachAudioHandlers(run)
run.audioStarted = false
this._stopPartialLoop(run)
if (!run.audioAcquired) return
// 현재 세션의 정상 정지면 캡처가 꼬리 오디오(SoX 버퍼에 남은 끝 음절)를 흘려보낼 때까지
// 핸들러를 붙여 둔다 — 먼저 떼면 그 끝부분을 잃는다. 취소·교체된 run 은 꼬리가 필요 없다.
// (그동안 audioStarted 를 유지해 _tryFlushAll 이 전사를 앞당기지 않게 한다)
const keepForTail = run.audioAcquired && this._isCurrent(run)
if (!keepForTail) {
this._detachAudioHandlers(run)
run.audioStarted = false
}
if (run.audioAcquired) {
run.audioAcquired = false
try {
await getAudioCaptureService().stop()
@ -614,6 +672,10 @@ class VoiceModeService extends EventEmitter {
}
}
this._detachAudioHandlers(run)
run.audioStarted = false
}
// ── 스크린 컨텍스트(선택 텍스트) ─────────────────────────
private _startSelectionCapture(run: SessionRun): void {
@ -629,18 +691,37 @@ class VoiceModeService extends EventEmitter {
})()
}
/** 선택 텍스트 캡처가 끝나기를 기다렸다가 세션 컨텍스트에 합친다. */
private async _applySelectionCapture(run: SessionRun): Promise<void> {
if (!run.selectionCapture) return
const selectedText = await run.selectionCapture
if (!selectedText || !this._isCurrent(run) || !this._session) return
const base: ScreenContext = this._session.screenContext ?? {
/**
* 활성 창 조회(녹음 시작 시점)와 선택 텍스트 캡처(정지 시점)가 끝나기를 기다렸다가
* 세션 컨텍스트로 합친다.
*/
private async _applyScreenContext(run: SessionRun): Promise<void> {
if (!run.windowContext && !run.selectionCapture) return
const [windowContext, selectedText] = await Promise.all([
run.windowContext ?? Promise.resolve(null),
run.selectionCapture ?? Promise.resolve(null),
])
if (!this._isCurrent(run) || !this._session) return
if (!windowContext && !selectedText) return
const base: ScreenContext = windowContext ?? this._session.screenContext ?? {
appName: null,
windowTitle: null,
selectedText: null,
capturedAt: Date.now(),
}
this._session.screenContext = { ...base, selectedText }
this._session.screenContext = selectedText ? { ...base, selectedText } : base
}
/** 세션 컨텍스트를 LLM 입력 접두사로 만든다 (없으면 빈 문자열) */
private async _buildContextPrefix(): Promise<string> {
const screenContext = this._session?.screenContext
if (!screenContext) return ''
try {
const { getScreenContextService } = await import('./ScreenContextService')
return getScreenContextService().buildContextPrompt(screenContext)
} catch {
return ''
}
}
// ── 실시간 부분 전사(미리보기) ─────────────────────────────
@ -806,7 +887,6 @@ class VoiceModeService extends EventEmitter {
// Phase 10.5: 음성 단축키 — 키워드 매칭으로 LLM 명령어 자동 선택
let effectiveText = result.text
let overrideAction: string | null = null
let overrideInstructionId: string | null = null
try {
const { getVoiceCommandService } = await import('./VoiceCommandService')
@ -815,7 +895,6 @@ class VoiceModeService extends EventEmitter {
const match = vcSvc.match(result.text)
if (match.matched && match.instructionId) {
effectiveText = match.cleanedText
overrideAction = 'custom'
overrideInstructionId = match.instructionId
logger.info(`Voice command matched: keyword="${match.matchedKeyword}", instruction=${match.instructionId}`)
}
@ -825,12 +904,9 @@ class VoiceModeService extends EventEmitter {
}
if (!this._isCurrent(run)) return
const llmAction = overrideAction ?? configGet('defaultLLMAction')
const backend = configGet('llmBackend')
const ollamaUnavailable = backend === 'local' && !getLocalLLMService().isAvailable()
const skipLLM = llmAction === 'none' || ollamaUnavailable
if (skipLLM) {
if (llmAction !== 'none' && ollamaUnavailable) {
const route = this._resolveLlmRoute(overrideInstructionId)
if (route.kind === 'insert-raw') {
if (route.reason === 'llm-unavailable') {
this.emit('error', {
error: new D3ROError(
ErrorCode.LLMServerUnreachable,
@ -842,7 +918,7 @@ class VoiceModeService extends EventEmitter {
}
void this._completeSession(run, effectiveText)
} else {
await this._processWithLLM(run, effectiveText, overrideInstructionId)
await this._processWithLLM(run, effectiveText, route)
}
} catch (error) {
if (!this._isCurrent(run)) return
@ -874,50 +950,42 @@ class VoiceModeService extends EventEmitter {
})
}
/** 현재 설정으로 LLM 후처리 경로를 결정한다 (정책은 @d3ro/core/voice-llm-route 가 소유) */
private _resolveLlmRoute(overrideInstructionId: string | null): VoiceLlmRoute {
const backend = configGet('llmBackend')
return resolveVoiceLlmRoute({
defaultAction: configGet('defaultLLMAction'),
overrideInstructionId,
activeChainId: configGet('activeChainId') ?? null,
activeInstructionId: configGet('activeInstructionId') ?? null,
localLlmAvailable: backend === 'local' ? getLocalLLMService().isAvailable() : true,
backend,
})
}
private async _processWithLLM(
run: SessionRun,
transcribedText: string,
overrideInstructionId?: string | null,
route: LlmProcessingRoute,
): Promise<void> {
if (!this._isCurrent(run)) return
try {
const configuredAction = configGet('defaultLLMAction')
// 음성 단축키가 특정 명령을 지목해 들어왔다면 기본 액션이 'none'이어도 처리한다.
// 명시적 요청을 기본 설정이 무효화하면 안 된다.
if (configuredAction === 'none' && !overrideInstructionId) {
void this._completeSession(run, transcribedText)
return
}
// 여기서 'none'이 남아 있다면 overrideInstructionId가 반드시 있다 → custom 경로.
const action: LLMAction = configuredAction === 'none' ? 'custom' : configuredAction
// Phase 10.2: 스크린 컨텍스트를 LLM 프롬프트에 주입 (선택 텍스트는 정지 시점 캡처)
await this._applySelectionCapture(run)
// Phase 10.2: 스크린 컨텍스트를 LLM 프롬프트에 주입 (선택 텍스트는 정지 이후 캡처).
// 정지 시점에 LLM 경로가 확실하지 않아 캡처를 미뤘다면 여기서 시작한다(멱등).
this._startSelectionCapture(run)
await this._applyScreenContext(run)
if (!this._isCurrent(run)) return
let contextPrefix = ''
if (this._session?.screenContext) {
try {
const { getScreenContextService } = await import('./ScreenContextService')
contextPrefix = getScreenContextService().buildContextPrompt(this._session.screenContext)
} catch {
// 무시
}
}
const userText = (await this._buildContextPrefix()) + transcribedText
// Phase 10.4: 체인 모드 처리
if (action === 'chain') {
if (route.kind === 'chain') {
try {
const { getChainService } = await import('./ChainService')
const activeChainId = configGet('activeChainId')
if (activeChainId) {
const chainResult = await getChainService().execute(activeChainId, contextPrefix + transcribedText)
const chainResult = await getChainService().execute(route.chainId, userText)
if (!this._isCurrent(run)) return
if (this._session) this._session.processedText = chainResult.finalText
void this._completeSession(run, chainResult.finalText)
return
}
} catch (error) {
if (!this._isCurrent(run)) return
logger.warn(`Chain execution failed: ${describe(error)}`)
@ -927,60 +995,14 @@ class VoiceModeService extends EventEmitter {
? error
: new D3ROError(ErrorCode.ChainExecutionFailed, `Chain execution failed: ${describe(error)}`),
)
}
return
}
}
let processedText: string
// 음성 단축키 오버라이드 또는 활성 명령어
const effectiveInstructionId = overrideInstructionId
?? configGet('activeInstructionId')
const targetLanguage = resolveTargetLanguage()
if (action === 'custom' || overrideInstructionId) {
const userText = contextPrefix + transcribedText
let invocationText = userText
let instructionPrompt: string | undefined
if (effectiveInstructionId) {
const { getCustomInstructionService } = await import('./CustomInstructionService')
const instruction = getCustomInstructionService().getById(effectiveInstructionId)
if (instruction) {
const invocation = buildInstructionInvocation(
instruction.prompt,
userText,
targetLanguage,
)
invocationText = invocation.text
instructionPrompt = invocation.systemPrompt
logger.info(`Using custom instruction: "${instruction.name}"`)
} else {
logger.warn(
`Custom instruction not found: ${effectiveInstructionId} — processing without an instruction`,
)
}
}
processedText = await this._runProcessorWithFallback(
invocationText,
'custom',
undefined,
instructionPrompt,
)
} else {
logger.info(
action === 'translate'
? `Processing with LLM (action: ${action}, targetLanguage: ${targetLanguage})`
: `Processing with LLM (action: ${action})`,
)
processedText = await this._runProcessorWithFallback(
contextPrefix + transcribedText,
action,
action === 'translate' ? targetLanguage : undefined,
)
}
const processedText =
route.kind === 'instruction'
? await this._processWithInstruction(userText, route.instructionId)
: await this._processWithAction(userText, route.action)
if (!this._isCurrent(run)) return
@ -1003,6 +1025,45 @@ class VoiceModeService extends EventEmitter {
}
}
/** 지시문(음성 단축키 또는 활성 지시문)으로 처리한다. 지시문이 없으면 지시문 없이 custom 처리 */
private async _processWithInstruction(userText: string, instructionId: string | null): Promise<string> {
let invocationText = userText
let instructionPrompt: string | undefined
if (instructionId) {
const { getCustomInstructionService } = await import('./CustomInstructionService')
const instruction = getCustomInstructionService().getById(instructionId)
if (instruction) {
const invocation = buildInstructionInvocation(instruction.prompt, userText, resolveTargetLanguage())
invocationText = invocation.text
instructionPrompt = invocation.systemPrompt
logger.info(`Using custom instruction: "${instruction.name}"`)
} else {
logger.warn(`Custom instruction not found: ${instructionId} — processing without an instruction`)
}
}
return this._runProcessorWithFallback(invocationText, 'custom', undefined, instructionPrompt)
}
/** 일반 액션(refine/translate/…)으로 처리한다 */
private async _processWithAction(
userText: string,
action: Extract<LlmProcessingRoute, { kind: 'action' }>['action'],
): Promise<string> {
const targetLanguage = resolveTargetLanguage()
logger.info(
action === 'translate'
? `Processing with LLM (action: ${action}, targetLanguage: ${targetLanguage})`
: `Processing with LLM (action: ${action})`,
)
return this._runProcessorWithFallback(
userText,
action,
action === 'translate' ? targetLanguage : undefined,
)
}
// ── 세션 완료/취소 ─────────────────────────────────────
private async _completeSession(run: SessionRun, finalText: string): Promise<void> {
@ -1161,10 +1222,8 @@ class VoiceModeService extends EventEmitter {
try {
switch (action.type) {
case 'press':
await this._handlePress(action)
break
case 'release':
await this._handleRelease(action)
await this._handleGesture(action)
break
case 'escape':
this.cancelSession()
@ -1175,48 +1234,29 @@ class VoiceModeService extends EventEmitter {
}
}
private async _handlePress(action: VoiceAction): Promise<void> {
if (action.mode === 'dictation') {
// Dictation: hold-to-talk — press로 시작
if (!this.isActive) {
await this.startSession('dictation')
}
} else if (action.mode === 'hands-free') {
// HandsFree: toggle
if (this.isActive) {
await this.stopSession()
} else {
await this.startSession('hands-free')
}
}
}
private async _handleRelease(action: VoiceAction): Promise<void> {
if (action.holdMode && action.mode === 'dictation' && this.isActive) {
// Dictation: hold-to-talk — release로 즉시 종료
// Speakly 패턴: 딜레이 없이 즉시 stop (딜레이가 race condition 유발)
await this.stopSession()
}
// HandsFree(토글): release 무시
}
// ── 유틸리티 ───────────────────────────────────────────
private _resolveMode(actionId: KeyBindingActionId, isDoublePress: boolean): VoiceMode {
if (actionId === 'hands-free' || isDoublePress) {
return 'hands-free'
}
return 'dictation'
}
/**
* 'command' 액션에는 아직 전용 핸들러가 없다.
* 현행 동작대로 dictation 파이프라인으로 fallback 하며, 그 경로는 hold-to-talk 이므로
* KEYBINDING_ACTIONS 의 holdMode(false)가 아니라 dictation 과 같은 값을 쓴다.
* 키 제스처를 세션 의도로 바꿔(voice-intent) 실행한다.
* 정지는 오디오 해제까지만 기다린다 — 전사·LLM 동안 큐가 풀려 있어야 처리 중에 들어온 토글이
* 'processing' 단계에서 무시되고, 세션이 끝난 뒤 새 녹음으로 재생되지 않는다.
*/
private _resolveHoldMode(actionId: KeyBindingActionId, specHoldMode: boolean): boolean {
if (actionId === 'command') return true
return specHoldMode
private async _handleGesture(action: VoiceAction): Promise<void> {
if (action.type === 'escape') return
const phase = deriveVoiceIntentPhase(this.isActive, this._run?.stopRequested ?? false)
const intent = resolveVoiceIntent(
{ type: action.type, mode: action.mode, holdMode: action.holdMode },
phase,
)
switch (intent.kind) {
case 'start':
await this.startSession(intent.mode)
break
case 'stop':
// Speakly 패턴: 딜레이 없이 즉시 stop (딜레이가 race condition 유발)
await this._requestStop()
break
case 'ignore':
break
}
}
// ── 자막 모드 토글 (Phase 10.1) ─────────────────────────

View file

@ -0,0 +1,156 @@
// src/main/services/foreground-window-port.ts
// 포그라운드(활성) 창 조회 포트와 플랫폼 어댑터.
//
// ScreenContextService(컨텍스트 정책)는 "지금 앞에 있는 앱이 무엇인가"만 알면 되고, 그것을 어떻게
// 읽는지(FFI, AppleScript)는 알 필요가 없다. 예전엔 서비스가 PowerShell + Add-Type 스크립트를 직접
// 들고 있었다 — koffi 리더(utils/win32-foreground.ts)와 중복이었고, 그 사본은 읽기 전용 $PID 에
// 대입하는 버그로 앱 이름이 늘 'powershell' 이었으며, 호출마다 0.7~3초가 걸렸다.
import { getLogger } from './LoggerService'
import { getForegroundWindowInfo, type ForegroundWindowInfo } from '../utils/win32-foreground'
const logger = getLogger('foreground-window')
export interface ActiveWindowInfo {
/** 프로세스 이름 (확장자 없음, 예: 'chrome'). 알 수 없으면 null */
appName: string | null
/** 창 제목. 알 수 없으면 null */
windowTitle: string | null
}
/** 포그라운드 창 조회 포트. macOS(osascript)는 비동기라 Promise 로 통일한다. */
export interface ForegroundWindowPort {
current(): Promise<ActiveWindowInfo>
}
const UNKNOWN: ActiveWindowInfo = { appName: null, windowTitle: null }
function blankToNull(value: string | null | undefined): string | null {
const trimmed = value?.trim()
return trimmed ? trimmed : null
}
/** 'Code.exe' → 'Code'. 예전 PowerShell 경로(Process.ProcessName)와 같은 모양을 유지한다. */
export function toProcessName(exeName: string): string {
return exeName.replace(/\.exe$/iu, '')
}
/** Windows: koffi FFI 리더(동기, 수 ms)를 감싼다. */
export function createWin32ForegroundWindowPort(
read: () => ForegroundWindowInfo | null = getForegroundWindowInfo,
): ForegroundWindowPort {
return {
async current() {
const info = read()
if (!info) return UNKNOWN
return {
appName: blankToNull(toProcessName(info.appName)),
windowTitle: blankToNull(info.title),
}
},
}
}
type ExecFileText = (
file: string,
args: string[],
options: { timeout: number },
) => Promise<{ stdout: string }>
async function defaultExecFileText(
file: string,
args: string[],
options: { timeout: number },
): Promise<{ stdout: string }> {
const { execFile } = await import('child_process')
const { promisify } = await import('util')
const { stdout } = await promisify(execFile)(file, args, { ...options, encoding: 'utf8' })
return { stdout: String(stdout) }
}
// AppleScript: 프로세스명과 앞 윈도우 타이틀을 2줄로 반환. 윈도우가 없는 앱도 있으므로 try-fallback.
const DARWIN_SCRIPT = `
try
tell application "System Events"
set frontApp to first process whose frontmost is true
set appName to name of frontApp
try
set winTitle to name of front window of frontApp
on error
set winTitle to ""
end try
return appName & linefeed & winTitle
end tell
on error errMsg
return "" & linefeed & ""
end try
`.trim()
/**
* macOS: osascript (System Events). Accessibility 권한이 필요하다 — 거부되면 { null, null }.
*/
export function createDarwinForegroundWindowPort(
execFileText: ExecFileText = defaultExecFileText,
): ForegroundWindowPort {
return {
async current() {
try {
const { stdout } = await execFileText('/usr/bin/osascript', ['-e', DARWIN_SCRIPT], { timeout: 3000 })
const lines = stdout.split('\n')
return { appName: blankToNull(lines[0]), windowTitle: blankToNull(lines[1]) }
} catch (error) {
logger.warn(
`osascript active window query failed: ${error instanceof Error ? error.message : String(error)}`,
)
return UNKNOWN
}
},
}
}
/** 미지원 플랫폼(Linux 등) */
export function createNullForegroundWindowPort(): ForegroundWindowPort {
return { current: async () => UNKNOWN }
}
export function createDefaultForegroundWindowPort(
platform: NodeJS.Platform = process.platform,
): ForegroundWindowPort {
if (platform === 'win32') return createWin32ForegroundWindowPort()
if (platform === 'darwin') return createDarwinForegroundWindowPort()
return createNullForegroundWindowPort()
}
// ── 선택 텍스트 캡처 안전성 ─────────────────────────────────
/**
* 콘솔/터미널 호스트(소문자 프로세스 이름). 이런 창에서 선택 없이 Ctrl+C 를 보내면 복사가 아니라
* SIGINT(실행 중 프로세스 중단)나 입력 줄 지우기가 된다 — 선택 텍스트 캡처를 건너뛴다.
*/
const TERMINAL_HOSTS = new Set([
'windowsterminal',
'openconsole',
'conhost',
'cmd',
'powershell',
'pwsh',
'powershell_ise',
'wezterm-gui',
'alacritty',
'mintty',
'conemu',
'conemu64',
'kitty',
'hyper',
'tabby',
'warp',
'terminal',
'iterm2',
'iterm',
'ghostty',
])
export function isTerminalHost(appName: string | null): boolean {
if (!appName) return false
return TERMINAL_HOSTS.has(toProcessName(appName).trim().toLowerCase())
}

View file

@ -0,0 +1,85 @@
// src/main/services/voice-intent.ts
// 키 제스처 → 음성 세션 의도(start/stop/ignore) 매핑 정책. 순수 함수만 둔다 (서비스·IO 의존 없음).
//
// 예전엔 이 규칙이 VoiceModeService 의 _handlePress / _handleRelease / _resolveMode /
// _resolveHoldMode 에 흩어져 있었고, '처리 중(processing)' 이라는 단계가 코드에 드러나지 않아
// 액션 큐의 직렬화 순서에 암묵적으로 기댔다. 정지가 전사·LLM 전체를 await 하는 동안 큐에 쌓인
// 입력이 세션이 끝난 뒤 다른 의미로 재생됐다(더블탭 정지의 두 번째 탭이 새 핸즈프리를 시작).
// 이제 단계를 명시적으로 받아 의도를 돌려준다. 처리 중의 토글·누름은 'ignore' 다.
//
// 남는 한계: 처리가 더블탭 간격(TIMING.DOUBLE_PRESS_DURATION)보다 빨리 끝나 세션이 이미 idle 이면
// 두 번째 탭은 여전히 새 핸즈프리 세션을 시작한다. 의도 판단에 시간 정보가 필요한 경우로, 이 모듈의
// 범위 밖이다.
import type { KeyBindingActionId, VoiceMode } from '@d3ro/core/types'
/** 의도 판단에 쓰는 세션 단계 */
export type VoiceIntentPhase = 'idle' | 'recording' | 'processing'
export interface VoiceGesture {
type: 'press' | 'release'
mode: VoiceMode
/** hold-to-talk 여부 — release 에서 세션을 끊을지 결정한다 */
holdMode: boolean
}
export type VoiceIntent =
| { kind: 'start'; mode: VoiceMode }
| { kind: 'stop' }
| { kind: 'ignore' }
const IGNORE: VoiceIntent = { kind: 'ignore' }
const STOP: VoiceIntent = { kind: 'stop' }
/**
* 세션 상태에서 의도 판단용 단계를 도출한다. 별도 상태 필드를 두지 않는다.
* @param active 세션이 살아 있다 (비터미널)
* @param stopRequested 이 세션에 정지가 이미 요청됐다 (오디오 해제 이후 = 전사/LLM 처리 중)
*/
export function deriveVoiceIntentPhase(active: boolean, stopRequested: boolean): VoiceIntentPhase {
if (!active) return 'idle'
return stopRequested ? 'processing' : 'recording'
}
/**
* 키 제스처를 세션 의도로 바꾼다.
*
* | 제스처 | idle | recording | processing |
* |---------------------------------|------------------|-----------|------------|
* | press dictation | start dictation | ignore | ignore |
* | press hands-free (토글) | start hands-free | stop | ignore |
* | release dictation + holdMode | ignore | stop | ignore |
* | release 그 외 (토글의 release) | ignore | ignore | ignore |
*
* 핸즈프리 세션도 dictation release(holdMode) 한 번으로 멈춘다 — 기본 바인딩에서 더블탭 정지의
* 첫 탭이 이 경로다(두 번째 탭은 processing 에서 ignore).
*/
export function resolveVoiceIntent(gesture: VoiceGesture, phase: VoiceIntentPhase): VoiceIntent {
if (phase === 'processing') return IGNORE
if (gesture.type === 'press') {
if (gesture.mode === 'dictation') {
return phase === 'idle' ? { kind: 'start', mode: 'dictation' } : IGNORE
}
return phase === 'idle' ? { kind: 'start', mode: 'hands-free' } : STOP
}
if (gesture.holdMode && gesture.mode === 'dictation' && phase === 'recording') return STOP
return IGNORE
}
/** 키바인딩 액션 → 세션 모드. 더블탭은 항상 핸즈프리다. */
export function resolveGestureMode(actionId: KeyBindingActionId, isDoublePress: boolean): VoiceMode {
if (actionId === 'hands-free' || isDoublePress) return 'hands-free'
return 'dictation'
}
/**
* 'command' 액션에는 아직 전용 핸들러가 없다.
* 현행 동작대로 dictation 파이프라인으로 fallback 하며, 그 경로는 hold-to-talk 이므로
* KEYBINDING_ACTIONS 의 holdMode(false)가 아니라 dictation 과 같은 값을 쓴다.
*/
export function resolveGestureHoldMode(actionId: KeyBindingActionId, specHoldMode: boolean): boolean {
if (actionId === 'command') return true
return specHoldMode
}

View file

@ -0,0 +1,230 @@
// tests/main/services/audio-capture-redteam-r2-7.test.ts
// AudioCaptureService / 마이크 테스트 IPC 회귀 테스트:
// - kill 된 옛 SoX 의 늦은 'close' 가 새 캡처를 오류로 끝내지 않는다
// - 마이크 테스트는 자기가 잡은 참조만 놓는다 (회의·자막의 캡처를 빼앗지 않는다)
// - 정지 시 SoX 를 곧바로 죽이지 않고 꼬리 오디오와 잔여 프레임을 흘려보낸다
// - SoX 버퍼를 줄여(--buffer 1024) 청크 지연·손실을 줄인다
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'
import { EventEmitter } from 'events'
import { ipcMain } from 'electron'
import { IPC_CHANNELS } from '@d3ro/core/ipc-channels'
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn(() => 'default'),
configSet: vi.fn(),
}))
vi.mock('../../../src/main/utils/paths', () => ({
getSoxPath: () => 'sox',
}))
vi.mock('../../../src/main/windows/WindowManager', () => ({
getMainWindow: () => null,
}))
class FakeProcess extends EventEmitter {
stdout = new EventEmitter()
stderr = new EventEmitter()
kill = vi.fn()
}
const spawned = vi.hoisted(() => ({ list: [] as unknown[], args: [] as string[][] }))
vi.mock('child_process', async (importOriginal) => {
const actual = await importOriginal<typeof import('child_process')>()
return {
...actual,
spawn: vi.fn((_cmd: string, args: string[]) => {
const proc = new FakeProcess()
spawned.list.push(proc)
spawned.args.push(args)
return proc
}),
}
})
type Module = typeof import('../../../src/main/services/AudioCaptureService')
let mod: Module
const FRAME = 1920 // 60ms
function proc(i: number): FakeProcess {
return spawned.list[i] as FakeProcess
}
beforeEach(async () => {
vi.resetModules()
vi.useRealTimers()
spawned.list = []
spawned.args = []
mod = await import('../../../src/main/services/AudioCaptureService')
mod.resetAudioCaptureServiceForTests()
})
afterEach(() => {
vi.useRealTimers()
})
describe('SoX 프로세스 수명', () => {
it("정지 후 곧바로 재시작하면 옛 SoX 의 늦은 'close' 가 새 캡처를 죽이지 않는다", async () => {
const audio = mod.getAudioCaptureService()
const stopped: string[] = []
audio.on('stopped', ({ reason }) => stopped.push(reason))
await audio.start()
await audio.stop()
await audio.start()
expect(spawned.list).toHaveLength(2)
// 옛 프로세스의 'close' 가 kill 몇 ms 뒤 도착
proc(0).emit('close', null)
proc(0).emit('error', new Error('late'))
expect(audio.state).toBe('capturing')
expect(proc(1).kill).not.toHaveBeenCalled()
expect(stopped).toEqual(['manual'])
})
it('SoX 인자에 --buffer 1024 가 들어간다', async () => {
await mod.getAudioCaptureService().start()
const args = spawned.args[0]
const i = args.indexOf('--buffer')
expect(i).toBeGreaterThanOrEqual(0)
expect(args[i + 1]).toBe(String(mod.SOX_BUFFER_BYTES))
expect(mod.SOX_BUFFER_BYTES).toBe(1024)
})
})
describe('정지 시 꼬리 오디오', () => {
it('정지 중 도착한 청크와 60ms 미만 잔여 바이트를 흘려보낸 뒤 SoX 를 끝낸다', async () => {
const audio = mod.getAudioCaptureService()
const received: number[] = []
audio.on('audio-data', ({ buffer }) => received.push(buffer.length))
await audio.start()
proc(0).stdout.emit('data', Buffer.alloc(FRAME + 100)) // 1 프레임 + 잔여 100
const stopping = audio.stop()
expect(proc(0).kill).not.toHaveBeenCalled()
// SoX 버퍼에 남아 있던 끝부분이 정지 직후 도착
proc(0).stdout.emit('data', Buffer.alloc(FRAME))
await stopping
expect(proc(0).kill).toHaveBeenCalledTimes(1)
// 1920 (첫 프레임) + 1920 (잔여 100 + 꼬리 1820) + 잔여 100 (마지막 부분 프레임)
expect(received).toEqual([FRAME, FRAME, 100])
expect(received.reduce((a, b) => a + b, 0)).toBe(FRAME * 2 + 100)
expect(audio.state).toBe('idle')
})
it('꼬리 수집 중 새 소비자가 오면 정지를 취소하고 같은 SoX 로 이어 간다', async () => {
const audio = mod.getAudioCaptureService()
const stopped: string[] = []
audio.on('stopped', ({ reason }) => stopped.push(reason))
await audio.start()
const stopping = audio.stop()
await audio.start()
await stopping
expect(spawned.list).toHaveLength(1)
expect(proc(0).kill).not.toHaveBeenCalled()
expect(audio.state).toBe('capturing')
expect(stopped).toEqual([])
await audio.stop()
expect(proc(0).kill).toHaveBeenCalledTimes(1)
expect(stopped).toEqual(['manual'])
})
})
describe('캡처 임대(lease)', () => {
it('release 는 멱등이고 다른 소비자의 참조를 깎지 않는다', async () => {
const audio = mod.getAudioCaptureService()
await audio.start() // 회의·자막이 잡은 참조
const lease = await audio.acquire()
await lease.release()
await lease.release()
expect(audio.state).toBe('capturing')
expect(proc(0).kill).not.toHaveBeenCalled()
})
it('캡처가 죽었다 다시 시작됐으면 옛 임대의 release 는 새 캡처를 멈추지 않는다', async () => {
const audio = mod.getAudioCaptureService()
const lease = await audio.acquire()
proc(0).emit('close', 1) // 장치 분리
expect(audio.state).toBe('error')
await audio.start() // 다른 소비자가 새로 시작
await lease.release()
expect(audio.state).toBe('capturing')
expect(proc(1).kill).not.toHaveBeenCalled()
})
})
describe('마이크 테스트 IPC', () => {
function handler(channel: string): () => Promise<unknown> {
const call = vi.mocked(ipcMain.handle).mock.calls.find(([ch]) => ch === channel)
if (!call) throw new Error(`handler not registered: ${channel}`)
return call[1] as unknown as () => Promise<unknown>
}
it('회의 캡처 중에 마이크 테스트를 해도 회의 캡처는 끊기지 않는다', async () => {
vi.mocked(ipcMain.handle).mockClear()
const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers')
registerAudioHandlers()
const audio = mod.getAudioCaptureService()
// 회의(자막) 가 캡처를 잡고 있다
await audio.start()
const meetingFrames: number[] = []
audio.on('audio-data', ({ buffer }) => meetingFrames.push(buffer.length))
// 설정 창에서 TEST → 곧바로 STOP
await handler(IPC_CHANNELS.AUDIO.TEST_DEVICE)()
await handler(IPC_CHANNELS.AUDIO.STOP_TEST)()
await new Promise((r) => setTimeout(r, mod.TAIL_FLUSH_MS + 20))
// 옛 SoX 를 죽이거나 새로 띄우지 않았고 회의 오디오가 계속 들어온다
expect(spawned.list).toHaveLength(1)
expect(proc(0).kill).not.toHaveBeenCalled()
expect(audio.state).toBe('capturing')
proc(0).stdout.emit('data', Buffer.alloc(FRAME))
expect(meetingFrames).toEqual([FRAME])
})
it('테스트가 돌고 있지 않을 때의 STOP 은 다른 소비자의 참조를 놓지 않는다', async () => {
vi.mocked(ipcMain.handle).mockClear()
const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers')
registerAudioHandlers()
const audio = mod.getAudioCaptureService()
await audio.start()
await handler(IPC_CHANNELS.AUDIO.STOP_TEST)()
await new Promise((r) => setTimeout(r, mod.TAIL_FLUSH_MS + 20))
expect(audio.state).toBe('capturing')
expect(proc(0).kill).not.toHaveBeenCalled()
})
it('테스트 단독이면 STOP 으로 캡처가 끝난다', async () => {
vi.mocked(ipcMain.handle).mockClear()
const { registerAudioHandlers } = await import('../../../src/main/ipc/audio-handlers')
registerAudioHandlers()
const audio = mod.getAudioCaptureService()
await handler(IPC_CHANNELS.AUDIO.TEST_DEVICE)()
expect(audio.state).toBe('capturing')
await handler(IPC_CHANNELS.AUDIO.STOP_TEST)()
await vi.waitFor(() => expect(audio.state).toBe('idle'))
expect(proc(0).kill).toHaveBeenCalledTimes(1)
})
})

View file

@ -0,0 +1,147 @@
// tests/main/services/screen-context-redteam-r2-7.test.ts
// ScreenContextService + ForegroundWindowPort 회귀 테스트:
// - 활성 창 조회는 포트(koffi/osascript 어댑터)에 맡기고 PowerShell 을 띄우지 않는다
// - Windows 어댑터는 실제 앱 이름을 돌려준다 (예전 PowerShell 경로는 $PID 버그로 늘 'powershell')
// - 콘솔/터미널이 포그라운드면 복사 단축키를 보내지 않는다 (SIGINT·입력 줄 지우기 방지)
// - macOS 에선 ⌘+C 를 보낸다
import { describe, it, expect, beforeEach, vi } from 'vitest'
import { clipboard } from 'electron'
import type { ForegroundWindowInfo } from '../../../src/main/utils/win32-foreground'
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn(() => true),
configSet: vi.fn(),
}))
vi.mock('../../../src/main/services/modifier-state', () => ({
waitForModifiersReleased: vi.fn(async () => true),
}))
const nut = vi.hoisted(() => ({ pressKey: vi.fn(async () => undefined), releaseKey: vi.fn(async () => undefined) }))
vi.mock('@nut-tree-fork/nut-js', () => ({
keyboard: nut,
Key: { LeftControl: 'LeftControl', LeftSuper: 'LeftSuper', C: 'C' },
}))
const childProcess = vi.hoisted(() => ({ execFile: vi.fn(), exec: vi.fn(), spawn: vi.fn() }))
vi.mock('child_process', async (importOriginal) => {
const actual = await importOriginal<typeof import('child_process')>()
return { ...actual, ...childProcess }
})
type ScreenModule = typeof import('../../../src/main/services/ScreenContextService')
type PortModule = typeof import('../../../src/main/services/foreground-window-port')
let screenMod: ScreenModule
let portMod: PortModule
Object.assign(clipboard, { clear: vi.fn() })
function fakePort(appName: string | null, windowTitle: string | null = 'Title'): { current: ReturnType<typeof vi.fn> } {
return { current: vi.fn(async () => ({ appName, windowTitle })) }
}
function info(patch: Partial<ForegroundWindowInfo>): ForegroundWindowInfo {
return { hwnd: 1, title: '', processId: 1, appName: '', appPath: null, bounds: null, ...patch }
}
beforeEach(async () => {
vi.resetModules()
vi.clearAllMocks()
screenMod = await import('../../../src/main/services/ScreenContextService')
portMod = await import('../../../src/main/services/foreground-window-port')
})
describe('captureContext — ForegroundWindowPort', () => {
it('포트가 준 앱·창 정보를 쓰고 자식 프로세스를 띄우지 않는다', async () => {
const port = fakePort('Code', 'main.ts - D3ROVoice')
const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' })
const { context, selectedTextAttempted } = await svc.captureContext(false)
expect(context.appName).toBe('Code')
expect(context.windowTitle).toBe('main.ts - D3ROVoice')
expect(selectedTextAttempted).toBe(false)
expect(childProcess.execFile).not.toHaveBeenCalled()
expect(childProcess.spawn).not.toHaveBeenCalled()
expect(svc.buildContextPrompt(context)).toContain('활성 앱: Code')
})
it('포트가 실패해도 컨텍스트 캡처는 계속된다', async () => {
const port = { current: vi.fn(async () => { throw new Error('ffi') }) }
const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' })
const { context } = await svc.captureContext(false)
expect(context.appName).toBeNull()
expect(svc.buildContextPrompt(context)).toBe('')
})
})
describe('Windows 포그라운드 어댑터 (koffi)', () => {
it('실행 파일 이름에서 .exe 를 떼어 실제 앱 이름을 돌려준다', async () => {
const port = portMod.createWin32ForegroundWindowPort(() =>
info({ appName: 'Code.exe', title: 'main.ts', appPath: 'C:\\Code\\Code.exe' }),
)
await expect(port.current()).resolves.toEqual({ appName: 'Code', windowTitle: 'main.ts' })
})
it('빈 값과 조회 실패는 null 이다', async () => {
await expect(portMod.createWin32ForegroundWindowPort(() => info({})).current()).resolves.toEqual({
appName: null,
windowTitle: null,
})
await expect(portMod.createWin32ForegroundWindowPort(() => null).current()).resolves.toEqual({
appName: null,
windowTitle: null,
})
})
})
describe('선택 텍스트 캡처 — 복사 단축키 안전성', () => {
it.each(['WindowsTerminal', 'conhost', 'pwsh', 'Terminal', 'iTerm2', 'WindowsTerminal.exe'])(
'포그라운드가 터미널(%s)이면 복사 단축키를 보내지 않는다',
async (appName) => {
const svc = new screenMod.ScreenContextService({ foreground: fakePort(appName), platform: 'win32' })
await expect(svc.captureSelectedText()).resolves.toBeNull()
expect(nut.pressKey).not.toHaveBeenCalled()
expect(clipboard.clear).not.toHaveBeenCalled()
},
)
it('Windows 에선 Ctrl+C 를 보낸다', async () => {
vi.mocked(clipboard.readText).mockReturnValueOnce('').mockReturnValueOnce('선택')
const svc = new screenMod.ScreenContextService({ foreground: fakePort('Code'), platform: 'win32' })
await expect(svc.captureSelectedText()).resolves.toBe('선택')
expect(nut.pressKey).toHaveBeenCalledWith('LeftControl', 'C')
})
it('macOS 에선 ⌘+C 를 보낸다', async () => {
vi.mocked(clipboard.readText).mockReturnValueOnce('').mockReturnValueOnce('선택')
const svc = new screenMod.ScreenContextService({ foreground: fakePort('Notes'), platform: 'darwin' })
await expect(svc.captureSelectedText()).resolves.toBe('선택')
expect(nut.pressKey).toHaveBeenCalledWith('LeftSuper', 'C')
expect(nut.pressKey).not.toHaveBeenCalledWith('LeftControl', 'C')
})
it('captureContext(true) 도 터미널이면 선택 텍스트를 건너뛴다', async () => {
const port = fakePort('WindowsTerminal')
const svc = new screenMod.ScreenContextService({ foreground: port, platform: 'win32' })
const { context, selectedTextAttempted } = await svc.captureContext(true)
expect(selectedTextAttempted).toBe(true)
expect(context.selectedText).toBeNull()
expect(nut.pressKey).not.toHaveBeenCalled()
// 포그라운드는 한 번만 조회한다
expect(port.current).toHaveBeenCalledTimes(1)
})
})
describe('isTerminalHost', () => {
it('대소문자·확장자와 무관하게 판별하고, 일반 앱은 아니다', () => {
expect(portMod.isTerminalHost('WINDOWSTERMINAL.EXE')).toBe(true)
expect(portMod.isTerminalHost('Code')).toBe(false)
expect(portMod.isTerminalHost(null)).toBe(false)
})
})

View file

@ -0,0 +1,84 @@
// tests/main/services/voice-intent.test.ts
// 키 제스처 → 세션 의도 표 테스트 (더블탭 on/off, hold, 처리 중 토글).
import { describe, it, expect } from 'vitest'
import {
deriveVoiceIntentPhase,
resolveGestureHoldMode,
resolveGestureMode,
resolveVoiceIntent,
type VoiceGesture,
type VoiceIntent,
type VoiceIntentPhase,
} from '../../../src/main/services/voice-intent'
const dictationPress: VoiceGesture = { type: 'press', mode: 'dictation', holdMode: true }
const dictationRelease: VoiceGesture = { type: 'release', mode: 'dictation', holdMode: true }
const handsFreePress: VoiceGesture = { type: 'press', mode: 'hands-free', holdMode: false }
const handsFreeRelease: VoiceGesture = { type: 'release', mode: 'hands-free', holdMode: false }
const START_DICTATION: VoiceIntent = { kind: 'start', mode: 'dictation' }
const START_HANDS_FREE: VoiceIntent = { kind: 'start', mode: 'hands-free' }
const STOP: VoiceIntent = { kind: 'stop' }
const IGNORE: VoiceIntent = { kind: 'ignore' }
describe('resolveVoiceIntent', () => {
it.each<[string, VoiceGesture, VoiceIntentPhase, VoiceIntent]>([
// hold-to-talk
['idle 에서 누르면 받아쓰기를 시작한다', dictationPress, 'idle', START_DICTATION],
['녹음 중 누름은 무시한다', dictationPress, 'recording', IGNORE],
['녹음 중 떼면 정지한다', dictationRelease, 'recording', STOP],
['idle 에서 뗌은 무시한다', dictationRelease, 'idle', IGNORE],
['처리 중 뗌은 무시한다', dictationRelease, 'processing', IGNORE],
// 핸즈프리 토글 (더블탭)
['idle 에서 더블탭은 핸즈프리를 켠다', handsFreePress, 'idle', START_HANDS_FREE],
['녹음 중 더블탭은 끈다', handsFreePress, 'recording', STOP],
['처리 중 더블탭은 새 녹음을 시작하지 않는다', handsFreePress, 'processing', IGNORE],
['토글의 release 는 늘 무시한다', handsFreeRelease, 'recording', IGNORE],
['토글의 release 는 idle 에서도 무시한다', handsFreeRelease, 'idle', IGNORE],
// 더블탭 정지의 첫 탭(dictation press/release)이 핸즈프리 세션을 멈춘다 (기존 동작 보존)
['핸즈프리 녹음 중 첫 탭 press 는 무시', dictationPress, 'recording', IGNORE],
['핸즈프리 녹음 중 첫 탭 release 가 정지', dictationRelease, 'recording', STOP],
['처리 중 받아쓰기 누름은 무시', dictationPress, 'processing', IGNORE],
])('%s', (_label, gesture, phase, expected) => {
expect(resolveVoiceIntent(gesture, phase)).toEqual(expected)
})
it('holdMode 가 아닌 dictation release 는 정지하지 않는다', () => {
expect(resolveVoiceIntent({ type: 'release', mode: 'dictation', holdMode: false }, 'recording')).toEqual(IGNORE)
})
it('더블탭 on → 더블탭 off 시퀀스: 두 번째 탭은 처리 중에 무시된다', () => {
// on
expect(resolveVoiceIntent(handsFreePress, 'idle')).toEqual(START_HANDS_FREE)
// off: 첫 탭 press/release, 정지 후 처리 중에 두 번째 탭
expect(resolveVoiceIntent(dictationPress, 'recording')).toEqual(IGNORE)
expect(resolveVoiceIntent(dictationRelease, 'recording')).toEqual(STOP)
expect(resolveVoiceIntent(handsFreePress, 'processing')).toEqual(IGNORE)
expect(resolveVoiceIntent(handsFreeRelease, 'processing')).toEqual(IGNORE)
})
})
describe('deriveVoiceIntentPhase', () => {
it('세션 상태에서 단계를 도출한다', () => {
expect(deriveVoiceIntentPhase(false, false)).toBe('idle')
expect(deriveVoiceIntentPhase(false, true)).toBe('idle')
expect(deriveVoiceIntentPhase(true, false)).toBe('recording')
expect(deriveVoiceIntentPhase(true, true)).toBe('processing')
})
})
describe('제스처 입력 정규화', () => {
it('더블탭이거나 hands-free 액션이면 핸즈프리다', () => {
expect(resolveGestureMode('dictation', false)).toBe('dictation')
expect(resolveGestureMode('dictation', true)).toBe('hands-free')
expect(resolveGestureMode('hands-free', false)).toBe('hands-free')
expect(resolveGestureMode('command', false)).toBe('dictation')
})
it("'command' 는 dictation 파이프라인 fallback 이라 hold-to-talk 이다", () => {
expect(resolveGestureHoldMode('command', false)).toBe(true)
expect(resolveGestureHoldMode('dictation', true)).toBe(true)
expect(resolveGestureHoldMode('hands-free', false)).toBe(false)
})
})

View file

@ -0,0 +1,373 @@
// tests/main/services/voice-mode-redteam-r2-7.test.ts
// VoiceModeService 회귀 테스트 (round 2 #7):
// - 더블탭으로 핸즈프리를 끄면 두 번째 탭이 처리 뒤 새 녹음을 시작하지 않는다 (큐가 막히지 않음)
// - 스크린 컨텍스트의 창 조회가 세션·마이크 시작을 막지 않는다
// - 원문 삽입 경로(none)에선 선택 텍스트용 복사 단축키를 보내지 않는다
// - '체인' 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다
// - 정지 시 캡처가 흘려보내는 꼬리 오디오를 받는다
import { describe, it, expect, beforeEach, vi } from 'vitest'
import { EventEmitter } from 'events'
import type { KeyBindingTriggerPayload } from '../../../src/main/services/KeyBindingService'
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
interface Deferred<T> {
promise: Promise<T>
resolve: (value: T) => void
}
function deferred<T>(): Deferred<T> {
let resolve!: (value: T) => void
const promise = new Promise<T>((res) => {
resolve = res
})
return { promise, resolve }
}
type SttResult = { text: string; segments: never[]; language: string; duration: number; processingTime: number }
const result = (text: string): SttResult => ({ text, segments: [], language: 'ko', duration: 1, processingTime: 1 })
const mockSTT = vi.hoisted(() => ({
initialize: vi.fn(),
transcribe: vi.fn(),
transcribePartial: vi.fn(async () => ''),
getModels: vi.fn(),
getStatus: vi.fn(() => ({ engineState: 'ready', activeModel: 'base', engineVersion: null, gpuAccelerated: false })),
}))
vi.mock('../../../src/main/services/LocalSTTService', () => ({
getLocalSTTService: () => mockSTT,
resetLocalSTTServiceForTests: () => undefined,
}))
const audioBus = new EventEmitter()
const mockAudio = vi.hoisted(() => ({ start: vi.fn(), stop: vi.fn(), on: vi.fn(), off: vi.fn() }))
vi.mock('../../../src/main/services/AudioCaptureService', () => ({
getAudioCaptureService: () => mockAudio,
}))
const keyBinding = vi.hoisted(() => ({
handler: null as ((payload: KeyBindingTriggerPayload) => void) | null,
}))
vi.mock('../../../src/main/services/KeyBindingService', () => ({
getKeyBindingService: () => ({
on: (_ev: string, fn: (payload: KeyBindingTriggerPayload) => void) => {
keyBinding.handler = fn
},
off: vi.fn(),
}),
}))
const config = vi.hoisted(() => ({ values: {} as Record<string, unknown> }))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn((key: string) => config.values[key]),
}))
const mockInsert = vi.hoisted(() => ({ insertText: vi.fn() }))
vi.mock('../../../src/main/services/TextInsertService', () => ({
getTextInsertService: () => mockInsert,
}))
vi.mock('../../../src/main/services/LocalLLMService', () => ({
getLocalLLMService: () => ({ isAvailable: () => true }),
}))
const gateway = vi.hoisted(() => ({ processText: vi.fn() }))
vi.mock('../../../src/main/services/llm/LlmGateway', () => ({
getLlmGateway: () => gateway,
}))
vi.mock('../../../src/main/services/CaptionService', () => ({
getCaptionService: () => ({ getState: () => 'inactive', stop: vi.fn(), start: vi.fn() }),
}))
const chain = vi.hoisted(() => ({ execute: vi.fn() }))
vi.mock('../../../src/main/services/ChainService', () => ({
getChainService: () => chain,
}))
const instructions = vi.hoisted(() => ({
byId: {} as Record<string, { id: string; name: string; prompt: string }>,
}))
vi.mock('../../../src/main/services/CustomInstructionService', () => ({
getCustomInstructionService: () => ({ getById: (id: string) => instructions.byId[id] ?? null }),
}))
const voiceCommand = vi.hoisted(() => ({ instructionId: null as string | null }))
vi.mock('../../../src/main/services/VoiceCommandService', () => ({
getVoiceCommandService: () => ({
isEnabled: () => voiceCommand.instructionId !== null,
match: (text: string) =>
voiceCommand.instructionId
? { matched: true, ruleId: 'r', instructionId: voiceCommand.instructionId, cleanedText: text, matchedKeyword: 'k' }
: { matched: false, ruleId: null, instructionId: null, cleanedText: text, matchedKeyword: null },
}),
}))
const screen = vi.hoisted(() => ({
enabled: false,
captureContext: vi.fn(),
captureSelectedText: vi.fn(),
}))
vi.mock('../../../src/main/services/ScreenContextService', () => ({
getScreenContextService: () => ({
isEnabled: () => screen.enabled,
captureContext: screen.captureContext,
captureSelectedText: screen.captureSelectedText,
buildContextPrompt: (ctx: { appName: string | null; selectedText: string | null }) =>
`[ctx app=${ctx.appName ?? '-'} sel=${ctx.selectedText ?? '-'}]\n`,
}),
}))
type VoiceModeModule = typeof import('../../../src/main/services/VoiceModeService')
let getVoiceModeService: VoiceModeModule['getVoiceModeService']
const SPEECH = Buffer.alloc(16000 * 2) // 1초
async function flush(times = 8): Promise<void> {
for (let i = 0; i < times; i++) await new Promise((r) => setImmediate(r))
}
function advanceClock(ms: number): void {
const base = Date.now()
vi.spyOn(Date, 'now').mockReturnValue(base + ms)
}
function key(
actionId: 'dictation' | 'hands-free',
type: 'pressed' | 'released',
isDoublePress: boolean,
): KeyBindingTriggerPayload {
return {
actionId,
type,
isDoublePress,
holdMode: actionId === 'dictation',
timestamp: Date.now(),
durationMs: 0,
} as unknown as KeyBindingTriggerPayload
}
function screenContext(appName: string): { context: { appName: string; windowTitle: string; selectedText: null; capturedAt: number }; selectedTextAttempted: boolean } {
return { context: { appName, windowTitle: 'w', selectedText: null, capturedAt: 0 }, selectedTextAttempted: false }
}
beforeEach(async () => {
vi.restoreAllMocks()
vi.resetModules()
audioBus.removeAllListeners()
config.values = { sttModelId: 'base', defaultLLMAction: 'none', autoInsert: true, llmBackend: 'local' }
keyBinding.handler = null
screen.enabled = false
voiceCommand.instructionId = null
instructions.byId = {}
for (const fn of [
mockSTT.initialize, mockSTT.transcribe, mockSTT.transcribePartial, mockSTT.getModels,
mockAudio.start, mockAudio.stop, mockAudio.on, mockAudio.off, mockInsert.insertText,
gateway.processText, chain.execute, screen.captureContext, screen.captureSelectedText,
]) {
fn.mockReset()
}
mockSTT.initialize.mockResolvedValue(undefined)
mockSTT.transcribe.mockResolvedValue(result('기본 전사'))
mockSTT.transcribePartial.mockResolvedValue('')
mockSTT.getModels.mockReturnValue([
{ id: 'base', name: 'Base', sizeBytes: 0, downloaded: true, languages: [], accuracy: 2, speed: 4 },
])
mockAudio.start.mockResolvedValue(undefined)
mockAudio.stop.mockResolvedValue(undefined)
mockAudio.on.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => {
audioBus.on(ev, fn)
})
mockAudio.off.mockImplementation((ev: string, fn: (...args: unknown[]) => void) => {
audioBus.off(ev, fn)
})
mockInsert.insertText.mockResolvedValue({ success: true, method: 'clipboard', textLength: 1, durationMs: 1 })
gateway.processText.mockImplementation(async (text: string) => `LLM(${text})`)
chain.execute.mockResolvedValue({ finalText: '체인 결과' })
screen.captureContext.mockResolvedValue(screenContext('Code'))
screen.captureSelectedText.mockResolvedValue('선택 문장')
const mod = await import('../../../src/main/services/VoiceModeService')
mod.resetVoiceModeServiceForTests()
getVoiceModeService = mod.getVoiceModeService
})
describe('더블탭 핸즈프리 정지', () => {
it('정지 더블탭의 두 번째 탭은 전사가 끝난 뒤 새 핸즈프리 녹음을 시작하지 않는다', async () => {
const stt = deferred<SttResult>()
mockSTT.transcribe.mockReturnValueOnce(stt.promise)
const svc = getVoiceModeService()
svc.connectKeyBindings()
const completed: string[] = []
svc.on('session-completed', ({ finalText }) => completed.push(finalText))
const emit = (p: KeyBindingTriggerPayload): void => keyBinding.handler?.(p)
// 더블탭으로 핸즈프리 켜기
emit(key('hands-free', 'pressed', true))
emit(key('hands-free', 'released', true))
await vi.waitFor(() => expect(mockAudio.start).toHaveBeenCalledTimes(1))
await vi.waitFor(() => expect(svc.isActive).toBe(true))
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(3000)
// 더블탭으로 끄기: 첫 탭(dictation press/release) → 정지, 두 번째 탭(hands-free press)
emit(key('dictation', 'pressed', false))
emit(key('dictation', 'released', false))
await vi.waitFor(() => expect(mockSTT.transcribe).toHaveBeenCalledTimes(1))
emit(key('hands-free', 'pressed', true))
emit(key('hands-free', 'released', true))
await flush()
// 전사 완료
stt.resolve(result('핸즈프리 받아쓰기'))
await vi.waitFor(() => expect(completed).toEqual(['핸즈프리 받아쓰기']))
await flush()
expect(mockAudio.start).toHaveBeenCalledTimes(1)
expect(svc.isActive).toBe(false)
})
it('공개 stopSession() 은 여전히 전사·삽입이 끝날 때까지 기다린다', async () => {
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)
expect(svc.isActive).toBe(false)
})
})
describe('스크린 컨텍스트 — 창 조회가 세션을 막지 않는다', () => {
it('창 조회가 끝나기 전에 세션과 마이크가 시작되고, 결과는 LLM 입력에 합쳐진다', async () => {
screen.enabled = true
config.values.defaultLLMAction = 'refine'
const probe = deferred<ReturnType<typeof screenContext>>()
screen.captureContext.mockReturnValueOnce(probe.promise)
const svc = getVoiceModeService()
const started: string[] = []
svc.on('session-started', ({ session }) => started.push(session.id))
const starting = svc.startSession('dictation')
await Promise.race([starting, new Promise((r) => setTimeout(r, 300))])
expect(started).toHaveLength(1)
expect(mockAudio.start).toHaveBeenCalledTimes(1)
expect(screen.captureContext).toHaveBeenCalledWith(false)
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
const stopping = svc.stopSession()
probe.resolve(screenContext('Code'))
await stopping
await starting
expect(gateway.processText).toHaveBeenCalledTimes(1)
const [text, action] = gateway.processText.mock.calls[0] as [string, string]
expect(action).toBe('refine')
expect(text).toBe('[ctx app=Code sel=선택 문장]\n기본 전사')
})
})
describe('스크린 컨텍스트 — 선택 텍스트 캡처', () => {
it("원문 삽입(none) 경로에선 복사 단축키(선택 텍스트 캡처)를 보내지 않는다", async () => {
screen.enabled = true
config.values.defaultLLMAction = 'none'
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
await flush()
expect(mockInsert.insertText).toHaveBeenCalledWith('기본 전사', undefined)
expect(screen.captureSelectedText).not.toHaveBeenCalled()
})
it('LLM 경로면 정지 시점에 선택 텍스트를 캡처한다', async () => {
screen.enabled = true
config.values.defaultLLMAction = 'refine'
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(screen.captureSelectedText).toHaveBeenCalledTimes(1)
})
it("none 이어도 음성 단축키로 LLM 경로가 되면 그때 선택 텍스트를 캡처해 쓴다", async () => {
screen.enabled = true
config.values.defaultLLMAction = 'none'
voiceCommand.instructionId = 'i1'
instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' }
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(screen.captureSelectedText).toHaveBeenCalledTimes(1)
expect(gateway.processText).toHaveBeenCalledTimes(1)
const [text, action] = gateway.processText.mock.calls[0] as [string, string]
expect(action).toBe('custom')
expect(text).toContain('[ctx app=Code sel=선택 문장]')
})
})
describe('LLM 경로 — 체인 + 음성 단축키', () => {
it('체인 기본 액션에서도 음성 단축키가 지목한 지시문이 이긴다', async () => {
config.values.defaultLLMAction = 'chain'
config.values.activeChainId = 'c1'
voiceCommand.instructionId = 'i1'
instructions.byId.i1 = { id: 'i1', name: '요약', prompt: '요약해줘' }
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(chain.execute).not.toHaveBeenCalled()
expect(gateway.processText).toHaveBeenCalledTimes(1)
const [, action, , systemPrompt] = gateway.processText.mock.calls[0] as [string, string, unknown, string]
expect(action).toBe('custom')
expect(systemPrompt).toContain('요약해줘')
})
it('음성 단축키가 없으면 활성 체인을 실행한다', async () => {
config.values.defaultLLMAction = 'chain'
config.values.activeChainId = 'c1'
const svc = getVoiceModeService()
const completed: string[] = []
svc.on('session-completed', ({ finalText }) => completed.push(finalText))
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(chain.execute).toHaveBeenCalledWith('c1', '기본 전사')
expect(gateway.processText).not.toHaveBeenCalled()
expect(completed).toEqual(['체인 결과'])
})
})
describe('정지 시 꼬리 오디오', () => {
it('캡처가 정지 중에 흘려보낸 꼬리 오디오까지 전사에 넣는다', async () => {
const TAIL = Buffer.alloc(3200) // 100ms
mockAudio.stop.mockImplementation(async () => {
// AudioCaptureService.stop() 은 SoX 를 죽이기 전에 남은 오디오를 emit 한다
audioBus.emit('audio-data', { buffer: TAIL })
})
const svc = getVoiceModeService()
await svc.startSession('dictation')
audioBus.emit('audio-data', { buffer: SPEECH })
advanceClock(1000)
await svc.stopSession()
expect(mockSTT.transcribe).toHaveBeenCalledTimes(1)
const [audio] = mockSTT.transcribe.mock.calls[0] as [Buffer]
expect(audio.length).toBe(SPEECH.length + TAIL.length)
expect(audioBus.listenerCount('audio-data')).toBe(0)
})
})

View file

@ -38,6 +38,19 @@ vi.mock('../../../src/main/services/PremiumLLMService', () => ({
getPremiumLLMService: vi.fn(),
}))
// 활성 창 조회는 koffi FFI 리더(ForegroundWindowPort)로 한다 — 자식 프로세스를 띄우지 않는다.
const foregroundReader = vi.hoisted(() => ({
getForegroundWindowInfo: vi.fn(() => ({
hwnd: 1,
title: 'Test window',
processId: 1,
appName: 'TestApp.exe',
appPath: null,
})),
}))
vi.mock('../../../src/main/utils/win32-foreground', () => foregroundReader)
function createProcess(args: unknown[]): EventEmitter & {
stdout: EventEmitter
stderr: EventEmitter
@ -128,8 +141,8 @@ describe('Windows child process visibility', () => {
expectWindowsHideOnAllCalls(childProcess.exec)
})
// 장치 목록·활성 창 조회는 win32 분기에서만 PowerShell 을 부른다.
it.skipIf(process.platform !== 'win32')('hides Windows device discovery and active-window PowerShell calls', async () => {
// 장치 목록 조회는 win32 분기에서만 PowerShell 을 부른다. 활성 창 조회는 FFI 로 하므로 프로세스가 없다.
it.skipIf(process.platform !== 'win32')('hides Windows device discovery and spawns no process for the active window', async () => {
const audioModule = await import('../../../src/main/services/AudioCaptureService')
const audioService = audioModule.getAudioCaptureService() as unknown as {
_getDevicesWindows(): Promise<unknown>
@ -145,10 +158,9 @@ describe('Windows child process visibility', () => {
timeout: 3000,
windowsHide: true,
})
expect(childProcess.execFileAsync.mock.calls[0]?.[2]).toMatchObject({
timeout: 3000,
windowsHide: true,
})
expect(foregroundReader.getForegroundWindowInfo).toHaveBeenCalled()
expect(childProcess.execFileAsync).not.toHaveBeenCalled()
expect(childProcess.execFile).not.toHaveBeenCalled()
})
it('declares hidden ffmpeg processes for conversion, duration, and chunk extraction', () => {

View file

@ -0,0 +1,68 @@
// packages/core/__tests__/voice-llm-route.test.ts
// 받아쓰기 LLM 경로 결정 우선순위 표 테스트 (override > chain > instruction > action > none).
import { describe, expect, it } from 'vitest'
import { resolveVoiceLlmRoute, routeUsesLlm, type VoiceLlmRouteInput } from '../src/voice-llm-route'
const base: VoiceLlmRouteInput = {
defaultAction: 'refine',
overrideInstructionId: null,
activeChainId: null,
activeInstructionId: null,
localLlmAvailable: true,
backend: 'local',
}
function route(patch: Partial<VoiceLlmRouteInput>): ReturnType<typeof resolveVoiceLlmRoute> {
return resolveVoiceLlmRoute({ ...base, ...patch })
}
describe('resolveVoiceLlmRoute — 우선순위 표', () => {
it.each([
// [설명, 입력, 기대 경로]
['override 는 체인보다 앞선다', { defaultAction: 'chain', activeChainId: 'c1', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }],
['override 는 none 을 이긴다', { defaultAction: 'none', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }],
['override 는 일반 액션을 이긴다', { defaultAction: 'translate', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }],
['override 는 활성 지시문보다 앞선다', { defaultAction: 'custom', activeInstructionId: 'i1', overrideInstructionId: 'i9' }, { kind: 'instruction', instructionId: 'i9', source: 'override' }],
['활성 체인이 있으면 체인', { defaultAction: 'chain', activeChainId: 'c1' }, { kind: 'chain', chainId: 'c1' }],
['체인인데 활성 체인이 없으면 일반 액션 chain (기존 동작)', { defaultAction: 'chain', activeChainId: null }, { kind: 'action', action: 'chain' }],
['custom 은 활성 지시문', { defaultAction: 'custom', activeInstructionId: 'i1' }, { kind: 'instruction', instructionId: 'i1', source: 'active' }],
['custom 인데 활성 지시문이 없으면 지시문 없이 custom', { defaultAction: 'custom', activeInstructionId: '' }, { kind: 'instruction', instructionId: null, source: 'active' }],
['활성 지시문은 일반 액션을 덮어쓰지 않는다', { defaultAction: 'refine', activeInstructionId: 'i1' }, { kind: 'action', action: 'refine' }],
['translate 는 일반 액션', { defaultAction: 'translate' }, { kind: 'action', action: 'translate' }],
['none 은 원문 삽입', { defaultAction: 'none' }, { kind: 'insert-raw', reason: 'none' }],
] as const)('%s', (_label, patch, expected) => {
expect(route(patch as Partial<VoiceLlmRouteInput>)).toEqual(expected)
})
})
describe('resolveVoiceLlmRoute — 로컬 LLM 가용성', () => {
it('로컬 백엔드에서 LLM 을 쓸 수 없으면 사유를 밝혀 원문 삽입', () => {
expect(route({ localLlmAvailable: false })).toEqual({ kind: 'insert-raw', reason: 'llm-unavailable' })
expect(route({ localLlmAvailable: false, overrideInstructionId: 'i9' })).toEqual({
kind: 'insert-raw',
reason: 'llm-unavailable',
})
expect(route({ localLlmAvailable: false, defaultAction: 'chain', activeChainId: 'c1' })).toEqual({
kind: 'insert-raw',
reason: 'llm-unavailable',
})
})
it("none 은 LLM 가용성과 무관하게 'none' 사유다 (경고 없음)", () => {
expect(route({ defaultAction: 'none', localLlmAvailable: false })).toEqual({ kind: 'insert-raw', reason: 'none' })
})
it('온라인 백엔드는 로컬 가용성을 보지 않는다', () => {
expect(route({ backend: 'online', localLlmAvailable: false })).toEqual({ kind: 'action', action: 'refine' })
expect(route({ backend: undefined, localLlmAvailable: false })).toEqual({ kind: 'action', action: 'refine' })
})
})
describe('routeUsesLlm', () => {
it('원문 삽입만 LLM 을 쓰지 않는다', () => {
expect(routeUsesLlm({ kind: 'insert-raw', reason: 'none' })).toBe(false)
expect(routeUsesLlm({ kind: 'chain', chainId: 'c' })).toBe(true)
expect(routeUsesLlm({ kind: 'action', action: 'refine' })).toBe(true)
})
})

View file

@ -0,0 +1,89 @@
// packages/core/src/voice-llm-route.ts
// 받아쓰기 세션의 LLM 후처리 경로 결정 — 순수 함수만 둔다 (IO·설정 저장소 의존 없음).
//
// 예전엔 같은 결정을 VoiceModeService._transcribe(스킵 여부)와 _processWithLLM(무엇을 실행할지)이
// 서로 다른 규칙으로 두 번 내렸다. 두 사본이 어긋나 '체인' 기본 액션에서 음성 단축키가 지목한
// 지시문이 조용히 무시됐다. 이제 경로는 여기서 한 번만 정하고, 서비스는 결과에 따라 IO 만 실행한다.
//
// 우선순위: 음성 단축키(override) > 체인 > 지시문(custom) > 일반 액션 > none
// - 지시문 경로는 기본 액션이 'custom' 이거나 override 가 있을 때만 탄다.
// (활성 지시문이 있어도 refine/translate 등 일반 액션을 덮어쓰지 않는다)
// - 체인을 골랐는데 활성 체인이 없으면 일반 액션 'chain' 으로 LLM 에 넘긴다(기존 동작 유지).
// - 로컬 백엔드인데 로컬 LLM 을 쓸 수 없으면 원문 삽입으로 떨어지고, 그 이유를 밝힌다
// (이 경우에만 호출자가 경고를 띄운다).
import type { LLMAction, LLMActionSelection } from './types'
export type VoiceLlmBackend = 'local' | 'online'
export interface VoiceLlmRouteInput {
/** 설정의 기본 후처리 액션 */
defaultAction: LLMActionSelection
/** 음성 단축키가 지목한 지시문 id (없으면 null) */
overrideInstructionId: string | null
/** 활성 체인 id */
activeChainId: string | null
/** 활성 지시문 id ('custom' 기본 액션에서만 쓴다) */
activeInstructionId: string | null
/** 로컬 LLM 사용 가능 여부 */
localLlmAvailable: boolean
/** LLM 백엔드 — 알 수 없으면 undefined (가용성 검사를 건너뛴다) */
backend: VoiceLlmBackend | undefined
}
export type VoiceLlmRoute =
/** LLM 없이 원문을 삽입한다 */
| { kind: 'insert-raw'; reason: 'none' | 'llm-unavailable' }
/** 활성 체인을 실행한다 */
| { kind: 'chain'; chainId: string }
/** 지시문('custom')으로 처리한다. instructionId 가 null 이면 지시문 없이 custom 처리 */
| { kind: 'instruction'; instructionId: string | null; source: 'override' | 'active' }
/** 일반 액션으로 처리한다 */
| { kind: 'action'; action: Exclude<LLMAction, 'custom'> }
function nonEmpty(value: string | null | undefined): string | null {
return typeof value === 'string' && value.length > 0 ? value : null
}
/** LLM 이 필요한 경로를 고른다 (가용성 검사 전). */
function resolveDesiredRoute(input: VoiceLlmRouteInput): VoiceLlmRoute {
const override = nonEmpty(input.overrideInstructionId)
if (override) {
// 명시적 음성 명령은 기본 설정(none·chain 포함)이 무효화하지 않는다
return { kind: 'instruction', instructionId: override, source: 'override' }
}
switch (input.defaultAction) {
case 'none':
return { kind: 'insert-raw', reason: 'none' }
case 'chain': {
const chainId = nonEmpty(input.activeChainId)
return chainId ? { kind: 'chain', chainId } : { kind: 'action', action: 'chain' }
}
case 'custom':
return {
kind: 'instruction',
instructionId: nonEmpty(input.activeInstructionId),
source: 'active',
}
default:
return { kind: 'action', action: input.defaultAction }
}
}
/** 이 경로가 LLM 호출을 필요로 하는가 */
export function routeUsesLlm(route: VoiceLlmRoute): boolean {
return route.kind !== 'insert-raw'
}
/**
* 받아쓰기 한 건의 LLM 후처리 경로를 결정한다.
*/
export function resolveVoiceLlmRoute(input: VoiceLlmRouteInput): VoiceLlmRoute {
const desired = resolveDesiredRoute(input)
if (!routeUsesLlm(desired)) return desired
if (input.backend === 'local' && !input.localLlmAvailable) {
return { kind: 'insert-raw', reason: 'llm-unavailable' }
}
return desired
}