fix(tts): pass Windows TTS text via env and let stop() only cancel its own playback

This commit is contained in:
Yun Chan 2026-09-28 00:53:46 +09:00
parent 9cd81b48c1
commit 91d4b4974c
3 changed files with 339 additions and 95 deletions

View file

@ -2,12 +2,20 @@
// Phase 13.1: TTS 재생 서비스 // Phase 13.1: TTS 재생 서비스
// 플랫폼별 로컬 TTS — macOS `say`, Windows SAPI(PowerShell). // 플랫폼별 로컬 TTS — macOS `say`, Windows SAPI(PowerShell).
// 온라인 불필요, 완전 로컬. 문장 단위 큐 재생. // 온라인 불필요, 완전 로컬. 문장 단위 큐 재생.
//
// 책임 분리: 이 서비스는 큐·프로세스 수명·이벤트만 다룬다. 플랫폼별 명령
// 조립(텍스트 전달 방식, rate 변환)은 ./tts/tts-command 의 순수 함수가 맡는다.
import { EventEmitter } from 'events' import { EventEmitter } from 'events'
import { spawn, type ChildProcess } from 'child_process' import { spawn, type ChildProcess } from 'child_process'
import { getLogger } from './LoggerService' import { getLogger } from './LoggerService'
import { configGet } from './ConfigService' import { configGet } from './ConfigService'
import { D3ROError, ErrorCode } from '@d3ro/core/errors' import { D3ROError, ErrorCode } from '@d3ro/core/errors'
import {
buildMacSayCommand,
buildWindowsSapiCommand,
type TTSCommand,
} from './tts/tts-command'
const logger = getLogger('TTSPlaybackService') const logger = getLogger('TTSPlaybackService')
@ -15,21 +23,27 @@ class TTSPlaybackService extends EventEmitter {
private _speaking = false private _speaking = false
private _queue: string[] = [] private _queue: string[] = []
private _currentProcess: ChildProcess | null = null private _currentProcess: ChildProcess | null = null
private _cancelled = false /**
* 재생 세대. stop() 이 올리면 그 이전 세대의 재생 루프·프로세스 콜백만
* 무효가 된다. 끈적한 cancelled 플래그와 달리 stop() 이후의 새 요청에는
* 영향이 없다.
*/
private _generation = 0
get isSpeaking(): boolean { get isSpeaking(): boolean {
return this._speaking return this._speaking
} }
/** /**
* 텍스트를 음성으로 재생 (Windows SAPI). * 텍스트를 음성으로 재생.
* 큐에 추가되어 순차 재생된다. * 큐에 추가되어 순차 재생된다.
*/ */
async speak(text: string): Promise<void> { async speak(text: string): Promise<void> {
if (!text.trim()) { const trimmed = text.trim()
if (!trimmed) {
throw new D3ROError(ErrorCode.TTSTextEmpty, 'TTS text is empty') throw new D3ROError(ErrorCode.TTSTextEmpty, 'TTS text is empty')
} }
this._queue.push(text.trim()) this._queue.push(trimmed)
if (!this._speaking) { if (!this._speaking) {
await this._processQueue() await this._processQueue()
} }
@ -37,12 +51,12 @@ class TTSPlaybackService extends EventEmitter {
/** /**
* 문장 배열을 순차 재생. * 문장 배열을 순차 재생.
* LLM 스트리밍에서 문장 단위로 호출한다. * LLM 스트리밍에서 문장 단위로 호출한다. 빈 문장은 건너뛴다.
*/ */
async speakSentences(sentences: string[]): Promise<void> { async speakSentences(sentences: string[]): Promise<void> {
for (const sentence of sentences) { for (const sentence of sentences) {
if (this._cancelled) break const trimmed = sentence.trim()
this._queue.push(sentence.trim()) if (trimmed) this._queue.push(trimmed)
} }
if (!this._speaking) { if (!this._speaking) {
await this._processQueue() await this._processQueue()
@ -53,7 +67,7 @@ class TTSPlaybackService extends EventEmitter {
* 재생 중단 + 큐 비우기. * 재생 중단 + 큐 비우기.
*/ */
stop(): void { stop(): void {
this._cancelled = true this._generation++
this._queue = [] this._queue = []
if (this._currentProcess) { if (this._currentProcess) {
this._currentProcess.kill() this._currentProcess.kill()
@ -64,11 +78,11 @@ class TTSPlaybackService extends EventEmitter {
} }
private async _processQueue(): Promise<void> { private async _processQueue(): Promise<void> {
const generation = this._generation
this._speaking = true this._speaking = true
this._cancelled = false
this.emit('started') this.emit('started')
while (this._queue.length > 0 && !this._cancelled) { while (this._queue.length > 0 && generation === this._generation) {
const text = this._queue.shift()! const text = this._queue.shift()!
try { try {
await this._speakOne(text) await this._speakOne(text)
@ -77,10 +91,11 @@ class TTSPlaybackService extends EventEmitter {
} }
} }
// stop() 으로 끊긴 세대는 이미 'stopped' 를 냈고 상태도 정리됐다.
// 여기서 _speaking 을 건드리면 그 사이 시작된 새 세대를 망가뜨린다.
if (generation !== this._generation) return
this._speaking = false this._speaking = false
if (!this._cancelled) { this.emit('finished')
this.emit('finished')
}
} }
/** /**
@ -102,100 +117,57 @@ class TTSPlaybackService extends EventEmitter {
) )
} }
/** /** macOS 내장 `say` 명령어로 단일 텍스트 재생. */
* macOS 내장 `say` 명령어로 단일 텍스트 재생.
* rate 단위: words per minute (기본 180).
* 텍스트는 argv로 직접 넘기므로 shell escape 불필요. `--` 구분자로
* `-`로 시작하는 텍스트가 플래그로 해석되는 것을 방지.
*/
private _speakOneMac(text: string): Promise<void> { private _speakOneMac(text: string): Promise<void> {
return new Promise((resolve, reject) => { return this._run(buildMacSayCommand(text, this._getSpeed()))
const rate = this._getMacRate()
this._currentProcess = spawn(
'say',
['-r', String(rate), '--', text],
{ stdio: 'pipe' },
)
this._bindProcessHandlers(resolve, reject)
})
} }
/** /**
* Windows PowerShell SAPI로 단일 텍스트 재생. * Windows PowerShell SAPI로 단일 텍스트 재생.
* 텍스트는 환경변수로 전달되어 스크립트 소스에 보간되지 않는다.
*/ */
private _speakOneWindows(text: string): Promise<void> { private _speakOneWindows(text: string): Promise<void> {
return this._run(buildWindowsSapiCommand(text, this._getSpeed()))
}
/**
* 명령을 띄우고 종료까지 기다린다.
* exit 0 이거나 stop() 으로 세대가 바뀌었으면 resolve, 그 외 reject.
*/
private _run(command: TTSCommand): Promise<void> {
const generation = this._generation
return new Promise((resolve, reject) => { return new Promise((resolve, reject) => {
// 텍스트를 PowerShell 안전 문자열로 이스케이프 const child = spawn(command.command, command.args, {
const escaped = text stdio: 'pipe',
.replace(/'/g, "''") windowsHide: command.windowsHide,
.replace(/\n/g, ' ') ...(command.extraEnv ? { env: { ...process.env, ...command.extraEnv } } : {}),
.replace(/\r/g, '') })
if (!child) {
const rate = this._getWindowsRate() reject(new D3ROError(ErrorCode.ConversationTTSFailed, 'TTS process not spawned'))
return
const script = `
Add-Type -AssemblyName System.Speech
$synth = New-Object System.Speech.Synthesis.SpeechSynthesizer
$synth.Rate = ${rate}
$synth.Speak('${escaped}')
$synth.Dispose()
`
this._currentProcess = spawn('powershell', [
'-NoProfile',
'-NonInteractive',
'-Command',
script,
], { stdio: 'pipe', windowsHide: true })
this._bindProcessHandlers(resolve, reject)
})
}
/**
* spawn된 `_currentProcess`에 close/error 핸들러를 바인딩.
* mac/win 공통 처리 — cancelled 상태나 exit 0은 resolve, 그 외 reject.
*/
private _bindProcessHandlers(
resolve: () => void,
reject: (err: Error) => void,
): void {
if (!this._currentProcess) {
reject(new D3ROError(ErrorCode.ConversationTTSFailed, 'TTS process not spawned'))
return
}
this._currentProcess.on('close', (code) => {
this._currentProcess = null
if (code === 0 || this._cancelled) {
resolve()
} else {
reject(new D3ROError(ErrorCode.ConversationTTSFailed, `TTS exited with code ${code}`))
} }
}) this._currentProcess = child
this._currentProcess.on('error', (err) => {
this._currentProcess = null const release = (): void => {
reject(new D3ROError(ErrorCode.ConversationTTSFailed, `TTS error: ${err.message}`)) if (this._currentProcess === child) this._currentProcess = null
}
child.on('close', (code) => {
release()
if (code === 0 || generation !== this._generation) {
resolve()
} else {
reject(new D3ROError(ErrorCode.ConversationTTSFailed, `TTS exited with code ${code}`))
}
})
child.on('error', (err) => {
release()
reject(new D3ROError(ErrorCode.ConversationTTSFailed, `TTS error: ${err.message}`))
})
}) })
} }
/** private _getSpeed(): number | undefined {
* macOS `say` rate: words per minute. 기본 180. return configGet('ttsSpeed') as number | undefined
* ttsSpeed 0.5→90, 1.0→180, 2.0→360
*/
private _getMacRate(): number {
const speed = configGet('ttsSpeed') as number | undefined
if (!speed) return 180
return Math.round(180 * speed)
}
/**
* Windows SAPI Rate: -10(매우 느림) ~ 10(매우 빠름), 기본 0
*/
private _getWindowsRate(): number {
const speed = configGet('ttsSpeed') as number | undefined
if (!speed || speed === 1.0) return 0
// 0.5 → -5, 1.0 → 0, 2.0 → 5
return Math.round((speed - 1.0) * 5)
} }
dispose(): void { dispose(): void {

View file

@ -0,0 +1,76 @@
// src/main/services/tts/tts-command.ts
// 플랫폼별 로컬 TTS 프로세스 명령 조립 — 순수 정책(IO 없음).
//
// TTSPlaybackService 는 큐·프로세스 수명만 책임지고, "어떤 명령을 어떤 인자로
// 띄우는가"는 여기서 결정한다. 새 플랫폼은 빌더 하나를 추가하면 된다.
//
// 보안 불변식: 재생할 텍스트는 절대 스크립트 소스에 보간하지 않는다.
// - macOS: `say` 의 argv 항목(`--` 뒤)으로 전달 → 셸 해석 없음.
// - Windows: 환경변수로 전달하고 스크립트는 `$env:...` 값으로만 읽는다.
// PowerShell 은 U+2018–U+201B 도 작은따옴표로 취급하므로, 문자열 리터럴
// 이스케이프로는 LLM 출력(예: "I don’t know")을 안전하게 담을 수 없다.
/** 텍스트를 담아 PowerShell 에 넘기는 환경변수 이름 */
export const WINDOWS_TTS_TEXT_ENV = 'D3RO_TTS_TEXT'
export interface TTSCommand {
command: string
args: string[]
/** 부모 환경 위에 덧씌울 변수. 없으면 부모 환경 그대로. */
extraEnv?: Record<string, string>
windowsHide: boolean
}
/**
* macOS `say` rate: words per minute. 기본 180.
* ttsSpeed 0.5→90, 1.0→180, 2.0→360
*/
export function macSayRate(speed: number | undefined): number {
if (!speed) return 180
return Math.round(180 * speed)
}
/**
* Windows SAPI Rate: -10(매우 느림) ~ 10(매우 빠름), 기본 0.
* 0.5 → -2, 1.0 → 0, 2.0 → 5
*/
export function windowsSapiRate(speed: number | undefined): number {
if (!speed || speed === 1.0 || !Number.isFinite(speed)) return 0
// SAPI 는 범위 밖 값에 예외를 던지므로 -10..10 으로 고정한다.
return Math.max(-10, Math.min(10, Math.round((speed - 1.0) * 5)))
}
/**
* macOS 내장 `say`. 텍스트는 argv 로 직접 넘기므로 escape 불필요하고,
* `--` 구분자로 `-`로 시작하는 텍스트가 플래그로 해석되는 것을 막는다.
*/
export function buildMacSayCommand(text: string, speed: number | undefined): TTSCommand {
return {
command: 'say',
args: ['-r', String(macSayRate(speed)), '--', text],
windowsHide: false,
}
}
/**
* Windows PowerShell SAPI. 스크립트는 고정 문자열이며 rate 는 정수만 들어간다.
* 텍스트는 환경변수 값으로만 읽으므로 어떤 문자(따옴표·세미콜론·개행)도
* 코드로 해석되지 않는다.
*/
export function buildWindowsSapiCommand(text: string, speed: number | undefined): TTSCommand {
const rate = windowsSapiRate(speed)
const script = [
'Add-Type -AssemblyName System.Speech',
'$synth = New-Object System.Speech.Synthesis.SpeechSynthesizer',
`$synth.Rate = ${rate}`,
`$synth.Speak([string]$env:${WINDOWS_TTS_TEXT_ENV})`,
'$synth.Dispose()',
].join('\n')
return {
command: 'powershell',
args: ['-NoProfile', '-NonInteractive', '-Command', script],
extraEnv: { [WINDOWS_TTS_TEXT_ENV]: text },
windowsHide: true,
}
}

View file

@ -0,0 +1,196 @@
import { EventEmitter } from 'events'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import {
WINDOWS_TTS_TEXT_ENV,
buildMacSayCommand,
buildWindowsSapiCommand,
macSayRate,
windowsSapiRate,
} from '../../../src/main/services/tts/tts-command'
const childProcess = vi.hoisted(() => ({ spawn: vi.fn() }))
vi.mock('child_process', () => childProcess)
vi.mock('../../../src/main/services/LoggerService', () => ({
getLogger: () => ({ info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }),
}))
vi.mock('../../../src/main/services/ConfigService', () => ({
configGet: vi.fn(() => undefined),
}))
type FakeProcess = EventEmitter & { kill: ReturnType<typeof vi.fn> }
interface SpawnCall {
command: string
args: string[]
options: { env?: NodeJS.ProcessEnv; windowsHide?: boolean }
proc: FakeProcess
}
let calls: SpawnCall[] = []
/** true 면 spawn 된 프로세스가 바로 exit 0 으로 닫힌다. */
let autoClose = true
function spokenText(call: SpawnCall): string | undefined {
return call.options.env?.[WINDOWS_TTS_TEXT_ENV]
}
const originalPlatform = process.platform
beforeEach(() => {
vi.resetModules()
calls = []
autoClose = true
Object.defineProperty(process, 'platform', { value: 'win32' })
childProcess.spawn.mockImplementation(
(command: string, args: string[], options: SpawnCall['options']) => {
const proc = Object.assign(new EventEmitter(), {
kill: vi.fn(() => {
queueMicrotask(() => proc.emit('close', null))
}),
})
calls.push({ command, args, options, proc })
if (autoClose) queueMicrotask(() => proc.emit('close', 0))
return proc
},
)
})
afterEach(() => {
Object.defineProperty(process, 'platform', { value: originalPlatform })
})
async function loadService() {
const mod = await import('../../../src/main/services/TTSPlaybackService')
mod.resetTTSPlaybackServiceForTests()
return mod.getTTSPlaybackService()
}
async function flush(): Promise<void> {
for (let i = 0; i < 5; i++) await Promise.resolve()
}
describe('tts-command (순수 정책)', () => {
it('Windows 스크립트에 텍스트를 보간하지 않고 환경변수로만 전달한다', () => {
const payload = "x’); Write-Output INJECTED; (’y"
const cmd = buildWindowsSapiCommand(payload, 1)
expect(cmd.command).toBe('powershell')
expect(cmd.windowsHide).toBe(true)
expect(cmd.args.join(' ')).not.toContain('INJECTED')
expect(cmd.args.join(' ')).toContain(`$env:${WINDOWS_TTS_TEXT_ENV}`)
expect(cmd.extraEnv).toEqual({ [WINDOWS_TTS_TEXT_ENV]: payload })
})
it.each(['I don’t know', 'It‘s', '‚odd‛', "plain 'ascii'", '줄1\n줄2'])(
'따옴표·개행이 섞인 텍스트(%s)도 원문 그대로 전달된다',
(text) => {
const cmd = buildWindowsSapiCommand(text, undefined)
expect(cmd.extraEnv?.[WINDOWS_TTS_TEXT_ENV]).toBe(text)
expect(cmd.args[3]).not.toContain(text)
},
)
it('macOS say 는 -- 뒤 argv 로 전달한다', () => {
const cmd = buildMacSayCommand('-v hi', 0.5)
expect(cmd).toEqual({ command: 'say', args: ['-r', '90', '--', '-v hi'], windowsHide: false })
})
it('rate 변환을 유지하고 SAPI 범위를 넘지 않는다', () => {
expect(macSayRate(undefined)).toBe(180)
expect(macSayRate(2)).toBe(360)
expect(windowsSapiRate(undefined)).toBe(0)
expect(windowsSapiRate(1)).toBe(0)
expect(windowsSapiRate(0.5)).toBe(-2)
expect(windowsSapiRate(2)).toBe(5)
expect(windowsSapiRate(10)).toBe(10)
expect(windowsSapiRate(Number.NaN)).toBe(0)
expect(windowsSapiRate(Number.POSITIVE_INFINITY)).toBe(0)
})
})
describe('TTSPlaybackService — Windows 텍스트 전달', () => {
it('타이포그래픽 따옴표가 든 문장을 스크립트가 아니라 환경변수로 넘긴다', async () => {
const tts = await loadService()
await tts.speak('I don’t know')
expect(calls).toHaveLength(1)
const [call] = calls
expect(call.command).toBe('powershell')
expect(call.options.windowsHide).toBe(true)
expect(call.args.join(' ')).not.toContain('don’t')
expect(spokenText(call)).toBe('I don’t know')
// 부모 환경은 유지된다
expect(call.options.env?.PATH ?? call.options.env?.Path).toBe(process.env.PATH ?? process.env.Path)
})
})
describe('TTSPlaybackService — stop() 이후 새 요청', () => {
it('stop() 직후 speakSentences 가 모든 문장을 재생한다', async () => {
const tts = await loadService()
const finished = vi.fn()
tts.on('finished', finished)
tts.stop()
await tts.speakSentences(['hello', 'world'])
expect(calls.map(spokenText)).toEqual(['hello', 'world'])
expect(finished).toHaveBeenCalledTimes(1)
})
it('빈 문장은 건너뛴다', async () => {
const tts = await loadService()
await tts.speakSentences([' ', 'a', ''])
expect(calls.map(spokenText)).toEqual(['a'])
})
it('재생 중 stop() 후 새 요청이 와도 이전 루프가 새 큐를 가로채지 않는다', async () => {
autoClose = false
const tts = await loadService()
const finished = vi.fn()
tts.on('finished', finished)
const first = tts.speakSentences(['a', 'b'])
await flush()
expect(calls.map(spokenText)).toEqual(['a'])
tts.stop()
expect(calls[0].proc.kill).toHaveBeenCalledTimes(1)
const second = tts.speakSentences(['c', 'd'])
await flush()
// 이전 프로세스(kill → close null)가 닫혀도 이전 루프는 종료돼야 한다
await first
expect(calls.map(spokenText)).toEqual(['a', 'c'])
expect(tts.isSpeaking).toBe(true)
calls[1].proc.emit('close', 0)
await flush()
expect(calls.map(spokenText)).toEqual(['a', 'c', 'd'])
calls[2].proc.emit('close', 0)
await second
expect(calls.map(spokenText)).toEqual(['a', 'c', 'd'])
expect(finished).toHaveBeenCalledTimes(1)
expect(tts.isSpeaking).toBe(false)
})
it('stop() 으로 끊긴 요청은 finished 를 내지 않는다', async () => {
autoClose = false
const tts = await loadService()
const finished = vi.fn()
const stopped = vi.fn()
tts.on('finished', finished)
tts.on('stopped', stopped)
const pending = tts.speakSentences(['a', 'b'])
await flush()
tts.stop()
await pending
expect(calls.map(spokenText)).toEqual(['a'])
expect(stopped).toHaveBeenCalledTimes(1)
expect(finished).not.toHaveBeenCalled()
expect(tts.isSpeaking).toBe(false)
})
})