Phase 4 구현: Ollama LLM 연동 (다듬기, 번역, 스트리밍)
- LocalLLMService: Ollama REST API, 스트리밍 NDJSON 파싱, 5초 가용성 폴링 - 시스템 프롬프트: refine/translate/summarize/grammar/expand/custom 6개 액션 - VoiceModeService: 전사→LLM 후처리→텍스트 삽입, LLM 실패 시 원본 폴백 - LLM IPC 핸들러: status/models/process/cancel/serverUrl 8개 - StatusBar: Ollama 연결 상태 + 활성 모델 표시 - Preload: llm API 섹션 추가 - Bootstrap: llm-polling 초기화 단계 추가
This commit is contained in:
parent
517210af2f
commit
4a5cf6c819
10 changed files with 663 additions and 9 deletions
433
src/main/services/LocalLLMService.ts
Normal file
433
src/main/services/LocalLLMService.ts
Normal file
|
|
@ -0,0 +1,433 @@
|
|||
// src/main/services/LocalLLMService.ts
|
||||
// Ollama REST API를 통해 로컬 LLM과 상호작용한다.
|
||||
// 설계서 01의 ILocalLLMService 구현.
|
||||
|
||||
import { EventEmitter } from 'events'
|
||||
import { getLogger } from './LoggerService'
|
||||
import { configGet } from './ConfigService'
|
||||
import { D3ROError, ErrorCode } from '@shared/errors'
|
||||
import type { LLMStatus, LLMModel, LLMAction, LLMConnectionState } from '@shared/types'
|
||||
|
||||
const logger = getLogger('LocalLLMService')
|
||||
|
||||
// ============================================================
|
||||
// 내부 타입
|
||||
// ============================================================
|
||||
|
||||
const enum LLMState {
|
||||
Unavailable = 'unavailable',
|
||||
Available = 'available',
|
||||
Generating = 'generating',
|
||||
Error = 'error'
|
||||
}
|
||||
|
||||
interface GenerateOptions {
|
||||
model?: string
|
||||
temperature?: number
|
||||
maxTokens?: number
|
||||
systemPrompt?: string
|
||||
stream?: boolean
|
||||
}
|
||||
|
||||
interface GenerateResult {
|
||||
text: string
|
||||
model: string
|
||||
promptTokens: number
|
||||
completionTokens: number
|
||||
totalDuration: number
|
||||
}
|
||||
|
||||
interface OllamaGenerateResponse {
|
||||
model: string
|
||||
response: string
|
||||
done: boolean
|
||||
total_duration?: number
|
||||
prompt_eval_count?: number
|
||||
eval_count?: number
|
||||
}
|
||||
|
||||
interface OllamaTagsResponse {
|
||||
models: Array<{
|
||||
name: string
|
||||
size: number
|
||||
parameter_size: string
|
||||
quantization_level: string
|
||||
modified_at: string
|
||||
}>
|
||||
}
|
||||
|
||||
interface LocalLLMEvents {
|
||||
token: (payload: { token: string; done: boolean }) => void
|
||||
complete: (payload: { result: GenerateResult }) => void
|
||||
'availability-changed': (payload: { available: boolean }) => void
|
||||
error: (payload: { error: D3ROError }) => void
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// 시스템 프롬프트 (설계서 Phase 4 참조)
|
||||
// ============================================================
|
||||
|
||||
const SYSTEM_PROMPTS: Record<string, string> = {
|
||||
refine: `다음 음성 전사 텍스트를 자연스럽고 격식 있는 문어체로 다듬어주세요.
|
||||
원래 의미를 유지하면서 문법 오류를 수정하고, 불필요한 반복이나 필러를 제거하세요.
|
||||
다듬어진 텍스트만 출력하세요. 설명이나 부가 문구를 붙이지 마세요.`,
|
||||
|
||||
translate: `다음 텍스트를 {{targetLanguage}}로 번역해주세요.
|
||||
자연스럽고 정확한 번역만 출력하세요. 원문이나 설명을 붙이지 마세요.`,
|
||||
|
||||
summarize: `다음 텍스트의 핵심 내용을 3줄 이내로 요약해주세요.
|
||||
요약문만 출력하세요.`,
|
||||
|
||||
grammar: `다음 텍스트의 문법 오류만 수정해주세요.
|
||||
원래 의미와 톤을 유지하면서 문법 오류만 수정하세요.
|
||||
수정된 텍스트만 출력하세요.`,
|
||||
|
||||
expand: `다음 텍스트를 더 자세하고 풍부하게 확장해주세요.
|
||||
확장된 텍스트만 출력하세요.`
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// LocalLLMService
|
||||
// ============================================================
|
||||
|
||||
class LocalLLMService extends EventEmitter {
|
||||
private _state = LLMState.Unavailable
|
||||
private _pollInterval: ReturnType<typeof setInterval> | null = null
|
||||
private _available = false
|
||||
private _abortController: AbortController | null = null
|
||||
private _disposed = false
|
||||
|
||||
get state(): LLMState {
|
||||
return this._state
|
||||
}
|
||||
|
||||
/**
|
||||
* Ollama 가용성 폴링을 시작한다 (5초 간격).
|
||||
*/
|
||||
startPolling(): void {
|
||||
this._checkAvailability()
|
||||
this._pollInterval = setInterval(() => {
|
||||
if (this._state !== LLMState.Generating) {
|
||||
this._checkAvailability()
|
||||
}
|
||||
}, 5000)
|
||||
logger.info('Ollama availability polling started')
|
||||
}
|
||||
|
||||
stopPolling(): void {
|
||||
if (this._pollInterval) {
|
||||
clearInterval(this._pollInterval)
|
||||
this._pollInterval = null
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 비스트리밍 텍스트 생성.
|
||||
*/
|
||||
async generate(prompt: string, options?: GenerateOptions): Promise<GenerateResult> {
|
||||
if (!this._available) {
|
||||
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama 서버에 연결할 수 없습니다')
|
||||
}
|
||||
|
||||
const serverUrl = configGet('ollamaServerUrl')
|
||||
const model = options?.model ?? configGet('llmModelId') ?? 'qwen3:4b'
|
||||
|
||||
this._state = LLMState.Generating
|
||||
|
||||
try {
|
||||
const response = await fetch(`${serverUrl}/api/generate`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model,
|
||||
prompt,
|
||||
system: options?.systemPrompt,
|
||||
stream: false,
|
||||
options: {
|
||||
temperature: options?.temperature ?? 0.3,
|
||||
num_predict: options?.maxTokens ?? 2048
|
||||
}
|
||||
}),
|
||||
signal: AbortSignal.timeout(120000)
|
||||
})
|
||||
|
||||
if (!response.ok) {
|
||||
throw new D3ROError(
|
||||
ErrorCode.LLMProcessingFailed,
|
||||
`Ollama responded with ${response.status}: ${response.statusText}`
|
||||
)
|
||||
}
|
||||
|
||||
const data = (await response.json()) as OllamaGenerateResponse
|
||||
|
||||
const result: GenerateResult = {
|
||||
text: data.response,
|
||||
model: data.model,
|
||||
promptTokens: data.prompt_eval_count ?? 0,
|
||||
completionTokens: data.eval_count ?? 0,
|
||||
totalDuration: data.total_duration ? data.total_duration / 1e6 : 0
|
||||
}
|
||||
|
||||
this._state = LLMState.Available
|
||||
this.emit('complete', { result })
|
||||
return result
|
||||
} catch (error) {
|
||||
this._state = LLMState.Available
|
||||
if (error instanceof D3ROError) throw error
|
||||
throw new D3ROError(
|
||||
ErrorCode.LLMProcessingFailed,
|
||||
`LLM generation failed: ${error instanceof Error ? error.message : String(error)}`
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 스트리밍 텍스트 생성. NDJSON 파싱.
|
||||
* 반환된 AbortController로 취소 가능.
|
||||
*/
|
||||
async *streamGenerate(
|
||||
prompt: string,
|
||||
options?: Omit<GenerateOptions, 'stream'>
|
||||
): AsyncGenerator<string, GenerateResult> {
|
||||
if (!this._available) {
|
||||
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama 서버에 연결할 수 없습니다')
|
||||
}
|
||||
|
||||
const serverUrl = configGet('ollamaServerUrl')
|
||||
const model = options?.model ?? configGet('llmModelId') ?? 'qwen3:4b'
|
||||
|
||||
this._state = LLMState.Generating
|
||||
this._abortController = new AbortController()
|
||||
|
||||
try {
|
||||
const response = await fetch(`${serverUrl}/api/generate`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model,
|
||||
prompt,
|
||||
system: options?.systemPrompt,
|
||||
stream: true,
|
||||
options: {
|
||||
temperature: options?.temperature ?? 0.3,
|
||||
num_predict: options?.maxTokens ?? 2048
|
||||
}
|
||||
}),
|
||||
signal: this._abortController.signal
|
||||
})
|
||||
|
||||
if (!response.ok || !response.body) {
|
||||
throw new D3ROError(
|
||||
ErrorCode.LLMProcessingFailed,
|
||||
`Ollama responded with ${response.status}`
|
||||
)
|
||||
}
|
||||
|
||||
const reader = response.body.getReader()
|
||||
const decoder = new TextDecoder()
|
||||
let buffer = ''
|
||||
let fullText = ''
|
||||
let lastChunk: OllamaGenerateResponse | null = null
|
||||
|
||||
while (true) {
|
||||
const { done, value } = await reader.read()
|
||||
if (done) break
|
||||
|
||||
buffer += decoder.decode(value, { stream: true })
|
||||
const lines = buffer.split('\n')
|
||||
buffer = lines.pop() ?? ''
|
||||
|
||||
for (const line of lines) {
|
||||
if (!line.trim()) continue
|
||||
try {
|
||||
const chunk = JSON.parse(line) as OllamaGenerateResponse
|
||||
fullText += chunk.response
|
||||
this.emit('token', { token: chunk.response, done: chunk.done })
|
||||
yield chunk.response
|
||||
|
||||
if (chunk.done) {
|
||||
lastChunk = chunk
|
||||
}
|
||||
} catch {
|
||||
logger.warn(`Failed to parse NDJSON line: ${line.substring(0, 100)}`)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
this._state = LLMState.Available
|
||||
this._abortController = null
|
||||
|
||||
const result: GenerateResult = {
|
||||
text: fullText,
|
||||
model: lastChunk?.model ?? model,
|
||||
promptTokens: lastChunk?.prompt_eval_count ?? 0,
|
||||
completionTokens: lastChunk?.eval_count ?? 0,
|
||||
totalDuration: lastChunk?.total_duration ? lastChunk.total_duration / 1e6 : 0
|
||||
}
|
||||
|
||||
this.emit('complete', { result })
|
||||
return result
|
||||
} catch (error) {
|
||||
this._state = LLMState.Available
|
||||
this._abortController = null
|
||||
if (error instanceof D3ROError) throw error
|
||||
if (error instanceof DOMException && error.name === 'AbortError') {
|
||||
throw new D3ROError(ErrorCode.LLMProcessingCancelled, 'LLM generation cancelled')
|
||||
}
|
||||
throw new D3ROError(
|
||||
ErrorCode.LLMProcessingFailed,
|
||||
`LLM streaming failed: ${error instanceof Error ? error.message : String(error)}`
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 텍스트를 LLM 액션에 따라 처리한다.
|
||||
*/
|
||||
async processText(
|
||||
text: string,
|
||||
action: LLMAction,
|
||||
targetLanguage?: string,
|
||||
customPrompt?: string
|
||||
): Promise<string> {
|
||||
let systemPrompt: string
|
||||
|
||||
if (action === 'custom' && customPrompt) {
|
||||
systemPrompt = customPrompt
|
||||
} else if (action === 'translate') {
|
||||
systemPrompt = SYSTEM_PROMPTS.translate.replace(
|
||||
'{{targetLanguage}}',
|
||||
targetLanguage ?? 'English'
|
||||
)
|
||||
} else {
|
||||
systemPrompt = SYSTEM_PROMPTS[action] ?? SYSTEM_PROMPTS.refine
|
||||
}
|
||||
|
||||
const result = await this.generate(text, { systemPrompt })
|
||||
return result.text.trim()
|
||||
}
|
||||
|
||||
cancelGeneration(): void {
|
||||
if (this._abortController) {
|
||||
this._abortController.abort()
|
||||
this._abortController = null
|
||||
logger.info('LLM generation cancelled')
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Ollama에 설치된 모델 목록을 조회한다.
|
||||
*/
|
||||
async getModels(): Promise<LLMModel[]> {
|
||||
const serverUrl = configGet('ollamaServerUrl')
|
||||
|
||||
try {
|
||||
const response = await fetch(`${serverUrl}/api/tags`, {
|
||||
signal: AbortSignal.timeout(5000)
|
||||
})
|
||||
|
||||
if (!response.ok) return []
|
||||
|
||||
const data = (await response.json()) as OllamaTagsResponse
|
||||
|
||||
return data.models.map((m) => ({
|
||||
id: m.name,
|
||||
name: m.name,
|
||||
sizeBytes: m.size,
|
||||
parameterSize: m.parameter_size ?? '',
|
||||
quantization: m.quantization_level ?? '',
|
||||
modifiedAt: m.modified_at
|
||||
}))
|
||||
} catch {
|
||||
return []
|
||||
}
|
||||
}
|
||||
|
||||
getStatus(): LLMStatus {
|
||||
const connectionState: LLMConnectionState = this._available
|
||||
? this._state === LLMState.Generating
|
||||
? 'connecting'
|
||||
: 'connected'
|
||||
: 'disconnected'
|
||||
|
||||
return {
|
||||
connectionState,
|
||||
serverUrl: configGet('ollamaServerUrl'),
|
||||
activeModel: configGet('llmModelId'),
|
||||
serverVersion: null
|
||||
}
|
||||
}
|
||||
|
||||
isAvailable(): boolean {
|
||||
return this._available
|
||||
}
|
||||
|
||||
dispose(): void {
|
||||
this._disposed = true
|
||||
this.stopPolling()
|
||||
this.cancelGeneration()
|
||||
this.removeAllListeners()
|
||||
logger.info('LocalLLMService disposed')
|
||||
}
|
||||
|
||||
// ── 가용성 체크 ────────────────────────────────────────
|
||||
|
||||
private async _checkAvailability(): Promise<void> {
|
||||
if (this._disposed) return
|
||||
const serverUrl = configGet('ollamaServerUrl')
|
||||
|
||||
try {
|
||||
const response = await fetch(`${serverUrl}/api/tags`, {
|
||||
signal: AbortSignal.timeout(3000)
|
||||
})
|
||||
|
||||
const wasAvailable = this._available
|
||||
this._available = response.ok
|
||||
|
||||
if (!wasAvailable && this._available) {
|
||||
this._state = LLMState.Available
|
||||
this.emit('availability-changed', { available: true })
|
||||
logger.info('Ollama server connected')
|
||||
} else if (wasAvailable && !this._available) {
|
||||
this._state = LLMState.Unavailable
|
||||
this.emit('availability-changed', { available: false })
|
||||
logger.warn('Ollama server disconnected')
|
||||
}
|
||||
} catch {
|
||||
if (this._available) {
|
||||
this._available = false
|
||||
this._state = LLMState.Unavailable
|
||||
this.emit('availability-changed', { available: false })
|
||||
logger.warn('Ollama server unreachable')
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── EventEmitter 타입 오버라이드 ───────────────────────
|
||||
|
||||
override on<K extends keyof LocalLLMEvents>(event: K, listener: LocalLLMEvents[K]): this {
|
||||
return super.on(event, listener)
|
||||
}
|
||||
|
||||
override off<K extends keyof LocalLLMEvents>(event: K, listener: LocalLLMEvents[K]): this {
|
||||
return super.off(event, listener)
|
||||
}
|
||||
|
||||
override emit<K extends keyof LocalLLMEvents>(
|
||||
event: K,
|
||||
...args: Parameters<LocalLLMEvents[K]>
|
||||
): boolean {
|
||||
return super.emit(event, ...args)
|
||||
}
|
||||
}
|
||||
|
||||
// ── 싱글톤 ─────────────────────────────────────────────
|
||||
|
||||
let instance: LocalLLMService | null = null
|
||||
|
||||
export function getLocalLLMService(): LocalLLMService {
|
||||
if (!instance) {
|
||||
instance = new LocalLLMService()
|
||||
}
|
||||
return instance
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue