// src/main/services/LocalLLMService.ts
// Ollama REST API를 통해 로컬 LLM과 상호작용한다.
// 설계서 01의 ILocalLLMService 구현.
import { EventEmitter } from 'events'
import { spawn } from 'child_process'
import * as fs from 'fs'
import * as path from 'path'
import { getLogger } from './LoggerService'
import { configGet } from './ConfigService'
import { D3ROError, ErrorCode } from '@d3ro/core/errors'
import type { LLMStatus, LLMModel, LLMAction, LLMConnectionState } from '@d3ro/core/types'
import { resolveSystemPrompt } from './llm-prompts'
const logger = getLogger('LocalLLMService')
// ============================================================
// 내부 타입
// ============================================================
const enum LLMState {
Unavailable = 'unavailable',
Available = 'available',
Generating = 'generating',
Error = 'error'
}
interface GenerateOptions {
model?: string
temperature?: number
maxTokens?: number
systemPrompt?: string
stream?: boolean
}
interface GenerateResult {
text: string
model: string
promptTokens: number
completionTokens: number
totalDuration: number
}
interface OllamaGenerateResponse {
model: string
response: string
done: boolean
total_duration?: number
prompt_eval_count?: number
eval_count?: number
}
interface OllamaTagsResponse {
models: Array<{
name: string
size: number
parameter_size: string
quantization_level: string
modified_at: string
}>
}
interface LocalLLMEvents {
token: (payload: { token: string; done: boolean }) => void
complete: (payload: { result: GenerateResult }) => void
'availability-changed': (payload: { available: boolean }) => void
error: (payload: { error: D3ROError }) => void
}
// ============================================================
// 시스템 프롬프트 (설계서 Phase 4 참조)
// ============================================================
// 기본 권장 모델은 `gemma4:e4b` (non-reasoning). 기본값으로 thinking mode가
// 꺼져 있어 추가 토큰이 필요 없지만, 사용자가 수동으로 qwen3/deepseek-r1 등
// reasoning 모델로 교체했을 때를 대비한 2중 방어:
// (1) 아래 `/no_think` 시스템 프롬프트 토큰 (qwen3 계열 전용 힌트, 타 모델은 무시)
// (2) Ollama 요청 body의 `think: false` 파라미터 (Ollama v0.20.0+)
// (3) `stripReasoningBlocks()` 출력 가드
const NO_THINK = '/no_think'
/**
* Reasoning model(qwen3, deepseek-r1 등)이 응답에 포함하는
* ... 블록을 제거한다. /no_think 토큰을 무시하는
* 모델에서도 안전하게 동작하도록.
*/
function stripReasoningBlocks(text: string): string {
return text
.replace(/[\s\S]*?<\/think>\s*/gi, '')
.replace(/[\s\S]*?<\/thinking>\s*/gi, '')
.trim()
}
// ============================================================
// LocalLLMService
// ============================================================
class LocalLLMService extends EventEmitter {
private _state = LLMState.Unavailable
private _pollInterval: ReturnType | null = null
private _available = false
private _abortController: AbortController | null = null
private _disposed = false
get state(): LLMState {
return this._state
}
/**
* Ollama 서버가 실행 중인지 확인하고, 설치되어 있는데 실행 중이 아니면 자동 실행한다.
*
* 반환값:
* - 'running': 이미 실행 중
* - 'starting': 바이너리를 찾아 detached 스폰 완료 (준비 확인은 폴링이 담당)
* - 'not-installed': Ollama 바이너리를 찾을 수 없음
* - 'failed': 스폰 시도 실패
*/
async ensureRunning(): Promise<'running' | 'starting' | 'not-installed' | 'failed'> {
if (await this._ping(1500)) {
logger.info('Ollama server already running')
return 'running'
}
const binaryPath = await this._findOllamaBinary()
if (!binaryPath) {
logger.warn('Ollama binary not found — install from https://ollama.com')
return 'not-installed'
}
logger.info(`Ollama not running, auto-starting from ${binaryPath}`)
try {
const child = spawn(binaryPath, ['serve'], {
detached: true,
stdio: 'ignore',
windowsHide: true
})
child.unref()
logger.info('Ollama serve spawned (detached) — polling will detect readiness')
return 'starting'
} catch (error) {
logger.error(
`Failed to spawn ollama serve: ${error instanceof Error ? error.message : String(error)}`
)
return 'failed'
}
}
/**
* Ollama /api/tags 엔드포인트로 가용성 핑. 지정 타임아웃 내 응답이 오면 true.
*/
private async _ping(timeoutMs: number): Promise {
const serverUrl = configGet('ollamaServerUrl')
try {
const response = await fetch(`${serverUrl}/api/tags`, {
signal: AbortSignal.timeout(timeoutMs)
})
return response.ok
} catch {
return false
}
}
/**
* 플랫폼별 기본 설치 경로 + PATH 탐색으로 ollama 바이너리 위치를 찾는다.
*/
private async _findOllamaBinary(): Promise {
const candidates: string[] = []
if (process.platform === 'win32') {
const localAppData = process.env.LOCALAPPDATA
if (localAppData) {
candidates.push(path.join(localAppData, 'Programs', 'Ollama', 'ollama.exe'))
}
const programFiles = process.env['ProgramFiles']
if (programFiles) {
candidates.push(path.join(programFiles, 'Ollama', 'ollama.exe'))
}
} else if (process.platform === 'darwin') {
candidates.push('/usr/local/bin/ollama', '/opt/homebrew/bin/ollama')
} else {
candidates.push('/usr/local/bin/ollama', '/usr/bin/ollama')
}
for (const candidate of candidates) {
try {
await fs.promises.access(candidate, fs.constants.X_OK)
return candidate
} catch {
// 다음 후보 시도
}
}
return await this._whichOllama()
}
/**
* `where ollama` (win) / `which ollama` (mac/linux)로 PATH에서 바이너리를 찾는다.
*/
private _whichOllama(): Promise {
return new Promise((resolve) => {
const cmd = process.platform === 'win32' ? 'where' : 'which'
const proc = spawn(cmd, ['ollama'], {
stdio: ['ignore', 'pipe', 'ignore'],
windowsHide: true
})
let out = ''
proc.stdout.on('data', (chunk: Buffer) => {
out += chunk.toString()
})
proc.on('close', (code) => {
if (code === 0) {
const firstLine = out.split(/\r?\n/).find((line) => line.trim().length > 0)
resolve(firstLine ? firstLine.trim() : null)
} else {
resolve(null)
}
})
proc.on('error', () => resolve(null))
})
}
/**
* Ollama 가용성 폴링을 시작한다 (5초 간격).
*/
startPolling(): void {
this._checkAvailability()
this._pollInterval = setInterval(() => {
if (this._state !== LLMState.Generating) {
this._checkAvailability()
}
}, 5000)
logger.info('Ollama availability polling started')
}
stopPolling(): void {
if (this._pollInterval) {
clearInterval(this._pollInterval)
this._pollInterval = null
}
}
/**
* 비스트리밍 텍스트 생성.
*/
async generate(prompt: string, options?: GenerateOptions): Promise {
if (!this._available) {
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama 서버에 연결할 수 없습니다')
}
const serverUrl = configGet('ollamaServerUrl')
const model = options?.model ?? configGet('llmModelId') ?? 'gemma4:e4b'
this._state = LLMState.Generating
try {
const response = await fetch(`${serverUrl}/api/generate`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model,
prompt,
system: options?.systemPrompt,
stream: false,
// Ollama v0.20+ think 파라미터: reasoning 모델에서 thinking 토큰 생성 중단.
// gemma4/llama3.2 등 non-reasoning 모델에서는 무시됨.
think: false,
options: {
temperature: options?.temperature ?? 0.3,
num_predict: options?.maxTokens ?? 2048
}
}),
signal: AbortSignal.timeout(120000)
})
if (!response.ok) {
throw new D3ROError(
ErrorCode.LLMProcessingFailed,
`Ollama responded with ${response.status}: ${response.statusText}`
)
}
const data = (await response.json()) as OllamaGenerateResponse
const result: GenerateResult = {
text: data.response,
model: data.model,
promptTokens: data.prompt_eval_count ?? 0,
completionTokens: data.eval_count ?? 0,
totalDuration: data.total_duration ? data.total_duration / 1e6 : 0
}
this._state = LLMState.Available
this.emit('complete', { result })
return result
} catch (error) {
this._state = LLMState.Available
if (error instanceof D3ROError) throw error
throw new D3ROError(
ErrorCode.LLMProcessingFailed,
`LLM generation failed: ${error instanceof Error ? error.message : String(error)}`
)
}
}
/**
* 스트리밍 텍스트 생성. NDJSON 파싱.
* 반환된 AbortController로 취소 가능.
*/
async *streamGenerate(
prompt: string,
options?: Omit
): AsyncGenerator {
if (!this._available) {
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama 서버에 연결할 수 없습니다')
}
const serverUrl = configGet('ollamaServerUrl')
const model = options?.model ?? configGet('llmModelId') ?? 'gemma4:e4b'
this._state = LLMState.Generating
this._abortController = new AbortController()
try {
const response = await fetch(`${serverUrl}/api/generate`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model,
prompt,
system: options?.systemPrompt,
stream: true,
think: false,
options: {
temperature: options?.temperature ?? 0.3,
num_predict: options?.maxTokens ?? 2048
}
}),
signal: this._abortController.signal
})
if (!response.ok || !response.body) {
throw new D3ROError(
ErrorCode.LLMProcessingFailed,
`Ollama responded with ${response.status}`
)
}
const reader = response.body.getReader()
const decoder = new TextDecoder()
let buffer = ''
let fullText = ''
let lastChunk: OllamaGenerateResponse | null = null
while (true) {
const { done, value } = await reader.read()
if (done) break
buffer += decoder.decode(value, { stream: true })
const lines = buffer.split('\n')
buffer = lines.pop() ?? ''
for (const line of lines) {
if (!line.trim()) continue
try {
const chunk = JSON.parse(line) as OllamaGenerateResponse
fullText += chunk.response
this.emit('token', { token: chunk.response, done: chunk.done })
yield chunk.response
if (chunk.done) {
lastChunk = chunk
}
} catch {
logger.warn(`Failed to parse NDJSON line: ${line.substring(0, 100)}`)
}
}
}
this._state = LLMState.Available
this._abortController = null
const result: GenerateResult = {
text: fullText,
model: lastChunk?.model ?? model,
promptTokens: lastChunk?.prompt_eval_count ?? 0,
completionTokens: lastChunk?.eval_count ?? 0,
totalDuration: lastChunk?.total_duration ? lastChunk.total_duration / 1e6 : 0
}
this.emit('complete', { result })
return result
} catch (error) {
this._state = LLMState.Available
this._abortController = null
if (error instanceof D3ROError) throw error
if (error instanceof DOMException && error.name === 'AbortError') {
throw new D3ROError(ErrorCode.LLMProcessingCancelled, 'LLM generation cancelled')
}
throw new D3ROError(
ErrorCode.LLMProcessingFailed,
`LLM streaming failed: ${error instanceof Error ? error.message : String(error)}`
)
}
}
/**
* 텍스트를 LLM 액션에 따라 처리한다.
*/
async processText(
text: string,
action: LLMAction,
targetLanguage?: string,
customPrompt?: string
): Promise {
// Phase 11: LLM 처리 쿼터 체크
try {
const { getLicenseService } = await import('./LicenseService')
const { Feature } = await import('@d3ro/core/types')
const license = getLicenseService()
const access = license.canUse(Feature.LLM_PROCESS)
if (!access.allowed) {
license.promptUpgrade(Feature.LLM_PROCESS, access.reason === 'quota_exceeded' ? 'quota_exceeded' : 'tier_required')
// LLM 처리 차단 시 원본 텍스트 반환 (폴백)
return text
}
license.consumeQuota(Feature.LLM_PROCESS)
} catch {
// LicenseService 미초기화 시 허용
}
const basePrompt = resolveSystemPrompt(action, targetLanguage, customPrompt)
const systemPrompt = `${NO_THINK}\n${basePrompt}`
const result = await this.generate(text, { systemPrompt })
const cleaned = stripReasoningBlocks(result.text)
// reasoning 블록 제거 후 빈 응답이면 원본 텍스트 폴백
// (모델이 thinking만 하고 출력은 안 한 경우 / 응답 파싱 실패 케이스)
if (cleaned.length === 0) {
logger.warn(
`LLM returned empty after reasoning strip — falling back to original transcript ` +
`(raw length=${result.text.length})`
)
return text
}
return cleaned
}
cancelGeneration(): void {
if (this._abortController) {
this._abortController.abort()
this._abortController = null
logger.info('LLM generation cancelled')
}
}
/**
* Ollama에 설치된 모델 목록을 조회한다.
*/
async getModels(): Promise {
const serverUrl = configGet('ollamaServerUrl')
try {
const response = await fetch(`${serverUrl}/api/tags`, {
signal: AbortSignal.timeout(5000)
})
if (!response.ok) return []
const data = (await response.json()) as OllamaTagsResponse
return data.models.map((m) => ({
id: m.name,
name: m.name,
sizeBytes: m.size,
parameterSize: m.parameter_size ?? '',
quantization: m.quantization_level ?? '',
modifiedAt: m.modified_at
}))
} catch {
return []
}
}
getStatus(): LLMStatus {
const connectionState: LLMConnectionState = this._available
? this._state === LLMState.Generating
? 'connecting'
: 'connected'
: 'disconnected'
return {
connectionState,
serverUrl: configGet('ollamaServerUrl'),
activeModel: configGet('llmModelId'),
serverVersion: null
}
}
isAvailable(): boolean {
return this._available
}
/**
* Ollama /api/chat 스트리밍 대화.
* messages 배열로 대화 히스토리를 전달한다.
* 각 토큰마다 yield, 완료 시 전체 응답 텍스트를 return.
*/
async *chatStream(
messages: Array<{ role: string; content: string }>,
options?: { model?: string; temperature?: number },
): AsyncGenerator {
if (!this._available) {
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama server not available')
}
const serverUrl = configGet('ollamaServerUrl')
const model = options?.model ?? configGet('llmModelId') ?? 'gemma4:e4b'
this._abortController = new AbortController()
this._state = LLMState.Generating
try {
const response = await fetch(`${serverUrl}/api/chat`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model,
messages,
stream: true,
think: false,
options: {
temperature: options?.temperature ?? 0.7,
},
}),
signal: this._abortController.signal,
})
if (!response.ok || !response.body) {
throw new D3ROError(ErrorCode.LLMProcessingFailed, `Chat API error: ${response.status}`)
}
const reader = response.body.getReader()
const decoder = new TextDecoder()
let buffer = ''
let accumulated = ''
while (true) {
const { done, value } = await reader.read()
if (done) break
buffer += decoder.decode(value, { stream: true })
const lines = buffer.split('\n')
buffer = lines.pop() ?? ''
for (const line of lines) {
if (!line.trim()) continue
try {
const chunk = JSON.parse(line) as { message?: { content: string }; done: boolean }
if (chunk.message?.content) {
accumulated += chunk.message.content
yield chunk.message.content
}
if (chunk.done) {
return accumulated
}
} catch {
// 불완전 JSON 무시
}
}
}
return accumulated
} finally {
this._state = this._available ? LLMState.Available : LLMState.Unavailable
this._abortController = null
}
}
dispose(): void {
this._disposed = true
this.stopPolling()
this.cancelGeneration()
this.removeAllListeners()
logger.info('LocalLLMService disposed')
}
// ── 가용성 체크 ────────────────────────────────────────
private async _checkAvailability(): Promise {
if (this._disposed) return
const serverUrl = configGet('ollamaServerUrl')
try {
const response = await fetch(`${serverUrl}/api/tags`, {
signal: AbortSignal.timeout(3000)
})
const wasAvailable = this._available
this._available = response.ok
if (!wasAvailable && this._available) {
this._state = LLMState.Available
this.emit('availability-changed', { available: true })
logger.info('Ollama server connected')
} else if (wasAvailable && !this._available) {
this._state = LLMState.Unavailable
this.emit('availability-changed', { available: false })
logger.warn('Ollama server disconnected')
}
} catch {
if (this._available) {
this._available = false
this._state = LLMState.Unavailable
this.emit('availability-changed', { available: false })
logger.warn('Ollama server unreachable')
}
}
}
// ── EventEmitter 타입 오버라이드 ───────────────────────
override on(event: K, listener: LocalLLMEvents[K]): this {
return super.on(event, listener)
}
override off(event: K, listener: LocalLLMEvents[K]): this {
return super.off(event, listener)
}
override emit(
event: K,
...args: Parameters
): boolean {
return super.emit(event, ...args)
}
}
// ── 싱글톤 ─────────────────────────────────────────────
let instance: LocalLLMService | null = null
export function getLocalLLMService(): LocalLLMService {
if (!instance) {
instance = new LocalLLMService()
}
return instance
}