증상:
- STT 426ms로 빠른데 LLM refine 단계가 39초 소요 후 빈 문자열 반환
- _completeSession('') → finalText.length === 0이라 paste 호출 자체 건너뜀
- 사용자에게는 '느리고 paste 안 됨'으로 보임
원인:
- 사용자가 ollama pull qwen3:4b 한 직후 첫 호출 (모델 cold start 일부 있음)
- qwen3는 reasoning model이라 응답에 <think>...</think> 블록을 길게 출력
- system prompt에 '/no_think' 토큰 없음 → reasoning mode ON
- generate()는 data.response.trim() 그대로 반환 → think 블록 + 빈 본문이면
trim 후 빈 문자열
- 빈 문자열에 대한 fallback이 없어서 그대로 _completeSession('')
수정:
- SYSTEM_PROMPTS 모두에 '/no_think' 헤더 추가
- qwen3 reasoning 비활성화 → 응답 속도 대폭 단축
- 다른 모델(llama, mistral, gemma)은 토큰 무시 → 호환성 OK
- stripReasoningBlocks() 추가
- <think>...</think> + <thinking>...</thinking> 블록 제거 (gi flag)
- /no_think를 무시하는 모델 + 응답에 think tag가 섞여 들어오는 케이스 안전망
- processText() 결과:
- stripReasoningBlocks(result.text)
- 빈 문자열이면 원본 transcript fallback + warn 로그
673 lines
20 KiB
TypeScript
673 lines
20 KiB
TypeScript
// src/main/services/LocalLLMService.ts
|
|
// Ollama REST API를 통해 로컬 LLM과 상호작용한다.
|
|
// 설계서 01의 ILocalLLMService 구현.
|
|
|
|
import { EventEmitter } from 'events'
|
|
import { spawn } from 'child_process'
|
|
import * as fs from 'fs'
|
|
import * as path from 'path'
|
|
import { getLogger } from './LoggerService'
|
|
import { configGet } from './ConfigService'
|
|
import { D3ROError, ErrorCode } from '@d3ro/core/errors'
|
|
import type { LLMStatus, LLMModel, LLMAction, LLMConnectionState } from '@d3ro/core/types'
|
|
|
|
const logger = getLogger('LocalLLMService')
|
|
|
|
// ============================================================
|
|
// 내부 타입
|
|
// ============================================================
|
|
|
|
const enum LLMState {
|
|
Unavailable = 'unavailable',
|
|
Available = 'available',
|
|
Generating = 'generating',
|
|
Error = 'error'
|
|
}
|
|
|
|
interface GenerateOptions {
|
|
model?: string
|
|
temperature?: number
|
|
maxTokens?: number
|
|
systemPrompt?: string
|
|
stream?: boolean
|
|
}
|
|
|
|
interface GenerateResult {
|
|
text: string
|
|
model: string
|
|
promptTokens: number
|
|
completionTokens: number
|
|
totalDuration: number
|
|
}
|
|
|
|
interface OllamaGenerateResponse {
|
|
model: string
|
|
response: string
|
|
done: boolean
|
|
total_duration?: number
|
|
prompt_eval_count?: number
|
|
eval_count?: number
|
|
}
|
|
|
|
interface OllamaTagsResponse {
|
|
models: Array<{
|
|
name: string
|
|
size: number
|
|
parameter_size: string
|
|
quantization_level: string
|
|
modified_at: string
|
|
}>
|
|
}
|
|
|
|
interface LocalLLMEvents {
|
|
token: (payload: { token: string; done: boolean }) => void
|
|
complete: (payload: { result: GenerateResult }) => void
|
|
'availability-changed': (payload: { available: boolean }) => void
|
|
error: (payload: { error: D3ROError }) => void
|
|
}
|
|
|
|
// ============================================================
|
|
// 시스템 프롬프트 (설계서 Phase 4 참조)
|
|
// ============================================================
|
|
|
|
// 시스템 프롬프트.
|
|
// `/no_think`는 qwen3 계열 reasoning model의 thinking mode를 비활성화하는 토큰.
|
|
// 다른 모델에서는 무시되므로 호환성에 문제 없음.
|
|
const NO_THINK = '/no_think'
|
|
|
|
const SYSTEM_PROMPTS: Record<string, string> = {
|
|
refine: `${NO_THINK}
|
|
다음 음성 전사 텍스트를 자연스럽고 격식 있는 문어체로 다듬어주세요.
|
|
원래 의미를 유지하면서 문법 오류를 수정하고, 불필요한 반복이나 필러를 제거하세요.
|
|
다듬어진 텍스트만 출력하세요. 설명이나 부가 문구를 붙이지 마세요.`,
|
|
|
|
translate: `${NO_THINK}
|
|
다음 텍스트를 {{targetLanguage}}로 번역해주세요.
|
|
자연스럽고 정확한 번역만 출력하세요. 원문이나 설명을 붙이지 마세요.`,
|
|
|
|
summarize: `${NO_THINK}
|
|
다음 텍스트의 핵심 내용을 3줄 이내로 요약해주세요.
|
|
요약문만 출력하세요.`,
|
|
|
|
grammar: `${NO_THINK}
|
|
다음 텍스트의 문법 오류만 수정해주세요.
|
|
원래 의미와 톤을 유지하면서 문법 오류만 수정하세요.
|
|
수정된 텍스트만 출력하세요.`,
|
|
|
|
expand: `${NO_THINK}
|
|
다음 텍스트를 더 자세하고 풍부하게 확장해주세요.
|
|
확장된 텍스트만 출력하세요.`
|
|
}
|
|
|
|
/**
|
|
* Reasoning model(qwen3, deepseek-r1 등)이 응답에 포함하는
|
|
* <think>...</think> 블록을 제거한다. /no_think 토큰을 무시하는
|
|
* 모델에서도 안전하게 동작하도록.
|
|
*/
|
|
function stripReasoningBlocks(text: string): string {
|
|
return text
|
|
.replace(/<think>[\s\S]*?<\/think>\s*/gi, '')
|
|
.replace(/<thinking>[\s\S]*?<\/thinking>\s*/gi, '')
|
|
.trim()
|
|
}
|
|
|
|
// ============================================================
|
|
// LocalLLMService
|
|
// ============================================================
|
|
|
|
class LocalLLMService extends EventEmitter {
|
|
private _state = LLMState.Unavailable
|
|
private _pollInterval: ReturnType<typeof setInterval> | null = null
|
|
private _available = false
|
|
private _abortController: AbortController | null = null
|
|
private _disposed = false
|
|
|
|
get state(): LLMState {
|
|
return this._state
|
|
}
|
|
|
|
/**
|
|
* Ollama 서버가 실행 중인지 확인하고, 설치되어 있는데 실행 중이 아니면 자동 실행한다.
|
|
*
|
|
* 반환값:
|
|
* - 'running': 이미 실행 중
|
|
* - 'starting': 바이너리를 찾아 detached 스폰 완료 (준비 확인은 폴링이 담당)
|
|
* - 'not-installed': Ollama 바이너리를 찾을 수 없음
|
|
* - 'failed': 스폰 시도 실패
|
|
*/
|
|
async ensureRunning(): Promise<'running' | 'starting' | 'not-installed' | 'failed'> {
|
|
if (await this._ping(1500)) {
|
|
logger.info('Ollama server already running')
|
|
return 'running'
|
|
}
|
|
|
|
const binaryPath = await this._findOllamaBinary()
|
|
if (!binaryPath) {
|
|
logger.warn('Ollama binary not found — install from https://ollama.com')
|
|
return 'not-installed'
|
|
}
|
|
|
|
logger.info(`Ollama not running, auto-starting from ${binaryPath}`)
|
|
|
|
try {
|
|
const child = spawn(binaryPath, ['serve'], {
|
|
detached: true,
|
|
stdio: 'ignore',
|
|
windowsHide: true
|
|
})
|
|
child.unref()
|
|
logger.info('Ollama serve spawned (detached) — polling will detect readiness')
|
|
return 'starting'
|
|
} catch (error) {
|
|
logger.error(
|
|
`Failed to spawn ollama serve: ${error instanceof Error ? error.message : String(error)}`
|
|
)
|
|
return 'failed'
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Ollama /api/tags 엔드포인트로 가용성 핑. 지정 타임아웃 내 응답이 오면 true.
|
|
*/
|
|
private async _ping(timeoutMs: number): Promise<boolean> {
|
|
const serverUrl = configGet('ollamaServerUrl')
|
|
try {
|
|
const response = await fetch(`${serverUrl}/api/tags`, {
|
|
signal: AbortSignal.timeout(timeoutMs)
|
|
})
|
|
return response.ok
|
|
} catch {
|
|
return false
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 플랫폼별 기본 설치 경로 + PATH 탐색으로 ollama 바이너리 위치를 찾는다.
|
|
*/
|
|
private async _findOllamaBinary(): Promise<string | null> {
|
|
const candidates: string[] = []
|
|
|
|
if (process.platform === 'win32') {
|
|
const localAppData = process.env.LOCALAPPDATA
|
|
if (localAppData) {
|
|
candidates.push(path.join(localAppData, 'Programs', 'Ollama', 'ollama.exe'))
|
|
}
|
|
const programFiles = process.env['ProgramFiles']
|
|
if (programFiles) {
|
|
candidates.push(path.join(programFiles, 'Ollama', 'ollama.exe'))
|
|
}
|
|
} else if (process.platform === 'darwin') {
|
|
candidates.push('/usr/local/bin/ollama', '/opt/homebrew/bin/ollama')
|
|
} else {
|
|
candidates.push('/usr/local/bin/ollama', '/usr/bin/ollama')
|
|
}
|
|
|
|
for (const candidate of candidates) {
|
|
try {
|
|
await fs.promises.access(candidate, fs.constants.X_OK)
|
|
return candidate
|
|
} catch {
|
|
// 다음 후보 시도
|
|
}
|
|
}
|
|
|
|
return await this._whichOllama()
|
|
}
|
|
|
|
/**
|
|
* `where ollama` (win) / `which ollama` (mac/linux)로 PATH에서 바이너리를 찾는다.
|
|
*/
|
|
private _whichOllama(): Promise<string | null> {
|
|
return new Promise((resolve) => {
|
|
const cmd = process.platform === 'win32' ? 'where' : 'which'
|
|
const proc = spawn(cmd, ['ollama'], {
|
|
stdio: ['ignore', 'pipe', 'ignore'],
|
|
windowsHide: true
|
|
})
|
|
let out = ''
|
|
proc.stdout.on('data', (chunk: Buffer) => {
|
|
out += chunk.toString()
|
|
})
|
|
proc.on('close', (code) => {
|
|
if (code === 0) {
|
|
const firstLine = out.split(/\r?\n/).find((line) => line.trim().length > 0)
|
|
resolve(firstLine ? firstLine.trim() : null)
|
|
} else {
|
|
resolve(null)
|
|
}
|
|
})
|
|
proc.on('error', () => resolve(null))
|
|
})
|
|
}
|
|
|
|
/**
|
|
* Ollama 가용성 폴링을 시작한다 (5초 간격).
|
|
*/
|
|
startPolling(): void {
|
|
this._checkAvailability()
|
|
this._pollInterval = setInterval(() => {
|
|
if (this._state !== LLMState.Generating) {
|
|
this._checkAvailability()
|
|
}
|
|
}, 5000)
|
|
logger.info('Ollama availability polling started')
|
|
}
|
|
|
|
stopPolling(): void {
|
|
if (this._pollInterval) {
|
|
clearInterval(this._pollInterval)
|
|
this._pollInterval = null
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 비스트리밍 텍스트 생성.
|
|
*/
|
|
async generate(prompt: string, options?: GenerateOptions): Promise<GenerateResult> {
|
|
if (!this._available) {
|
|
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama 서버에 연결할 수 없습니다')
|
|
}
|
|
|
|
const serverUrl = configGet('ollamaServerUrl')
|
|
const model = options?.model ?? configGet('llmModelId') ?? 'qwen3:4b'
|
|
|
|
this._state = LLMState.Generating
|
|
|
|
try {
|
|
const response = await fetch(`${serverUrl}/api/generate`, {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
model,
|
|
prompt,
|
|
system: options?.systemPrompt,
|
|
stream: false,
|
|
options: {
|
|
temperature: options?.temperature ?? 0.3,
|
|
num_predict: options?.maxTokens ?? 2048
|
|
}
|
|
}),
|
|
signal: AbortSignal.timeout(120000)
|
|
})
|
|
|
|
if (!response.ok) {
|
|
throw new D3ROError(
|
|
ErrorCode.LLMProcessingFailed,
|
|
`Ollama responded with ${response.status}: ${response.statusText}`
|
|
)
|
|
}
|
|
|
|
const data = (await response.json()) as OllamaGenerateResponse
|
|
|
|
const result: GenerateResult = {
|
|
text: data.response,
|
|
model: data.model,
|
|
promptTokens: data.prompt_eval_count ?? 0,
|
|
completionTokens: data.eval_count ?? 0,
|
|
totalDuration: data.total_duration ? data.total_duration / 1e6 : 0
|
|
}
|
|
|
|
this._state = LLMState.Available
|
|
this.emit('complete', { result })
|
|
return result
|
|
} catch (error) {
|
|
this._state = LLMState.Available
|
|
if (error instanceof D3ROError) throw error
|
|
throw new D3ROError(
|
|
ErrorCode.LLMProcessingFailed,
|
|
`LLM generation failed: ${error instanceof Error ? error.message : String(error)}`
|
|
)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 스트리밍 텍스트 생성. NDJSON 파싱.
|
|
* 반환된 AbortController로 취소 가능.
|
|
*/
|
|
async *streamGenerate(
|
|
prompt: string,
|
|
options?: Omit<GenerateOptions, 'stream'>
|
|
): AsyncGenerator<string, GenerateResult> {
|
|
if (!this._available) {
|
|
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama 서버에 연결할 수 없습니다')
|
|
}
|
|
|
|
const serverUrl = configGet('ollamaServerUrl')
|
|
const model = options?.model ?? configGet('llmModelId') ?? 'qwen3:4b'
|
|
|
|
this._state = LLMState.Generating
|
|
this._abortController = new AbortController()
|
|
|
|
try {
|
|
const response = await fetch(`${serverUrl}/api/generate`, {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
model,
|
|
prompt,
|
|
system: options?.systemPrompt,
|
|
stream: true,
|
|
options: {
|
|
temperature: options?.temperature ?? 0.3,
|
|
num_predict: options?.maxTokens ?? 2048
|
|
}
|
|
}),
|
|
signal: this._abortController.signal
|
|
})
|
|
|
|
if (!response.ok || !response.body) {
|
|
throw new D3ROError(
|
|
ErrorCode.LLMProcessingFailed,
|
|
`Ollama responded with ${response.status}`
|
|
)
|
|
}
|
|
|
|
const reader = response.body.getReader()
|
|
const decoder = new TextDecoder()
|
|
let buffer = ''
|
|
let fullText = ''
|
|
let lastChunk: OllamaGenerateResponse | null = null
|
|
|
|
while (true) {
|
|
const { done, value } = await reader.read()
|
|
if (done) break
|
|
|
|
buffer += decoder.decode(value, { stream: true })
|
|
const lines = buffer.split('\n')
|
|
buffer = lines.pop() ?? ''
|
|
|
|
for (const line of lines) {
|
|
if (!line.trim()) continue
|
|
try {
|
|
const chunk = JSON.parse(line) as OllamaGenerateResponse
|
|
fullText += chunk.response
|
|
this.emit('token', { token: chunk.response, done: chunk.done })
|
|
yield chunk.response
|
|
|
|
if (chunk.done) {
|
|
lastChunk = chunk
|
|
}
|
|
} catch {
|
|
logger.warn(`Failed to parse NDJSON line: ${line.substring(0, 100)}`)
|
|
}
|
|
}
|
|
}
|
|
|
|
this._state = LLMState.Available
|
|
this._abortController = null
|
|
|
|
const result: GenerateResult = {
|
|
text: fullText,
|
|
model: lastChunk?.model ?? model,
|
|
promptTokens: lastChunk?.prompt_eval_count ?? 0,
|
|
completionTokens: lastChunk?.eval_count ?? 0,
|
|
totalDuration: lastChunk?.total_duration ? lastChunk.total_duration / 1e6 : 0
|
|
}
|
|
|
|
this.emit('complete', { result })
|
|
return result
|
|
} catch (error) {
|
|
this._state = LLMState.Available
|
|
this._abortController = null
|
|
if (error instanceof D3ROError) throw error
|
|
if (error instanceof DOMException && error.name === 'AbortError') {
|
|
throw new D3ROError(ErrorCode.LLMProcessingCancelled, 'LLM generation cancelled')
|
|
}
|
|
throw new D3ROError(
|
|
ErrorCode.LLMProcessingFailed,
|
|
`LLM streaming failed: ${error instanceof Error ? error.message : String(error)}`
|
|
)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 텍스트를 LLM 액션에 따라 처리한다.
|
|
*/
|
|
async processText(
|
|
text: string,
|
|
action: LLMAction,
|
|
targetLanguage?: string,
|
|
customPrompt?: string
|
|
): Promise<string> {
|
|
// Phase 11: LLM 처리 쿼터 체크
|
|
try {
|
|
const { getLicenseService } = await import('./LicenseService')
|
|
const { Feature } = await import('@d3ro/core/types')
|
|
const license = getLicenseService()
|
|
const access = license.canUse(Feature.LLM_PROCESS)
|
|
if (!access.allowed) {
|
|
license.promptUpgrade(Feature.LLM_PROCESS, access.reason === 'quota_exceeded' ? 'quota_exceeded' : 'tier_required')
|
|
// LLM 처리 차단 시 원본 텍스트 반환 (폴백)
|
|
return text
|
|
}
|
|
license.consumeQuota(Feature.LLM_PROCESS)
|
|
} catch {
|
|
// LicenseService 미초기화 시 허용
|
|
}
|
|
|
|
let systemPrompt: string
|
|
|
|
if (action === 'custom' && customPrompt) {
|
|
systemPrompt = customPrompt
|
|
} else if (action === 'translate') {
|
|
systemPrompt = SYSTEM_PROMPTS.translate.replace(
|
|
'{{targetLanguage}}',
|
|
targetLanguage ?? 'English'
|
|
)
|
|
} else {
|
|
systemPrompt = SYSTEM_PROMPTS[action] ?? SYSTEM_PROMPTS.refine
|
|
}
|
|
|
|
const result = await this.generate(text, { systemPrompt })
|
|
const cleaned = stripReasoningBlocks(result.text)
|
|
// reasoning 블록 제거 후 빈 응답이면 원본 텍스트 폴백
|
|
// (모델이 thinking만 하고 출력은 안 한 경우 / 응답 파싱 실패 케이스)
|
|
if (cleaned.length === 0) {
|
|
logger.warn(
|
|
`LLM returned empty after reasoning strip — falling back to original transcript ` +
|
|
`(raw length=${result.text.length})`
|
|
)
|
|
return text
|
|
}
|
|
return cleaned
|
|
}
|
|
|
|
cancelGeneration(): void {
|
|
if (this._abortController) {
|
|
this._abortController.abort()
|
|
this._abortController = null
|
|
logger.info('LLM generation cancelled')
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Ollama에 설치된 모델 목록을 조회한다.
|
|
*/
|
|
async getModels(): Promise<LLMModel[]> {
|
|
const serverUrl = configGet('ollamaServerUrl')
|
|
|
|
try {
|
|
const response = await fetch(`${serverUrl}/api/tags`, {
|
|
signal: AbortSignal.timeout(5000)
|
|
})
|
|
|
|
if (!response.ok) return []
|
|
|
|
const data = (await response.json()) as OllamaTagsResponse
|
|
|
|
return data.models.map((m) => ({
|
|
id: m.name,
|
|
name: m.name,
|
|
sizeBytes: m.size,
|
|
parameterSize: m.parameter_size ?? '',
|
|
quantization: m.quantization_level ?? '',
|
|
modifiedAt: m.modified_at
|
|
}))
|
|
} catch {
|
|
return []
|
|
}
|
|
}
|
|
|
|
getStatus(): LLMStatus {
|
|
const connectionState: LLMConnectionState = this._available
|
|
? this._state === LLMState.Generating
|
|
? 'connecting'
|
|
: 'connected'
|
|
: 'disconnected'
|
|
|
|
return {
|
|
connectionState,
|
|
serverUrl: configGet('ollamaServerUrl'),
|
|
activeModel: configGet('llmModelId'),
|
|
serverVersion: null
|
|
}
|
|
}
|
|
|
|
isAvailable(): boolean {
|
|
return this._available
|
|
}
|
|
|
|
/**
|
|
* Ollama /api/chat 스트리밍 대화.
|
|
* messages 배열로 대화 히스토리를 전달한다.
|
|
* 각 토큰마다 yield, 완료 시 전체 응답 텍스트를 return.
|
|
*/
|
|
async *chatStream(
|
|
messages: Array<{ role: string; content: string }>,
|
|
options?: { model?: string; temperature?: number },
|
|
): AsyncGenerator<string, string> {
|
|
if (!this._available) {
|
|
throw new D3ROError(ErrorCode.LLMServerUnreachable, 'Ollama server not available')
|
|
}
|
|
|
|
const serverUrl = configGet('ollamaServerUrl')
|
|
const model = options?.model ?? configGet('llmModelId') ?? 'qwen3:4b'
|
|
|
|
this._abortController = new AbortController()
|
|
this._state = LLMState.Generating
|
|
|
|
try {
|
|
const response = await fetch(`${serverUrl}/api/chat`, {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
model,
|
|
messages,
|
|
stream: true,
|
|
options: {
|
|
temperature: options?.temperature ?? 0.7,
|
|
},
|
|
}),
|
|
signal: this._abortController.signal,
|
|
})
|
|
|
|
if (!response.ok || !response.body) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, `Chat API error: ${response.status}`)
|
|
}
|
|
|
|
const reader = response.body.getReader()
|
|
const decoder = new TextDecoder()
|
|
let buffer = ''
|
|
let accumulated = ''
|
|
|
|
while (true) {
|
|
const { done, value } = await reader.read()
|
|
if (done) break
|
|
|
|
buffer += decoder.decode(value, { stream: true })
|
|
const lines = buffer.split('\n')
|
|
buffer = lines.pop() ?? ''
|
|
|
|
for (const line of lines) {
|
|
if (!line.trim()) continue
|
|
try {
|
|
const chunk = JSON.parse(line) as { message?: { content: string }; done: boolean }
|
|
if (chunk.message?.content) {
|
|
accumulated += chunk.message.content
|
|
yield chunk.message.content
|
|
}
|
|
if (chunk.done) {
|
|
return accumulated
|
|
}
|
|
} catch {
|
|
// 불완전 JSON 무시
|
|
}
|
|
}
|
|
}
|
|
|
|
return accumulated
|
|
} finally {
|
|
this._state = this._available ? LLMState.Available : LLMState.Unavailable
|
|
this._abortController = null
|
|
}
|
|
}
|
|
|
|
dispose(): void {
|
|
this._disposed = true
|
|
this.stopPolling()
|
|
this.cancelGeneration()
|
|
this.removeAllListeners()
|
|
logger.info('LocalLLMService disposed')
|
|
}
|
|
|
|
// ── 가용성 체크 ────────────────────────────────────────
|
|
|
|
private async _checkAvailability(): Promise<void> {
|
|
if (this._disposed) return
|
|
const serverUrl = configGet('ollamaServerUrl')
|
|
|
|
try {
|
|
const response = await fetch(`${serverUrl}/api/tags`, {
|
|
signal: AbortSignal.timeout(3000)
|
|
})
|
|
|
|
const wasAvailable = this._available
|
|
this._available = response.ok
|
|
|
|
if (!wasAvailable && this._available) {
|
|
this._state = LLMState.Available
|
|
this.emit('availability-changed', { available: true })
|
|
logger.info('Ollama server connected')
|
|
} else if (wasAvailable && !this._available) {
|
|
this._state = LLMState.Unavailable
|
|
this.emit('availability-changed', { available: false })
|
|
logger.warn('Ollama server disconnected')
|
|
}
|
|
} catch {
|
|
if (this._available) {
|
|
this._available = false
|
|
this._state = LLMState.Unavailable
|
|
this.emit('availability-changed', { available: false })
|
|
logger.warn('Ollama server unreachable')
|
|
}
|
|
}
|
|
}
|
|
|
|
// ── EventEmitter 타입 오버라이드 ───────────────────────
|
|
|
|
override on<K extends keyof LocalLLMEvents>(event: K, listener: LocalLLMEvents[K]): this {
|
|
return super.on(event, listener)
|
|
}
|
|
|
|
override off<K extends keyof LocalLLMEvents>(event: K, listener: LocalLLMEvents[K]): this {
|
|
return super.off(event, listener)
|
|
}
|
|
|
|
override emit<K extends keyof LocalLLMEvents>(
|
|
event: K,
|
|
...args: Parameters<LocalLLMEvents[K]>
|
|
): boolean {
|
|
return super.emit(event, ...args)
|
|
}
|
|
}
|
|
|
|
// ── 싱글톤 ─────────────────────────────────────────────
|
|
|
|
let instance: LocalLLMService | null = null
|
|
|
|
export function getLocalLLMService(): LocalLLMService {
|
|
if (!instance) {
|
|
instance = new LocalLLMService()
|
|
}
|
|
return instance
|
|
}
|