// src/main/services/PremiumLLMService.ts // Phase 3.2: Anthropic Claude 프리미엄 LLM 서비스. // 내부적으로는 온라인 API를 호출한다. // Supabase Edge Function(`llm-proxy`)을 경유해 Claude Messages API를 호출. // // 특징: // - 싱글톤 + EventEmitter (설계서 01 패턴) // - processText(), chatStream() — 시그니처 동일 // - 네트워크 실패 / 401 / 429 / 5xx 감지 시 에러 throw → LLMRouterService가 local로 fallback // - quota-warning / upgrade-required / fallback-triggered 이벤트 emit import { EventEmitter } from 'events' import { getLogger } from './LoggerService' import { getCloudSyncService } from './CloudSyncService' import { D3ROError, ErrorCode } from '@d3ro/core/errors' import type { LLMAction } from '@d3ro/core/types' import { resolveSystemPrompt } from './llm-prompts' import { parseAnthropicSSE, AnthropicStreamError } from '../utils/sse-parser' import { fitChatRequest, LLM_PROXY_CHAT_LIMITS, toChatRequest, type ChatRequest, type ChatStreamOptions, type ChatTurn, type RoleMessage, } from '@d3ro/core/llm-chat' const logger = getLogger('PremiumLLMService') // ============================================================ // 내부 타입 // ============================================================ /** llm-proxy Edge Function 요청 body */ interface LlmProxyRequest { messages: ChatTurn[] system?: string max_tokens?: number model?: string stream?: boolean } /** Claude Messages API 비스트리밍 응답 */ interface ClaudeMessageResponse { id: string model: string role: 'assistant' content: Array<{ type: 'text'; text: string }> stop_reason: string usage: { input_tokens: number; output_tokens: number } } export interface QuotaUsageSnapshot { tier: 'free' | 'pro' | 'pro_plus' current: number limit: number overageCredits: number } type UpgradeReason = 'quota_exceeded' | 'model_not_allowed' | 'auth_required' /** 호출 단위 취소·기한 상태. cancelGeneration() 은 활성 호출 전부를 취소한다. */ type AbortCause = 'timeout' | 'cancelled' interface PremiumCall { controller: AbortController abortCause: AbortCause | null abort: (cause: AbortCause) => void close: () => void } /** llm-proxy 출력 토큰 상한 (llm-contract MAX_OUTPUT_TOKENS) */ const PROXY_MAX_OUTPUT_TOKENS = 4096 const DEFAULT_CHAT_MAX_TOKENS = 2048 /** 클라이언트 측 채팅 기한. 프록시가 공급자 호출에 45초 기한을 두므로 여유 있게 잡는다. */ const DEFAULT_CHAT_TIMEOUT_MS = 120_000 /** * llm-proxy 오류 메시지를 D3ROError 로 분류한다 (순수 함수). * upgradeReason 이 있으면 호출자가 'upgrade-required' 를 emit 한다. */ export function classifyProxyError(message: string): { error: D3ROError; upgradeReason: UpgradeReason | null } { if (message.includes('401') || message.includes('Unauthorized') || message.includes('auth')) { return { error: new D3ROError(ErrorCode.LLMServerUnreachable, `인증 실패: ${message}`), upgradeReason: 'auth_required', } } if (message.includes('quota_exceeded') || message.includes('429')) { return { error: new D3ROError(ErrorCode.LLMProcessingFailed, `쿼터 초과: ${message}`), upgradeReason: 'quota_exceeded', } } if (message.includes('model_not_allowed') || message.includes('403')) { return { error: new D3ROError(ErrorCode.LLMInvalidAction, `모델 권한 없음: ${message}`), upgradeReason: 'model_not_allowed', } } return { error: new D3ROError(ErrorCode.LLMProcessingFailed, `llm-proxy: ${message}`), upgradeReason: null, } } /** * invokeFunctionStream 오류가 프록시의 HTTP 응답(`: `)인지 판별한다. * 프록시가 응답했다면 같은 요청을 비스트리밍으로 다시 보내도 같은 실패(와 쿼터 소비)만 * 되풀이되므로, 비스트리밍 폴백은 HTTP 상태가 없는 전송 계층 실패에서만 쓴다. */ export function proxyHttpStatus(message: string): number | null { const match = /^(\d{3}):/.exec(message) return match ? Number(match[1]) : null } function resolveMaxTokens(maxTokens: number | undefined): number { if (maxTokens === undefined) return DEFAULT_CHAT_MAX_TOKENS if (!Number.isSafeInteger(maxTokens) || maxTokens <= 0) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'maxTokens must be a positive safe integer') } return Math.min(maxTokens, PROXY_MAX_OUTPUT_TOKENS) } function resolveTimeoutMs(timeoutMs: number | undefined): number { if (timeoutMs === undefined) return DEFAULT_CHAT_TIMEOUT_MS if (!Number.isSafeInteger(timeoutMs) || timeoutMs <= 0) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'timeoutMs must be a positive safe integer') } return timeoutMs } /** ChatRequest → llm-proxy body. system 은 최상위 system 필드로 보낸다. */ function toProxyBody( request: ChatRequest, maxTokens: number, model: string | undefined, stream: boolean, ): LlmProxyRequest { const body: LlmProxyRequest = { messages: request.turns, max_tokens: maxTokens, stream, } if (request.system !== undefined) body.system = request.system if (model !== undefined) body.model = model return body } function firstText(response: ClaudeMessageResponse): string { const firstBlock = response.content?.[0] return firstBlock?.type === 'text' ? firstBlock.text : '' } interface PremiumLLMEvents { 'quota-warning': (payload: { current: number; limit: number; overageCredits: number }) => void 'upgrade-required': (payload: { reason: UpgradeReason }) => void 'fallback-triggered': (payload: { reason: string }) => void } // ============================================================ // PremiumLLMService // ============================================================ class PremiumLLMService extends EventEmitter { private _activeCalls = new Set() private _disposed = false private _lastQuota: QuotaUsageSnapshot | null = null /** * 사용 가능 여부 — Supabase URL + access token 모두 있어야 true. * (LLMRouter가 local/premium 분기 시 호출) */ isAvailable(): boolean { const cloud = getCloudSyncService() return cloud.isEnabled() && cloud.isAuthenticated() } /** * 마지막 응답에서 서버가 알려준 쿼터 상태. * Settings UI에서 "오늘 X/250" 표시에 사용. */ getLastQuota(): QuotaUsageSnapshot | null { return this._lastQuota } /** * disposed / 미인증 상태 공통 가드. * 문제 시 D3ROError throw, upgrade-required emit. */ private _ensureAuth(): void { if (this._disposed) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'PremiumLLMService disposed') } const cloud = getCloudSyncService() if (!cloud.isEnabled() || !cloud.isAuthenticated()) { this.emit('upgrade-required', { reason: 'auth_required' }) throw new D3ROError( ErrorCode.LLMServerUnreachable, 'Premium LLM 사용 전 로그인 필요', ) } } /** * 텍스트 액션 처리. * Phase 3.2 MVP는 비스트리밍 (stream=false). */ async processText( text: string, action: LLMAction, targetLanguage?: string, customPrompt?: string, ): Promise { this._ensureAuth() const systemPrompt = resolveSystemPrompt(action, targetLanguage, customPrompt) const body: LlmProxyRequest = { messages: [{ role: 'user', content: text }], system: systemPrompt, max_tokens: 2048, stream: false, } try { const response = await this._invokeProxy(body) // Claude 응답 → text 추출 const firstBlock = response.content?.[0] if (!firstBlock || firstBlock.type !== 'text' || !firstBlock.text.trim()) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM returned empty content') } return firstBlock.text.trim() } catch (err) { // LLMRouter가 local fallback 처리. 여기서는 에러 전파. if (err instanceof D3ROError) throw err throw new D3ROError( ErrorCode.LLMProcessingFailed, `Premium LLM failed: ${err instanceof Error ? err.message : String(err)}`, ) } } /** * 스트리밍 대화 (Voice Conversation / 회의 채팅용). * SSE 스트리밍: llm-proxy에 stream=true로 요청, Anthropic SSE를 토큰 단위 yield. * * @d3ro/core/llm-chat 계약을 따른다: * - role:'system' 메시지는 버리지 않고 body.system 으로 보낸다 (한도 초과 시 앞부분 보존). * - message_stop 없이 끊긴 스트림, 공급자 오류 이벤트, 빈 응답은 D3ROError 로 throw 한다. * - signal / timeoutMs / maxTokens 는 호출 단위로 적용된다. temperature 는 프록시가 받지 않는다. * - 비스트리밍 폴백은 전송 계층 실패에서만 쓴다 (프록시 HTTP 오류는 그대로 전파). */ async *chatStream( messages: RoleMessage[], options?: ChatStreamOptions, ): AsyncGenerator { this._ensureAuth() const request = fitChatRequest(toChatRequest(messages), LLM_PROXY_CHAT_LIMITS) if (request.turns.length === 0) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium chat requires a user message') } const body = toProxyBody(request, resolveMaxTokens(options?.maxTokens), options?.model, true) const call = this._beginCall(options?.signal, resolveTimeoutMs(options?.timeoutMs)) let completed = false try { const cloud = getCloudSyncService() const { stream, error } = await cloud.invokeFunctionStream( 'llm-proxy', body as unknown as Record, call.controller.signal, ) if (error || !stream) { const msg = error?.message ?? 'Stream unavailable' if (call.controller.signal.aborted) { throw new D3ROError(ErrorCode.LLMProcessingCancelled, msg) } logger.error(`SSE stream failed: ${msg}`) if (proxyHttpStatus(msg) !== null) { // 프록시가 응답한 오류 — 재요청은 같은 실패와 쿼터 소비만 되풀이한다. throw this._rejectProxyError(msg) } // 전송 계층 실패 시 비스트리밍 fallback (같은 system/maxTokens/signal 적용) logger.info('Falling back to non-streaming Premium LLM') const response = await this._invokeProxy({ ...body, stream: false }, call.controller.signal) const text = firstText(response) if (!text.trim()) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM returned empty content') } completed = true yield text return text } let accumulated = '' for await (const token of parseAnthropicSSE(stream)) { accumulated += token yield token } completed = true if (!accumulated.trim()) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM returned empty content') } return accumulated } catch (err) { throw this._toChatError(err, call) } finally { // 완료 전 종료(소비자 break, 오류) 시 연결을 끊어 프록시 스트림을 정리한다. if (!completed) call.abort('cancelled') call.close() } } /** * 제목/요약/액션 플랜 등 자유 생성. processText 와 같은 프록시 경로를 탄다. */ async generate( text: string, options?: { systemPrompt?: string; temperature?: number; maxTokens?: number }, ): Promise<{ text: string }> { this._ensureAuth() const response = await this._invokeProxy({ messages: [{ role: 'user', content: text }], system: options?.systemPrompt, max_tokens: options?.maxTokens ?? 2048, stream: false, }) const firstBlock = response.content?.[0] if (!firstBlock || firstBlock.type !== 'text' || !firstBlock.text.trim()) { throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM generate returned empty content') } return { text: firstBlock.text.trim() } } cancelGeneration(): void { if (this._activeCalls.size === 0) return for (const call of [...this._activeCalls]) { call.abort('cancelled') } logger.info('Premium LLM generation cancelled') } dispose(): void { this._disposed = true this.cancelGeneration() this.removeAllListeners() logger.info('PremiumLLMService disposed') } // ── 내부: Edge Function 호출 ───────────────────────────── private async _invokeProxy(body: LlmProxyRequest, signal?: AbortSignal): Promise { const cloud = getCloudSyncService() // Supabase JS 클라이언트의 functions.invoke() 사용 — auth 헤더를 올바르게 처리. // raw fetch + Authorization: Bearer 방식은 Supabase gateway가 401로 거부. const { data, error } = signal ? await cloud.invokeFunction('llm-proxy', body as unknown as Record, { signal }) : await cloud.invokeFunction('llm-proxy', body as unknown as Record) if (error) { const msg = error.message ?? 'Edge Function error' logger.error(`llm-proxy error: ${msg}`) throw this._rejectProxyError(msg) } // functions.invoke는 response body를 자동 파싱해서 data에 넣음 const result = data as ClaudeMessageResponse if (!result?.content) { logger.warn('llm-proxy returned unexpected shape — falling back') throw new D3ROError(ErrorCode.LLMProcessingFailed, 'llm-proxy returned invalid response') } return result } /** 프록시 오류를 분류하고 필요 시 upgrade-required 를 알린다. */ private _rejectProxyError(message: string): D3ROError { const { error, upgradeReason } = classifyProxyError(message) if (upgradeReason) this.emit('upgrade-required', { reason: upgradeReason }) return error } /** 호출 단위 AbortController 를 만들고 외부 signal·기한·cancelGeneration 에 연결한다. */ private _beginCall(externalSignal: AbortSignal | undefined, timeoutMs: number): PremiumCall { const controller = new AbortController() const onExternalAbort = (): void => call.abort('cancelled') const timer = setTimeout(() => call.abort('timeout'), timeoutMs) const call: PremiumCall = { controller, abortCause: null, abort: (cause) => { if (call.abortCause !== null) return call.abortCause = cause controller.abort() }, close: () => { clearTimeout(timer) externalSignal?.removeEventListener('abort', onExternalAbort) this._activeCalls.delete(call) }, } this._activeCalls.add(call) if (externalSignal?.aborted) { onExternalAbort() } else { externalSignal?.addEventListener('abort', onExternalAbort, { once: true }) } return call } /** 채팅 스트림 실패를 호출 단위 원인에 맞는 D3ROError 로 정규화한다. */ private _toChatError(error: unknown, call: PremiumCall): D3ROError { if (call.abortCause === 'timeout') { return new D3ROError(ErrorCode.LLMProcessingTimeout, 'Premium LLM generation timed out') } if (call.abortCause === 'cancelled' || call.controller.signal.aborted) { return new D3ROError(ErrorCode.LLMProcessingCancelled, 'Premium LLM generation cancelled') } if (error instanceof D3ROError) return error if (error instanceof AnthropicStreamError) { logger.warn(`Premium SSE stream failed (${error.kind}): ${error.message}`) return new D3ROError(ErrorCode.LLMProcessingFailed, `Premium LLM stream failed: ${error.message}`) } return new D3ROError( ErrorCode.LLMProcessingFailed, `Premium LLM failed: ${error instanceof Error ? error.message : String(error)}`, ) } // ── EventEmitter 타입 오버라이드 ─────────────────────── on(event: K, listener: PremiumLLMEvents[K]): this { return super.on(event, listener) } emit( event: K, ...args: Parameters ): boolean { return super.emit(event, ...args) } } // ── 싱글톤 ───────────────────────────────────────────────── let _instance: PremiumLLMService | null = null export function resetPremiumLLMServiceForTests(): void { if (_instance) _instance.removeAllListeners() _instance = null } export function getPremiumLLMService(): PremiumLLMService { if (!_instance) { _instance = new PremiumLLMService() } return _instance } export type { PremiumLLMService }