468 lines
16 KiB
TypeScript
468 lines
16 KiB
TypeScript
// src/main/services/PremiumLLMService.ts
|
|
// Phase 3.2: Anthropic Claude 프리미엄 LLM 서비스.
|
|
// 내부적으로는 온라인 API를 호출한다.
|
|
// Supabase Edge Function(`llm-proxy`)을 경유해 Claude Messages API를 호출.
|
|
//
|
|
// 특징:
|
|
// - 싱글톤 + EventEmitter (설계서 01 패턴)
|
|
// - processText(), chatStream() — 시그니처 동일
|
|
// - 네트워크 실패 / 401 / 429 / 5xx 감지 시 에러 throw → LLMRouterService가 local로 fallback
|
|
// - quota-warning / upgrade-required / fallback-triggered 이벤트 emit
|
|
|
|
import { EventEmitter } from 'events'
|
|
import { getLogger } from './LoggerService'
|
|
import { getCloudSyncService } from './CloudSyncService'
|
|
import { D3ROError, ErrorCode } from '@d3ro/core/errors'
|
|
import type { LLMAction } from '@d3ro/core/types'
|
|
import { resolveSystemPrompt } from './llm-prompts'
|
|
import { parseAnthropicSSE, AnthropicStreamError } from '../utils/sse-parser'
|
|
import {
|
|
fitChatRequest,
|
|
LLM_PROXY_CHAT_LIMITS,
|
|
toChatRequest,
|
|
type ChatRequest,
|
|
type ChatStreamOptions,
|
|
type ChatTurn,
|
|
type RoleMessage,
|
|
} from '@d3ro/core/llm-chat'
|
|
|
|
const logger = getLogger('PremiumLLMService')
|
|
|
|
// ============================================================
|
|
// 내부 타입
|
|
// ============================================================
|
|
|
|
/** llm-proxy Edge Function 요청 body */
|
|
interface LlmProxyRequest {
|
|
messages: ChatTurn[]
|
|
system?: string
|
|
max_tokens?: number
|
|
model?: string
|
|
stream?: boolean
|
|
}
|
|
|
|
/** Claude Messages API 비스트리밍 응답 */
|
|
interface ClaudeMessageResponse {
|
|
id: string
|
|
model: string
|
|
role: 'assistant'
|
|
content: Array<{ type: 'text'; text: string }>
|
|
stop_reason: string
|
|
usage: { input_tokens: number; output_tokens: number }
|
|
}
|
|
|
|
export interface QuotaUsageSnapshot {
|
|
tier: 'free' | 'pro' | 'pro_plus'
|
|
current: number
|
|
limit: number
|
|
overageCredits: number
|
|
}
|
|
|
|
type UpgradeReason = 'quota_exceeded' | 'model_not_allowed' | 'auth_required'
|
|
|
|
/** 호출 단위 취소·기한 상태. cancelGeneration() 은 활성 호출 전부를 취소한다. */
|
|
type AbortCause = 'timeout' | 'cancelled'
|
|
|
|
interface PremiumCall {
|
|
controller: AbortController
|
|
abortCause: AbortCause | null
|
|
abort: (cause: AbortCause) => void
|
|
close: () => void
|
|
}
|
|
|
|
/** llm-proxy 출력 토큰 상한 (llm-contract MAX_OUTPUT_TOKENS) */
|
|
const PROXY_MAX_OUTPUT_TOKENS = 4096
|
|
const DEFAULT_CHAT_MAX_TOKENS = 2048
|
|
/** 클라이언트 측 채팅 기한. 프록시가 공급자 호출에 45초 기한을 두므로 여유 있게 잡는다. */
|
|
const DEFAULT_CHAT_TIMEOUT_MS = 120_000
|
|
|
|
/**
|
|
* llm-proxy 오류 메시지를 D3ROError 로 분류한다 (순수 함수).
|
|
* upgradeReason 이 있으면 호출자가 'upgrade-required' 를 emit 한다.
|
|
*/
|
|
export function classifyProxyError(message: string): { error: D3ROError; upgradeReason: UpgradeReason | null } {
|
|
if (message.includes('401') || message.includes('Unauthorized') || message.includes('auth')) {
|
|
return {
|
|
error: new D3ROError(ErrorCode.LLMServerUnreachable, `인증 실패: ${message}`),
|
|
upgradeReason: 'auth_required',
|
|
}
|
|
}
|
|
if (message.includes('quota_exceeded') || message.includes('429')) {
|
|
return {
|
|
error: new D3ROError(ErrorCode.LLMProcessingFailed, `쿼터 초과: ${message}`),
|
|
upgradeReason: 'quota_exceeded',
|
|
}
|
|
}
|
|
if (message.includes('model_not_allowed') || message.includes('403')) {
|
|
return {
|
|
error: new D3ROError(ErrorCode.LLMInvalidAction, `모델 권한 없음: ${message}`),
|
|
upgradeReason: 'model_not_allowed',
|
|
}
|
|
}
|
|
return {
|
|
error: new D3ROError(ErrorCode.LLMProcessingFailed, `llm-proxy: ${message}`),
|
|
upgradeReason: null,
|
|
}
|
|
}
|
|
|
|
/**
|
|
* invokeFunctionStream 오류가 프록시의 HTTP 응답(`<status>: <body>`)인지 판별한다.
|
|
* 프록시가 응답했다면 같은 요청을 비스트리밍으로 다시 보내도 같은 실패(와 쿼터 소비)만
|
|
* 되풀이되므로, 비스트리밍 폴백은 HTTP 상태가 없는 전송 계층 실패에서만 쓴다.
|
|
*/
|
|
export function proxyHttpStatus(message: string): number | null {
|
|
const match = /^(\d{3}):/.exec(message)
|
|
return match ? Number(match[1]) : null
|
|
}
|
|
|
|
function resolveMaxTokens(maxTokens: number | undefined): number {
|
|
if (maxTokens === undefined) return DEFAULT_CHAT_MAX_TOKENS
|
|
if (!Number.isSafeInteger(maxTokens) || maxTokens <= 0) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'maxTokens must be a positive safe integer')
|
|
}
|
|
return Math.min(maxTokens, PROXY_MAX_OUTPUT_TOKENS)
|
|
}
|
|
|
|
function resolveTimeoutMs(timeoutMs: number | undefined): number {
|
|
if (timeoutMs === undefined) return DEFAULT_CHAT_TIMEOUT_MS
|
|
if (!Number.isSafeInteger(timeoutMs) || timeoutMs <= 0) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'timeoutMs must be a positive safe integer')
|
|
}
|
|
return timeoutMs
|
|
}
|
|
|
|
/** ChatRequest → llm-proxy body. system 은 최상위 system 필드로 보낸다. */
|
|
function toProxyBody(
|
|
request: ChatRequest,
|
|
maxTokens: number,
|
|
model: string | undefined,
|
|
stream: boolean,
|
|
): LlmProxyRequest {
|
|
const body: LlmProxyRequest = {
|
|
messages: request.turns,
|
|
max_tokens: maxTokens,
|
|
stream,
|
|
}
|
|
if (request.system !== undefined) body.system = request.system
|
|
if (model !== undefined) body.model = model
|
|
return body
|
|
}
|
|
|
|
function firstText(response: ClaudeMessageResponse): string {
|
|
const firstBlock = response.content?.[0]
|
|
return firstBlock?.type === 'text' ? firstBlock.text : ''
|
|
}
|
|
|
|
|
|
interface PremiumLLMEvents {
|
|
'quota-warning': (payload: { current: number; limit: number; overageCredits: number }) => void
|
|
'upgrade-required': (payload: { reason: UpgradeReason }) => void
|
|
'fallback-triggered': (payload: { reason: string }) => void
|
|
}
|
|
|
|
// ============================================================
|
|
// PremiumLLMService
|
|
// ============================================================
|
|
|
|
class PremiumLLMService extends EventEmitter {
|
|
private _activeCalls = new Set<PremiumCall>()
|
|
private _disposed = false
|
|
private _lastQuota: QuotaUsageSnapshot | null = null
|
|
|
|
/**
|
|
* 사용 가능 여부 — Supabase URL + access token 모두 있어야 true.
|
|
* (LLMRouter가 local/premium 분기 시 호출)
|
|
*/
|
|
isAvailable(): boolean {
|
|
const cloud = getCloudSyncService()
|
|
return cloud.isEnabled() && cloud.isAuthenticated()
|
|
}
|
|
|
|
/**
|
|
* 마지막 응답에서 서버가 알려준 쿼터 상태.
|
|
* Settings UI에서 "오늘 X/250" 표시에 사용.
|
|
*/
|
|
getLastQuota(): QuotaUsageSnapshot | null {
|
|
return this._lastQuota
|
|
}
|
|
|
|
/**
|
|
* disposed / 미인증 상태 공통 가드.
|
|
* 문제 시 D3ROError throw, upgrade-required emit.
|
|
*/
|
|
private _ensureAuth(): void {
|
|
if (this._disposed) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'PremiumLLMService disposed')
|
|
}
|
|
const cloud = getCloudSyncService()
|
|
if (!cloud.isEnabled() || !cloud.isAuthenticated()) {
|
|
this.emit('upgrade-required', { reason: 'auth_required' })
|
|
throw new D3ROError(
|
|
ErrorCode.LLMServerUnreachable,
|
|
'Premium LLM 사용 전 로그인 필요',
|
|
)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 텍스트 액션 처리.
|
|
* Phase 3.2 MVP는 비스트리밍 (stream=false).
|
|
*/
|
|
async processText(
|
|
text: string,
|
|
action: LLMAction,
|
|
targetLanguage?: string,
|
|
customPrompt?: string,
|
|
): Promise<string> {
|
|
this._ensureAuth()
|
|
|
|
const systemPrompt = resolveSystemPrompt(action, targetLanguage, customPrompt)
|
|
|
|
const body: LlmProxyRequest = {
|
|
messages: [{ role: 'user', content: text }],
|
|
system: systemPrompt,
|
|
max_tokens: 2048,
|
|
stream: false,
|
|
}
|
|
|
|
try {
|
|
const response = await this._invokeProxy(body)
|
|
// Claude 응답 → text 추출
|
|
const firstBlock = response.content?.[0]
|
|
if (!firstBlock || firstBlock.type !== 'text' || !firstBlock.text.trim()) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM returned empty content')
|
|
}
|
|
return firstBlock.text.trim()
|
|
} catch (err) {
|
|
// LLMRouter가 local fallback 처리. 여기서는 에러 전파.
|
|
if (err instanceof D3ROError) throw err
|
|
throw new D3ROError(
|
|
ErrorCode.LLMProcessingFailed,
|
|
`Premium LLM failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 스트리밍 대화 (Voice Conversation / 회의 채팅용).
|
|
* SSE 스트리밍: llm-proxy에 stream=true로 요청, Anthropic SSE를 토큰 단위 yield.
|
|
*
|
|
* @d3ro/core/llm-chat 계약을 따른다:
|
|
* - role:'system' 메시지는 버리지 않고 body.system 으로 보낸다 (한도 초과 시 앞부분 보존).
|
|
* - message_stop 없이 끊긴 스트림, 공급자 오류 이벤트, 빈 응답은 D3ROError 로 throw 한다.
|
|
* - signal / timeoutMs / maxTokens 는 호출 단위로 적용된다. temperature 는 프록시가 받지 않는다.
|
|
* - 비스트리밍 폴백은 전송 계층 실패에서만 쓴다 (프록시 HTTP 오류는 그대로 전파).
|
|
*/
|
|
async *chatStream(
|
|
messages: RoleMessage[],
|
|
options?: ChatStreamOptions,
|
|
): AsyncGenerator<string, string> {
|
|
this._ensureAuth()
|
|
|
|
const request = fitChatRequest(toChatRequest(messages), LLM_PROXY_CHAT_LIMITS)
|
|
if (request.turns.length === 0) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium chat requires a user message')
|
|
}
|
|
const body = toProxyBody(request, resolveMaxTokens(options?.maxTokens), options?.model, true)
|
|
const call = this._beginCall(options?.signal, resolveTimeoutMs(options?.timeoutMs))
|
|
let completed = false
|
|
|
|
try {
|
|
const cloud = getCloudSyncService()
|
|
const { stream, error } = await cloud.invokeFunctionStream(
|
|
'llm-proxy',
|
|
body as unknown as Record<string, unknown>,
|
|
call.controller.signal,
|
|
)
|
|
|
|
if (error || !stream) {
|
|
const msg = error?.message ?? 'Stream unavailable'
|
|
if (call.controller.signal.aborted) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingCancelled, msg)
|
|
}
|
|
logger.error(`SSE stream failed: ${msg}`)
|
|
|
|
if (proxyHttpStatus(msg) !== null) {
|
|
// 프록시가 응답한 오류 — 재요청은 같은 실패와 쿼터 소비만 되풀이한다.
|
|
throw this._rejectProxyError(msg)
|
|
}
|
|
|
|
// 전송 계층 실패 시 비스트리밍 fallback (같은 system/maxTokens/signal 적용)
|
|
logger.info('Falling back to non-streaming Premium LLM')
|
|
const response = await this._invokeProxy({ ...body, stream: false }, call.controller.signal)
|
|
const text = firstText(response)
|
|
if (!text.trim()) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM returned empty content')
|
|
}
|
|
completed = true
|
|
yield text
|
|
return text
|
|
}
|
|
|
|
let accumulated = ''
|
|
for await (const token of parseAnthropicSSE(stream)) {
|
|
accumulated += token
|
|
yield token
|
|
}
|
|
completed = true
|
|
if (!accumulated.trim()) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM returned empty content')
|
|
}
|
|
return accumulated
|
|
} catch (err) {
|
|
throw this._toChatError(err, call)
|
|
} finally {
|
|
// 완료 전 종료(소비자 break, 오류) 시 연결을 끊어 프록시 스트림을 정리한다.
|
|
if (!completed) call.abort('cancelled')
|
|
call.close()
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 제목/요약/액션 플랜 등 자유 생성. processText 와 같은 프록시 경로를 탄다.
|
|
*/
|
|
async generate(
|
|
text: string,
|
|
options?: { systemPrompt?: string; temperature?: number; maxTokens?: number },
|
|
): Promise<{ text: string }> {
|
|
this._ensureAuth()
|
|
const response = await this._invokeProxy({
|
|
messages: [{ role: 'user', content: text }],
|
|
system: options?.systemPrompt,
|
|
max_tokens: options?.maxTokens ?? 2048,
|
|
stream: false,
|
|
})
|
|
const firstBlock = response.content?.[0]
|
|
if (!firstBlock || firstBlock.type !== 'text' || !firstBlock.text.trim()) {
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'Premium LLM generate returned empty content')
|
|
}
|
|
return { text: firstBlock.text.trim() }
|
|
}
|
|
|
|
cancelGeneration(): void {
|
|
if (this._activeCalls.size === 0) return
|
|
for (const call of [...this._activeCalls]) {
|
|
call.abort('cancelled')
|
|
}
|
|
logger.info('Premium LLM generation cancelled')
|
|
}
|
|
|
|
dispose(): void {
|
|
this._disposed = true
|
|
this.cancelGeneration()
|
|
this.removeAllListeners()
|
|
logger.info('PremiumLLMService disposed')
|
|
}
|
|
|
|
// ── 내부: Edge Function 호출 ─────────────────────────────
|
|
|
|
private async _invokeProxy(body: LlmProxyRequest, signal?: AbortSignal): Promise<ClaudeMessageResponse> {
|
|
const cloud = getCloudSyncService()
|
|
|
|
// Supabase JS 클라이언트의 functions.invoke() 사용 — auth 헤더를 올바르게 처리.
|
|
// raw fetch + Authorization: Bearer 방식은 Supabase gateway가 401로 거부.
|
|
const { data, error } = signal
|
|
? await cloud.invokeFunction('llm-proxy', body as unknown as Record<string, unknown>, { signal })
|
|
: await cloud.invokeFunction('llm-proxy', body as unknown as Record<string, unknown>)
|
|
|
|
if (error) {
|
|
const msg = error.message ?? 'Edge Function error'
|
|
logger.error(`llm-proxy error: ${msg}`)
|
|
throw this._rejectProxyError(msg)
|
|
}
|
|
|
|
// functions.invoke는 response body를 자동 파싱해서 data에 넣음
|
|
const result = data as ClaudeMessageResponse
|
|
if (!result?.content) {
|
|
logger.warn('llm-proxy returned unexpected shape — falling back')
|
|
throw new D3ROError(ErrorCode.LLMProcessingFailed, 'llm-proxy returned invalid response')
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
/** 프록시 오류를 분류하고 필요 시 upgrade-required 를 알린다. */
|
|
private _rejectProxyError(message: string): D3ROError {
|
|
const { error, upgradeReason } = classifyProxyError(message)
|
|
if (upgradeReason) this.emit('upgrade-required', { reason: upgradeReason })
|
|
return error
|
|
}
|
|
|
|
/** 호출 단위 AbortController 를 만들고 외부 signal·기한·cancelGeneration 에 연결한다. */
|
|
private _beginCall(externalSignal: AbortSignal | undefined, timeoutMs: number): PremiumCall {
|
|
const controller = new AbortController()
|
|
const onExternalAbort = (): void => call.abort('cancelled')
|
|
const timer = setTimeout(() => call.abort('timeout'), timeoutMs)
|
|
const call: PremiumCall = {
|
|
controller,
|
|
abortCause: null,
|
|
abort: (cause) => {
|
|
if (call.abortCause !== null) return
|
|
call.abortCause = cause
|
|
controller.abort()
|
|
},
|
|
close: () => {
|
|
clearTimeout(timer)
|
|
externalSignal?.removeEventListener('abort', onExternalAbort)
|
|
this._activeCalls.delete(call)
|
|
},
|
|
}
|
|
this._activeCalls.add(call)
|
|
if (externalSignal?.aborted) {
|
|
onExternalAbort()
|
|
} else {
|
|
externalSignal?.addEventListener('abort', onExternalAbort, { once: true })
|
|
}
|
|
return call
|
|
}
|
|
|
|
/** 채팅 스트림 실패를 호출 단위 원인에 맞는 D3ROError 로 정규화한다. */
|
|
private _toChatError(error: unknown, call: PremiumCall): D3ROError {
|
|
if (call.abortCause === 'timeout') {
|
|
return new D3ROError(ErrorCode.LLMProcessingTimeout, 'Premium LLM generation timed out')
|
|
}
|
|
if (call.abortCause === 'cancelled' || call.controller.signal.aborted) {
|
|
return new D3ROError(ErrorCode.LLMProcessingCancelled, 'Premium LLM generation cancelled')
|
|
}
|
|
if (error instanceof D3ROError) return error
|
|
if (error instanceof AnthropicStreamError) {
|
|
logger.warn(`Premium SSE stream failed (${error.kind}): ${error.message}`)
|
|
return new D3ROError(ErrorCode.LLMProcessingFailed, `Premium LLM stream failed: ${error.message}`)
|
|
}
|
|
return new D3ROError(
|
|
ErrorCode.LLMProcessingFailed,
|
|
`Premium LLM failed: ${error instanceof Error ? error.message : String(error)}`,
|
|
)
|
|
}
|
|
|
|
// ── EventEmitter 타입 오버라이드 ───────────────────────
|
|
|
|
on<K extends keyof PremiumLLMEvents>(event: K, listener: PremiumLLMEvents[K]): this {
|
|
return super.on(event, listener)
|
|
}
|
|
|
|
emit<K extends keyof PremiumLLMEvents>(
|
|
event: K,
|
|
...args: Parameters<PremiumLLMEvents[K]>
|
|
): boolean {
|
|
return super.emit(event, ...args)
|
|
}
|
|
}
|
|
|
|
// ── 싱글톤 ─────────────────────────────────────────────────
|
|
|
|
let _instance: PremiumLLMService | null = null
|
|
|
|
export function resetPremiumLLMServiceForTests(): void {
|
|
if (_instance) _instance.removeAllListeners()
|
|
_instance = null
|
|
}
|
|
|
|
export function getPremiumLLMService(): PremiumLLMService {
|
|
if (!_instance) {
|
|
_instance = new PremiumLLMService()
|
|
}
|
|
return _instance
|
|
}
|
|
|
|
export type { PremiumLLMService }
|