336 lines
12 KiB
TypeScript
336 lines
12 KiB
TypeScript
// server/supabase/functions/llm-proxy/handler.ts
|
|
// Use case: proxy one Anthropic Messages API call for an authenticated user.
|
|
//
|
|
// Order of work:
|
|
// 1. authenticate, validate the request, resolve the tier and model
|
|
// 2. reserve one quota unit (atomic, counts in-flight requests)
|
|
// 3. call the provider only while holding the reservation
|
|
// 4. settle the reservation: completed when the user received the answer,
|
|
// released (refunded) when the provider failed, timed out, or the stream
|
|
// ended before message_stop
|
|
//
|
|
// IO goes through ports (auth, quota store, receipt issuer, provider fetch) so
|
|
// the quota ordering and the stream deadlines are testable without Supabase or
|
|
// Anthropic. index.ts wires the real adapters.
|
|
|
|
import { corsHeaders, handleCorsPreflightRequest } from '../_shared/cors.ts'
|
|
import { authErrorResponse, type AuthError } from '../_shared/auth.ts'
|
|
import {
|
|
getQuotaPolicy,
|
|
modelToQuotaKey,
|
|
type QuotaFeature,
|
|
type QuotaPeriod,
|
|
type QuotaReservation,
|
|
type QuotaReservationFinalStatus,
|
|
type Tier,
|
|
} from '../_shared/quota.ts'
|
|
import {
|
|
hasAssistantText,
|
|
LlmRequestError,
|
|
parseLlmRequest,
|
|
} from '../_shared/llm-contract.ts'
|
|
import {
|
|
GENERATION_ID_HEADER,
|
|
GENERATION_PURPOSE_HEADER,
|
|
GenerationReceiptError,
|
|
type GenerationPurpose,
|
|
parseGenerationPurpose,
|
|
} from '../_shared/generation-receipt.ts'
|
|
import { buildAnthropicSystemBlocks } from '../_shared/generative-ai-safety.ts'
|
|
import {
|
|
createProviderCallDeadline,
|
|
DEFAULT_LLM_PROVIDER_TIMEOUTS,
|
|
type LlmProviderTimeouts,
|
|
relayProviderStream,
|
|
type StreamRelayOutcome,
|
|
} from './provider-deadline.ts'
|
|
|
|
export const ANTHROPIC_MESSAGES_URL = 'https://api.anthropic.com/v1/messages'
|
|
|
|
/** 티어별 허용 모델 — free는 Haiku만, pro는 +Sonnet, pro_plus는 +Opus, team/enterprise는 전 모델 */
|
|
export const TIER_MODELS: Readonly<Record<Tier, readonly string[]>> = {
|
|
free: ['claude-haiku-4-5-20251001'],
|
|
pro: ['claude-haiku-4-5-20251001', 'claude-sonnet-4-6', 'claude-opus-4-6'],
|
|
pro_plus: ['claude-haiku-4-5-20251001', 'claude-sonnet-4-6', 'claude-opus-4-6'],
|
|
team: ['claude-haiku-4-5-20251001', 'claude-sonnet-4-6', 'claude-opus-4-6'],
|
|
enterprise: ['claude-haiku-4-5-20251001', 'claude-sonnet-4-6', 'claude-opus-4-6'],
|
|
}
|
|
|
|
export const DEFAULT_MODEL: Readonly<Record<Tier, string>> = {
|
|
free: 'claude-haiku-4-5-20251001',
|
|
pro: 'claude-sonnet-4-6',
|
|
pro_plus: 'claude-sonnet-4-6',
|
|
team: 'claude-sonnet-4-6',
|
|
enterprise: 'claude-sonnet-4-6',
|
|
}
|
|
|
|
/** Port over the LLM quota tables (subscriptions, daily_usage, llm_quota_reservations). */
|
|
export interface LlmQuotaStore {
|
|
readTier(userId: string): Promise<Tier>
|
|
reserve(
|
|
userId: string,
|
|
reservationId: string,
|
|
feature: QuotaFeature,
|
|
baseLimit: number,
|
|
period: QuotaPeriod,
|
|
): Promise<QuotaReservation>
|
|
finalize(reservationId: string, succeeded: boolean): Promise<QuotaReservationFinalStatus>
|
|
}
|
|
|
|
/** Request passed to the provider port. */
|
|
export interface ProviderRequest {
|
|
method: 'POST'
|
|
headers: Record<string, string>
|
|
body: string
|
|
signal: AbortSignal
|
|
}
|
|
|
|
/** Port over the provider HTTP call (global fetch in production). */
|
|
export type ProviderFetch = (url: string, init: ProviderRequest) => Promise<Response>
|
|
|
|
export interface LlmProxyDeps {
|
|
authenticate(req: Request): Promise<{ id: string }>
|
|
/** null when the provider key is not configured. */
|
|
providerKey(): string | null
|
|
quotaStore(): LlmQuotaStore
|
|
/** Issue a content-generation receipt; throws GenerationReceiptError when unavailable. */
|
|
issueGenerationReceipt(userId: string, purpose: GenerationPurpose, model: string): Promise<string>
|
|
fetchProvider?: ProviderFetch
|
|
newReservationId?(): string
|
|
timeouts?: Partial<LlmProviderTimeouts>
|
|
}
|
|
|
|
const NO_STORE_JSON = { ...corsHeaders, 'Content-Type': 'application/json', 'Cache-Control': 'no-store' }
|
|
|
|
function json(status: number, body: Record<string, unknown>, noStore = true): Response {
|
|
return new Response(JSON.stringify(body), {
|
|
status,
|
|
headers: noStore ? NO_STORE_JSON : { ...corsHeaders, 'Content-Type': 'application/json' },
|
|
})
|
|
}
|
|
|
|
/** 쓰지 않을 공급자 응답 본문을 닫아 연결과 생성을 정리한다. */
|
|
async function discardBody(resp: Response): Promise<void> {
|
|
if (!resp.body || resp.bodyUsed) return
|
|
try {
|
|
await resp.body.cancel()
|
|
} catch {
|
|
// 이미 닫힌 스트림 — 무시
|
|
}
|
|
}
|
|
|
|
function generationHeaders(generationId: string | null): Record<string, string> {
|
|
return generationId === null ? {} : { [GENERATION_ID_HEADER]: generationId }
|
|
}
|
|
|
|
function isAuthLikeError(err: unknown): err is AuthError {
|
|
return Boolean(err && typeof err === 'object' && 'status' in err && 'message' in err)
|
|
}
|
|
|
|
/**
|
|
* Settle a reservation exactly once. A failed finalize is logged; an unsettled
|
|
* reservation is released when its lease expires.
|
|
*/
|
|
function reservationSettler(store: LlmQuotaStore, reservationId: string): (succeeded: boolean) => Promise<void> {
|
|
let done = false
|
|
return async (succeeded: boolean) => {
|
|
if (done) return
|
|
done = true
|
|
try {
|
|
await store.finalize(reservationId, succeeded)
|
|
} catch {
|
|
console.error('LLM quota finalization failed', { succeeded })
|
|
}
|
|
}
|
|
}
|
|
|
|
export function createLlmProxyHandler(deps: LlmProxyDeps): (req: Request) => Promise<Response> {
|
|
const fetchProvider: ProviderFetch = deps.fetchProvider ?? ((url, init) => fetch(url, init))
|
|
const newReservationId = deps.newReservationId ?? (() => crypto.randomUUID())
|
|
const timeouts: LlmProviderTimeouts = { ...DEFAULT_LLM_PROVIDER_TIMEOUTS, ...deps.timeouts }
|
|
|
|
return async (req: Request): Promise<Response> => {
|
|
const preflight = handleCorsPreflightRequest(req)
|
|
if (preflight) return preflight
|
|
|
|
if (req.method !== 'POST') {
|
|
return json(405, { error: 'Method not allowed' }, false)
|
|
}
|
|
|
|
let settleReservation: ((succeeded: boolean) => Promise<void>) | null = null
|
|
try {
|
|
const user = await deps.authenticate(req)
|
|
|
|
let rawBody: unknown
|
|
try {
|
|
rawBody = await req.json()
|
|
} catch {
|
|
throw new LlmRequestError('Invalid JSON body')
|
|
}
|
|
const body = parseLlmRequest(rawBody)
|
|
const generationPurpose = parseGenerationPurpose(req.headers.get(GENERATION_PURPOSE_HEADER))
|
|
|
|
// A deployment without a provider must not consume quota or fabricate an answer.
|
|
const anthropicKey = deps.providerKey()
|
|
if (!anthropicKey) return json(503, { error: 'provider_unavailable' })
|
|
|
|
const quota = deps.quotaStore()
|
|
|
|
// 1단계: 티어 조회 + 모델 검증
|
|
const tier = await quota.readTier(user.id)
|
|
const requestedModel = body.model ?? DEFAULT_MODEL[tier]
|
|
if (!TIER_MODELS[tier].includes(requestedModel)) {
|
|
return json(403, {
|
|
error: 'model_not_allowed',
|
|
tier,
|
|
requested: requestedModel,
|
|
allowed: TIER_MODELS[tier],
|
|
}, false)
|
|
}
|
|
|
|
// 2단계: 공급자 호출 전에 쿼터 한 단위를 원자적으로 예약한다.
|
|
// 진행 중인 요청도 사용량에 포함되므로 병렬 요청이 한도를 넘어 공급자 비용을 쓰지 못한다.
|
|
const quotaKey = modelToQuotaKey(requestedModel)
|
|
const policy = getQuotaPolicy(tier, quotaKey)
|
|
const reservation = await quota.reserve(
|
|
user.id,
|
|
newReservationId(),
|
|
quotaKey,
|
|
policy.limit,
|
|
policy.period,
|
|
)
|
|
if (!reservation.allowed || reservation.reservationId === null) {
|
|
return json(429, {
|
|
error: 'quota_exceeded',
|
|
model: requestedModel,
|
|
current: reservation.current,
|
|
limit: reservation.limit,
|
|
period: reservation.period,
|
|
tier: reservation.tier,
|
|
overage_credits: reservation.overageCredits,
|
|
}, false)
|
|
}
|
|
const settle = reservationSettler(quota, reservation.reservationId)
|
|
settleReservation = settle
|
|
|
|
// 3단계: 공급자 호출 (Prompt Caching 2024-07-31 활성화).
|
|
// 공급자 5xx/과부하/타임아웃으로 답을 주지 못한 요청은 예약을 해제(환불)한다.
|
|
const deadline = createProviderCallDeadline(timeouts.ttfbMs, req.signal)
|
|
let anthropicResp: Response
|
|
try {
|
|
anthropicResp = await fetchProvider(ANTHROPIC_MESSAGES_URL, {
|
|
method: 'POST',
|
|
headers: {
|
|
'Content-Type': 'application/json',
|
|
'x-api-key': anthropicKey,
|
|
'anthropic-version': '2023-06-01',
|
|
'anthropic-beta': 'prompt-caching-2024-07-31',
|
|
},
|
|
body: JSON.stringify({
|
|
model: requestedModel,
|
|
max_tokens: body.max_tokens,
|
|
system: buildAnthropicSystemBlocks(body.system),
|
|
messages: body.messages,
|
|
stream: body.stream,
|
|
}),
|
|
signal: deadline.signal,
|
|
})
|
|
} catch (err) {
|
|
deadline.dispose()
|
|
throw err
|
|
}
|
|
|
|
if (!anthropicResp.ok) {
|
|
deadline.dispose()
|
|
console.error('Anthropic request failed', { status: anthropicResp.status })
|
|
await discardBody(anthropicResp)
|
|
await settle(false)
|
|
return json(502, { error: 'provider_request_failed' })
|
|
}
|
|
|
|
if (body.stream && anthropicResp.body) {
|
|
// The first-byte deadline covers the provider call only; the streamed
|
|
// body gets its own idle/total deadlines.
|
|
deadline.headersReceived()
|
|
let generationId: string | null
|
|
try {
|
|
generationId = generationPurpose === null
|
|
? null
|
|
: await deps.issueGenerationReceipt(user.id, generationPurpose, requestedModel)
|
|
} catch (err) {
|
|
deadline.abort(err)
|
|
deadline.dispose()
|
|
await discardBody(anthropicResp)
|
|
throw err
|
|
}
|
|
const relayed = relayProviderStream(anthropicResp.body, {
|
|
idleMs: timeouts.streamIdleMs,
|
|
totalMs: timeouts.streamTotalMs,
|
|
abortUpstream: (reason) => deadline.abort(reason),
|
|
onSettled: async (outcome: StreamRelayOutcome) => {
|
|
deadline.dispose()
|
|
if (outcome === 'failed') console.error('Anthropic stream did not complete')
|
|
// A client that stops reading keeps the charge: the provider already
|
|
// generated what was sent. Only provider-side failures are refunded.
|
|
await settle(outcome !== 'failed')
|
|
},
|
|
})
|
|
return new Response(relayed, {
|
|
status: 200,
|
|
headers: {
|
|
...corsHeaders,
|
|
...generationHeaders(generationId),
|
|
'Content-Type': 'text/event-stream',
|
|
'Cache-Control': 'no-store',
|
|
Connection: 'keep-alive',
|
|
},
|
|
})
|
|
}
|
|
|
|
// 비스트리밍: 본문 전체가 답이므로 첫 바이트 기한을 본문 읽기까지 유지한다.
|
|
let nonStreamData: unknown
|
|
try {
|
|
nonStreamData = await anthropicResp.json()
|
|
} finally {
|
|
deadline.dispose()
|
|
}
|
|
if (!hasAssistantText(nonStreamData)) {
|
|
console.error('Anthropic returned an invalid response shape')
|
|
await settle(false)
|
|
return json(502, { error: 'provider_invalid_response' })
|
|
}
|
|
|
|
const generationId = generationPurpose === null
|
|
? null
|
|
: await deps.issueGenerationReceipt(user.id, generationPurpose, requestedModel)
|
|
await settle(true)
|
|
return new Response(JSON.stringify(nonStreamData), {
|
|
status: 200,
|
|
headers: {
|
|
...corsHeaders,
|
|
...generationHeaders(generationId),
|
|
'Content-Type': 'application/json',
|
|
'Cache-Control': 'no-store',
|
|
},
|
|
})
|
|
} catch (err) {
|
|
// Anything that failed while a reservation was held gives the unit back.
|
|
await settleReservation?.(false)
|
|
if (err instanceof GenerationReceiptError) {
|
|
const invalidPurpose = err.code === 'invalid_generation_purpose'
|
|
return json(invalidPurpose ? 400 : 503, {
|
|
error: invalidPurpose ? 'invalid_request' : 'generation_receipt_unavailable',
|
|
})
|
|
}
|
|
if (err instanceof LlmRequestError) {
|
|
return json(err.status, { error: 'invalid_request', message: err.message })
|
|
}
|
|
if (isAuthLikeError(err)) {
|
|
return authErrorResponse(err, corsHeaders)
|
|
}
|
|
const timedOut = err instanceof DOMException && err.name === 'TimeoutError'
|
|
console.error('llm-proxy failed', { kind: timedOut ? 'provider_timeout' : 'internal_error' })
|
|
return json(timedOut ? 504 : 500, { error: timedOut ? 'provider_timeout' : 'internal_error' })
|
|
}
|
|
}
|
|
}
|