fix(edge): meter and size-cap knowledge embeddings in embed-chunks and search-knowledge

This commit is contained in:
Yun Chan 2026-09-28 00:53:48 +09:00
parent b35676c75c
commit 1afaea7214
12 changed files with 1111 additions and 270 deletions

View file

@ -0,0 +1,91 @@
// server/supabase/functions/search-knowledge/handler.ts
// Use case: embed a search query and return the caller's matching chunks.
// IO is injected through ports so the quota guard is testable offline.
import { handleCorsPreflightRequest } from '../_shared/cors.ts'
import { errorResponse, jsonResponse as json } from '../_shared/json-response.ts'
import { type EmbeddingProvider, isEmbedding } from '../_shared/openai-embeddings.ts'
import {
EMBEDDING_LIMITS,
type EmbeddingReservation,
type EmbeddingUsageStore,
embeddingUnitsFor,
quotaExceededBody,
refundEmbeddingUnits,
reserveEmbeddingUnits,
} from '../_shared/embedding-quota.ts'
export const DEFAULT_MATCH_COUNT = 5
export const MAX_MATCH_COUNT = 20
export const SIMILARITY_THRESHOLD = 0.5
export type ChunkSearchOutcome =
| { ok: true; results: unknown[] }
| { ok: false; reason: 'unavailable' | 'failed' }
/** Port: similarity search run with the caller's own JWT (RLS scopes the rows). */
export type ChunkSearcher = (req: Request, embedding: number[], count: number) => Promise<ChunkSearchOutcome>
export interface SearchKnowledgeDeps {
authenticate(req: Request): Promise<{ id: string }>
/** null when the provider key is not configured. */
embeddingProvider(): EmbeddingProvider | null
usageStore(): EmbeddingUsageStore
searchChunks: ChunkSearcher
now?(): Date
}
export function createSearchKnowledgeHandler(deps: SearchKnowledgeDeps): (req: Request) => Promise<Response> {
const now = () => deps.now?.() ?? new Date()
return async (req: Request): Promise<Response> => {
const preflight = handleCorsPreflightRequest(req)
if (preflight) return preflight
if (req.method !== 'POST') return json(405, { error: 'method_not_allowed' })
try {
const user = await deps.authenticate(req)
const body = await req.json().catch(() => null) as {
query?: unknown
count?: unknown
} | null
const query = typeof body?.query === 'string' ? body.query.trim() : ''
const count = body?.count === undefined ? DEFAULT_MATCH_COUNT : body.count
if (!query || query.length > EMBEDDING_LIMITS.maxQueryChars) return json(400, { error: 'invalid_query' })
if (typeof count !== 'number' || !Number.isInteger(count) || count < 1 || count > MAX_MATCH_COUNT) {
return json(400, { error: 'invalid_count' })
}
const provider = deps.embeddingProvider()
if (!provider) return json(503, { error: 'embedding_provider_unavailable' })
const usage = deps.usageStore()
const units = embeddingUnitsFor(query)
let reservation: EmbeddingReservation
try {
reservation = await reserveEmbeddingUnits(usage, user.id, units, now())
} catch {
return json(503, { error: 'quota_unavailable' })
}
if (!reservation.allowed) return json(429, quotaExceededBody(reservation, units))
const outcome = await provider.embed(query)
if (!outcome.ok && outcome.reason === 'upstream') {
await refundEmbeddingUnits(usage, user.id, reservation, units, now())
return json(502, { error: 'embedding_upstream_failed' })
}
const queryEmbedding = outcome.ok ? outcome.data[0]?.embedding : undefined
if (!isEmbedding(queryEmbedding)) return json(502, { error: 'embedding_response_invalid' })
const result = await deps.searchChunks(req, queryEmbedding, count)
if (!result.ok) {
return result.reason === 'unavailable'
? json(503, { error: 'knowledge_storage_unavailable' })
: json(500, { error: 'knowledge_search_failed' })
}
return json(200, { results: result.results })
} catch (error) {
return errorResponse(error)
}
}
}