fix(llm): keep system prompts on premium chat, reject incomplete streams, stop double-charging quota

This commit is contained in:
Yun Chan 2026-09-28 00:53:44 +09:00
parent d96601a283
commit d311e8123f
10 changed files with 1225 additions and 237 deletions

View file

@ -0,0 +1,100 @@
// LocalLLMService.chatStream 이 @d3ro/core/llm-chat 계약을 따르는지 (red-team r1-6)
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { ErrorCode } from '@d3ro/core/errors'
import { initInMemoryConfig, resetInMemoryConfig } from '../../../src/main/services/ConfigService'
import {
getLocalLLMService,
resetLocalLLMServiceForTests,
} from '../../../src/main/services/LocalLLMService'
const encoder = new TextEncoder()
function ndjsonResponse(parts: string[]): Response {
return new Response(new ReadableStream<Uint8Array>({
start(controller) {
for (const part of parts) controller.enqueue(encoder.encode(part))
controller.close()
},
}), { status: 200 })
}
async function collect(gen: AsyncGenerator<string, string>): Promise<{ tokens: string[]; result: string }> {
const tokens: string[] = []
for (let next = await gen.next(); ; next = await gen.next()) {
if (next.done) return { tokens, result: next.value }
tokens.push(next.value)
}
}
describe('LocalLLMService chat contract', () => {
beforeEach(() => {
initInMemoryConfig()
resetLocalLLMServiceForTests()
;(getLocalLLMService() as unknown as { _available: boolean })._available = true
})
afterEach(() => {
resetLocalLLMServiceForTests()
resetInMemoryConfig()
vi.unstubAllGlobals()
})
it('sends the system prompt as a single leading system message', async () => {
let sent: Array<{ role: string; content: string }> = []
vi.stubGlobal('fetch', vi.fn(async (_input: RequestInfo | URL, init?: RequestInit) => {
sent = (JSON.parse(String(init?.body)) as { messages: typeof sent }).messages
return ndjsonResponse([
`${JSON.stringify({ message: { content: 'hi' }, done: false })}\n`,
JSON.stringify({ message: { content: '' }, done: true }),
])
}))
const { tokens, result } = await collect(getLocalLLMService().chatStream([
{ role: 'system', content: 'persona' },
{ role: 'user', content: 'hello' },
]))
expect(tokens).toEqual(['hi'])
expect(result).toBe('hi')
expect(sent).toEqual([
{ role: 'system', content: 'persona' },
{ role: 'user', content: 'hello' },
])
})
it('rejects a stream that ends without a done frame', async () => {
vi.stubGlobal('fetch', vi.fn(async () => ndjsonResponse([
`${JSON.stringify({ message: { content: 'partial' }, done: false })}\n`,
])))
await expect(collect(getLocalLLMService().chatStream([{ role: 'user', content: 'q' }])))
.rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed })
})
it('rejects malformed NDJSON frames, including non-object JSON', async () => {
vi.stubGlobal('fetch', vi.fn(async () => ndjsonResponse(['{broken\n'])))
await expect(collect(getLocalLLMService().chatStream([{ role: 'user', content: 'q' }])))
.rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed, message: 'Ollama returned malformed NDJSON' })
vi.stubGlobal('fetch', vi.fn(async () => ndjsonResponse(['null\n'])))
await expect(collect(getLocalLLMService().chatStream([{ role: 'user', content: 'q' }])))
.rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed, message: 'Ollama returned malformed NDJSON' })
})
it('streamGenerate still completes on a trailing done frame without newline', async () => {
vi.stubGlobal('fetch', vi.fn(async () => ndjsonResponse([
`${JSON.stringify({ model: 'm', response: 'a', done: false })}\n`,
JSON.stringify({ model: 'm', response: 'b', done: true, eval_count: 2 }),
])))
const gen = getLocalLLMService().streamGenerate('p')
const tokens: string[] = []
let next = await gen.next()
while (!next.done) {
tokens.push(next.value)
next = await gen.next()
}
expect(tokens).toEqual(['a', 'b'])
expect(next.value).toMatchObject({ text: 'ab', completionTokens: 2 })
})
})

View file

@ -0,0 +1,304 @@
// PremiumLLMService.chatStream 계약 회귀 테스트 (red-team r1-6)
// - system 메시지 보존 (회의 전사 / 음성 대화 페르소나)
// - 끊긴 스트림·오류 이벤트·빈 응답은 성공이 아니다
// - 프록시 HTTP 오류는 비스트리밍으로 재요청하지 않는다 (쿼터 이중 소비 방지)
// - 호출 단위 취소
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { ErrorCode } from '@d3ro/core/errors'
import { parseLlmRequest } from '../../../../../server/supabase/functions/_shared/llm-contract'
const cloud = vi.hoisted(() => ({
isEnabled: (): boolean => true,
isAuthenticated: (): boolean => true,
invokeFunctionStream: vi.fn(),
invokeFunction: vi.fn(),
}))
vi.mock('../../../src/main/services/CloudSyncService', () => ({
getCloudSyncService: () => cloud,
}))
import {
getPremiumLLMService,
resetPremiumLLMServiceForTests,
proxyHttpStatus,
} from '../../../src/main/services/PremiumLLMService'
const encoder = new TextEncoder()
interface SseFrame {
event: string
data: unknown
}
function sseText(frames: SseFrame[]): string {
return frames.map((f) => `event: ${f.event}\ndata: ${JSON.stringify(f.data)}\n\n`).join('')
}
function sseStream(frames: SseFrame[]): ReadableStream<Uint8Array> {
return new ReadableStream({
start(controller) {
controller.enqueue(encoder.encode(sseText(frames)))
controller.close()
},
})
}
const delta = (text: string): SseFrame => ({
event: 'content_block_delta',
data: { type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text } },
})
const messageStart: SseFrame = { event: 'message_start', data: { type: 'message_start', message: {} } }
const messageStop: SseFrame = { event: 'message_stop', data: { type: 'message_stop' } }
const overloaded: SseFrame = {
event: 'error',
data: { type: 'error', error: { type: 'overloaded_error', message: 'Overloaded' } },
}
/** 신호가 abort 되기 전까지 끝나지 않는 SSE 스트림 */
function hangingStream(signal: AbortSignal): ReadableStream<Uint8Array> {
return new ReadableStream({
start(controller) {
controller.enqueue(encoder.encode(sseText([messageStart, delta('partial ')])))
signal.addEventListener('abort', () => {
controller.error(new DOMException('Aborted', 'AbortError'))
}, { once: true })
},
})
}
async function collect(gen: AsyncGenerator<string, string>): Promise<{ tokens: string[]; result: string }> {
const tokens: string[] = []
for (let next = await gen.next(); ; next = await gen.next()) {
if (next.done) return { tokens, result: next.value }
tokens.push(next.value)
}
}
function lastStreamBody(): Record<string, unknown> {
const call = cloud.invokeFunctionStream.mock.calls.at(-1)
if (!call) throw new Error('invokeFunctionStream was not called')
return call[1] as Record<string, unknown>
}
describe('PremiumLLMService.chatStream contract', () => {
beforeEach(() => {
resetPremiumLLMServiceForTests()
cloud.invokeFunctionStream.mockReset()
cloud.invokeFunction.mockReset()
})
afterEach(() => {
resetPremiumLLMServiceForTests()
})
it('sends role:system content as body.system instead of dropping it', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: sseStream([messageStart, delta('short answer.'), messageStop]),
error: null,
})
const { tokens, result } = await collect(getPremiumLLMService().chatStream([
{ role: 'system', content: 'Keep answers brief (2-3 sentences).' },
{ role: 'user', content: 'hello' },
]))
expect(tokens).toEqual(['short answer.'])
expect(result).toBe('short answer.')
const body = lastStreamBody()
expect(body.system).toBe('Keep answers brief (2-3 sentences).')
expect(body.messages).toEqual([{ role: 'user', content: 'hello' }])
expect(body.stream).toBe(true)
})
it('keeps a meeting transcript system prompt within the llm-proxy contract', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: sseStream([delta('answer'), messageStop]),
error: null,
})
const transcript = '회의 발언 '.repeat(4_000)
const systemPrompt = `당신은 회의 내용을 분석하는 AI 어시스턴트입니다.\n\n## 회의 전사\n${transcript}`
const history = Array.from({ length: 60 }, (_, i) => ({
role: i % 2 === 0 ? 'user' : 'assistant',
content: `turn ${i}`,
}))
await collect(getPremiumLLMService().chatStream([
{ role: 'system', content: systemPrompt },
...history,
{ role: 'user', content: '결정 사항은?' },
]))
const body = lastStreamBody()
const validated = parseLlmRequest(body)
expect(validated.system?.startsWith('당신은 회의 내용을 분석하는 AI 어시스턴트입니다.')).toBe(true)
expect(validated.system).toContain('## 회의 전사')
expect(validated.messages.at(-1)).toEqual({ role: 'user', content: '결정 사항은?' })
})
it('removes an empty assistant turn so a previous failure cannot poison later turns', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: sseStream([delta('ok'), messageStop]),
error: null,
})
await collect(getPremiumLLMService().chatStream([
{ role: 'system', content: 'persona' },
{ role: 'user', content: 'first' },
{ role: 'assistant', content: '' },
{ role: 'user', content: 'second' },
]))
const body = lastStreamBody()
expect(() => parseLlmRequest(body)).not.toThrow()
expect(body.messages).toEqual([
{ role: 'user', content: 'first' },
{ role: 'user', content: 'second' },
])
})
it('rejects when the provider sends an error event mid-stream', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: sseStream([messageStart, delta('partial '), overloaded]),
error: null,
})
const tokens: string[] = []
await expect((async () => {
for await (const token of getPremiumLLMService().chatStream([{ role: 'user', content: 'q' }])) {
tokens.push(token)
}
})()).rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed })
expect(tokens).toEqual(['partial '])
})
it('rejects when the stream ends without message_stop', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: sseStream([messageStart, delta('cut off')]),
error: null,
})
await expect(collect(getPremiumLLMService().chatStream([{ role: 'user', content: 'q' }])))
.rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed })
})
it('rejects an empty completed reply instead of returning ""', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: sseStream([messageStart, messageStop]),
error: null,
})
await expect(collect(getPremiumLLMService().chatStream([{ role: 'user', content: 'q' }])))
.rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed })
})
it('does not resend a proxy 502 as a non-streaming request', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: null,
error: { message: '502: {"error":"provider_request_failed"}' },
})
await expect(collect(getPremiumLLMService().chatStream([{ role: 'user', content: 'q' }])))
.rejects.toMatchObject({ code: ErrorCode.LLMProcessingFailed })
expect(cloud.invokeFunction).not.toHaveBeenCalled()
})
it('reports quota_exceeded from the stream call without a second request', async () => {
cloud.invokeFunctionStream.mockResolvedValue({
stream: null,
error: { message: '429: {"error":"quota_exceeded"}' },
})
const upgrade = vi.fn()
const service = getPremiumLLMService()
service.on('upgrade-required', upgrade)
await expect(collect(service.chatStream([{ role: 'user', content: 'q' }]))).rejects.toBeTruthy()
expect(upgrade).toHaveBeenCalledWith({ reason: 'quota_exceeded' })
expect(cloud.invokeFunction).not.toHaveBeenCalled()
})
it('falls back to non-streaming only on transport failure, keeping system and max_tokens', async () => {
cloud.invokeFunctionStream.mockResolvedValue({ stream: null, error: { message: 'fetch failed' } })
cloud.invokeFunction.mockResolvedValue({
data: {
id: 'm', model: 'claude', role: 'assistant',
content: [{ type: 'text', text: 'fallback answer' }],
stop_reason: 'end_turn', usage: { input_tokens: 1, output_tokens: 1 },
},
error: null,
})
const { tokens, result } = await collect(getPremiumLLMService().chatStream(
[{ role: 'system', content: 'persona' }, { role: 'user', content: 'q' }],
{ maxTokens: 512 },
))
expect(tokens).toEqual(['fallback answer'])
expect(result).toBe('fallback answer')
const [name, body, options] = cloud.invokeFunction.mock.calls[0] as [string, Record<string, unknown>, { signal?: AbortSignal }]
expect(name).toBe('llm-proxy')
expect(body).toMatchObject({ system: 'persona', stream: false, max_tokens: 512 })
expect(options?.signal).toBeInstanceOf(AbortSignal)
})
it('cancelGeneration cancels every active call, not only the latest one', async () => {
cloud.invokeFunctionStream.mockImplementation(async (_n: string, _b: unknown, signal: AbortSignal) => ({
stream: hangingStream(signal),
error: null,
}))
const service = getPremiumLLMService()
const first = service.chatStream([{ role: 'user', content: 'a' }])
const second = service.chatStream([{ role: 'user', content: 'b' }])
expect((await first.next()).value).toBe('partial ')
expect((await second.next()).value).toBe('partial ')
const firstRest = first.next()
const secondRest = second.next()
service.cancelGeneration()
await expect(firstRest).rejects.toMatchObject({ code: ErrorCode.LLMProcessingCancelled })
await expect(secondRest).rejects.toMatchObject({ code: ErrorCode.LLMProcessingCancelled })
})
it('honors a caller signal and a per-call timeout', async () => {
cloud.invokeFunctionStream.mockImplementation(async (_n: string, _b: unknown, signal: AbortSignal) => ({
stream: hangingStream(signal),
error: null,
}))
const service = getPremiumLLMService()
const external = new AbortController()
const cancelled = service.chatStream([{ role: 'user', content: 'a' }], { signal: external.signal })
await cancelled.next()
const pending = cancelled.next()
external.abort()
await expect(pending).rejects.toMatchObject({ code: ErrorCode.LLMProcessingCancelled })
const timed = service.chatStream([{ role: 'user', content: 'b' }], { timeoutMs: 20 })
await timed.next()
await expect(timed.next()).rejects.toMatchObject({ code: ErrorCode.LLMProcessingTimeout })
})
it('aborts the underlying request when the consumer stops early', async () => {
const signals: AbortSignal[] = []
cloud.invokeFunctionStream.mockImplementation(async (_n: string, _b: unknown, signal: AbortSignal) => {
signals.push(signal)
return { stream: hangingStream(signal), error: null }
})
for await (const token of getPremiumLLMService().chatStream([{ role: 'user', content: 'q' }])) {
expect(token).toBe('partial ')
break
}
expect(signals[0].aborted).toBe(true)
})
})
describe('proxyHttpStatus', () => {
it('detects proxy HTTP responses and ignores transport errors', () => {
expect(proxyHttpStatus('502: {"error":"provider_request_failed"}')).toBe(502)
expect(proxyHttpStatus('429: {"error":"quota_exceeded"}')).toBe(429)
expect(proxyHttpStatus('fetch failed')).toBeNull()
expect(proxyHttpStatus('No active session — 로그인 필요')).toBeNull()
})
})