d3ro-voice/packages/core/__tests__/transcript-chunking-redteam-r3-3.test.ts
Yun Chan ba9ef9741e fix: red-team round 3 hardening across desktop, mobile, core and server
Batch of red-team r3 fixes that were in the working tree before the
2026-09-28 design overhaul, committed as one unit with their tests.

- desktop main: STT timeouts and sidecar, voice recording store, sync
  (credentials, audio, knowledge reindex, push gates), runtime
  provisioner, update policy, AltGr keybindings, voice-command policy,
  dictionary file codec/limits, meeting transcript condensing and a
  local recording ledger so interrupted-session recovery only closes
  meetings this device recorded (a phone's live meeting is left alone).
- mobile: login CSRF via implicit token callbacks rejected, account
  deletion/retention, durable queue retention, knowledge realtime
  without unfiltered DELETE, meeting re-record failure paths, cloud STT
  client, preferences store/resync.
- core: text chunking splits long unbroken transcripts to fit, template
  field policy, dictionary limits, meeting markdown inline handling.
- server: payple webhook policy and cancellation order scope, meeting
  document generation quota, team RPC null-role guard, unified LLM
  quota in-flight accounting, knowledge chunk vector index, meeting
  re-record failure paths (migrations 20260929*).
- ci: portable/runtime feed gates, update-policy schema, Forgejo file
  delete and alias planning.

Four older tests are updated to the new contracts rather than the old
behavior: token-pair auth callbacks are rejected, knowledge realtime no
longer subscribes to DELETE, long transcript lines are split, and
meeting recovery requires the local recording ledger for empty rows.
2026-09-28 20:45:52 +09:00

107 lines
5.1 KiB
TypeScript

// 줄바꿈 없는 긴 전사(자막·파일 전사 히스토리, 폰 회의 raw_transcript)도 한도 이하 조각으로 나뉘어야 한다.
import { describe, expect, it } from 'vitest'
import { splitTextToFit } from '../src/text-chunking'
import { chunkTranscriptByLines, looksTruncatedRewrite } from '../src/meeting-transcript'
import {
MEETING_MAP_CHUNK_CHARS,
planTranscriptCondense,
selectTranscriptExcerpts,
} from '../src/meeting-llm-input'
/** 자막 세션 fullText 처럼 세그먼트를 공백으로 이어 붙인 한 줄 전사 (약 20,000자) */
function singleLineTranscript(targetChars: number): string {
const sentences: string[] = []
let length = 0
for (let i = 0; length < targetChars; i++) {
const sentence =
i === 437 ? '출시 담당자는 김민수로 결정했습니다.' : `오늘 회의의 ${i}번째 안건에 대해 이야기를 나눴습니다.`
sentences.push(sentence)
length += sentence.length + 1
}
return sentences.join(' ')
}
describe('splitTextToFit', () => {
it('들어가면 그대로 둔다', () => {
expect(splitTextToFit(' 짧은 문장. ', 100)).toEqual(['짧은 문장.'])
expect(splitTextToFit(' ', 100)).toEqual([])
})
it('문장 경계를 먼저 쓴다', () => {
expect(splitTextToFit('첫 문장입니다. 둘째 문장입니다. 셋째', 20)).toEqual(['첫 문장입니다. 둘째 문장입니다.', '셋째'])
})
it('소수점은 문장 경계가 아니다', () => {
const pieces = splitTextToFit('매출은 3.5 억 원 증가 예상입니다', 12)
expect(pieces.every((p) => p.length <= 12)).toBe(true)
expect(pieces.join(' ')).toBe('매출은 3.5 억 원 증가 예상입니다')
expect(pieces.some((p) => p.endsWith('3.'))).toBe(false)
})
it('문장 경계가 너무 앞에 있으면 공백에서 자른다', () => {
expect(splitTextToFit('네. 그러면 다음 주 월요일까지 초안을 보내겠습니다', 20)).toEqual([
'네. 그러면 다음 주 월요일까지',
'초안을 보내겠습니다',
])
})
it('전각 문장 부호는 뒤에 공백이 없어도 경계다', () => {
expect(splitTextToFit('今日は晴れです。明日は雨です。', 9)).toEqual(['今日は晴れです。', '明日は雨です。'])
})
it('경계가 없으면 글자 수로 자르고 서로게이트 쌍은 가르지 않는다', () => {
expect(splitTextToFit('x'.repeat(25), 10)).toEqual(['x'.repeat(10), 'x'.repeat(10), 'x'.repeat(5)])
const emoji = `${'a'.repeat(4)}😀${'b'.repeat(4)}`
const pieces = splitTextToFit(emoji, 5)
expect(pieces.join('')).toBe(emoji)
expect(pieces.every((p) => p.length <= 5)).toBe(true)
})
})
describe('chunkTranscriptByLines — 줄바꿈 없는 긴 줄', () => {
it('단일 줄 20,000자 전사도 한도 이하 조각 여러 개로 나뉜다', () => {
const transcript = singleLineTranscript(20_000)
expect(transcript.includes('\n')).toBe(false)
const chunks = chunkTranscriptByLines(transcript, 1_500)
expect(chunks.length).toBeGreaterThanOrEqual(14)
expect(chunks.every((c) => c.length <= 1_500)).toBe(true)
// 내용은 잃지 않는다(조각 경계의 공백만 줄바꿈으로 바뀐다)
expect(chunks.join(' ').replace(/\s+/g, ' ')).toBe(transcript)
// 문장 경계에서 끊었다
expect(chunks.every((c) => c.endsWith('.'))).toBe(true)
})
it('긴 시각 줄을 자르면 조각마다 [MM:SS] [화자] 머리를 붙인다', () => {
const line = `[12:34] [화자 1] ${'가나다라 마바사. '.repeat(40).trim()}`
const chunks = chunkTranscriptByLines(`[00:01] 짧은 줄\n${line}`, 120)
expect(chunks.every((c) => c.length <= 120)).toBe(true)
const pieces = chunks.flatMap((c) => c.split('\n')).slice(1)
expect(pieces.length).toBeGreaterThan(1)
expect(pieces.every((p) => p.startsWith('[12:34] [화자 1] '))).toBe(true)
})
it('다시 쓰기 조각 하나를 그대로 돌려받으면 잘린 것으로 보지 않는다', () => {
const chunks = chunkTranscriptByLines(singleLineTranscript(20_000), 1_500)
for (const chunk of chunks) expect(looksTruncatedRewrite(chunk, chunk)).toBe(false)
})
})
describe('meeting-llm-input — 단일 줄 전사', () => {
it('planTranscriptCondense 는 프록시 한도 안쪽 조각 여러 개를 낸다', () => {
const transcript = singleLineTranscript(20_000)
const chunks = planTranscriptCondense(transcript, 8_000)
expect(chunks.length).toBeGreaterThanOrEqual(4)
expect(chunks.every((c) => c.length <= MEETING_MAP_CHUNK_CHARS)).toBe(true)
})
it('selectTranscriptExcerpts 는 질문과 관련된 구간을 실제로 넣는다(빈 발췌 금지)', () => {
const transcript = singleLineTranscript(20_000)
const excerpt = selectTranscriptExcerpts(transcript, '출시 담당자는 누구로 결정했나요?', 7_000)
expect(excerpt.complete).toBe(false)
expect(excerpt.totalChunks).toBeGreaterThan(1)
expect(excerpt.includedChunks).toBeGreaterThan(0)
expect(excerpt.text.length).toBeGreaterThan(0)
expect(excerpt.text.length).toBeLessThanOrEqual(7_000)
expect(excerpt.text).toContain('출시 담당자는 김민수로 결정했습니다.')
})
})