d3ro-voice/packages/core/src/meeting-llm-input.ts
Yun Chan ba9ef9741e fix: red-team round 3 hardening across desktop, mobile, core and server
Batch of red-team r3 fixes that were in the working tree before the
2026-09-28 design overhaul, committed as one unit with their tests.

- desktop main: STT timeouts and sidecar, voice recording store, sync
  (credentials, audio, knowledge reindex, push gates), runtime
  provisioner, update policy, AltGr keybindings, voice-command policy,
  dictionary file codec/limits, meeting transcript condensing and a
  local recording ledger so interrupted-session recovery only closes
  meetings this device recorded (a phone's live meeting is left alone).
- mobile: login CSRF via implicit token callbacks rejected, account
  deletion/retention, durable queue retention, knowledge realtime
  without unfiltered DELETE, meeting re-record failure paths, cloud STT
  client, preferences store/resync.
- core: text chunking splits long unbroken transcripts to fit, template
  field policy, dictionary limits, meeting markdown inline handling.
- server: payple webhook policy and cancellation order scope, meeting
  document generation quota, team RPC null-role guard, unified LLM
  quota in-flight accounting, knowledge chunk vector index, meeting
  re-record failure paths (migrations 20260929*).
- ci: portable/runtime feed gates, update-policy schema, Forgejo file
  delete and alias planning.

Four older tests are updated to the new contracts rather than the old
behavior: token-pair auth callbacks are rejected, knowledge realtime no
longer subscribes to DELETE, long transcript lines are split, and
meeting recovery requires the local recording ledger for empty rows.
2026-09-28 20:45:52 +09:00

99 lines
4.1 KiB
TypeScript

// packages/core/src/meeting-llm-input.ts
// 긴 회의 전사를 LLM 입력 한도 안에 넣는 순수 정책.
//
// llm-proxy 는 메시지·system 하나당 8,000자를 넘으면 400 으로 거부하고, 채팅의 system 은 앞부분만
// 남기고 잘린다. 30분 넘는 회의 전사는 쉽게 이 한도를 넘어, 문서 생성·요약이 실패하고 채팅은 회의
// 앞부분만 보고 "확인되지 않습니다"라고 답했다. 여기서는
// - 문서·요약: 구간별로 줄인 뒤(map) 합쳐 한 번에 보낸다(reduce) — 조각 계획만 정한다.
// - 채팅: 질문과 관련 있는 구간만 골라 한도 안에 넣고, 잘렸다는 사실을 함께 돌려준다.
import { chunkTranscriptByLines } from './meeting-transcript'
/** 구간 요약(map) 한 번에 보낼 전사 조각 크기 — 프록시 메시지 한도(8,000자) 안쪽 */
export const MEETING_MAP_CHUNK_CHARS = 6_000
/** 채팅 발췌를 고를 때의 조각 크기 */
export const MEETING_EXCERPT_CHUNK_CHARS = 1_200
const EXCERPT_SEPARATOR = '\n…\n'
/**
* 전사가 budgetChars 를 넘으면 구간 요약용 조각으로 나눈다. 넘지 않으면 빈 배열(나눌 필요 없음).
* 조각은 모두 조각 크기 이하다 — 줄바꿈 없는 긴 줄은 문장 경계 → 공백 → 글자 수 순으로 자른다.
*/
export function planTranscriptCondense(
transcript: string,
budgetChars: number,
chunkChars: number = MEETING_MAP_CHUNK_CHARS,
): string[] {
if (transcript.trim().length <= budgetChars) return []
return chunkTranscriptByLines(transcript, Math.max(1, Math.min(chunkChars, budgetChars)))
}
/** 구간 요약들을 순서 표시와 함께 이어 붙인다(reduce 입력). */
export function joinCondensedSections(sections: readonly string[]): string {
const total = sections.length
return sections.map((section, i) => `### ${i + 1}/${total}\n${section.trim()}`).join('\n\n')
}
/** 문자·숫자만 남긴 소문자 문자 bigram — 조사가 붙은 한국어("결정은"/"결정을")도 겹치게 한다. */
function bigrams(text: string): Set<string> {
const normalized = text.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ')
const grams = new Set<string>()
for (const word of normalized.split(' ')) {
const chars = [...word]
if (chars.length === 1) grams.add(chars[0])
for (let i = 0; i + 1 < chars.length; i++) grams.add(chars[i] + chars[i + 1])
}
return grams
}
export interface TranscriptExcerpt {
text: string
/** 전사를 통째로 넣었는가 (false 면 일부 구간만 넣었다) */
complete: boolean
includedChunks: number
totalChunks: number
}
/**
* 채팅 질문에 쓸 전사를 budgetChars 안으로 고른다.
* - 전사 전체가 들어가면 그대로 쓴다.
* - 아니면 질문과 겹치는 bigram 이 많은 구간부터 넣고(동점이면 앞 구간), 시간 순으로 되돌려 이어 붙인다.
*/
export function selectTranscriptExcerpts(
transcript: string,
question: string,
budgetChars: number,
chunkChars: number = MEETING_EXCERPT_CHUNK_CHARS,
): TranscriptExcerpt {
const text = transcript.trim()
if (text.length <= budgetChars) {
return { text, complete: true, includedChunks: text ? 1 : 0, totalChunks: text ? 1 : 0 }
}
const chunks = chunkTranscriptByLines(text, Math.max(1, Math.min(chunkChars, budgetChars)))
const query = bigrams(question)
const scored = chunks.map((chunk, index) => {
const grams = bigrams(chunk)
let score = 0
for (const gram of query) if (grams.has(gram)) score++
return { index, chunk, score }
})
scored.sort((a, b) => b.score - a.score || a.index - b.index)
const chosen: Array<{ index: number; chunk: string }> = []
let used = 0
for (const candidate of scored) {
const cost = candidate.chunk.length + (chosen.length > 0 ? EXCERPT_SEPARATOR.length : 0)
if (used + cost > budgetChars) continue
chosen.push(candidate)
used += cost
}
chosen.sort((a, b) => a.index - b.index)
return {
text: chosen.map((c) => c.chunk).join(EXCERPT_SEPARATOR),
complete: chosen.length === chunks.length,
includedChunks: chosen.length,
totalChunks: chunks.length,
}
}