// packages/core/src/meeting-llm-input.ts // 긴 회의 전사를 LLM 입력 한도 안에 넣는 순수 정책. // // llm-proxy 는 메시지·system 하나당 8,000자를 넘으면 400 으로 거부하고, 채팅의 system 은 앞부분만 // 남기고 잘린다. 30분 넘는 회의 전사는 쉽게 이 한도를 넘어, 문서 생성·요약이 실패하고 채팅은 회의 // 앞부분만 보고 "확인되지 않습니다"라고 답했다. 여기서는 // - 문서·요약: 구간별로 줄인 뒤(map) 합쳐 한 번에 보낸다(reduce) — 조각 계획만 정한다. // - 채팅: 질문과 관련 있는 구간만 골라 한도 안에 넣고, 잘렸다는 사실을 함께 돌려준다. import { chunkTranscriptByLines } from './meeting-transcript' /** 구간 요약(map) 한 번에 보낼 전사 조각 크기 — 프록시 메시지 한도(8,000자) 안쪽 */ export const MEETING_MAP_CHUNK_CHARS = 6_000 /** 채팅 발췌를 고를 때의 조각 크기 */ export const MEETING_EXCERPT_CHUNK_CHARS = 1_200 const EXCERPT_SEPARATOR = '\n…\n' /** * 전사가 budgetChars 를 넘으면 구간 요약용 조각으로 나눈다. 넘지 않으면 빈 배열(나눌 필요 없음). * 조각은 모두 조각 크기 이하다 — 줄바꿈 없는 긴 줄은 문장 경계 → 공백 → 글자 수 순으로 자른다. */ export function planTranscriptCondense( transcript: string, budgetChars: number, chunkChars: number = MEETING_MAP_CHUNK_CHARS, ): string[] { if (transcript.trim().length <= budgetChars) return [] return chunkTranscriptByLines(transcript, Math.max(1, Math.min(chunkChars, budgetChars))) } /** 구간 요약들을 순서 표시와 함께 이어 붙인다(reduce 입력). */ export function joinCondensedSections(sections: readonly string[]): string { const total = sections.length return sections.map((section, i) => `### ${i + 1}/${total}\n${section.trim()}`).join('\n\n') } /** 문자·숫자만 남긴 소문자 문자 bigram — 조사가 붙은 한국어("결정은"/"결정을")도 겹치게 한다. */ function bigrams(text: string): Set { const normalized = text.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ') const grams = new Set() for (const word of normalized.split(' ')) { const chars = [...word] if (chars.length === 1) grams.add(chars[0]) for (let i = 0; i + 1 < chars.length; i++) grams.add(chars[i] + chars[i + 1]) } return grams } export interface TranscriptExcerpt { text: string /** 전사를 통째로 넣었는가 (false 면 일부 구간만 넣었다) */ complete: boolean includedChunks: number totalChunks: number } /** * 채팅 질문에 쓸 전사를 budgetChars 안으로 고른다. * - 전사 전체가 들어가면 그대로 쓴다. * - 아니면 질문과 겹치는 bigram 이 많은 구간부터 넣고(동점이면 앞 구간), 시간 순으로 되돌려 이어 붙인다. */ export function selectTranscriptExcerpts( transcript: string, question: string, budgetChars: number, chunkChars: number = MEETING_EXCERPT_CHUNK_CHARS, ): TranscriptExcerpt { const text = transcript.trim() if (text.length <= budgetChars) { return { text, complete: true, includedChunks: text ? 1 : 0, totalChunks: text ? 1 : 0 } } const chunks = chunkTranscriptByLines(text, Math.max(1, Math.min(chunkChars, budgetChars))) const query = bigrams(question) const scored = chunks.map((chunk, index) => { const grams = bigrams(chunk) let score = 0 for (const gram of query) if (grams.has(gram)) score++ return { index, chunk, score } }) scored.sort((a, b) => b.score - a.score || a.index - b.index) const chosen: Array<{ index: number; chunk: string }> = [] let used = 0 for (const candidate of scored) { const cost = candidate.chunk.length + (chosen.length > 0 ? EXCERPT_SEPARATOR.length : 0) if (used + cost > budgetChars) continue chosen.push(candidate) used += cost } chosen.sort((a, b) => a.index - b.index) return { text: chosen.map((c) => c.chunk).join(EXCERPT_SEPARATOR), complete: chosen.length === chunks.length, includedChunks: chosen.length, totalChunks: chunks.length, } }