99 lines
4.1 KiB
TypeScript
99 lines
4.1 KiB
TypeScript
// packages/core/src/meeting-llm-input.ts
|
|
// 긴 회의 전사를 LLM 입력 한도 안에 넣는 순수 정책.
|
|
//
|
|
// llm-proxy 는 메시지·system 하나당 8,000자를 넘으면 400 으로 거부하고, 채팅의 system 은 앞부분만
|
|
// 남기고 잘린다. 30분 넘는 회의 전사는 쉽게 이 한도를 넘어, 문서 생성·요약이 실패하고 채팅은 회의
|
|
// 앞부분만 보고 "확인되지 않습니다"라고 답했다. 여기서는
|
|
// - 문서·요약: 구간별로 줄인 뒤(map) 합쳐 한 번에 보낸다(reduce) — 조각 계획만 정한다.
|
|
// - 채팅: 질문과 관련 있는 구간만 골라 한도 안에 넣고, 잘렸다는 사실을 함께 돌려준다.
|
|
|
|
import { chunkTranscriptByLines } from './meeting-transcript'
|
|
|
|
/** 구간 요약(map) 한 번에 보낼 전사 조각 크기 — 프록시 메시지 한도(8,000자) 안쪽 */
|
|
export const MEETING_MAP_CHUNK_CHARS = 6_000
|
|
|
|
/** 채팅 발췌를 고를 때의 조각 크기 */
|
|
export const MEETING_EXCERPT_CHUNK_CHARS = 1_200
|
|
|
|
const EXCERPT_SEPARATOR = '\n…\n'
|
|
|
|
/**
|
|
* 전사가 budgetChars 를 넘으면 구간 요약용 조각으로 나눈다. 넘지 않으면 빈 배열(나눌 필요 없음).
|
|
* 한 줄이 조각 크기보다 길면 그 줄 하나가 한 조각이 된다(줄 중간은 자르지 않는다).
|
|
*/
|
|
export function planTranscriptCondense(
|
|
transcript: string,
|
|
budgetChars: number,
|
|
chunkChars: number = MEETING_MAP_CHUNK_CHARS,
|
|
): string[] {
|
|
if (transcript.trim().length <= budgetChars) return []
|
|
return chunkTranscriptByLines(transcript, Math.max(1, Math.min(chunkChars, budgetChars)))
|
|
}
|
|
|
|
/** 구간 요약들을 순서 표시와 함께 이어 붙인다(reduce 입력). */
|
|
export function joinCondensedSections(sections: readonly string[]): string {
|
|
const total = sections.length
|
|
return sections.map((section, i) => `### ${i + 1}/${total}\n${section.trim()}`).join('\n\n')
|
|
}
|
|
|
|
/** 문자·숫자만 남긴 소문자 문자 bigram — 조사가 붙은 한국어("결정은"/"결정을")도 겹치게 한다. */
|
|
function bigrams(text: string): Set<string> {
|
|
const normalized = text.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ')
|
|
const grams = new Set<string>()
|
|
for (const word of normalized.split(' ')) {
|
|
const chars = [...word]
|
|
if (chars.length === 1) grams.add(chars[0])
|
|
for (let i = 0; i + 1 < chars.length; i++) grams.add(chars[i] + chars[i + 1])
|
|
}
|
|
return grams
|
|
}
|
|
|
|
export interface TranscriptExcerpt {
|
|
text: string
|
|
/** 전사를 통째로 넣었는가 (false 면 일부 구간만 넣었다) */
|
|
complete: boolean
|
|
includedChunks: number
|
|
totalChunks: number
|
|
}
|
|
|
|
/**
|
|
* 채팅 질문에 쓸 전사를 budgetChars 안으로 고른다.
|
|
* - 전사 전체가 들어가면 그대로 쓴다.
|
|
* - 아니면 질문과 겹치는 bigram 이 많은 구간부터 넣고(동점이면 앞 구간), 시간 순으로 되돌려 이어 붙인다.
|
|
*/
|
|
export function selectTranscriptExcerpts(
|
|
transcript: string,
|
|
question: string,
|
|
budgetChars: number,
|
|
chunkChars: number = MEETING_EXCERPT_CHUNK_CHARS,
|
|
): TranscriptExcerpt {
|
|
const text = transcript.trim()
|
|
if (text.length <= budgetChars) {
|
|
return { text, complete: true, includedChunks: text ? 1 : 0, totalChunks: text ? 1 : 0 }
|
|
}
|
|
const chunks = chunkTranscriptByLines(text, Math.max(1, Math.min(chunkChars, budgetChars)))
|
|
const query = bigrams(question)
|
|
const scored = chunks.map((chunk, index) => {
|
|
const grams = bigrams(chunk)
|
|
let score = 0
|
|
for (const gram of query) if (grams.has(gram)) score++
|
|
return { index, chunk, score }
|
|
})
|
|
scored.sort((a, b) => b.score - a.score || a.index - b.index)
|
|
|
|
const chosen: Array<{ index: number; chunk: string }> = []
|
|
let used = 0
|
|
for (const candidate of scored) {
|
|
const cost = candidate.chunk.length + (chosen.length > 0 ? EXCERPT_SEPARATOR.length : 0)
|
|
if (used + cost > budgetChars) continue
|
|
chosen.push(candidate)
|
|
used += cost
|
|
}
|
|
chosen.sort((a, b) => a.index - b.index)
|
|
return {
|
|
text: chosen.map((c) => c.chunk).join(EXCERPT_SEPARATOR),
|
|
complete: chosen.length === chunks.length,
|
|
includedChunks: chosen.length,
|
|
totalChunks: chunks.length,
|
|
}
|
|
}
|