fix(sync,rag): keep restored rows, reconcile after tombstone pruning, reject partial knowledge docs, surface RAG indexing failures

This commit is contained in:
Yun Chan 2026-09-28 02:16:14 +09:00
parent bbaf1e0a99
commit 3a46437f28
13 changed files with 1630 additions and 448 deletions

View file

@ -0,0 +1,94 @@
// src/main/services/rag/pdf-text.ts
// PDF 바이너리에서 텍스트를 뽑는다 — 순수 JS + zlib, 외부 의존 없음(RAGService에서 옮김, 동작 동일).
// FlateDecode 스트림을 풀고 BT...ET 블록 안의 Tj/TJ/' 연산자 문자열을 모은다.
import { inflateSync } from 'zlib'
const STREAM_MARKER_CRLF = 'stream\r\n'
const STREAM_MARKER_LF = 'stream\n'
const END_MARKER = 'endstream'
/** 텍스트 블록(BT...ET) 하나에서 문자열 조각을 모은다. */
export function extractTextOperators(content: string, textParts: string[]): void {
const btRegex = /BT([\s\S]*?)ET/g
let btMatch: RegExpExecArray | null = null
while ((btMatch = btRegex.exec(content)) !== null) {
const block = btMatch[1]
// Tj: (text) Tj
const tjRegex = /\(([^)]*)\)\s*Tj/g
let tjMatch: RegExpExecArray | null = null
while ((tjMatch = tjRegex.exec(block)) !== null) {
if (tjMatch[1].trim()) textParts.push(tjMatch[1])
}
// TJ: [(text) num (text)] TJ
const tjArrayRegex = /\[(.*?)\]\s*TJ/g
let tjArrayMatch: RegExpExecArray | null = null
while ((tjArrayMatch = tjArrayRegex.exec(block)) !== null) {
const items = tjArrayMatch[1]
const itemRegex = /\(([^)]*)\)/g
let itemMatch: RegExpExecArray | null = null
while ((itemMatch = itemRegex.exec(items)) !== null) {
if (itemMatch[1].trim()) textParts.push(itemMatch[1])
}
}
// ' 연산자: (text) '
const quoteRegex = /\(([^)]*)\)\s*'/g
let quoteMatch: RegExpExecArray | null = null
while ((quoteMatch = quoteRegex.exec(block)) !== null) {
if (quoteMatch[1].trim()) textParts.push(quoteMatch[1])
}
}
}
/** PDF 문자열 이스케이프를 풀고 표시할 수 없는 문자를 공백으로 정리한다. */
export function normalizePdfText(parts: readonly string[]): string {
return parts
.join(' ')
.replace(/\\n/g, '\n')
.replace(/\\r/g, '\r')
.replace(/\\t/g, '\t')
.replace(/\\\(/g, '(')
.replace(/\\\)/g, ')')
.replace(/\\\\/g, '\\')
.replace(/[^\x20-\x7E\u00A0-\u00FF\u3000-\u9FFF\uAC00-\uD7AF\n]/g, ' ')
.replace(/\s+/g, ' ')
.trim()
}
/** PDF 바이너리에서 텍스트 추출. 읽을 텍스트가 없으면 빈 문자열. */
export function extractPdfText(buffer: Buffer): string {
const raw = buffer.toString('binary')
const textParts: string[] = []
let pos = 0
while (pos < raw.length) {
let streamStart = raw.indexOf(STREAM_MARKER_CRLF, pos)
let offset = STREAM_MARKER_CRLF.length
if (streamStart === -1) {
streamStart = raw.indexOf(STREAM_MARKER_LF, pos)
offset = STREAM_MARKER_LF.length
}
if (streamStart === -1) break
const dataStart = streamStart + offset
const streamEnd = raw.indexOf(END_MARKER, dataStart)
if (streamEnd === -1) break
const streamData = Buffer.from(raw.slice(dataStart, streamEnd), 'binary')
pos = streamEnd + END_MARKER.length
let decoded: string
try {
decoded = inflateSync(streamData).toString('binary')
} catch {
// 압축 안 된 스트림
decoded = streamData.toString('binary')
}
extractTextOperators(decoded, textParts)
}
return normalizePdfText(textParts)
}