fix(sync,rag): keep restored rows, reconcile after tombstone pruning, reject partial knowledge docs, surface RAG indexing failures
This commit is contained in:
parent
bbaf1e0a99
commit
3a46437f28
13 changed files with 1630 additions and 448 deletions
94
apps/desktop/src/main/services/rag/pdf-text.ts
Normal file
94
apps/desktop/src/main/services/rag/pdf-text.ts
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
// src/main/services/rag/pdf-text.ts
|
||||
// PDF 바이너리에서 텍스트를 뽑는다 — 순수 JS + zlib, 외부 의존 없음(RAGService에서 옮김, 동작 동일).
|
||||
// FlateDecode 스트림을 풀고 BT...ET 블록 안의 Tj/TJ/' 연산자 문자열을 모은다.
|
||||
|
||||
import { inflateSync } from 'zlib'
|
||||
|
||||
const STREAM_MARKER_CRLF = 'stream\r\n'
|
||||
const STREAM_MARKER_LF = 'stream\n'
|
||||
const END_MARKER = 'endstream'
|
||||
|
||||
/** 텍스트 블록(BT...ET) 하나에서 문자열 조각을 모은다. */
|
||||
export function extractTextOperators(content: string, textParts: string[]): void {
|
||||
const btRegex = /BT([\s\S]*?)ET/g
|
||||
let btMatch: RegExpExecArray | null = null
|
||||
while ((btMatch = btRegex.exec(content)) !== null) {
|
||||
const block = btMatch[1]
|
||||
|
||||
// Tj: (text) Tj
|
||||
const tjRegex = /\(([^)]*)\)\s*Tj/g
|
||||
let tjMatch: RegExpExecArray | null = null
|
||||
while ((tjMatch = tjRegex.exec(block)) !== null) {
|
||||
if (tjMatch[1].trim()) textParts.push(tjMatch[1])
|
||||
}
|
||||
|
||||
// TJ: [(text) num (text)] TJ
|
||||
const tjArrayRegex = /\[(.*?)\]\s*TJ/g
|
||||
let tjArrayMatch: RegExpExecArray | null = null
|
||||
while ((tjArrayMatch = tjArrayRegex.exec(block)) !== null) {
|
||||
const items = tjArrayMatch[1]
|
||||
const itemRegex = /\(([^)]*)\)/g
|
||||
let itemMatch: RegExpExecArray | null = null
|
||||
while ((itemMatch = itemRegex.exec(items)) !== null) {
|
||||
if (itemMatch[1].trim()) textParts.push(itemMatch[1])
|
||||
}
|
||||
}
|
||||
|
||||
// ' 연산자: (text) '
|
||||
const quoteRegex = /\(([^)]*)\)\s*'/g
|
||||
let quoteMatch: RegExpExecArray | null = null
|
||||
while ((quoteMatch = quoteRegex.exec(block)) !== null) {
|
||||
if (quoteMatch[1].trim()) textParts.push(quoteMatch[1])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** PDF 문자열 이스케이프를 풀고 표시할 수 없는 문자를 공백으로 정리한다. */
|
||||
export function normalizePdfText(parts: readonly string[]): string {
|
||||
return parts
|
||||
.join(' ')
|
||||
.replace(/\\n/g, '\n')
|
||||
.replace(/\\r/g, '\r')
|
||||
.replace(/\\t/g, '\t')
|
||||
.replace(/\\\(/g, '(')
|
||||
.replace(/\\\)/g, ')')
|
||||
.replace(/\\\\/g, '\\')
|
||||
.replace(/[^\x20-\x7E\u00A0-\u00FF\u3000-\u9FFF\uAC00-\uD7AF\n]/g, ' ')
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim()
|
||||
}
|
||||
|
||||
/** PDF 바이너리에서 텍스트 추출. 읽을 텍스트가 없으면 빈 문자열. */
|
||||
export function extractPdfText(buffer: Buffer): string {
|
||||
const raw = buffer.toString('binary')
|
||||
const textParts: string[] = []
|
||||
|
||||
let pos = 0
|
||||
while (pos < raw.length) {
|
||||
let streamStart = raw.indexOf(STREAM_MARKER_CRLF, pos)
|
||||
let offset = STREAM_MARKER_CRLF.length
|
||||
if (streamStart === -1) {
|
||||
streamStart = raw.indexOf(STREAM_MARKER_LF, pos)
|
||||
offset = STREAM_MARKER_LF.length
|
||||
}
|
||||
if (streamStart === -1) break
|
||||
|
||||
const dataStart = streamStart + offset
|
||||
const streamEnd = raw.indexOf(END_MARKER, dataStart)
|
||||
if (streamEnd === -1) break
|
||||
|
||||
const streamData = Buffer.from(raw.slice(dataStart, streamEnd), 'binary')
|
||||
pos = streamEnd + END_MARKER.length
|
||||
|
||||
let decoded: string
|
||||
try {
|
||||
decoded = inflateSync(streamData).toString('binary')
|
||||
} catch {
|
||||
// 압축 안 된 스트림
|
||||
decoded = streamData.toString('binary')
|
||||
}
|
||||
extractTextOperators(decoded, textParts)
|
||||
}
|
||||
|
||||
return normalizePdfText(textParts)
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue