94 lines
3.2 KiB
TypeScript
94 lines
3.2 KiB
TypeScript
// src/main/services/rag/pdf-text.ts
|
|
// PDF 바이너리에서 텍스트를 뽑는다 — 순수 JS + zlib, 외부 의존 없음(RAGService에서 옮김, 동작 동일).
|
|
// FlateDecode 스트림을 풀고 BT...ET 블록 안의 Tj/TJ/' 연산자 문자열을 모은다.
|
|
|
|
import { inflateSync } from 'zlib'
|
|
|
|
const STREAM_MARKER_CRLF = 'stream\r\n'
|
|
const STREAM_MARKER_LF = 'stream\n'
|
|
const END_MARKER = 'endstream'
|
|
|
|
/** 텍스트 블록(BT...ET) 하나에서 문자열 조각을 모은다. */
|
|
export function extractTextOperators(content: string, textParts: string[]): void {
|
|
const btRegex = /BT([\s\S]*?)ET/g
|
|
let btMatch: RegExpExecArray | null = null
|
|
while ((btMatch = btRegex.exec(content)) !== null) {
|
|
const block = btMatch[1]
|
|
|
|
// Tj: (text) Tj
|
|
const tjRegex = /\(([^)]*)\)\s*Tj/g
|
|
let tjMatch: RegExpExecArray | null = null
|
|
while ((tjMatch = tjRegex.exec(block)) !== null) {
|
|
if (tjMatch[1].trim()) textParts.push(tjMatch[1])
|
|
}
|
|
|
|
// TJ: [(text) num (text)] TJ
|
|
const tjArrayRegex = /\[(.*?)\]\s*TJ/g
|
|
let tjArrayMatch: RegExpExecArray | null = null
|
|
while ((tjArrayMatch = tjArrayRegex.exec(block)) !== null) {
|
|
const items = tjArrayMatch[1]
|
|
const itemRegex = /\(([^)]*)\)/g
|
|
let itemMatch: RegExpExecArray | null = null
|
|
while ((itemMatch = itemRegex.exec(items)) !== null) {
|
|
if (itemMatch[1].trim()) textParts.push(itemMatch[1])
|
|
}
|
|
}
|
|
|
|
// ' 연산자: (text) '
|
|
const quoteRegex = /\(([^)]*)\)\s*'/g
|
|
let quoteMatch: RegExpExecArray | null = null
|
|
while ((quoteMatch = quoteRegex.exec(block)) !== null) {
|
|
if (quoteMatch[1].trim()) textParts.push(quoteMatch[1])
|
|
}
|
|
}
|
|
}
|
|
|
|
/** PDF 문자열 이스케이프를 풀고 표시할 수 없는 문자를 공백으로 정리한다. */
|
|
export function normalizePdfText(parts: readonly string[]): string {
|
|
return parts
|
|
.join(' ')
|
|
.replace(/\\n/g, '\n')
|
|
.replace(/\\r/g, '\r')
|
|
.replace(/\\t/g, '\t')
|
|
.replace(/\\\(/g, '(')
|
|
.replace(/\\\)/g, ')')
|
|
.replace(/\\\\/g, '\\')
|
|
.replace(/[^\x20-\x7E\u00A0-\u00FF\u3000-\u9FFF\uAC00-\uD7AF\n]/g, ' ')
|
|
.replace(/\s+/g, ' ')
|
|
.trim()
|
|
}
|
|
|
|
/** PDF 바이너리에서 텍스트 추출. 읽을 텍스트가 없으면 빈 문자열. */
|
|
export function extractPdfText(buffer: Buffer): string {
|
|
const raw = buffer.toString('binary')
|
|
const textParts: string[] = []
|
|
|
|
let pos = 0
|
|
while (pos < raw.length) {
|
|
let streamStart = raw.indexOf(STREAM_MARKER_CRLF, pos)
|
|
let offset = STREAM_MARKER_CRLF.length
|
|
if (streamStart === -1) {
|
|
streamStart = raw.indexOf(STREAM_MARKER_LF, pos)
|
|
offset = STREAM_MARKER_LF.length
|
|
}
|
|
if (streamStart === -1) break
|
|
|
|
const dataStart = streamStart + offset
|
|
const streamEnd = raw.indexOf(END_MARKER, dataStart)
|
|
if (streamEnd === -1) break
|
|
|
|
const streamData = Buffer.from(raw.slice(dataStart, streamEnd), 'binary')
|
|
pos = streamEnd + END_MARKER.length
|
|
|
|
let decoded: string
|
|
try {
|
|
decoded = inflateSync(streamData).toString('binary')
|
|
} catch {
|
|
// 압축 안 된 스트림
|
|
decoded = streamData.toString('binary')
|
|
}
|
|
extractTextOperators(decoded, textParts)
|
|
}
|
|
|
|
return normalizePdfText(textParts)
|
|
}
|