43 lines
2 KiB
TypeScript
43 lines
2 KiB
TypeScript
// tests/main/services/rag-docx-redteam-r1-3.test.ts
|
|
// Word 가 만든(deflate 압축) .docx 에서 본문을 뽑는지, ZIP 이 아니면 바이너리 잡음을
|
|
// 색인하지 않고 실패하는지 확인한다.
|
|
|
|
import { describe, it, expect } from 'vitest'
|
|
import { Document, Packer, Paragraph, TextRun } from 'docx'
|
|
import {
|
|
extractDocxText,
|
|
decodeXmlEntities,
|
|
wordXmlToText,
|
|
DocxExtractionError,
|
|
} from '../../../src/main/services/rag/docx-text'
|
|
|
|
async function buildDocx(paragraphs: string[]): Promise<Buffer> {
|
|
const doc = new Document({
|
|
sections: [{ children: paragraphs.map((p) => new Paragraph({ children: [new TextRun(p)] })) }],
|
|
})
|
|
return Buffer.from(await Packer.toBuffer(doc))
|
|
}
|
|
|
|
describe('extractDocxText', () => {
|
|
it('압축된 word/document.xml 에서 문단 텍스트를 뽑는다', async () => {
|
|
const buffer = await buildDocx(['분기 매출 보고서', 'List<String> 타입을 쓰고 A & B 를 비교한다'])
|
|
// 원본 ZIP 바이트에는 본문이 평문으로 보이지 않는다 (예전 정규식이 실패한 이유)
|
|
expect(buffer.toString('utf-8')).not.toContain('분기 매출 보고서')
|
|
|
|
const text = extractDocxText(buffer)
|
|
expect(text).toBe('분기 매출 보고서\nList<String> 타입을 쓰고 A & B 를 비교한다')
|
|
})
|
|
|
|
it('ZIP 이 아니면 잡음을 돌려주지 않고 실패한다', () => {
|
|
const garbage = Buffer.from('PK\u0003\u0004 this is not really a zip archive at all, just junk bytes', 'utf-8')
|
|
expect(() => extractDocxText(garbage)).toThrow(DocxExtractionError)
|
|
})
|
|
|
|
it('XML 엔티티와 탭/줄바꿈을 복원한다', () => {
|
|
expect(decodeXmlEntities('a & b <c> AB')).toBe('a & b <c> AB')
|
|
const xml =
|
|
'<w:body><w:p><w:r><w:t>하나</w:t><w:tab/><w:t xml:space="preserve"> 둘</w:t></w:r></w:p>' +
|
|
'<w:p><w:pPr><w:pStyle w:val="x"/></w:pPr><w:r><w:t>셋</w:t><w:br/><w:t>넷</w:t></w:r></w:p><w:p/></w:body>'
|
|
expect(wordXmlToText(xml)).toBe('하나\t 둘\n셋\n넷')
|
|
})
|
|
})
|