// tests/main/services/rag-docx-redteam-r1-3.test.ts // Word 가 만든(deflate 압축) .docx 에서 본문을 뽑는지, ZIP 이 아니면 바이너리 잡음을 // 색인하지 않고 실패하는지 확인한다. import { describe, it, expect } from 'vitest' import { Document, Packer, Paragraph, TextRun } from 'docx' import { extractDocxText, decodeXmlEntities, wordXmlToText, DocxExtractionError, } from '../../../src/main/services/rag/docx-text' async function buildDocx(paragraphs: string[]): Promise { const doc = new Document({ sections: [{ children: paragraphs.map((p) => new Paragraph({ children: [new TextRun(p)] })) }], }) return Buffer.from(await Packer.toBuffer(doc)) } describe('extractDocxText', () => { it('압축된 word/document.xml 에서 문단 텍스트를 뽑는다', async () => { const buffer = await buildDocx(['분기 매출 보고서', 'List 타입을 쓰고 A & B 를 비교한다']) // 원본 ZIP 바이트에는 본문이 평문으로 보이지 않는다 (예전 정규식이 실패한 이유) expect(buffer.toString('utf-8')).not.toContain('분기 매출 보고서') const text = extractDocxText(buffer) expect(text).toBe('분기 매출 보고서\nList 타입을 쓰고 A & B 를 비교한다') }) it('ZIP 이 아니면 잡음을 돌려주지 않고 실패한다', () => { const garbage = Buffer.from('PK\u0003\u0004 this is not really a zip archive at all, just junk bytes', 'utf-8') expect(() => extractDocxText(garbage)).toThrow(DocxExtractionError) }) it('XML 엔티티와 탭/줄바꿈을 복원한다', () => { expect(decodeXmlEntities('a & b <c> AB')).toBe('a & b AB') const xml = '하나 둘' + '셋넷' expect(wordXmlToText(xml)).toBe('하나\t 둘\n셋\n넷') }) })