59 lines
2.9 KiB
TypeScript
59 lines
2.9 KiB
TypeScript
import { describe, expect, it } from 'vitest'
|
|
|
|
import { extractTerms, jaccard, tokenizeTerms } from '../src/personal-graph'
|
|
|
|
// 회귀: 예전 분리 정규식은 ASCII/한글/가나/한자만 문자로 봤다.
|
|
// 키릴·태국·그리스·아랍 문장은 terms=[] 가 되어 shares_terms 엣지가 생기지 않았고,
|
|
// 악센트 라틴 단어는 'pr'/'ller' 같은 조각으로 쪼개져 거짓 겹침을 만들었다.
|
|
describe('extractTerms — 다국어 문자 체계', () => {
|
|
it('러시아어(키릴) 단어를 용어로 뽑는다', () => {
|
|
const terms = extractTerms('Отправлю отчёт завтра утром')
|
|
expect(terms).toEqual(expect.arrayContaining(['отправлю', 'отчёт', 'завтра', 'утром']))
|
|
})
|
|
|
|
it('독일어 악센트 단어를 조각내지 않는다', () => {
|
|
const terms = extractTerms('Müller schickt die Präsentation')
|
|
expect(terms).toContain('müller')
|
|
expect(terms).toContain('präsentation')
|
|
expect(terms).not.toContain('pr')
|
|
expect(terms).not.toContain('ller')
|
|
expect(terms).not.toContain('sentation')
|
|
})
|
|
|
|
it('프랑스어 악센트 단어를 온전히 유지한다', () => {
|
|
expect(extractTerms('café résumé')).toEqual(['café', 'résumé'])
|
|
})
|
|
|
|
it('그리스어·아랍어 단어를 용어로 뽑는다', () => {
|
|
expect(extractTerms('Στέλνω την αναφορά αύριο')).toEqual(
|
|
expect.arrayContaining(['στέλνω', 'την', 'αναφορά', 'αύριο'])
|
|
)
|
|
expect(extractTerms('سأرسل التقرير غدا')).toEqual(expect.arrayContaining(['سأرسل', 'التقرير', 'غدا']))
|
|
})
|
|
|
|
it('결합 문자(NFD)로 들어온 악센트도 같은 용어가 된다', () => {
|
|
const composed = extractTerms('Präsentation')
|
|
const decomposed = extractTerms('Präsentation')
|
|
expect(decomposed).toEqual(composed)
|
|
})
|
|
|
|
it('띄어쓰기 없는 태국어 문장도 용어를 만든다', () => {
|
|
const terms = extractTerms('ส่งรายงานพรุ่งนี้เช้า')
|
|
expect(terms.length).toBeGreaterThan(0)
|
|
if (typeof Intl !== 'undefined' && typeof Intl.Segmenter === 'function') {
|
|
expect(terms).toContain('รายงาน')
|
|
}
|
|
})
|
|
|
|
it('같은 용어를 공유하는 러시아어 문장은 자카드가 0 보다 크다', () => {
|
|
const a = extractTerms('Отправлю отчёт завтра утром')
|
|
const b = extractTerms('Проверю отчёт завтра вечером')
|
|
expect(jaccard(a, b)).toBeGreaterThan(0)
|
|
})
|
|
|
|
it('한국어·영어·일본어·중국어 토큰화는 기존과 같다', () => {
|
|
expect(tokenizeTerms('오늘 회의에서, Report 정리!')).toEqual(['오늘', '회의에서', 'report', '정리'])
|
|
expect(tokenizeTerms('会議の資料を送ります')).toEqual(['会議の資料を送ります'])
|
|
expect(tokenizeTerms('明天发送报告')).toEqual(['明天发送报告'])
|
|
})
|
|
})
|