fix(core): extract personal-graph terms from any script
This commit is contained in:
parent
102f3ef934
commit
bbaf1e0a99
2 changed files with 111 additions and 4 deletions
|
|
@ -93,15 +93,63 @@ const STOPWORDS: ReadonlySet<string> = new Set([
|
|||
/** 최소 길이 (1~2자 토큰은 노이즈가 많다). */
|
||||
const MIN_TERM_CHARS = 2
|
||||
|
||||
/**
|
||||
* 용어 경계 — 어떤 문자 체계든 문자(L)·숫자(N)·결합 부호(M)가 아닌 것.
|
||||
*
|
||||
* 예전에는 ASCII/한글/가나/한자만 허용해서 키릴·그리스·아랍·태국 문자는 통째로 사라지고
|
||||
* 'Präsentation' 같은 악센트 라틴 단어는 'pr' + 'sentation' 조각으로 쪼개졌다.
|
||||
*/
|
||||
const TERM_SEPARATOR = /[^\p{L}\p{N}\p{M}]+/u
|
||||
|
||||
/** 띄어쓰기 없이 쓰는 문자 체계 — 경계 분리만으로는 문장 전체가 한 토큰이 된다. */
|
||||
const UNSPACED_SCRIPT = /[\p{Script=Thai}\p{Script=Lao}\p{Script=Khmer}\p{Script=Myanmar}]/u
|
||||
|
||||
let wordSegmenter: Intl.Segmenter | null | undefined
|
||||
|
||||
/** 런타임이 지원할 때만 단어 분할기를 만든다 (Hermes 등에는 없을 수 있다). */
|
||||
function getWordSegmenter(): Intl.Segmenter | null {
|
||||
if (wordSegmenter !== undefined) return wordSegmenter
|
||||
wordSegmenter =
|
||||
typeof Intl !== 'undefined' && typeof Intl.Segmenter === 'function'
|
||||
? new Intl.Segmenter(undefined, { granularity: 'word' })
|
||||
: null
|
||||
return wordSegmenter
|
||||
}
|
||||
|
||||
/** 띄어쓰기 없는 문자 체계의 토큰을 사전 기반 단어로 나눈다. 분할기가 없으면 그대로 둔다. */
|
||||
function segmentUnspacedToken(token: string): string[] {
|
||||
if (!UNSPACED_SCRIPT.test(token)) return [token]
|
||||
const segmenter = getWordSegmenter()
|
||||
if (!segmenter) return [token]
|
||||
const words: string[] = []
|
||||
for (const part of segmenter.segment(token)) {
|
||||
if (part.isWordLike) words.push(part.segment)
|
||||
}
|
||||
return words.length > 0 ? words : [token]
|
||||
}
|
||||
|
||||
/**
|
||||
* 텍스트를 비교용 토큰으로 자른다 (불용어/길이 필터 전 단계).
|
||||
*
|
||||
* NFC 정규화 → 소문자화 → 문자/숫자/결합 부호 경계로 분리 → 띄어쓰기 없는 문자 체계는 단어 분할.
|
||||
* 한글 음절·ASCII·가나·한자로 된 토큰은 예전과 같은 결과를 낸다 (저장된 노드의 terms 와 호환).
|
||||
*/
|
||||
export function tokenizeTerms(text: string): string[] {
|
||||
return text
|
||||
.normalize('NFC')
|
||||
.toLowerCase()
|
||||
.split(TERM_SEPARATOR)
|
||||
.filter((token) => token.length > 0)
|
||||
.flatMap(segmentUnspacedToken)
|
||||
}
|
||||
|
||||
/**
|
||||
* 문장에서 비교용 용어를 뽑는다.
|
||||
*
|
||||
* 소문자화 → 문자/숫자 경계로 분리 → 불용어/짧은 토큰 제거 → 빈도 순.
|
||||
* 토큰화 → 불용어/짧은 토큰 제거 → 빈도 순.
|
||||
*/
|
||||
export function extractTerms(text: string, limit = 8): string[] {
|
||||
const tokens = text
|
||||
.toLowerCase()
|
||||
.split(/[^0-9a-z\uac00-\ud7a3\u3040-\u30ff\u4e00-\u9fff]+/u)
|
||||
const tokens = tokenizeTerms(text)
|
||||
.filter((token) => token.length >= MIN_TERM_CHARS)
|
||||
.filter((token) => !STOPWORDS.has(token))
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue