feat(desktop): sync knowledge, recordings and shared settings; play any recording
Knowledge documents travel as source-text chunks; each surface embeds them with its own model, the server index is requested through embed-chunks, and documents from the phone are stored without a file and indexed from their chunks. Chunk text is now kept when local embedding fails, so reindexing no longer needs the original file. Recordings upload to the mobile storage contract (audio bucket under the user's folder plus an audio_files row, 50 MiB cap, a Settings > Cloud toggle) and are removed with their record. The history card gains a play button that uses the local file or, for phone recordings, a signed URL. Language (ko/en), system/light/dark theme, auto-polish and the active user command follow the phone's user_settings with its revision rule; changes that arrive from the phone reach the open window.
This commit is contained in:
parent
cee4ab9317
commit
9a8f7e6aa6
34 changed files with 1405 additions and 64 deletions
|
|
@ -4,13 +4,15 @@
|
|||
|
||||
import { EventEmitter } from 'events'
|
||||
import path from 'path'
|
||||
import { eq } from 'drizzle-orm'
|
||||
import fs from 'fs'
|
||||
import { asc, eq } from 'drizzle-orm'
|
||||
import { getLogger } from './LoggerService'
|
||||
import { getPremiumLLMService } from './PremiumLLMService'
|
||||
import { getOllamaServerUrl } from './LocalLLMService'
|
||||
import { getDatabase } from '../db'
|
||||
import { ragDocuments, ragChunks } from '../db/schema'
|
||||
import { getMainWindow } from '../windows/WindowManager'
|
||||
import { getCloudSyncService } from './CloudSyncService'
|
||||
import { IPC_CHANNELS } from '@d3ro/core/ipc-channels'
|
||||
import { D3ROError, ErrorCode } from '@d3ro/core/errors'
|
||||
import type {
|
||||
|
|
@ -133,8 +135,12 @@ class RAGService extends EventEmitter {
|
|||
addedAt: Date.now(),
|
||||
}).run()
|
||||
|
||||
// 청크 원문을 먼저 저장한다 — 임베딩이 실패해도 원문은 남아 재색인·기기 간 동기화가 가능하다.
|
||||
this._storeChunks(docId, chunks)
|
||||
getCloudSyncService().pushOne('knowledge_documents', docId)
|
||||
|
||||
// 비동기 인덱싱 (임베딩 생성)
|
||||
this._indexDocument(docId, fileName, chunks).catch((err) => {
|
||||
this._embedStoredChunks(docId, fileName).catch((err) => {
|
||||
logger.error(`Indexing failed for ${fileName}:`, err)
|
||||
})
|
||||
|
||||
|
|
@ -154,10 +160,59 @@ class RAGService extends EventEmitter {
|
|||
* 문서 제거 (청크 포함)
|
||||
*/
|
||||
removeDocument(documentId: string): void {
|
||||
this.removeRemote(documentId)
|
||||
getCloudSyncService().pushDelete('knowledge_documents', documentId)
|
||||
logger.info(`RAG document removed: ${documentId}`)
|
||||
}
|
||||
|
||||
/** 동기화: 다른 기기에서 지운 문서를 지운다(outbox에 넣지 않는다). */
|
||||
removeRemote(documentId: string): boolean {
|
||||
const db = getDatabase()
|
||||
db.delete(ragChunks).where(eq(ragChunks.documentId, documentId)).run()
|
||||
db.delete(ragDocuments).where(eq(ragDocuments.id, documentId)).run()
|
||||
logger.info(`RAG document removed: ${documentId}`)
|
||||
return db.delete(ragDocuments).where(eq(ragDocuments.id, documentId)).run().changes > 0
|
||||
}
|
||||
|
||||
/**
|
||||
* 동기화: 다른 기기(모바일·웹)의 지식 문서를 원문 청크로 받아 저장하고, 이 기기의 임베딩 모델로 색인한다.
|
||||
* 임베딩 공간이 기기마다 달라 벡터는 옮기지 않는다. 원본 파일은 없으므로 filePath는 비워 둔다.
|
||||
*/
|
||||
applyRemoteDocument(doc: {
|
||||
id: string
|
||||
fileName: string
|
||||
fileType: RAGDocument['fileType']
|
||||
chunks: string[]
|
||||
addedAt: number
|
||||
}): boolean {
|
||||
const db = getDatabase()
|
||||
if (db.select({ id: ragDocuments.id }).from(ragDocuments).where(eq(ragDocuments.id, doc.id)).get()) return false
|
||||
const chunks = doc.chunks.filter((c) => c.trim().length > 0)
|
||||
if (chunks.length === 0) return false
|
||||
db.insert(ragDocuments).values({
|
||||
id: doc.id,
|
||||
fileName: doc.fileName,
|
||||
filePath: '',
|
||||
fileType: doc.fileType,
|
||||
chunkCount: chunks.length,
|
||||
indexed: false,
|
||||
indexedAt: null,
|
||||
addedAt: doc.addedAt,
|
||||
}).run()
|
||||
this._storeChunks(doc.id, chunks)
|
||||
this._embedStoredChunks(doc.id, doc.fileName).catch((err) => {
|
||||
logger.warn(`Synced document ${doc.fileName} is stored but not embedded yet:`, err)
|
||||
})
|
||||
return true
|
||||
}
|
||||
|
||||
/** 저장된 청크 원문(chunkIndex 순) — 동기화 업로드용 */
|
||||
getStoredChunks(documentId: string): string[] {
|
||||
return getDatabase()
|
||||
.select({ content: ragChunks.content })
|
||||
.from(ragChunks)
|
||||
.where(eq(ragChunks.documentId, documentId))
|
||||
.orderBy(asc(ragChunks.chunkIndex))
|
||||
.all()
|
||||
.map((r) => r.content)
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -171,8 +226,11 @@ class RAGService extends EventEmitter {
|
|||
}
|
||||
const doc = rows[0]
|
||||
|
||||
// 기존 청크 삭제
|
||||
db.delete(ragChunks).where(eq(ragChunks.documentId, documentId)).run()
|
||||
// 원본 파일이 없으면(다른 기기에서 동기화된 문서 등) 저장된 원문 청크로 다시 임베딩한다.
|
||||
if (!doc.filePath || !fs.existsSync(doc.filePath)) {
|
||||
await this._embedStoredChunks(documentId, doc.fileName, { force: true })
|
||||
return
|
||||
}
|
||||
|
||||
// 텍스트 재추출 + 재인덱싱
|
||||
const content = await this._extractText(doc.filePath, doc.fileType as RAGDocument['fileType'])
|
||||
|
|
@ -183,7 +241,9 @@ class RAGService extends EventEmitter {
|
|||
.where(eq(ragDocuments.id, documentId))
|
||||
.run()
|
||||
|
||||
await this._indexDocument(documentId, doc.fileName, chunks)
|
||||
this._storeChunks(documentId, chunks)
|
||||
getCloudSyncService().pushOne('knowledge_documents', documentId)
|
||||
await this._embedStoredChunks(documentId, doc.fileName)
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -194,7 +254,8 @@ class RAGService extends EventEmitter {
|
|||
|
||||
try {
|
||||
const db = getDatabase()
|
||||
const allChunks = db.select().from(ragChunks).all()
|
||||
// 원문만 있고 아직 임베딩되지 않은 청크는 검색 대상이 아니다.
|
||||
const allChunks = db.select().from(ragChunks).all().filter((c) => c.embedding.length > 0)
|
||||
if (allChunks.length === 0) {
|
||||
throw new D3ROError(ErrorCode.RAGQueryFailed, 'No indexed chunks to query')
|
||||
}
|
||||
|
|
@ -245,9 +306,37 @@ ${context}`
|
|||
|
||||
// ── 내부 메서드 ──
|
||||
|
||||
private async _indexDocument(docId: string, fileName: string, chunks: string[]): Promise<void> {
|
||||
/** 청크 원문을 (다시) 저장한다. 임베딩은 비워 두고 _embedStoredChunks 가 채운다. */
|
||||
private _storeChunks(docId: string, chunks: string[]): void {
|
||||
const db = getDatabase()
|
||||
db.transaction((tx) => {
|
||||
tx.delete(ragChunks).where(eq(ragChunks.documentId, docId)).run()
|
||||
chunks.forEach((content, index) => {
|
||||
tx.insert(ragChunks).values({
|
||||
id: crypto.randomUUID(),
|
||||
documentId: docId,
|
||||
content,
|
||||
embedding: '',
|
||||
chunkIndex: index,
|
||||
}).run()
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/** 저장된 청크 중 임베딩이 없는 것(force면 전부)을 이 기기의 임베딩 모델로 채운다. */
|
||||
private async _embedStoredChunks(
|
||||
docId: string,
|
||||
fileName: string,
|
||||
options: { force?: boolean } = {}
|
||||
): Promise<void> {
|
||||
this._state = 'indexing'
|
||||
const db = getDatabase()
|
||||
const chunks = db
|
||||
.select()
|
||||
.from(ragChunks)
|
||||
.where(eq(ragChunks.documentId, docId))
|
||||
.orderBy(asc(ragChunks.chunkIndex))
|
||||
.all()
|
||||
|
||||
logger.info(`RAG indexing started: ${fileName} (${chunks.length} chunks)`)
|
||||
|
||||
|
|
@ -262,21 +351,21 @@ ${context}`
|
|||
|
||||
let successCount = 0
|
||||
for (let i = 0; i < chunks.length; i++) {
|
||||
const chunk = chunks[i]
|
||||
if (chunk.embedding.length > 0 && !options.force) {
|
||||
successCount++
|
||||
continue
|
||||
}
|
||||
try {
|
||||
const embedding = await this._embed(chunks[i])
|
||||
|
||||
db.insert(ragChunks).values({
|
||||
id: crypto.randomUUID(),
|
||||
documentId: docId,
|
||||
content: chunks[i],
|
||||
embedding: JSON.stringify(embedding),
|
||||
chunkIndex: i,
|
||||
}).run()
|
||||
|
||||
const embedding = await this._embed(chunk.content)
|
||||
db.update(ragChunks)
|
||||
.set({ embedding: JSON.stringify(embedding) })
|
||||
.where(eq(ragChunks.id, chunk.id))
|
||||
.run()
|
||||
successCount++
|
||||
} catch (err) {
|
||||
logger.warn(`RAG embedding failed for chunk ${i}/${chunks.length} of ${fileName}:`, err)
|
||||
// 개별 청크 실패는 건너뛰고 계속 진행
|
||||
// 개별 청크 실패는 건너뛰고 계속 진행 — 원문은 남아 있어 재색인할 수 있다
|
||||
}
|
||||
|
||||
const progress: RAGIndexProgress = {
|
||||
|
|
@ -307,9 +396,9 @@ ${context}`
|
|||
)
|
||||
}
|
||||
|
||||
// 인덱싱 완료 표시
|
||||
// 인덱싱 완료 표시 (chunkCount는 원문 청크 수 — 서버·모바일과 같은 기준)
|
||||
db.update(ragDocuments)
|
||||
.set({ indexed: true, indexedAt: Date.now(), chunkCount: successCount })
|
||||
.set({ indexed: true, indexedAt: Date.now(), chunkCount: chunks.length })
|
||||
.where(eq(ragDocuments.id, docId))
|
||||
.run()
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue