import { mkdtempSync, existsSync } from 'fs' import { rm } from 'fs/promises' import { tmpdir } from 'os' import { join } from 'path' import { DatabaseSync } from 'node:sqlite' import { afterEach, describe, expect, it } from 'vitest' import { DEFAULT_KNOWLEDGE_CHUNKER, type KnowledgeFtsConfig } from '../../src/shared/knowledge' import { chunkConversation } from '../../src/main/knowledge/chunker' import { estimateKnowledgeCapacityPreflight, getKnowledgeDatabasePath, KnowledgeStore, removeKnowledgeDatabase } from '../../src/main/knowledge/knowledge-store' import { normalizeKnowledgeMessage } from '../../src/main/knowledge/normalizer' import { createSyntheticConversation, FIXTURE_ACCOUNT_A, FIXTURE_ACCOUNT_B } from '../fixtures/knowledge-rag' const roots: string[] = [] const fts: KnowledgeFtsConfig = { profileId: 'test-trigram-external-full', tokenizer: 'trigram', contentMode: 'external', detail: 'full', columnsize: 1 } function makeRoot(): string { const root = mkdtempSync(join(tmpdir(), 'wxe-knowledge-')) roots.push(root) return root } afterEach(async () => { await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) }) describe('knowledge normalizer and chunker', () => { it('indexes text, attachment metadata and existing voice transcripts without paths or binary data', () => { const normalized = normalizeKnowledgeMessage({ accountId: FIXTURE_ACCOUNT_A, conversationId: 'conversation-a', messageId: 'message-a', createTime: 1, kind: 'voice', text: ' 原始说明 ', attachment: { name: 'plan.txt', kind: 'file' }, voiceTranscript: ' 已完成语音转写 ' }) expect(normalized.searchableText).toContain('原始说明') expect(normalized.searchableText).toContain('附件:plan.txt') expect(normalized.searchableText).toContain('语音转写:已完成语音转写') }) it('cuts on time gaps and preserves message evidence ids', () => { const source = createSyntheticConversation( FIXTURE_ACCOUNT_A, 'conversation-a', 0, 4, 'short' ).messages source[3].createTime += 20 * 60 * 1000 const chunks = chunkConversation(source.map(normalizeKnowledgeMessage), { ...DEFAULT_KNOWLEDGE_CHUNKER, maxMessages: 12 }) expect(chunks).toHaveLength(2) expect(chunks.flatMap((chunk) => chunk.messageIds)).toEqual( source.map((item) => item.messageId) ) }) }) describe('knowledge sqlite', () => { it('is idempotent, supports FTS evidence lookup, and does not mix accounts', async () => { const root = makeRoot() const source = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'conversation-a', 0, 25, 'mixed') const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const first = await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) const second = await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect(first.updatedChunks).toBeGreaterThan(0) expect(second.updatedChunks).toBe(0) expect(second.unchangedConversations).toBe(1) const evidence = store.search({ accountId: FIXTURE_ACCOUNT_A, text: '本地知识库', limit: 10 }) expect(evidence).not.toHaveLength(0) expect(evidence[0]).toMatchObject({ messageId: expect.stringMatching(/^synthetic-mixed-/), conversationId: 'conversation-a', sender: expect.any(String), timestamp: expect.any(Number) }) expect( evidence.every((item) => item.messageIds.every((id) => id.startsWith('synthetic-mixed-'))) ).toBe(true) expect(() => store.search({ accountId: FIXTURE_ACCOUNT_B, text: '本地知识库', limit: 10 }) ).toThrow(/account/) store.close() }) it('recovers safely after cancellation and only removes the derived database', async () => { const root = makeRoot() const source = createSyntheticConversation( FIXTURE_ACCOUNT_A, 'conversation-a', 0, 2_000, 'mixed' ) const controller = new AbortController() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const cancelled = await store.index( { conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }, controller.signal, (progress) => { if (progress.processedMessages >= 501) controller.abort() } ) expect(cancelled.cancelled).toBe(true) const resumed = await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect(resumed.cancelled).toBe(false) const databasePath = getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A) store.close() expect(existsSync(databasePath)).toBe(true) removeKnowledgeDatabase(root, FIXTURE_ACCOUNT_A) expect(existsSync(databasePath)).toBe(false) }) /** * 取消之后 `run_state` 不允许留下一个假的 `indexing`,而且已提交的分片必须保留 * (取消 ≠ 回滚),下一次索引从断点继续。 */ it('records a cancelled pass as cancelled and keeps the committed chunks queryable', async () => { const root = makeRoot() // 两个会话:取消发生在第二个会话内部,于是第一个会话必须已经提交。 // (生产里索引是 per-conversation 事务,取消 ≠ 回滚。) const committed = createSyntheticConversation( FIXTURE_ACCOUNT_A, 'conversation-a', 0, 300, 'mixed' ) const interrupted = createSyntheticConversation( FIXTURE_ACCOUNT_A, 'conversation-b', 10_000, 2_000, 'mixed' ) const controller = new AbortController() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const cancelled = await store.index( { conversations: [committed, interrupted], chunker: DEFAULT_KNOWLEDGE_CHUNKER }, controller.signal, (progress) => { if (progress.conversationId === 'conversation-b' && progress.processedMessages >= 900) { controller.abort() } } ) expect(cancelled.cancelled).toBe(true) // 取消是真实的终态,不能被当成"还在跑"。 const status = store.getRuntimeStatus() expect(status.state).toBe('cancelled') expect(status.state).not.toBe('indexing') // 已经提交的会话仍然可查 —— 取消不等于回滚。 const evidence = store.search({ accountId: FIXTURE_ACCOUNT_A, text: '本地知识库', limit: 5 }) expect(evidence.length).toBeGreaterThan(0) expect(new Set(evidence.map((item) => item.conversationId))).toEqual( new Set(['conversation-a']) ) store.close() }) it('does not report an interrupted pass as an unusable index', async () => { const root = makeRoot() const source = createSyntheticConversation( FIXTURE_ACCOUNT_A, 'conversation-a', 0, 200, 'mixed' ) const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect(store.getRuntimeStatus().state).toBe('ready') store.close() // 模拟进程被杀:派生库里分片齐全,但 run_state 残留 'indexing'。 const databasePath = getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A) const raw = new DatabaseSync(databasePath) raw .prepare(`UPDATE knowledge_meta SET value = 'indexing' WHERE key = 'run_state'`) .run() raw.close() const reopened = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) // 「可查询」不能因为一次中断残留就变成「不可用」。 expect(reopened.getRuntimeStatus().state).toBe('ready') reopened.close() }) it('uses a bounded exact fallback for two-character Chinese queries with the trigram profile', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) await store.index({ conversations: [ { conversationId: 'short-query', completeSnapshot: true, messages: [ { accountId: FIXTURE_ACCOUNT_A, conversationId: 'short-query', messageId: 'short-query-message', createTime: Date.UTC(2026, 7, 5), senderId: 'fixture-member', senderName: '脱敏成员', kind: 'text', text: '收到,明早十点。' } ] } ], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect( store.search({ accountId: FIXTURE_ACCOUNT_A, text: '十点', terms: ['十点'], limit: 10 }) ).toEqual([ expect.objectContaining({ messageId: 'short-query-message', conversationId: 'short-query' }) ]) store.close() }) it('keeps equal message ids from different conversations as separate Evidence', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) await store.index({ conversations: ['conversation-a', 'conversation-b'].map((conversationId) => ({ conversationId, completeSnapshot: true, messages: [ { accountId: FIXTURE_ACCOUNT_A, conversationId, messageId: 'shared-message-id', createTime: Date.UTC(2026, 7, 5), senderId: `${conversationId}-sender`, senderName: conversationId, kind: 'text', text: '今天去健身。' } ] })), chunker: DEFAULT_KNOWLEDGE_CHUNKER }) const result = store.searchWithStatus({ accountId: FIXTURE_ACCOUNT_A, text: '去健身', terms: ['去健身'], limit: 10 }) const evidence = result.evidence expect(evidence).toHaveLength(2) expect(evidence.map((item) => `${item.conversationId}:${item.messageId}`).sort()).toEqual([ 'conversation-a:shared-message-id', 'conversation-b:shared-message-id' ]) expect(result.timings).toMatchObject({ workerIpcMs: 0, ftsMs: expect.any(Number), messageLoadMs: expect.any(Number), chunkExpandMs: expect.any(Number), rankingMs: expect.any(Number), totalMs: expect.any(Number) }) expect(result.timings.totalMs).toBeGreaterThanOrEqual(result.timings.ftsMs) store.close() }) it('marks voice Evidence and reports scoped transcript coverage without indexing error text', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) await store.index({ conversations: [ { conversationId: 'voice-coverage', completeSnapshot: true, messages: [ { accountId: FIXTURE_ACCOUNT_A, conversationId: 'voice-coverage', messageId: 'voice-ready', createTime: Date.UTC(2026, 7, 5, 9), senderName: '成员甲', kind: 'voice', text: '[语音消息]', voiceTranscript: '语音里确认今天去健身。', voiceTranscriptState: 'transcribed' }, { accountId: FIXTURE_ACCOUNT_A, conversationId: 'voice-coverage', messageId: 'voice-failed', createTime: Date.UTC(2026, 7, 5, 10), senderName: '成员乙', kind: 'voice', text: '[语音消息]', voiceTranscriptState: 'failed' } ] } ], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) const result = store.searchWithStatus({ accountId: FIXTURE_ACCOUNT_A, text: '去健身', terms: ['去健身'], conversationIds: ['voice-coverage'], limit: 10 }) expect(result.evidence[0]).toMatchObject({ messageId: 'voice-ready', sourceKind: 'voice' }) expect(result.voiceCoverage).toEqual({ voiceMessageCount: 2, transcribedVoiceCount: 1, failedVoiceCount: 1, voiceCoverageComplete: false }) expect(result.evidence[0].text).not.toContain('失败') store.close() }) it('keeps truthful count snapshots off the repeated-search hot path', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const source = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'stats-snapshot', 0, 12, 'mixed') await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) const first = store.searchWithStatus({ accountId: FIXTURE_ACCOUNT_A, text: '本地知识库', terms: ['本地知识库'], limit: 10 }) const second = store.searchWithStatus({ accountId: FIXTURE_ACCOUNT_A, text: '本地知识库', terms: ['本地知识库'], limit: 10 }) expect(first).toMatchObject({ indexedMessageCount: 12, indexedChunkCount: expect.any(Number) }) expect(first.timings.globalCountMs).toBeGreaterThanOrEqual(0) expect(second.timings).toMatchObject({ globalCountMs: 0, voiceCoverageMs: expect.any(Number), workerExecutionMs: expect.any(Number) }) expect(second.indexedMessageCount).toBe(first.indexedMessageCount) expect(second.indexedChunkCount).toBe(first.indexedChunkCount) store.close() const reopened = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) expect(reopened.getSearchStatus()).toMatchObject({ indexedMessageCount: 12, indexedChunkCount: first.indexedChunkCount }) reopened.close() }) it('refreshes statistics only on the final request of a complete source pass', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const first = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'stats-first', 0, 4, 'mixed') const second = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'stats-second', 4, 3, 'mixed') await store.index({ conversations: [first], chunker: DEFAULT_KNOWLEDGE_CHUNKER, sourceMessageCount: 4 }) const inspect = new DatabaseSync(getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A)) const readMeta = (key: string): string | undefined => ( inspect.prepare('SELECT value FROM knowledge_meta WHERE key = ?').get(key) as | { value: string } | undefined )?.value expect(readMeta('stats_state')).toBe('fresh') expect(readMeta('stats_message_count')).toBe('4') await store.index({ conversations: [second], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect(readMeta('stats_state')).toBe('stale') expect(readMeta('stats_message_count')).toBe('4') await store.index({ conversations: [second], chunker: DEFAULT_KNOWLEDGE_CHUNKER, sourceMessageCount: 7 }) expect(readMeta('stats_state')).toBe('fresh') expect(readMeta('stats_message_count')).toBe('7') inspect.close() store.close() }) it('把 per-conversation 的源侧覆盖边界当作索引覆盖口径(meta / high_water_time 只作回退)', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const conversation = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'coverage', 0, 4, 'mixed') const lastCreateTime = Math.max(...conversation.messages.map((message) => message.createTime)) await store.index({ conversations: [conversation], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) // 既没有 per-conversation 源侧边界、也没有完整 pass 口径时,退化为 // "最新被索引的消息时间"(只会偏旧、不会冒充更新)。 expect(store.getRuntimeStatus().indexLatestAt).toBe(lastCreateTime) await store.index({ conversations: [conversation], chunker: DEFAULT_KNOWLEDGE_CHUNKER, sourceMessageCount: conversation.messages.length, // 源数据边界可以比"被索引建模的最新消息"更新(例如最新一条是不可建模的图片)。 sourceLatestAt: lastCreateTime + 60_000 }) const status = store.getRuntimeStatus() expect(status.indexLatestAt).toBe(lastCreateTime + 60_000) // 派生库自己看不到源数据,sourceLatestAt 由 KnowledgeSearchService 填。 expect(status.sourceLatestAt).toBeNull() // per-conversation 的源侧边界一旦存在就是**权威口径**:它是每个成功处理的会话 // 立刻持久化的聚合值,增量 pass 也能推进它。而 `source_latest_at` meta 只在 // 「整遍零跳过」时才写 —— 增量世界里这让它永久冻结(冻结值落后真实 checkpoint, // 导致 isKnowledgeFresh() 恒为 false)。 await store.index({ conversations: [{ ...conversation, sourceHighWaterTime: lastCreateTime + 180_000 }], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect(store.getRuntimeStatus().indexLatestAt).toBe(lastCreateTime + 180_000) // checkpoint 单调:一个"看起来更旧"的源侧边界不得让覆盖率回退, // 否则已经追到最新的会话会被重新打回"有新消息",每遍都白读。 await store.index({ conversations: [{ ...conversation, sourceHighWaterTime: lastCreateTime + 1_000 }], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect(store.getRuntimeStatus().indexLatestAt).toBe(lastCreateTime + 180_000) store.close() }) it('keeps conversation, sender and time filters when a participant question has no topic terms', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) await store.index({ conversations: [ { conversationId: 'participant-query', completeSnapshot: true, messages: [ { accountId: FIXTURE_ACCOUNT_A, conversationId: 'participant-query', messageId: 'participant-a', createTime: Date.UTC(2026, 7, 5, 9), senderId: 'member-a', senderName: '成员甲', kind: 'text', text: '第一条讨论。' }, { accountId: FIXTURE_ACCOUNT_A, conversationId: 'participant-query', messageId: 'participant-b', createTime: Date.UTC(2026, 7, 5, 10), senderId: 'member-b', senderName: '成员乙', kind: 'text', text: '第二条讨论。' } ] } ], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) expect( store.search({ accountId: FIXTURE_ACCOUNT_A, text: '成员甲最近聊了什么', terms: [], conversationIds: ['participant-query'], senderIds: ['member-a'], startTime: Date.UTC(2026, 7, 5, 8), limit: 10 }) ).toEqual([expect.objectContaining({ messageId: 'participant-a', sender: '成员甲' })]) store.close() }) it('compresses a single-conversation recap into time chunks and deprioritizes system messages', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const base = Date.UTC(2026, 6, 1) await store.index({ conversations: [ { conversationId: 'recap-query', completeSnapshot: true, messages: Array.from({ length: 48 }, (_, index) => ({ accountId: FIXTURE_ACCOUNT_A, conversationId: 'recap-query', messageId: `recap-${index}`, createTime: base + Math.floor(index / 12) * 3 * 3600 * 1000 + (index % 12) * 60_000, senderId: 'fixture-member', senderName: '脱敏成员', kind: index % 11 === 0 ? ('system' as const) : ('text' as const), text: index % 11 === 0 ? '对方撤回了一条消息' : `第 ${index} 条健身计划和饮食安排讨论。` })) } ], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) const result = store.searchWithStatus({ accountId: FIXTURE_ACCOUNT_A, text: '我和张三最近聊了什么', terms: [], conversationIds: ['recap-query'], startTime: base, limit: 100 }) expect(result.conversationRetrieval).toMatchObject({ totalMessages: 48, chunkCount: 4, complete: true }) expect(result.evidence.length).toBeLessThan(48) expect(new Set(result.evidence.map((item) => item.chunkId)).size).toBeGreaterThan(1) expect(result.evidence.filter((item) => item.text.includes('撤回')).length).toBeLessThan(5) store.close() }) it('keeps late conversation slices when the recap candidate budget is reached', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const base = Date.UTC(2026, 6, 1) await store.index({ conversations: [ { conversationId: 'long-recap-query', completeSnapshot: true, messages: Array.from({ length: 90 }, (_, index) => ({ accountId: FIXTURE_ACCOUNT_A, conversationId: 'long-recap-query', messageId: `long-recap-${index}`, createTime: base + Math.floor(index / 3) * 3 * 3600 * 1000 + (index % 3) * 60_000, senderId: 'fixture-member', senderName: '脱敏成员', kind: 'text' as const, text: `第 ${index} 条近期聊天内容。` })) } ], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) const result = store.searchWithStatus({ accountId: FIXTURE_ACCOUNT_A, text: '我和张三最近聊了什么', terms: [], conversationIds: ['long-recap-query'], startTime: base, limit: 100 }) expect(result.conversationRetrieval).toMatchObject({ chunkCount: 30, candidateMessages: 60 }) expect(Math.max(...result.evidence.map((item) => item.timestamp))).toBeGreaterThan( base + 28 * 3 * 3600 * 1000 ) store.close() }) it('provides a read-only capacity preflight before a database exists', async () => { const root = makeRoot() const source = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'conversation-a', 0, 20, 'long') const result = await estimateKnowledgeCapacityPreflight({ accountId: FIXTURE_ACCOUNT_A, databaseRoot: root, conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER, availableDiskBytes: 1 }) expect(result.sourceMessageCount).toBe(20) expect(result.voiceTranscriptCount).toBeGreaterThan(0) expect(result.hasSufficientDiskSpace).toBe(false) expect(existsSync(getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A))).toBe(false) }) it('indexes 100,000 desensitized messages without touching the main process database', async () => { const root = makeRoot() const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts) const started = performance.now() for (let batch = 0; batch < 10; batch += 1) { const source = createSyntheticConversation( FIXTURE_ACCOUNT_A, `performance-${batch}`, batch * 10_000, 10_000, 'mixed' ) await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER }) } const stats = store.getStorageStats() expect(stats.databaseBytes).toBeGreaterThan(0) expect(performance.now() - started).toBeLessThan(60_000) store.close() }, 70_000) })