Files
WechatExplorer/tests/unit/knowledge-store.test.ts
Wxw-Gu ab3b3731b7 feat: 重构问问微信并完善知识库增量检索
统一问问微信与 Agent Hub 的查询链路
完善查询覆盖度、新鲜度和部分结果表达,避免索引滞后产生错误结论
基于会话实现真正的增量追新与历史补齐
支持后台同步、取消恢复、重启续传以及同步期间继续查询
优化知识库跨会话检索、同步状态、进度展示和侧栏布局
2026-09-11 17:44:48 +08:00

641 lines
23 KiB
TypeScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { mkdtempSync, existsSync } from 'fs'
import { rm } from 'fs/promises'
import { tmpdir } from 'os'
import { join } from 'path'
import { DatabaseSync } from 'node:sqlite'
import { afterEach, describe, expect, it } from 'vitest'
import { DEFAULT_KNOWLEDGE_CHUNKER, type KnowledgeFtsConfig } from '../../src/shared/knowledge'
import { chunkConversation } from '../../src/main/knowledge/chunker'
import {
estimateKnowledgeCapacityPreflight,
getKnowledgeDatabasePath,
KnowledgeStore,
removeKnowledgeDatabase
} from '../../src/main/knowledge/knowledge-store'
import { normalizeKnowledgeMessage } from '../../src/main/knowledge/normalizer'
import {
createSyntheticConversation,
FIXTURE_ACCOUNT_A,
FIXTURE_ACCOUNT_B
} from '../fixtures/knowledge-rag'
const roots: string[] = []
const fts: KnowledgeFtsConfig = {
profileId: 'test-trigram-external-full',
tokenizer: 'trigram',
contentMode: 'external',
detail: 'full',
columnsize: 1
}
function makeRoot(): string {
const root = mkdtempSync(join(tmpdir(), 'wxe-knowledge-'))
roots.push(root)
return root
}
afterEach(async () => {
await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true })))
})
describe('knowledge normalizer and chunker', () => {
it('indexes text, attachment metadata and existing voice transcripts without paths or binary data', () => {
const normalized = normalizeKnowledgeMessage({
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'conversation-a',
messageId: 'message-a',
createTime: 1,
kind: 'voice',
text: ' 原始说明 ',
attachment: { name: 'plan.txt', kind: 'file' },
voiceTranscript: ' 已完成语音转写 '
})
expect(normalized.searchableText).toContain('原始说明')
expect(normalized.searchableText).toContain('附件:plan.txt')
expect(normalized.searchableText).toContain('语音转写:已完成语音转写')
})
it('cuts on time gaps and preserves message evidence ids', () => {
const source = createSyntheticConversation(
FIXTURE_ACCOUNT_A,
'conversation-a',
0,
4,
'short'
).messages
source[3].createTime += 20 * 60 * 1000
const chunks = chunkConversation(source.map(normalizeKnowledgeMessage), {
...DEFAULT_KNOWLEDGE_CHUNKER,
maxMessages: 12
})
expect(chunks).toHaveLength(2)
expect(chunks.flatMap((chunk) => chunk.messageIds)).toEqual(
source.map((item) => item.messageId)
)
})
})
describe('knowledge sqlite', () => {
it('is idempotent, supports FTS evidence lookup, and does not mix accounts', async () => {
const root = makeRoot()
const source = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'conversation-a', 0, 25, 'mixed')
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const first = await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER })
const second = await store.index({
conversations: [source],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
expect(first.updatedChunks).toBeGreaterThan(0)
expect(second.updatedChunks).toBe(0)
expect(second.unchangedConversations).toBe(1)
const evidence = store.search({ accountId: FIXTURE_ACCOUNT_A, text: '本地知识库', limit: 10 })
expect(evidence).not.toHaveLength(0)
expect(evidence[0]).toMatchObject({
messageId: expect.stringMatching(/^synthetic-mixed-/),
conversationId: 'conversation-a',
sender: expect.any(String),
timestamp: expect.any(Number)
})
expect(
evidence.every((item) => item.messageIds.every((id) => id.startsWith('synthetic-mixed-')))
).toBe(true)
expect(() =>
store.search({ accountId: FIXTURE_ACCOUNT_B, text: '本地知识库', limit: 10 })
).toThrow(/account/)
store.close()
})
it('recovers safely after cancellation and only removes the derived database', async () => {
const root = makeRoot()
const source = createSyntheticConversation(
FIXTURE_ACCOUNT_A,
'conversation-a',
0,
2_000,
'mixed'
)
const controller = new AbortController()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const cancelled = await store.index(
{ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER },
controller.signal,
(progress) => {
if (progress.processedMessages >= 501) controller.abort()
}
)
expect(cancelled.cancelled).toBe(true)
const resumed = await store.index({
conversations: [source],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
expect(resumed.cancelled).toBe(false)
const databasePath = getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A)
store.close()
expect(existsSync(databasePath)).toBe(true)
removeKnowledgeDatabase(root, FIXTURE_ACCOUNT_A)
expect(existsSync(databasePath)).toBe(false)
})
/**
* 取消之后 `run_state` 不允许留下一个假的 `indexing`,而且已提交的分片必须保留
* (取消 ≠ 回滚),下一次索引从断点继续。
*/
it('records a cancelled pass as cancelled and keeps the committed chunks queryable', async () => {
const root = makeRoot()
// 两个会话:取消发生在第二个会话内部,于是第一个会话必须已经提交。
// (生产里索引是 per-conversation 事务,取消 ≠ 回滚。)
const committed = createSyntheticConversation(
FIXTURE_ACCOUNT_A,
'conversation-a',
0,
300,
'mixed'
)
const interrupted = createSyntheticConversation(
FIXTURE_ACCOUNT_A,
'conversation-b',
10_000,
2_000,
'mixed'
)
const controller = new AbortController()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const cancelled = await store.index(
{ conversations: [committed, interrupted], chunker: DEFAULT_KNOWLEDGE_CHUNKER },
controller.signal,
(progress) => {
if (progress.conversationId === 'conversation-b' && progress.processedMessages >= 900) {
controller.abort()
}
}
)
expect(cancelled.cancelled).toBe(true)
// 取消是真实的终态,不能被当成"还在跑"。
const status = store.getRuntimeStatus()
expect(status.state).toBe('cancelled')
expect(status.state).not.toBe('indexing')
// 已经提交的会话仍然可查 —— 取消不等于回滚。
const evidence = store.search({ accountId: FIXTURE_ACCOUNT_A, text: '本地知识库', limit: 5 })
expect(evidence.length).toBeGreaterThan(0)
expect(new Set(evidence.map((item) => item.conversationId))).toEqual(
new Set(['conversation-a'])
)
store.close()
})
it('does not report an interrupted pass as an unusable index', async () => {
const root = makeRoot()
const source = createSyntheticConversation(
FIXTURE_ACCOUNT_A,
'conversation-a',
0,
200,
'mixed'
)
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER })
expect(store.getRuntimeStatus().state).toBe('ready')
store.close()
// 模拟进程被杀:派生库里分片齐全,但 run_state 残留 'indexing'。
const databasePath = getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A)
const raw = new DatabaseSync(databasePath)
raw
.prepare(`UPDATE knowledge_meta SET value = 'indexing' WHERE key = 'run_state'`)
.run()
raw.close()
const reopened = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
// 「可查询」不能因为一次中断残留就变成「不可用」。
expect(reopened.getRuntimeStatus().state).toBe('ready')
reopened.close()
})
it('uses a bounded exact fallback for two-character Chinese queries with the trigram profile', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
await store.index({
conversations: [
{
conversationId: 'short-query',
completeSnapshot: true,
messages: [
{
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'short-query',
messageId: 'short-query-message',
createTime: Date.UTC(2026, 7, 5),
senderId: 'fixture-member',
senderName: '脱敏成员',
kind: 'text',
text: '收到,明早十点。'
}
]
}
],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
expect(
store.search({ accountId: FIXTURE_ACCOUNT_A, text: '十点', terms: ['十点'], limit: 10 })
).toEqual([
expect.objectContaining({ messageId: 'short-query-message', conversationId: 'short-query' })
])
store.close()
})
it('keeps equal message ids from different conversations as separate Evidence', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
await store.index({
conversations: ['conversation-a', 'conversation-b'].map((conversationId) => ({
conversationId,
completeSnapshot: true,
messages: [
{
accountId: FIXTURE_ACCOUNT_A,
conversationId,
messageId: 'shared-message-id',
createTime: Date.UTC(2026, 7, 5),
senderId: `${conversationId}-sender`,
senderName: conversationId,
kind: 'text',
text: '今天去健身。'
}
]
})),
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
const result = store.searchWithStatus({
accountId: FIXTURE_ACCOUNT_A,
text: '去健身',
terms: ['去健身'],
limit: 10
})
const evidence = result.evidence
expect(evidence).toHaveLength(2)
expect(evidence.map((item) => `${item.conversationId}:${item.messageId}`).sort()).toEqual([
'conversation-a:shared-message-id',
'conversation-b:shared-message-id'
])
expect(result.timings).toMatchObject({
workerIpcMs: 0,
ftsMs: expect.any(Number),
messageLoadMs: expect.any(Number),
chunkExpandMs: expect.any(Number),
rankingMs: expect.any(Number),
totalMs: expect.any(Number)
})
expect(result.timings.totalMs).toBeGreaterThanOrEqual(result.timings.ftsMs)
store.close()
})
it('marks voice Evidence and reports scoped transcript coverage without indexing error text', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
await store.index({
conversations: [
{
conversationId: 'voice-coverage',
completeSnapshot: true,
messages: [
{
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'voice-coverage',
messageId: 'voice-ready',
createTime: Date.UTC(2026, 7, 5, 9),
senderName: '成员甲',
kind: 'voice',
text: '[语音消息]',
voiceTranscript: '语音里确认今天去健身。',
voiceTranscriptState: 'transcribed'
},
{
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'voice-coverage',
messageId: 'voice-failed',
createTime: Date.UTC(2026, 7, 5, 10),
senderName: '成员乙',
kind: 'voice',
text: '[语音消息]',
voiceTranscriptState: 'failed'
}
]
}
],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
const result = store.searchWithStatus({
accountId: FIXTURE_ACCOUNT_A,
text: '去健身',
terms: ['去健身'],
conversationIds: ['voice-coverage'],
limit: 10
})
expect(result.evidence[0]).toMatchObject({
messageId: 'voice-ready',
sourceKind: 'voice'
})
expect(result.voiceCoverage).toEqual({
voiceMessageCount: 2,
transcribedVoiceCount: 1,
failedVoiceCount: 1,
voiceCoverageComplete: false
})
expect(result.evidence[0].text).not.toContain('失败')
store.close()
})
it('keeps truthful count snapshots off the repeated-search hot path', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const source = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'stats-snapshot', 0, 12, 'mixed')
await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER })
const first = store.searchWithStatus({
accountId: FIXTURE_ACCOUNT_A,
text: '本地知识库',
terms: ['本地知识库'],
limit: 10
})
const second = store.searchWithStatus({
accountId: FIXTURE_ACCOUNT_A,
text: '本地知识库',
terms: ['本地知识库'],
limit: 10
})
expect(first).toMatchObject({
indexedMessageCount: 12,
indexedChunkCount: expect.any(Number)
})
expect(first.timings.globalCountMs).toBeGreaterThanOrEqual(0)
expect(second.timings).toMatchObject({
globalCountMs: 0,
voiceCoverageMs: expect.any(Number),
workerExecutionMs: expect.any(Number)
})
expect(second.indexedMessageCount).toBe(first.indexedMessageCount)
expect(second.indexedChunkCount).toBe(first.indexedChunkCount)
store.close()
const reopened = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
expect(reopened.getSearchStatus()).toMatchObject({
indexedMessageCount: 12,
indexedChunkCount: first.indexedChunkCount
})
reopened.close()
})
it('refreshes statistics only on the final request of a complete source pass', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const first = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'stats-first', 0, 4, 'mixed')
const second = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'stats-second', 4, 3, 'mixed')
await store.index({
conversations: [first],
chunker: DEFAULT_KNOWLEDGE_CHUNKER,
sourceMessageCount: 4
})
const inspect = new DatabaseSync(getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A))
const readMeta = (key: string): string | undefined =>
(
inspect.prepare('SELECT value FROM knowledge_meta WHERE key = ?').get(key) as
| { value: string }
| undefined
)?.value
expect(readMeta('stats_state')).toBe('fresh')
expect(readMeta('stats_message_count')).toBe('4')
await store.index({ conversations: [second], chunker: DEFAULT_KNOWLEDGE_CHUNKER })
expect(readMeta('stats_state')).toBe('stale')
expect(readMeta('stats_message_count')).toBe('4')
await store.index({
conversations: [second],
chunker: DEFAULT_KNOWLEDGE_CHUNKER,
sourceMessageCount: 7
})
expect(readMeta('stats_state')).toBe('fresh')
expect(readMeta('stats_message_count')).toBe('7')
inspect.close()
store.close()
})
it('把 per-conversation 的源侧覆盖边界当作索引覆盖口径(meta / high_water_time 只作回退)', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const conversation = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'coverage', 0, 4, 'mixed')
const lastCreateTime = Math.max(...conversation.messages.map((message) => message.createTime))
await store.index({ conversations: [conversation], chunker: DEFAULT_KNOWLEDGE_CHUNKER })
// 既没有 per-conversation 源侧边界、也没有完整 pass 口径时,退化为
// "最新被索引的消息时间"(只会偏旧、不会冒充更新)。
expect(store.getRuntimeStatus().indexLatestAt).toBe(lastCreateTime)
await store.index({
conversations: [conversation],
chunker: DEFAULT_KNOWLEDGE_CHUNKER,
sourceMessageCount: conversation.messages.length,
// 源数据边界可以比"被索引建模的最新消息"更新(例如最新一条是不可建模的图片)。
sourceLatestAt: lastCreateTime + 60_000
})
const status = store.getRuntimeStatus()
expect(status.indexLatestAt).toBe(lastCreateTime + 60_000)
// 派生库自己看不到源数据,sourceLatestAt 由 KnowledgeSearchService 填。
expect(status.sourceLatestAt).toBeNull()
// per-conversation 的源侧边界一旦存在就是**权威口径**:它是每个成功处理的会话
// 立刻持久化的聚合值,增量 pass 也能推进它。而 `source_latest_at` meta 只在
// 「整遍零跳过」时才写 —— 增量世界里这让它永久冻结(冻结值落后真实 checkpoint,
// 导致 isKnowledgeFresh() 恒为 false)。
await store.index({
conversations: [{ ...conversation, sourceHighWaterTime: lastCreateTime + 180_000 }],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
expect(store.getRuntimeStatus().indexLatestAt).toBe(lastCreateTime + 180_000)
// checkpoint 单调:一个"看起来更旧"的源侧边界不得让覆盖率回退,
// 否则已经追到最新的会话会被重新打回"有新消息",每遍都白读。
await store.index({
conversations: [{ ...conversation, sourceHighWaterTime: lastCreateTime + 1_000 }],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
expect(store.getRuntimeStatus().indexLatestAt).toBe(lastCreateTime + 180_000)
store.close()
})
it('keeps conversation, sender and time filters when a participant question has no topic terms', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
await store.index({
conversations: [
{
conversationId: 'participant-query',
completeSnapshot: true,
messages: [
{
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'participant-query',
messageId: 'participant-a',
createTime: Date.UTC(2026, 7, 5, 9),
senderId: 'member-a',
senderName: '成员甲',
kind: 'text',
text: '第一条讨论。'
},
{
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'participant-query',
messageId: 'participant-b',
createTime: Date.UTC(2026, 7, 5, 10),
senderId: 'member-b',
senderName: '成员乙',
kind: 'text',
text: '第二条讨论。'
}
]
}
],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
expect(
store.search({
accountId: FIXTURE_ACCOUNT_A,
text: '成员甲最近聊了什么',
terms: [],
conversationIds: ['participant-query'],
senderIds: ['member-a'],
startTime: Date.UTC(2026, 7, 5, 8),
limit: 10
})
).toEqual([expect.objectContaining({ messageId: 'participant-a', sender: '成员甲' })])
store.close()
})
it('compresses a single-conversation recap into time chunks and deprioritizes system messages', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const base = Date.UTC(2026, 6, 1)
await store.index({
conversations: [
{
conversationId: 'recap-query',
completeSnapshot: true,
messages: Array.from({ length: 48 }, (_, index) => ({
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'recap-query',
messageId: `recap-${index}`,
createTime: base + Math.floor(index / 12) * 3 * 3600 * 1000 + (index % 12) * 60_000,
senderId: 'fixture-member',
senderName: '脱敏成员',
kind: index % 11 === 0 ? ('system' as const) : ('text' as const),
text: index % 11 === 0 ? '对方撤回了一条消息' : `第 ${index} 条健身计划和饮食安排讨论。`
}))
}
],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
const result = store.searchWithStatus({
accountId: FIXTURE_ACCOUNT_A,
text: '我和张三最近聊了什么',
terms: [],
conversationIds: ['recap-query'],
startTime: base,
limit: 100
})
expect(result.conversationRetrieval).toMatchObject({
totalMessages: 48,
chunkCount: 4,
complete: true
})
expect(result.evidence.length).toBeLessThan(48)
expect(new Set(result.evidence.map((item) => item.chunkId)).size).toBeGreaterThan(1)
expect(result.evidence.filter((item) => item.text.includes('撤回')).length).toBeLessThan(5)
store.close()
})
it('keeps late conversation slices when the recap candidate budget is reached', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const base = Date.UTC(2026, 6, 1)
await store.index({
conversations: [
{
conversationId: 'long-recap-query',
completeSnapshot: true,
messages: Array.from({ length: 90 }, (_, index) => ({
accountId: FIXTURE_ACCOUNT_A,
conversationId: 'long-recap-query',
messageId: `long-recap-${index}`,
createTime: base + Math.floor(index / 3) * 3 * 3600 * 1000 + (index % 3) * 60_000,
senderId: 'fixture-member',
senderName: '脱敏成员',
kind: 'text' as const,
text: `第 ${index} 条近期聊天内容。`
}))
}
],
chunker: DEFAULT_KNOWLEDGE_CHUNKER
})
const result = store.searchWithStatus({
accountId: FIXTURE_ACCOUNT_A,
text: '我和张三最近聊了什么',
terms: [],
conversationIds: ['long-recap-query'],
startTime: base,
limit: 100
})
expect(result.conversationRetrieval).toMatchObject({ chunkCount: 30, candidateMessages: 60 })
expect(Math.max(...result.evidence.map((item) => item.timestamp))).toBeGreaterThan(
base + 28 * 3 * 3600 * 1000
)
store.close()
})
it('provides a read-only capacity preflight before a database exists', async () => {
const root = makeRoot()
const source = createSyntheticConversation(FIXTURE_ACCOUNT_A, 'conversation-a', 0, 20, 'long')
const result = await estimateKnowledgeCapacityPreflight({
accountId: FIXTURE_ACCOUNT_A,
databaseRoot: root,
conversations: [source],
chunker: DEFAULT_KNOWLEDGE_CHUNKER,
availableDiskBytes: 1
})
expect(result.sourceMessageCount).toBe(20)
expect(result.voiceTranscriptCount).toBeGreaterThan(0)
expect(result.hasSufficientDiskSpace).toBe(false)
expect(existsSync(getKnowledgeDatabasePath(root, FIXTURE_ACCOUNT_A))).toBe(false)
})
it('indexes 100,000 desensitized messages without touching the main process database', async () => {
const root = makeRoot()
const store = new KnowledgeStore(root, FIXTURE_ACCOUNT_A, fts)
const started = performance.now()
for (let batch = 0; batch < 10; batch += 1) {
const source = createSyntheticConversation(
FIXTURE_ACCOUNT_A,
`performance-${batch}`,
batch * 10_000,
10_000,
'mixed'
)
await store.index({ conversations: [source], chunker: DEFAULT_KNOWLEDGE_CHUNKER })
}
const stats = store.getStorageStats()
expect(stats.databaseBytes).toBeGreaterThan(0)
expect(performance.now() - started).toBeLessThan(60_000)
store.close()
}, 70_000)
})