mirror of
https://wget.la/https://github.com/Wxw-Gu/WechatExplorer
synced 2026-10-05 04:20:34 +08:00
feat: 新增微信图片文字索引与问问微信图片检索能力
This commit is contained in:
@@ -0,0 +1,283 @@
|
||||
/**
|
||||
* 事故回归:**"4.5 万张全部失败,UI 却说已建立"** 这一整套语义。
|
||||
*
|
||||
* 真机现场(派生库实测):
|
||||
* total = 45,707 / 全部 binding = decrypt_failed 45,479 / artifacts = 0 行
|
||||
* 根因是解密服务在回填时不存在(只在 db:getImage 里懒加载),每张图都在
|
||||
* `processOne` 第一步就失败。这里把"不许再发生"的四件事钉死:
|
||||
* 1. 前置依赖缺失时必须**一条记录都不写**(preflight);
|
||||
* 2. 处理过但一条没成功 = **异常**,不是"已建立";
|
||||
* 3. 百分比不许四舍五入到 100(45,479 / 45,707);
|
||||
* 4. 重置失败记录**不能**动已经成功的记录。
|
||||
*/
|
||||
import { mkdtempSync } from 'node:fs'
|
||||
import { rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import type * as chat from '../../src/main/services/chat-service'
|
||||
import { ImageTextIndexService } from '../../src/main/services/image-text-index-service'
|
||||
import {
|
||||
ImageTextIndexStore,
|
||||
getImageTextIndexDatabasePath
|
||||
} from '../../src/main/services/image-text-index-store'
|
||||
import {
|
||||
describeImageTextCoverage,
|
||||
imageTextCoverageState,
|
||||
imageTextProcessedPercent,
|
||||
type ImageTextIndexCoverage
|
||||
} from '../../src/shared/image-text-index'
|
||||
|
||||
const ACCOUNT = 'wxid_incident_fixture'
|
||||
const CONVERSATION = 'md5-incident'
|
||||
const roots: string[] = []
|
||||
|
||||
function makeRoot(): string {
|
||||
const root = mkdtempSync(join(tmpdir(), 'tm-image-incident-'))
|
||||
roots.push(root)
|
||||
return root
|
||||
}
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true })))
|
||||
})
|
||||
|
||||
function imageMessage(localId: number): chat.FormattedMessage {
|
||||
return {
|
||||
localId: String(localId),
|
||||
createTime: 1_700_000_000 + localId,
|
||||
content: '[图片]',
|
||||
contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` }
|
||||
} as unknown as chat.FormattedMessage
|
||||
}
|
||||
|
||||
function coverageOf(overrides: Partial<ImageTextIndexCoverage>): ImageTextIndexCoverage {
|
||||
return {
|
||||
totalImageMessages: 0,
|
||||
processed: 0,
|
||||
indexed: 0,
|
||||
empty: 0,
|
||||
missing: 0,
|
||||
failed: 0,
|
||||
runtimeUnavailable: 0,
|
||||
pending: 0,
|
||||
established: false,
|
||||
complete: false,
|
||||
systemicFailure: false,
|
||||
countedAt: null,
|
||||
...overrides
|
||||
}
|
||||
}
|
||||
|
||||
describe('事故语义:全失败不能叫"已建立"', () => {
|
||||
it('indexed/empty/missing 全为 0 而 failed 不为 0 → 异常,且 complete 必为 false', () => {
|
||||
const coverage = coverageOf({
|
||||
totalImageMessages: 45_707,
|
||||
processed: 45_479,
|
||||
failed: 45_479,
|
||||
pending: 228,
|
||||
established: true,
|
||||
countedAt: 1_789_516_520_246,
|
||||
complete: false, // 服务侧已经算出 false;这里验证状态与文案
|
||||
systemicFailure: true
|
||||
})
|
||||
|
||||
expect(imageTextCoverageState(coverage)).toBe('failed')
|
||||
expect(describeImageTextCoverage(coverage)).toContain('当前异常')
|
||||
expect(describeImageTextCoverage(coverage)).not.toContain('已覆盖全部')
|
||||
})
|
||||
|
||||
it('45,479 / 45,707 不能显示成 100%', () => {
|
||||
// Math.round(45479 / 45707 * 100) === 100 —— 这正是"仅完成 100%"的来源。
|
||||
expect(Math.round((45_479 / 45_707) * 100)).toBe(100)
|
||||
// 正确口径:保留 1 位小数,未完成时封顶 99.9。
|
||||
expect(imageTextProcessedPercent(45_479, 45_707)).toBe(99.5)
|
||||
expect(imageTextProcessedPercent(45_707, 45_707)).toBe(100)
|
||||
expect(imageTextProcessedPercent(0, 0)).toBe(0)
|
||||
})
|
||||
|
||||
it('服务侧:一条都没成功时不给 complete,并把状态判成 failed', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
store.writeCountedTotal({ total: 100, countedAt: 1, complete: true })
|
||||
for (let index = 1; index <= 30; index += 1) {
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: `local:${index}`,
|
||||
createTime: index,
|
||||
imageIdentity: '',
|
||||
artifactKey: `unavailable|${index}`,
|
||||
state: 'decrypt_failed',
|
||||
updatedAt: index
|
||||
})
|
||||
}
|
||||
store.close()
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({ databaseRoot, resolveAccountId: () => ACCOUNT })
|
||||
const status = await service.getStatus()
|
||||
|
||||
expect(status.coverage.processed).toBe(30)
|
||||
expect(status.coverage.failed).toBe(30)
|
||||
expect(status.coverage.systemicFailure).toBe(true)
|
||||
expect(status.coverage.complete).toBe(false)
|
||||
expect(imageTextCoverageState(status.coverage)).toBe('failed')
|
||||
// 收干净句柄:Windows 上没关连接会让临时目录清理 EBUSY。
|
||||
service.resetAccount()
|
||||
})
|
||||
|
||||
it('运行时不可用不计入 processed,并且阻断 complete', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
store.writeCountedTotal({ total: 10, countedAt: 1, complete: true })
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: 'local:1',
|
||||
createTime: 1,
|
||||
imageIdentity: '',
|
||||
artifactKey: 'unavailable|1',
|
||||
state: 'decrypt_unavailable',
|
||||
updatedAt: 1
|
||||
})
|
||||
store.close()
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({ databaseRoot, resolveAccountId: () => ACCOUNT })
|
||||
const status = await service.getStatus()
|
||||
|
||||
expect(status.coverage.runtimeUnavailable).toBe(1)
|
||||
expect(status.coverage.processed).toBe(0)
|
||||
expect(status.coverage.complete).toBe(false)
|
||||
service.resetAccount()
|
||||
})
|
||||
|
||||
it('部分成功 + 部分图片缺失 → 仍然是正常的"部分完成"(不误判成异常)', () => {
|
||||
const coverage = coverageOf({
|
||||
totalImageMessages: 100,
|
||||
processed: 100,
|
||||
indexed: 60,
|
||||
empty: 20,
|
||||
missing: 20,
|
||||
established: true,
|
||||
countedAt: 1,
|
||||
complete: true,
|
||||
systemicFailure: false
|
||||
})
|
||||
expect(imageTextCoverageState(coverage)).toBe('complete')
|
||||
expect(describeImageTextCoverage(coverage)).toContain('已覆盖全部')
|
||||
})
|
||||
})
|
||||
|
||||
describe('事故防线:前置依赖缺失时一条记录都不写', () => {
|
||||
it('解密服务不可用 → pass 直接报错,不写任何 binding', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [
|
||||
{ md5: CONVERSATION, m_nsUsrName: 'incident', type: 'group' as const }
|
||||
],
|
||||
listMessages: async () => [imageMessage(1)],
|
||||
countConversationImages: async () => ({ count: 1, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ count: 1, maxLocalId: 1 }),
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN'
|
||||
}),
|
||||
// 关键:没有解密服务(本次事故的根因形态)
|
||||
decryptService: () => null
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
|
||||
const status = await service.getStatus()
|
||||
// 这一条就是整场事故的防线:宁可一次都不跑,也不要写 45,479 条假失败。
|
||||
expect(status.progress.state).toBe('error')
|
||||
expect(status.progress.lastError).toContain('解密服务')
|
||||
expect(status.coverage.processed).toBe(0)
|
||||
expect(status.coverage.failed).toBe(0)
|
||||
expect(status.coverage.runtimeUnavailable).toBe(0)
|
||||
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
expect(store.countByState()).toEqual({})
|
||||
store.close()
|
||||
service.resetAccount()
|
||||
})
|
||||
})
|
||||
|
||||
describe('事故收尾:重置失败记录不能动成功记录', () => {
|
||||
it('只删失败绑定与它们的 checkpoint,indexed 一条不动', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
store.writeCountedTotal({ total: 3, countedAt: 1, complete: true })
|
||||
store.putArtifact({
|
||||
accountId: ACCOUNT,
|
||||
artifactKey: 'good|1',
|
||||
imageIdentity: 'sha256:good',
|
||||
state: 'indexed',
|
||||
text: 'TRACE_KEEP_ME',
|
||||
charCount: 13,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN',
|
||||
createdAt: 1,
|
||||
updatedAt: 1
|
||||
})
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: 'local:1',
|
||||
createTime: 1,
|
||||
imageIdentity: 'sha256:good',
|
||||
artifactKey: 'good|1',
|
||||
state: 'indexed',
|
||||
updatedAt: 1
|
||||
})
|
||||
for (const id of [2, 3]) {
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: `local:${id}`,
|
||||
createTime: id,
|
||||
imageIdentity: '',
|
||||
artifactKey: `bad|${id}`,
|
||||
state: 'decrypt_failed',
|
||||
updatedAt: id
|
||||
})
|
||||
}
|
||||
store.writeScanState({
|
||||
conversationId: CONVERSATION,
|
||||
state: 'done',
|
||||
imageTotal: 3,
|
||||
imageProcessed: 3,
|
||||
maxLocalId: 3
|
||||
})
|
||||
store.close()
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({ databaseRoot, resolveAccountId: () => ACCOUNT })
|
||||
const result = await service.resetRetriableFailures()
|
||||
expect(result.reset).toBe(2)
|
||||
|
||||
// 成功记录与它的 artifact 必须原封不动。
|
||||
const after = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
expect(after.countByState()).toEqual({ indexed: 1 })
|
||||
expect(after.getArtifact('good|1')?.text).toBe('TRACE_KEEP_ME')
|
||||
// checkpoint 也要清掉,否则下一轮会以"该会话已完成"直接跳过(点了重试却没反应)。
|
||||
expect(after.readScanState().size).toBe(0)
|
||||
after.close()
|
||||
service.resetAccount()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,312 @@
|
||||
/**
|
||||
* 两条**成本**契约(都在真机上被踩过):
|
||||
*
|
||||
* 1. 「更新图片文字索引」必须是增量的。
|
||||
* 真机已经跑了 1 小时、留下 14,342 条 OCR 文本 + 31,126 条 empty 终态。
|
||||
* OCR 是昂贵产物,Knowledge / FTS / binding 才是可重建派生层 ——
|
||||
* 更新时只能处理 new / pending / retryable,**已有终态一条都不许重算**。
|
||||
*
|
||||
* 2. 修索引问题不许重跑 OCR(Derived Index Repair)。
|
||||
* 只重建 L3(Knowledge 派生条目 / FTS),数据来源是已有 L1/L2;
|
||||
* `ocrExecutions: 0` 是写进类型字面量的契约,不是"期望值"。
|
||||
*/
|
||||
import { mkdtempSync } from 'node:fs'
|
||||
import { rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import type * as chat from '../../src/main/services/chat-service'
|
||||
import { ImageTextIndexService } from '../../src/main/services/image-text-index-service'
|
||||
import {
|
||||
ImageTextIndexStore,
|
||||
getImageTextIndexDatabasePath
|
||||
} from '../../src/main/services/image-text-index-store'
|
||||
|
||||
const ACCOUNT = 'wxid_incremental_fixture'
|
||||
const CONVERSATION = 'md5-incremental'
|
||||
const roots: string[] = []
|
||||
|
||||
function makeRoot(): string {
|
||||
const root = mkdtempSync(join(tmpdir(), 'tm-image-incremental-'))
|
||||
roots.push(root)
|
||||
return root
|
||||
}
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true })))
|
||||
})
|
||||
|
||||
/** 合法的 PNG 头(`detectSystemOcrImageFormat` 只认前 4 字节),尾部塞一个唯一序号。 */
|
||||
function pngBytes(seed: number): Buffer {
|
||||
return Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, seed & 0xff, (seed >> 8) & 0xff])
|
||||
}
|
||||
|
||||
function imageMessage(localId: number): chat.FormattedMessage {
|
||||
return {
|
||||
id: String(localId),
|
||||
localId: String(localId),
|
||||
from: 'user',
|
||||
type: '图片',
|
||||
content: '',
|
||||
isSender: false,
|
||||
name: '对方',
|
||||
contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` },
|
||||
createTime: 1_700_000_000 + localId
|
||||
} as unknown as chat.FormattedMessage
|
||||
}
|
||||
|
||||
/** 假的派生库预热:写入 `count` 条已完成的 OCR 记录(成功终态)。 */
|
||||
function seedTerminal(store: ImageTextIndexStore, count: number, state: 'indexed' | 'empty'): void {
|
||||
for (let index = 1; index <= count; index += 1) {
|
||||
const artifactKey = `seeded|${index}`
|
||||
store.putArtifact({
|
||||
accountId: ACCOUNT,
|
||||
artifactKey,
|
||||
imageIdentity: `sha256:seeded-${index}`,
|
||||
state,
|
||||
text: state === 'indexed' ? `TRACE_SEEDED_${index}` : '',
|
||||
charCount: state === 'indexed' ? 16 : 0,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN',
|
||||
createdAt: index,
|
||||
updatedAt: index
|
||||
})
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: `local:${index}`,
|
||||
createTime: index,
|
||||
imageIdentity: `sha256:seeded-${index}`,
|
||||
artifactKey,
|
||||
state,
|
||||
updatedAt: index
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
describe('增量更新:已有终态直接复用,只对新增/未完成做 OCR', () => {
|
||||
it('已有 100 条终态 + 新增 10 张 + 2 条未完成 → OCR 只跑 12 次', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
seedTerminal(store, 100, 'indexed')
|
||||
// 2 条"未完成":binding 状态不是终态 → 允许重试。
|
||||
for (const id of [111, 112]) {
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: `local:${id}`,
|
||||
createTime: id,
|
||||
imageIdentity: '',
|
||||
artifactKey: `pending|${id}`,
|
||||
state: 'pending',
|
||||
updatedAt: id
|
||||
})
|
||||
}
|
||||
// 旧 checkpoint:当时该会话只有 100 张、水位 100。
|
||||
store.writeScanState({
|
||||
conversationId: CONVERSATION,
|
||||
state: 'done',
|
||||
imageTotal: 100,
|
||||
imageProcessed: 100,
|
||||
maxLocalId: 100
|
||||
})
|
||||
store.close()
|
||||
|
||||
// 现在源数据变成 112 张(1..100 已有终态,101..110 全新,111..112 之前没跑完)。
|
||||
const messages = Array.from({ length: 112 }, (_, index) => imageMessage(index + 1))
|
||||
const recognize = vi.fn(async () => ({
|
||||
success: true,
|
||||
text: 'TRACE_FRESH_TEXT',
|
||||
language: 'zh-Hans-CN'
|
||||
}))
|
||||
const decryptImage = vi.fn((path: string) => {
|
||||
const seed = Number(String(path).replace(/\D/g, '')) || 0
|
||||
return pngBytes(seed)
|
||||
})
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [
|
||||
{ md5: CONVERSATION, m_nsUsrName: 'incremental', type: 'user' as const }
|
||||
],
|
||||
listMessages: async () => messages,
|
||||
countConversationImages: async () => ({ count: 112, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ count: 112, maxLocalId: 112 }),
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN'
|
||||
}),
|
||||
decryptService: () => ({ findImageFile: (md5) => `C:/fake/${md5}.dat`, decryptImage }) as never,
|
||||
recognize
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
|
||||
// 关键断言:12 次,而不是 112 次。
|
||||
expect(recognize).toHaveBeenCalledTimes(12)
|
||||
// 已有终态那 100 条连解密都不该碰。
|
||||
expect(decryptImage).toHaveBeenCalledTimes(12)
|
||||
|
||||
const after = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
// 100 条旧终态 + 100 个旧 artifact 一条不少,文本原样保留(不重算、不覆盖)。
|
||||
expect(after.countByState().indexed).toBe(112)
|
||||
expect(after.getArtifact('seeded|1')?.text).toBe('TRACE_SEEDED_1')
|
||||
expect(after.getArtifact('seeded|100')?.text).toBe('TRACE_SEEDED_100')
|
||||
after.close()
|
||||
service.resetAccount()
|
||||
})
|
||||
|
||||
it('水位完全没变 → 整个会话直接跳过,OCR 一次都不调', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
seedTerminal(store, 20, 'empty')
|
||||
store.writeScanState({
|
||||
conversationId: CONVERSATION,
|
||||
state: 'done',
|
||||
imageTotal: 20,
|
||||
imageProcessed: 20,
|
||||
maxLocalId: 20
|
||||
})
|
||||
store.close()
|
||||
|
||||
const recognize = vi.fn(async () => ({ success: true, text: 'X', language: null }))
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [
|
||||
{ md5: CONVERSATION, m_nsUsrName: 'incremental', type: 'user' as const }
|
||||
],
|
||||
listMessages: async () => Array.from({ length: 20 }, (_, index) => imageMessage(index + 1)),
|
||||
countConversationImages: async () => ({ count: 20, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ count: 20, maxLocalId: 20 }),
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: null
|
||||
}),
|
||||
decryptService: () => ({ findImageFile: () => null, decryptImage: () => null }) as never,
|
||||
recognize
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
|
||||
expect(recognize).not.toHaveBeenCalled()
|
||||
service.resetAccount()
|
||||
})
|
||||
})
|
||||
|
||||
describe('派生索引修复:只重建 L3,绝不重跑 OCR', () => {
|
||||
it('只重建"有 OCR 文本"的会话,且 recognize 一次都不被调用', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
seedTerminal(store, 3, 'indexed')
|
||||
// 另一个会话只有 empty(没有派生文本可修)→ 不该被重建,白读一遍 WCDB。
|
||||
store.putArtifact({
|
||||
accountId: ACCOUNT,
|
||||
artifactKey: 'empty-only|1',
|
||||
imageIdentity: 'sha256:empty-only',
|
||||
state: 'empty',
|
||||
text: '',
|
||||
charCount: 0,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: null,
|
||||
createdAt: 1,
|
||||
updatedAt: 1
|
||||
})
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: 'md5-empty-only',
|
||||
messageId: 'local:1',
|
||||
createTime: 1,
|
||||
imageIdentity: 'sha256:empty-only',
|
||||
artifactKey: 'empty-only|1',
|
||||
state: 'empty',
|
||||
updatedAt: 1
|
||||
})
|
||||
store.close()
|
||||
|
||||
const recognize = vi.fn(async () => ({ success: true, text: 'X', language: null }))
|
||||
const onConversationIndexed = vi.fn(async () => undefined)
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
recognize,
|
||||
onConversationIndexed
|
||||
})
|
||||
|
||||
const result = await service.repairKnowledgeIndex()
|
||||
|
||||
expect(result.skipped).toBe(false)
|
||||
expect(result.conversations).toBe(1)
|
||||
// 契约:修复路径的定义就是"不调 OCR"。类型上写死成字面量 0。
|
||||
expect(result.ocrExecutions).toBe(0)
|
||||
expect(recognize).not.toHaveBeenCalled()
|
||||
expect(onConversationIndexed).toHaveBeenCalledTimes(1)
|
||||
expect(onConversationIndexed).toHaveBeenCalledWith(CONVERSATION)
|
||||
service.resetAccount()
|
||||
})
|
||||
|
||||
it('索引任务正在跑时拒绝并发修复(避免读到半程 binding)', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
seedTerminal(store, 1, 'indexed')
|
||||
store.close()
|
||||
|
||||
let release: (() => void) | null = null
|
||||
const gate = new Promise<void>((resolve) => {
|
||||
release = resolve
|
||||
})
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [
|
||||
{ md5: CONVERSATION, m_nsUsrName: 'incremental', type: 'user' as const }
|
||||
],
|
||||
listMessages: async () => [imageMessage(1)],
|
||||
countConversationImages: async () => ({ count: 1, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ count: 1, maxLocalId: 1 }),
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: null
|
||||
}),
|
||||
decryptService: () => ({ findImageFile: () => null, decryptImage: () => null }) as never,
|
||||
// 卡住 pass,让 running 保持为真。
|
||||
interactiveIdle: () => gate
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(true))
|
||||
|
||||
const during = await service.repairKnowledgeIndex()
|
||||
expect(during.skipped).toBe(true)
|
||||
expect(during.conversations).toBe(0)
|
||||
|
||||
release?.()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
service.resetAccount()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,283 @@
|
||||
/**
|
||||
* §2 / §3 的硬条件:清理图片文字索引必须让 **Knowledge 里已经产生的 OCR 派生文字**一起失效。
|
||||
*
|
||||
* 背景:OCR 文本经 normalizer 的固定前缀 `图片文字:` 拼进 `searchableText`,
|
||||
* 再进 chunks / FTS。所以"清理成功"不能只等于"派生 SQLite 删掉了" ——
|
||||
* 用户执行设置里的「清理图片文字索引」之后,`search_messages` 必须搜不到那些图片文字,
|
||||
* 同时**普通文字消息必须一条不少地留着**。
|
||||
*
|
||||
* 本文件分两部分:
|
||||
* - A:Knowledge 侧的失效机制本身成立(内容变了 / 消息被移除都会被重建替换);
|
||||
* - B:生产路径真的触发了它(`imageTextIndexService.clear()` 会逐会话重建)。
|
||||
*/
|
||||
import { mkdtempSync } from 'node:fs'
|
||||
import { rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
DEFAULT_KNOWLEDGE_CHUNKER,
|
||||
type KnowledgeFtsConfig,
|
||||
type KnowledgeSourceMessage
|
||||
} from '../../src/shared/knowledge'
|
||||
import { KnowledgeStore } from '../../src/main/knowledge/knowledge-store'
|
||||
import { ImageTextIndexService } from '../../src/main/services/image-text-index-service'
|
||||
import { getImageTextIndexDatabasePath } from '../../src/main/services/image-text-index-store'
|
||||
import type * as chat from '../../src/main/services/chat-service'
|
||||
|
||||
const ACCOUNT = 'fixture-account-image-ocr'
|
||||
const CONVERSATION = 'conversation-image-ocr'
|
||||
const OCR_TOKEN = 'TRACE_IMAGE_OCR_UNIQUE_2026'
|
||||
const PLAIN_TEXT = '普通聊天内容保留'
|
||||
const IMAGE_MESSAGE_ID = 'local:9001'
|
||||
const TEXT_MESSAGE_ID = 'local:9002'
|
||||
|
||||
const fts: KnowledgeFtsConfig = {
|
||||
profileId: 'test-trigram-external-full',
|
||||
tokenizer: 'trigram',
|
||||
contentMode: 'external',
|
||||
detail: 'full',
|
||||
columnsize: 1
|
||||
}
|
||||
|
||||
const roots: string[] = []
|
||||
|
||||
function makeRoot(): string {
|
||||
const root = mkdtempSync(join(tmpdir(), 'wxe-image-ocr-invalidation-'))
|
||||
roots.push(root)
|
||||
return root
|
||||
}
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true })))
|
||||
})
|
||||
|
||||
function textMessage(): KnowledgeSourceMessage {
|
||||
return {
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: TEXT_MESSAGE_ID,
|
||||
createTime: Date.UTC(2026, 8, 1, 10, 0),
|
||||
senderId: 'fixture-member-1',
|
||||
senderName: '张三',
|
||||
kind: 'text',
|
||||
text: PLAIN_TEXT
|
||||
}
|
||||
}
|
||||
|
||||
/** 带 OCR 派生文本的图片消息(这是 OCR 索引建立后的状态)。 */
|
||||
function imageMessageWithOcr(caption?: string): KnowledgeSourceMessage {
|
||||
return {
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: IMAGE_MESSAGE_ID,
|
||||
createTime: Date.UTC(2026, 8, 1, 10, 5),
|
||||
senderId: 'fixture-member-2',
|
||||
senderName: '李四',
|
||||
kind: 'image',
|
||||
...(caption ? { text: caption } : {}),
|
||||
imageOcrText: OCR_TOKEN,
|
||||
imageOcrState: 'indexed'
|
||||
}
|
||||
}
|
||||
|
||||
/** 同一张图片,但 OCR 派生文本已经不存在(= 派生库被清掉后 resolver 拿不到东西)。 */
|
||||
function imageMessageWithoutOcr(caption?: string): KnowledgeSourceMessage {
|
||||
return {
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: IMAGE_MESSAGE_ID,
|
||||
createTime: Date.UTC(2026, 8, 1, 10, 5),
|
||||
senderId: 'fixture-member-2',
|
||||
senderName: '李四',
|
||||
kind: 'image',
|
||||
...(caption ? { text: caption } : {})
|
||||
}
|
||||
}
|
||||
|
||||
function searchTokens(store: KnowledgeStore, text: string): string[] {
|
||||
return store
|
||||
.search({ accountId: ACCOUNT, text, limit: 20 })
|
||||
.map((item) => item.messageId)
|
||||
}
|
||||
|
||||
function evidenceFor(store: KnowledgeStore, text: string) {
|
||||
return store.search({ accountId: ACCOUNT, text, limit: 20 })
|
||||
}
|
||||
|
||||
async function indexConversation(
|
||||
store: KnowledgeStore,
|
||||
messages: KnowledgeSourceMessage[]
|
||||
): Promise<void> {
|
||||
await store.index({
|
||||
conversations: [{ conversationId: CONVERSATION, completeSnapshot: true, messages }],
|
||||
chunker: DEFAULT_KNOWLEDGE_CHUNKER
|
||||
})
|
||||
}
|
||||
|
||||
describe('§2-A Knowledge 侧的失效机制:OCR 派生文字必须能真的消失', () => {
|
||||
it('图片消息仍然存在、只是 OCR 文本没了 → 旧 OCR 文字搜不到,普通文字不受影响', async () => {
|
||||
const store = new KnowledgeStore(makeRoot(), ACCOUNT, fts)
|
||||
|
||||
// 1) 建立图片 OCR 派生记录 + 完成索引
|
||||
await indexConversation(store, [textMessage(), imageMessageWithOcr()])
|
||||
|
||||
// 2) 必须能搜到,并且命中的是**原始图片消息**
|
||||
const before = evidenceFor(store, OCR_TOKEN)
|
||||
expect(before.length).toBeGreaterThan(0)
|
||||
expect(before[0].messageId).toBe(IMAGE_MESSAGE_ID)
|
||||
expect(before[0].sourceKind).toBe('image')
|
||||
|
||||
// 3) OCR 文本被清掉(模拟「清理图片文字索引」后重建)
|
||||
await indexConversation(store, [textMessage(), imageMessageWithoutOcr()])
|
||||
|
||||
// 4) 旧 OCR 文字必须彻底搜不到
|
||||
expect(searchTokens(store, OCR_TOKEN)).toEqual([])
|
||||
|
||||
// 5) 普通文字消息必须仍然命中 —— 不能清掉普通 Knowledge
|
||||
expect(searchTokens(store, PLAIN_TEXT)).toContain(TEXT_MESSAGE_ID)
|
||||
|
||||
store.close()
|
||||
})
|
||||
|
||||
it('无文字图片消息在 OCR 清掉后被整体移除 → 旧 OCR 文字同样搜不到', async () => {
|
||||
const store = new KnowledgeStore(makeRoot(), ACCOUNT, fts)
|
||||
|
||||
// 这条图片消息除了 OCR 文本之外没有任何内容;OCR 一清,它就不该再进索引。
|
||||
await indexConversation(store, [textMessage(), imageMessageWithOcr()])
|
||||
expect(searchTokens(store, OCR_TOKEN).length).toBeGreaterThan(0)
|
||||
|
||||
await indexConversation(store, [textMessage()])
|
||||
|
||||
expect(searchTokens(store, OCR_TOKEN)).toEqual([])
|
||||
expect(searchTokens(store, PLAIN_TEXT)).toContain(TEXT_MESSAGE_ID)
|
||||
|
||||
store.close()
|
||||
})
|
||||
|
||||
it('§3:OCR 文本变化(state 仍是 indexed)也必须让旧文本失效', async () => {
|
||||
const store = new KnowledgeStore(makeRoot(), ACCOUNT, fts)
|
||||
|
||||
await indexConversation(store, [textMessage(), imageMessageWithOcr()])
|
||||
expect(searchTokens(store, OCR_TOKEN).length).toBeGreaterThan(0)
|
||||
|
||||
// 同一个 state(indexed),内容换成了另一段文字 —— 例如换了 OCR 运行时后重新识别。
|
||||
const replaced = {
|
||||
...imageMessageWithOcr(),
|
||||
imageOcrText: 'TRACE_IMAGE_OCR_REPLACED_2026'
|
||||
}
|
||||
await indexConversation(store, [textMessage(), replaced])
|
||||
|
||||
expect(searchTokens(store, OCR_TOKEN)).toEqual([])
|
||||
expect(searchTokens(store, 'TRACE_IMAGE_OCR_REPLACED_2026')).toContain(IMAGE_MESSAGE_ID)
|
||||
|
||||
store.close()
|
||||
})
|
||||
})
|
||||
|
||||
describe('§2-B 生产路径:清理必须逐会话重建 Knowledge', () => {
|
||||
function imageMessage(localId: number, conversationId: string): chat.FormattedMessage {
|
||||
return {
|
||||
localId: String(localId),
|
||||
createTime: 1_700_000_000 + localId,
|
||||
content: '[图片]',
|
||||
contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` },
|
||||
sessionId: conversationId
|
||||
} as unknown as chat.FormattedMessage
|
||||
}
|
||||
|
||||
it('clear() 对每个有 OCR 派生文本的会话都触发一次重建,而不是清空整个 Knowledge', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const conversations = ['conv-alpha', 'conv-beta']
|
||||
const onConversationIndexed = vi.fn(async () => undefined)
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () =>
|
||||
conversations.map((md5) => ({ md5, m_nsUsrName: md5, type: 'group' as const })),
|
||||
listMessages: async (conversationId) => [imageMessage(1, conversationId)],
|
||||
countConversationImages: async () => ({ count: 1, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ count: 1, maxLocalId: 1 }),
|
||||
// 没有解密服务 → 图片判为 image_missing;这不影响"这条会话有没有 OCR 绑定"。
|
||||
decryptService: () => ({ findImageFile: () => null, decryptImage: () => null }) as never,
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN'
|
||||
}),
|
||||
onConversationIndexed
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
|
||||
// 两个会话都真的产生了绑定。
|
||||
const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
expect(onConversationIndexed).toHaveBeenCalledTimes(2)
|
||||
|
||||
onConversationIndexed.mockClear()
|
||||
const result = await service.clear()
|
||||
|
||||
expect(result.removed).toBe(true)
|
||||
// 关键:清理必须重建这两个会话,否则 Knowledge 里还留着图片文字。
|
||||
expect(onConversationIndexed).toHaveBeenCalledTimes(2)
|
||||
expect(onConversationIndexed.mock.calls.map((call) => call[0]).sort()).toEqual(
|
||||
[...conversations].sort()
|
||||
)
|
||||
expect(service.lastInvalidatedConversations).toBe(2)
|
||||
expect(databasePath).toBeTruthy()
|
||||
})
|
||||
|
||||
it('prepareForCacheClear() 同样会做失效("清理全部"路径不能漏)', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const onConversationIndexed = vi.fn(async () => undefined)
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [
|
||||
{ md5: CONVERSATION, m_nsUsrName: CONVERSATION, type: 'group' as const }
|
||||
],
|
||||
listMessages: async () => [imageMessage(1, CONVERSATION)],
|
||||
countConversationImages: async () => ({ count: 1, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ count: 1, maxLocalId: 1 }),
|
||||
decryptService: () => ({ findImageFile: () => null, decryptImage: () => null }) as never,
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN'
|
||||
}),
|
||||
onConversationIndexed
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
onConversationIndexed.mockClear()
|
||||
|
||||
await service.prepareForCacheClear()
|
||||
|
||||
expect(onConversationIndexed).toHaveBeenCalledWith(CONVERSATION)
|
||||
})
|
||||
|
||||
it('派生库里没有 OCR 绑定时,清理不触发任何无意义的重建', async () => {
|
||||
const databaseRoot = makeRoot()
|
||||
const onConversationIndexed = vi.fn(async () => undefined)
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
onConversationIndexed
|
||||
})
|
||||
|
||||
const result = await service.clear()
|
||||
expect(result.removed).toBe(true)
|
||||
expect(onConversationIndexed).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,409 @@
|
||||
/**
|
||||
* 「图片文字索引」的 P0 语义测试。
|
||||
*
|
||||
* 这里覆盖的都是**不能用 UI 数字糊过去**的硬约束:
|
||||
* - 覆盖度必须在重启后依然诚实(派生库只知道处理过什么,不知道源数据一共多少);
|
||||
* - 增量判据必须能发现「总数没变但集合变了」;
|
||||
* - 清理必须真的把文件删掉,删不掉要如实上报;
|
||||
* - 涉及图片的问题在索引未完成时,答案语义里必须带"不能因为没搜到就说没有"。
|
||||
*/
|
||||
import { existsSync, mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import type * as chat from '../../src/main/services/chat-service'
|
||||
import { ImageTextIndexService } from '../../src/main/services/image-text-index-service'
|
||||
import {
|
||||
ImageTextIndexStore,
|
||||
getImageTextIndexDatabasePath
|
||||
} from '../../src/main/services/image-text-index-store'
|
||||
import { buildImageOcrCoverage } from '../../src/main/services/local-query-api-service'
|
||||
import type { ImageTextIndexCoverage } from '../../src/shared/image-text-index'
|
||||
|
||||
const ACCOUNT = 'wxid_fixture_account'
|
||||
const CONVERSATION = 'conversation-md5-fixture'
|
||||
const roots: string[] = []
|
||||
|
||||
function makeDatabaseRoot(): string {
|
||||
const root = mkdtempSync(join(tmpdir(), 'tm-image-text-index-'))
|
||||
roots.push(root)
|
||||
return root
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
for (const root of roots.splice(0)) {
|
||||
try {
|
||||
rmSync(root, { recursive: true, force: true })
|
||||
} catch {
|
||||
// 测试收尾尽力而为。
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
/** 只带图片索引需要的字段;其余字段与本测试无关。 */
|
||||
function imageMessage(localId: number, createTime: number): chat.FormattedMessage {
|
||||
return {
|
||||
localId: String(localId),
|
||||
createTime,
|
||||
content: '[图片]',
|
||||
contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` }
|
||||
} as unknown as chat.FormattedMessage
|
||||
}
|
||||
|
||||
type Harness = {
|
||||
service: ImageTextIndexService
|
||||
databaseRoot: string
|
||||
listMessages: ReturnType<typeof vi.fn>
|
||||
watermark: { count: number; maxLocalId: number }
|
||||
databasePath: string
|
||||
}
|
||||
|
||||
function makeHarness(options: { messages?: chat.FormattedMessage[] } = {}): Harness {
|
||||
const databaseRoot = makeDatabaseRoot()
|
||||
const listMessages = vi.fn(async () => options.messages ?? [])
|
||||
const watermark = { count: 0, maxLocalId: 0 }
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
resolveAccountRoot: () => 'C:/fixture/account',
|
||||
listContacts: async () => [
|
||||
{ md5: CONVERSATION, m_nsUsrName: 'fixture', type: 'group' as const }
|
||||
],
|
||||
listMessages,
|
||||
countConversationImages: async () => ({ count: watermark.count, typeColumn: 'local_type' }),
|
||||
imageWatermark: async () => ({ ...watermark }),
|
||||
// 没有解密服务 → 每张图片都会被判成 image_missing。这样测试完全不碰真实图片。
|
||||
decryptService: () => ({ findImageFile: () => null, decryptImage: () => null }) as never,
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN'
|
||||
}),
|
||||
recognize: async () => ({ success: true, text: '', language: 'zh-Hans-CN' })
|
||||
})
|
||||
return {
|
||||
service,
|
||||
databaseRoot,
|
||||
listMessages,
|
||||
watermark,
|
||||
databasePath: getImageTextIndexDatabasePath(databaseRoot, ACCOUNT)
|
||||
}
|
||||
}
|
||||
|
||||
describe('§2 增量水位:只比条数会漏掉「等量替换」', () => {
|
||||
it('水位(条数 + 最大插入序)都没变时才跳过,不读 WCDB', async () => {
|
||||
const harness = makeHarness({ messages: [imageMessage(10, 1000), imageMessage(20, 2000)] })
|
||||
harness.watermark.count = 2
|
||||
harness.watermark.maxLocalId = 20
|
||||
|
||||
await harness.service.startPass()
|
||||
// 走到完成态需要等内部 promise 收敛。
|
||||
await vi.waitFor(() => expect(harness.service.isRunning()).toBe(false))
|
||||
expect(harness.listMessages).toHaveBeenCalledTimes(1)
|
||||
|
||||
// 第二遍:水位完全一致 → 跳过,不再读会话消息。
|
||||
await harness.service.startPass()
|
||||
await vi.waitFor(() => expect(harness.service.isRunning()).toBe(false))
|
||||
expect(harness.listMessages).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it('总数相同但最大插入序前进 → 必须重扫(撤回一张旧图 + 新增一张新图)', async () => {
|
||||
const harness = makeHarness({ messages: [imageMessage(10, 1000), imageMessage(20, 2000)] })
|
||||
harness.watermark.count = 2
|
||||
harness.watermark.maxLocalId = 20
|
||||
|
||||
await harness.service.startPass()
|
||||
await vi.waitFor(() => expect(harness.service.isRunning()).toBe(false))
|
||||
expect(harness.listMessages).toHaveBeenCalledTimes(1)
|
||||
|
||||
// 集合变了、条数没变:localId 10 被撤回,新增 localId 30。
|
||||
harness.listMessages.mockImplementation(async () => [
|
||||
imageMessage(20, 2000),
|
||||
imageMessage(30, 3000)
|
||||
])
|
||||
harness.watermark.maxLocalId = 30
|
||||
|
||||
await harness.service.startPass()
|
||||
await vi.waitFor(() => expect(harness.service.isRunning()).toBe(false))
|
||||
// 只看 count 的实现会在这里静默跳过 —— 那正是会漏掉新图片的洞。
|
||||
expect(harness.listMessages).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
|
||||
it('水位不可用(数据库不支持该聚合)时一律重扫,宁可慢也不漏', async () => {
|
||||
const harness = makeHarness({ messages: [imageMessage(10, 1000)] })
|
||||
harness.watermark.count = 1
|
||||
const service = new ImageTextIndexService()
|
||||
const listMessages = vi.fn(async () => [imageMessage(10, 1000)])
|
||||
service.bind({
|
||||
databaseRoot: harness.databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [{ md5: CONVERSATION, m_nsUsrName: 'fixture', type: 'group' }],
|
||||
listMessages,
|
||||
countConversationImages: async () => ({ count: 1, typeColumn: 'local_type' }),
|
||||
// 关键:不提供 imageWatermark
|
||||
decryptService: () => ({ findImageFile: () => null, decryptImage: () => null }) as never,
|
||||
capability: async () => ({
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: null,
|
||||
language: null
|
||||
})
|
||||
})
|
||||
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
await service.startPass()
|
||||
await vi.waitFor(() => expect(service.isRunning()).toBe(false))
|
||||
expect(listMessages).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
})
|
||||
|
||||
describe('§1 覆盖度诚实性', () => {
|
||||
it('重启后仍是 partial:分母来自落盘统计,不会退化成 processed', async () => {
|
||||
const { databaseRoot, databasePath } = makeHarness()
|
||||
// 先按「已建立过索引」写库:总数 100,实际只处理了 30 条。
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
store.writeCountedTotal({ total: 100, countedAt: 1_700_000_000_000, complete: true })
|
||||
for (let index = 0; index < 30; index += 1) {
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: `local:${index}`,
|
||||
createTime: index,
|
||||
imageIdentity: `sha256:${index}`,
|
||||
artifactKey: `sha256:${index}|fake`,
|
||||
state: 'indexed',
|
||||
updatedAt: index
|
||||
})
|
||||
}
|
||||
store.close()
|
||||
|
||||
// 全新 service 实例 = 模拟应用重启(内存计数器归零)。
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({ databaseRoot, resolveAccountId: () => ACCOUNT })
|
||||
const status = await service.getStatus()
|
||||
|
||||
expect(status.coverage.totalImageMessages).toBe(100)
|
||||
expect(status.coverage.processed).toBe(30)
|
||||
expect(status.coverage.established).toBe(true)
|
||||
// 修复前这里会因为 total 退化成 processed 而变成 true(把 30% 谎报成 100%)。
|
||||
expect(status.coverage.complete).toBe(false)
|
||||
expect(status.coverage.countedAt).toBe(1_700_000_000_000)
|
||||
})
|
||||
|
||||
it('统计时有会话没数上 → 分母不完整,不允许声称 complete', async () => {
|
||||
const { databaseRoot, databasePath } = makeHarness()
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
store.writeCountedTotal({ total: 10, countedAt: 1, complete: false })
|
||||
store.putBinding({
|
||||
accountId: ACCOUNT,
|
||||
conversationId: CONVERSATION,
|
||||
messageId: 'local:1',
|
||||
createTime: 1,
|
||||
imageIdentity: 'sha256:1',
|
||||
artifactKey: 'sha256:1|fake',
|
||||
state: 'indexed',
|
||||
updatedAt: 1
|
||||
})
|
||||
store.close()
|
||||
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({ databaseRoot, resolveAccountId: () => ACCOUNT })
|
||||
const status = await service.getStatus()
|
||||
expect(status.coverage.processed).toBe(1)
|
||||
expect(status.coverage.complete).toBe(false)
|
||||
})
|
||||
|
||||
it('从未统计过 → 不算已建立,且查询路径不为看覆盖度凭空建库', async () => {
|
||||
const { databaseRoot, databasePath } = makeHarness()
|
||||
const service = new ImageTextIndexService()
|
||||
service.bind({ databaseRoot, resolveAccountId: () => ACCOUNT })
|
||||
expect(service.getCoverageSnapshot()).toBeNull()
|
||||
expect(existsSync(databasePath)).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('§5 清理:删得掉才算成功', () => {
|
||||
it('清理后派生库文件消失,覆盖度回到未建立', async () => {
|
||||
const { service, databasePath } = makeHarness()
|
||||
// 建一份有内容的派生数据(建库 + 写 artifact/binding/水位 + 落盘总数)。
|
||||
const store = new ImageTextIndexStore(databasePath, ACCOUNT)
|
||||
store.writeCountedTotal({ total: 5, countedAt: 1, complete: true })
|
||||
store.putArtifact({
|
||||
accountId: ACCOUNT,
|
||||
artifactKey: 'k',
|
||||
imageIdentity: 'sha256:x',
|
||||
state: 'indexed',
|
||||
text: 'fixture',
|
||||
charCount: 7,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN',
|
||||
createdAt: 1,
|
||||
updatedAt: 1
|
||||
})
|
||||
store.close()
|
||||
expect(existsSync(databasePath)).toBe(true)
|
||||
|
||||
const result = await service.clear()
|
||||
expect(result.removed).toBe(true)
|
||||
expect(existsSync(databasePath)).toBe(false)
|
||||
|
||||
const status = await service.getStatus()
|
||||
expect(status.coverage.established).toBe(false)
|
||||
expect(status.coverage.totalImageMessages).toBe(0)
|
||||
expect(status.coverage.countedAt).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
describe('§1/§7 查询层:覆盖度必须是独立维度且带零结果诚实性', () => {
|
||||
const coverageOf = (input: Partial<ImageTextIndexCoverage>): ImageTextIndexCoverage => ({
|
||||
totalImageMessages: 0,
|
||||
processed: 0,
|
||||
indexed: 0,
|
||||
empty: 0,
|
||||
missing: 0,
|
||||
failed: 0,
|
||||
pending: 0,
|
||||
established: false,
|
||||
complete: false,
|
||||
countedAt: null,
|
||||
...input
|
||||
})
|
||||
|
||||
it('partial:必须明确「不能因为没搜到就回答没有」并给出真实比例', () => {
|
||||
const built = buildImageOcrCoverage(
|
||||
coverageOf({
|
||||
totalImageMessages: 100,
|
||||
processed: 30,
|
||||
indexed: 28,
|
||||
empty: 2,
|
||||
established: true,
|
||||
countedAt: 1_700_000_000_000
|
||||
})
|
||||
)
|
||||
expect(built?.state).toBe('partial')
|
||||
expect(built?.totalImageMessages).toBe(100)
|
||||
expect(built?.processed).toBe(30)
|
||||
expect(built?.summary).toContain('30')
|
||||
expect(built?.summary).toContain('100')
|
||||
expect(built?.summary).toContain('不能因为没搜到就回答')
|
||||
})
|
||||
|
||||
it('complete:不附加零结果约束,但仍带上统计时刻', () => {
|
||||
const built = buildImageOcrCoverage(
|
||||
coverageOf({
|
||||
totalImageMessages: 100,
|
||||
processed: 100,
|
||||
indexed: 90,
|
||||
empty: 10,
|
||||
established: true,
|
||||
complete: true,
|
||||
countedAt: 1_700_000_000_000
|
||||
})
|
||||
)
|
||||
expect(built?.state).toBe('complete')
|
||||
expect(built?.summary).not.toContain('不能因为没搜到就回答')
|
||||
})
|
||||
|
||||
it('not_built:说明图片里的文字目前搜不到,且同样禁止凭零结果下"没有"', () => {
|
||||
const built = buildImageOcrCoverage(coverageOf({}))
|
||||
expect(built?.state).toBe('not_built')
|
||||
expect(built?.summary).toContain('尚未建立')
|
||||
expect(built?.summary).toContain('不能因为没搜到就回答')
|
||||
})
|
||||
|
||||
it('没有覆盖度(库都不存在)时不下发该字段,不制造假维度', () => {
|
||||
expect(buildImageOcrCoverage(null)).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('图片数量统计:必须区分「0 张」与「统计失败」', () => {
|
||||
function bindCounting(
|
||||
service: ImageTextIndexService,
|
||||
databaseRoot: string,
|
||||
probe: () => Promise<{ count: number | null; typeColumn: string | null; error?: string }>
|
||||
): void {
|
||||
service.bind({
|
||||
databaseRoot,
|
||||
resolveAccountId: () => ACCOUNT,
|
||||
listContacts: async () => [
|
||||
{ md5: 'conv-a', m_nsUsrName: 'a', type: 'group' as const },
|
||||
{ md5: 'conv-b', m_nsUsrName: 'b', type: 'group' as const }
|
||||
],
|
||||
countConversationImages: probe
|
||||
})
|
||||
}
|
||||
|
||||
it('真的 0 张:scanned=2 / failed=0,可以放心说 0', async () => {
|
||||
const service = new ImageTextIndexService()
|
||||
bindCounting(service, makeDatabaseRoot(), async () => ({
|
||||
count: 0,
|
||||
typeColumn: 'local_type'
|
||||
}))
|
||||
|
||||
const result = await service.countImageMessages()
|
||||
expect(result.totalImageMessages).toBe(0)
|
||||
expect(result.scannedConversations).toBe(2)
|
||||
expect(result.failedConversations).toBe(0)
|
||||
expect(result.typeColumn).toBe('local_type')
|
||||
expect(result.error).toBeUndefined()
|
||||
})
|
||||
|
||||
it('统计全部失败:不得表现为 0 张,且必须给出原因', async () => {
|
||||
const service = new ImageTextIndexService()
|
||||
bindCounting(service, makeDatabaseRoot(), async () => ({
|
||||
count: null,
|
||||
typeColumn: null,
|
||||
error: '读取消息分片失败'
|
||||
}))
|
||||
|
||||
const result = await service.countImageMessages()
|
||||
expect(result.totalImageMessages).toBe(0)
|
||||
// 关键区分:一个会话都没数成。
|
||||
expect(result.scannedConversations).toBe(0)
|
||||
expect(result.failedConversations).toBe(2)
|
||||
expect(result.error).toBe('读取消息分片失败')
|
||||
// 统计失败 → 分母不成立 → 不允许声称已建立覆盖。
|
||||
const status = await service.getStatus()
|
||||
expect(status.coverage.established).toBe(false)
|
||||
expect(status.coverage.complete).toBe(false)
|
||||
expect(status.coverage.countedAt).not.toBeNull()
|
||||
})
|
||||
|
||||
it('部分失败:总数偏小,coverage 不允许声称 complete', async () => {
|
||||
const service = new ImageTextIndexService()
|
||||
let call = 0
|
||||
bindCounting(service, makeDatabaseRoot(), async () => {
|
||||
call += 1
|
||||
return call === 1
|
||||
? { count: 10, typeColumn: 'local_type' }
|
||||
: { count: null, typeColumn: null, error: '图片消息统计查询失败' }
|
||||
})
|
||||
|
||||
const result = await service.countImageMessages()
|
||||
expect(result.totalImageMessages).toBe(10)
|
||||
expect(result.scannedConversations).toBe(1)
|
||||
expect(result.failedConversations).toBe(1)
|
||||
|
||||
const status = await service.getStatus()
|
||||
expect(status.coverage.totalImageMessages).toBe(10)
|
||||
expect(status.coverage.complete).toBe(false)
|
||||
})
|
||||
|
||||
it('探测到的类型列名会向上透出(列名不一致时是唯一线索)', async () => {
|
||||
const service = new ImageTextIndexService()
|
||||
bindCounting(service, makeDatabaseRoot(), async () => ({
|
||||
count: 3,
|
||||
typeColumn: 'msg_type'
|
||||
}))
|
||||
|
||||
const result = await service.countImageMessages()
|
||||
expect(result.typeColumn).toBe('msg_type')
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,389 @@
|
||||
/**
|
||||
* §2:图片 OCR 来源语义的 **deterministic synthetic E2E**。
|
||||
*
|
||||
* 硬要求是"不依赖真实线上 AI 模型也能 PASS",所以这里把两个外部边界**确定性**地固定住:
|
||||
* - WCDB(chat-service)→ 用合成联系人 / 合成消息;
|
||||
* - Knowledge 检索 → 用 fake 直接返回合成证据(形状与真实 `KnowledgeEvidence` 一致,
|
||||
* 包括**未清理**的 `searchable_text`,用来验证内部前缀确实被剥掉)。
|
||||
*
|
||||
* 链路上真正的被测代码仍然是生产实现:
|
||||
* LocalQueryApiService.search() ← 真实 scope 解析 / 证据映射 / 前缀剥离
|
||||
* createLocalQueryToolExecutor() ← 真实 Tool 执行
|
||||
* QueryAgentService.run() ← 真实 Agent 循环 / tool result 组装
|
||||
*
|
||||
* 断言的 10 项对应需求:FOUND=YES / sourceKind=image / derived source=image_ocr /
|
||||
* conversation scope=技术交流群 / Evidence messageRef=原始图片消息 /
|
||||
* Evidence UI=图片文字 / jump target=原始图片消息 / 不产生虚构 OCR 消息。
|
||||
*/
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { decodeMessageRef } from '../../src/shared/local-query-api'
|
||||
|
||||
process.env.TZ = 'Asia/Shanghai'
|
||||
|
||||
const GROUP_MD5 = 'md5-tech-group'
|
||||
const GROUP_NAME = '技术交流群'
|
||||
const IMAGE_MESSAGE_ID = 'local:9001'
|
||||
const TEXT_MESSAGE_ID = 'local:9002'
|
||||
const OCR_TEXT = 'OpenAI ChatGPT Plus $20 Pro $200'
|
||||
/** Knowledge 侧的原始 searchable_text:带内部标签,绝不该出现在 Evidence 里。 */
|
||||
const RAW_SEARCHABLE = `图片文字:${OCR_TEXT}`
|
||||
|
||||
const fixture = vi.hoisted(() => {
|
||||
const imageTimestamp = Date.parse('2026-09-03T14:32:00+08:00')
|
||||
const textTimestamp = Date.parse('2026-09-03T14:30:00+08:00')
|
||||
return {
|
||||
imageTimestamp,
|
||||
textTimestamp,
|
||||
contacts: [
|
||||
{
|
||||
m_nsUsrName: 'wxid-tech-group',
|
||||
m_nsNickName: '技术交流群',
|
||||
md5: 'md5-tech-group',
|
||||
type: 'group' as const
|
||||
}
|
||||
],
|
||||
messages: [
|
||||
{
|
||||
id: '9002',
|
||||
localId: '9002',
|
||||
from: 'user',
|
||||
type: '文本',
|
||||
datetime: '2026/9/3 14:30:00',
|
||||
content: '今天正常讨论一下 API',
|
||||
isSender: false,
|
||||
name: '张三',
|
||||
createTime: Math.floor(textTimestamp / 1000)
|
||||
},
|
||||
{
|
||||
id: '9001',
|
||||
localId: '9001',
|
||||
from: 'user',
|
||||
type: '图片',
|
||||
datetime: '2026/9/3 14:32:00',
|
||||
content: '',
|
||||
contentData: { type: 'image', md5: 'image-md5-fixture', datName: 'dat-fixture' },
|
||||
isSender: false,
|
||||
name: '张三',
|
||||
createTime: Math.floor(imageTimestamp / 1000)
|
||||
}
|
||||
]
|
||||
}
|
||||
})
|
||||
|
||||
const IMAGE_TIMESTAMP = fixture.imageTimestamp
|
||||
|
||||
vi.mock('../../src/main/services/chat-service', () => ({
|
||||
isReady: () => true,
|
||||
listContactsAsync: vi.fn(async () => fixture.contacts),
|
||||
listMessagesAsync: vi.fn(async () => fixture.messages)
|
||||
}))
|
||||
|
||||
import { LocalQueryApiService } from '../../src/main/services/local-query-api-service'
|
||||
import { createLocalQueryToolExecutor } from '../../src/main/services/local-query-tool-executor'
|
||||
import {
|
||||
QueryAgentService,
|
||||
type QueryAgentProvider
|
||||
} from '../../src/main/services/query-agent-service'
|
||||
|
||||
/** 与真实 Knowledge 检索返回的证据形状一致(含原始未清理文本)。 */
|
||||
function syntheticKnowledgeEvidence() {
|
||||
return [
|
||||
{
|
||||
chunkId: 'chunk-1',
|
||||
conversationId: GROUP_MD5,
|
||||
startTime: IMAGE_TIMESTAMP,
|
||||
endTime: IMAGE_TIMESTAMP,
|
||||
messageId: IMAGE_MESSAGE_ID,
|
||||
senderId: 'fixture-member',
|
||||
sender: '张三',
|
||||
timestamp: IMAGE_TIMESTAMP,
|
||||
messageIds: [IMAGE_MESSAGE_ID],
|
||||
sourceKind: 'image' as const,
|
||||
text: RAW_SEARCHABLE,
|
||||
imageOcrText: OCR_TEXT,
|
||||
derivedSource: 'image_ocr' as const
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
function makeKnowledge() {
|
||||
return {
|
||||
search: vi.fn(async () => ({
|
||||
state: 'ready',
|
||||
evidence: syntheticKnowledgeEvidence(),
|
||||
conversationRetrieval: { totalMessages: 2, chunkCount: 1, complete: true },
|
||||
voiceCoverage: undefined
|
||||
})),
|
||||
requestCatchUp: vi.fn(() => ({ triggered: false, inProgress: false })),
|
||||
waitForIndexingComplete: vi.fn(async () => false),
|
||||
lastPassDurationMs: vi.fn(() => 0),
|
||||
beginInteractiveQuery: vi.fn(),
|
||||
endInteractiveQuery: vi.fn()
|
||||
} as never
|
||||
}
|
||||
|
||||
const NOW = new Date('2026-09-16T09:00:00+08:00')
|
||||
|
||||
describe('§2 图片文字索引 synthetic E2E(确定性,不依赖真模型)', () => {
|
||||
let knowledge: ReturnType<typeof makeKnowledge>
|
||||
let service: LocalQueryApiService
|
||||
|
||||
beforeEach(() => {
|
||||
knowledge = makeKnowledge()
|
||||
service = new LocalQueryApiService(knowledge, () => NOW)
|
||||
})
|
||||
|
||||
it('Question Tool 链路:命中图片文字的 Evidence 指向原始图片消息,且不泄露内部前缀', async () => {
|
||||
const result = await service.search({
|
||||
target: { query: GROUP_NAME },
|
||||
timeRange: { kind: 'all' },
|
||||
query: 'ChatGPT 价格',
|
||||
variants: ['ChatGPT']
|
||||
})
|
||||
|
||||
expect(result.status).toBe('completed')
|
||||
|
||||
// FOUND = YES
|
||||
expect(result.evidenceCount).toBe(1)
|
||||
expect(result.evidence).toHaveLength(1)
|
||||
const evidence = result.evidence![0]
|
||||
|
||||
// sourceKind = image(原始消息是什么)
|
||||
expect(evidence.sourceKind).toBe('image')
|
||||
// derived source = image_ocr(靠什么搜到的)
|
||||
expect(evidence.derivedSource).toBe('image_ocr')
|
||||
// OCR 片段只作命中解释
|
||||
expect(evidence.imageOcrText).toBe(OCR_TEXT)
|
||||
|
||||
// conversation scope = 技术交流群:target 把检索范围真正收敛到这一个会话
|
||||
expect(result.target).toEqual({ displayName: GROUP_NAME, type: 'group' })
|
||||
expect(evidence.conversationName).toBe(GROUP_NAME)
|
||||
expect(evidence.conversationType).toBe('group')
|
||||
expect(knowledge.search).toHaveBeenCalledTimes(2)
|
||||
for (const call of knowledge.search.mock.calls) {
|
||||
expect((call[0] as { conversationIds?: string[] }).conversationIds).toEqual([GROUP_MD5])
|
||||
}
|
||||
|
||||
// sender / createTime 来自原始消息
|
||||
expect(evidence.sender).toBe('张三')
|
||||
expect(evidence.timestamp).toBe(IMAGE_TIMESTAMP)
|
||||
|
||||
// Evidence messageRef = 原始 image message(jump target 就是它)。
|
||||
// 注意 `local:` 只是 WCDB 侧的本地 id 装饰,不属于身份本身,所以还原后是裸 id。
|
||||
const identity = decodeMessageRef(evidence.messageRef)
|
||||
expect(identity).toEqual({ conversationId: GROUP_MD5, messageId: '9001' })
|
||||
// 不能产生"OCR 消息":证据集合里不存在任何非原始消息的身份
|
||||
expect(result.evidence!.every((item) => decodeMessageRef(item.messageRef)?.messageId === '9001')).toBe(true)
|
||||
expect(result.evidence!.some((item) => decodeMessageRef(item.messageRef)?.messageId === '9002')).toBe(false)
|
||||
|
||||
// 内部前缀绝不泄露给用户(模型侧与 UI 侧都不允许)
|
||||
expect(evidence.text).not.toContain('图片文字:')
|
||||
expect(evidence.text).not.toContain('OCR:')
|
||||
expect(evidence.text).not.toContain('system-ocr')
|
||||
expect(evidence.text).toContain(OCR_TEXT)
|
||||
})
|
||||
|
||||
it('Query Agent 链路:来源语义进入 tool result,OCR 片段不进模型上下文', async () => {
|
||||
const executor = createLocalQueryToolExecutor(service)
|
||||
const responses: Array<Awaited<ReturnType<QueryAgentProvider['chatWithTools']>>> = [
|
||||
{
|
||||
success: true,
|
||||
toolCalls: [
|
||||
{
|
||||
id: 'call-1',
|
||||
name: 'search_messages',
|
||||
arguments: JSON.stringify({
|
||||
target: { query: GROUP_NAME },
|
||||
timeRange: { kind: 'all' },
|
||||
queries: ['ChatGPT 价格']
|
||||
})
|
||||
}
|
||||
]
|
||||
},
|
||||
{ success: true, data: '找到了:技术交流群发过一张 ChatGPT 价格的图片。' }
|
||||
]
|
||||
const provider: QueryAgentProvider = {
|
||||
getRuntimeConfig: () => ({
|
||||
configured: true,
|
||||
providerName: 'Fixture Provider',
|
||||
model: 'fixture-model',
|
||||
modelName: 'Fixture Model'
|
||||
}),
|
||||
chatWithTools: vi.fn(async () => responses.shift() || { success: true, data: 'done' })
|
||||
}
|
||||
|
||||
const agentResult = await new QueryAgentService(provider, executor).run(
|
||||
'技术交流群之前是不是发过 ChatGPT 价格的图片?'
|
||||
)
|
||||
|
||||
// 模型实际看到的 tool result
|
||||
const toolMessage = vi
|
||||
.mocked(provider.chatWithTools)
|
||||
.mock.calls[1]?.[0].find((message) => message.role === 'tool')
|
||||
const presented = JSON.parse(String(toolMessage?.content)) as Record<string, any>
|
||||
const presentedEvidence = presented.evidence?.[0]
|
||||
|
||||
expect(presentedEvidence.sourceKind).toBe('image')
|
||||
expect(presentedEvidence.derivedSource).toBe('image_ocr')
|
||||
// 片段的内容已经在 text 里,不再重复塞进上下文(避免无谓 token)。
|
||||
expect(presentedEvidence.imageOcrText).toBeUndefined()
|
||||
expect(presentedEvidence.text).not.toContain('图片文字:')
|
||||
// messageRef 指向原始图片消息(模型只拿到 opaque ref,看不到会话身份)。
|
||||
expect(decodeMessageRef(presentedEvidence.messageRef)).toEqual({
|
||||
conversationId: GROUP_MD5,
|
||||
messageId: '9001'
|
||||
})
|
||||
|
||||
// 暴露给 UI 的证据保留来源语义与片段
|
||||
const uiEvidence = agentResult.evidence.find((item) => item.messageRef === presentedEvidence.messageRef)
|
||||
expect(uiEvidence?.messageType).toBe('image')
|
||||
expect(uiEvidence?.derivedSource).toBe('image_ocr')
|
||||
expect(uiEvidence?.imageOcrText).toBe(OCR_TEXT)
|
||||
expect(uiEvidence?.text).not.toContain('图片文字:')
|
||||
expect(uiEvidence?.conversationName).toBe(GROUP_NAME)
|
||||
})
|
||||
})
|
||||
|
||||
describe('§3 partial coverage honesty(确定性,不依赖真模型)', () => {
|
||||
const NOT_INDEXED_KEYWORD = 'TRACE_NOT_YET_INDEXED_IMAGE'
|
||||
|
||||
function partialImageCoverage() {
|
||||
return {
|
||||
totalImageMessages: 100,
|
||||
processed: 30,
|
||||
indexed: 28,
|
||||
empty: 2,
|
||||
missing: 0,
|
||||
failed: 0,
|
||||
pending: 70,
|
||||
established: true,
|
||||
complete: false,
|
||||
countedAt: Date.parse('2026-09-16T08:00:00+08:00')
|
||||
}
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks()
|
||||
})
|
||||
|
||||
it('已处理的 30 张里搜不到关键词时,覆盖度必须带上"不能断言没有"的语义', async () => {
|
||||
const knowledge = makeKnowledge()
|
||||
// 关键:已建立的 30 张里确实没有这个关键词 → 检索结果为空。
|
||||
knowledge.search.mockImplementation(async () => ({
|
||||
state: 'ready',
|
||||
evidence: [],
|
||||
// 文字索引这一维是**完整**的(噪音):证明图片维度不会被文字维度"带过"。
|
||||
indexLatestAt: NOW.getTime(),
|
||||
sourceLatestAt: NOW.getTime(),
|
||||
conversationRetrieval: { totalMessages: 2, chunkCount: 1, complete: true },
|
||||
voiceCoverage: undefined
|
||||
}))
|
||||
const service = new LocalQueryApiService(knowledge, () => NOW)
|
||||
// 图片文字索引建立过,但只完成 30 / 100。
|
||||
service.setImageTextCoverageProvider(() => partialImageCoverage())
|
||||
|
||||
const result = await service.search({
|
||||
target: { query: GROUP_NAME },
|
||||
timeRange: { kind: 'all' },
|
||||
query: NOT_INDEXED_KEYWORD
|
||||
})
|
||||
|
||||
expect(result.status).toBe('completed')
|
||||
expect(result.evidenceCount).toBe(0)
|
||||
// 文字索引这一维是完整的(噪音),图片这一维才是缺口。
|
||||
expect(result.coverage).toEqual({ state: 'complete' })
|
||||
expect(result.imageOcrCoverage).toMatchObject({
|
||||
state: 'partial',
|
||||
totalImageMessages: 100,
|
||||
processed: 30,
|
||||
pending: 70
|
||||
})
|
||||
const summary = result.imageOcrCoverage!.summary
|
||||
expect(summary).toContain('30')
|
||||
expect(summary).toContain('100')
|
||||
expect(summary).toContain('不能因为没搜到就回答')
|
||||
|
||||
// 覆盖度必须真的进入 Query Agent 的上下文,而不是只留在 Engine 里。
|
||||
const executor = createLocalQueryToolExecutor(service)
|
||||
const responses: Array<Awaited<ReturnType<QueryAgentProvider['chatWithTools']>>> = [
|
||||
{
|
||||
success: true,
|
||||
toolCalls: [
|
||||
{
|
||||
id: 'call-1',
|
||||
name: 'search_messages',
|
||||
arguments: JSON.stringify({
|
||||
target: { query: GROUP_NAME },
|
||||
timeRange: { kind: 'all' },
|
||||
queries: [NOT_INDEXED_KEYWORD]
|
||||
})
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
success: true,
|
||||
data: '图片文字索引目前只处理 30 / 100 条图片消息,当前结果不完整,无法确认全部历史图片。'
|
||||
}
|
||||
]
|
||||
const provider: QueryAgentProvider = {
|
||||
getRuntimeConfig: () => ({
|
||||
configured: true,
|
||||
providerName: 'Fixture Provider',
|
||||
model: 'fixture-model',
|
||||
modelName: 'Fixture Model'
|
||||
}),
|
||||
chatWithTools: vi.fn(async () => responses.shift() || { success: true, data: 'done' })
|
||||
}
|
||||
const agentResult = await new QueryAgentService(provider, executor).run(
|
||||
`之前是不是有张图片写着 ${NOT_INDEXED_KEYWORD}?`
|
||||
)
|
||||
|
||||
const calls = vi.mocked(provider.chatWithTools).mock.calls
|
||||
// 提示词里写死了零结果诚实性规则(不能指望模型自己想到)。
|
||||
expect(String(calls[0]?.[0]?.[0]?.content)).toContain('imageOcrCoverage')
|
||||
const presented = JSON.parse(
|
||||
String(calls[1]?.[0].find((message) => message.role === 'tool')?.content)
|
||||
) as Record<string, any>
|
||||
expect(presented.evidenceCount).toBe(0)
|
||||
expect(presented.imageOcrCoverage).toMatchObject({
|
||||
state: 'partial',
|
||||
totalImageMessages: 100,
|
||||
processed: 30,
|
||||
pending: 70
|
||||
})
|
||||
expect(presented.imageOcrCoverage.summary).toContain('不能因为没搜到就回答')
|
||||
|
||||
// 最终回答本身必须是"覆盖不完整",不是"没有"。
|
||||
expect(agentResult.answer).toContain('30')
|
||||
expect(agentResult.answer).toContain('100')
|
||||
expect(agentResult.answer).not.toBe('没有')
|
||||
})
|
||||
|
||||
it('图片索引完整时不下发零结果约束(避免模型机械附加警告)', async () => {
|
||||
const knowledge = makeKnowledge()
|
||||
knowledge.search.mockImplementation(async () => ({
|
||||
state: 'ready',
|
||||
evidence: [],
|
||||
conversationRetrieval: { totalMessages: 2, chunkCount: 1, complete: true },
|
||||
voiceCoverage: undefined
|
||||
}))
|
||||
const service = new LocalQueryApiService(knowledge, () => NOW)
|
||||
service.setImageTextCoverageProvider(() => ({
|
||||
...partialImageCoverage(),
|
||||
processed: 100,
|
||||
indexed: 98,
|
||||
empty: 2,
|
||||
pending: 0,
|
||||
complete: true
|
||||
}))
|
||||
|
||||
const result = await service.search({
|
||||
target: { query: GROUP_NAME },
|
||||
timeRange: { kind: 'all' },
|
||||
query: NOT_INDEXED_KEYWORD
|
||||
})
|
||||
|
||||
expect(result.imageOcrCoverage?.state).toBe('complete')
|
||||
expect(result.imageOcrCoverage?.summary).not.toContain('不能因为没搜到就回答')
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,326 @@
|
||||
/**
|
||||
* 精确读消息(`query_messages`)必须能读到**图片里识别出的文字**。
|
||||
*
|
||||
* 真机回归:问「我今早给文件传输助手发的那张图片里写了什么」,Query Agent 准确找到了
|
||||
* 原始图片消息(2026/9/16 07:30:15、sender=self、type=image),却回答
|
||||
* 「查询只返回图片附件,没有取得 OCR 文字」,甚至反过来建议用户"建立图片文字索引后再查"。
|
||||
*
|
||||
* 真机派生库 + Knowledge 实测结论(CASE A):
|
||||
* L1 artifact state=indexed / char_count=17
|
||||
* L2 binding state=indexed
|
||||
* L3 Knowledge image_ocr_text 与 artifact 文本**逐字相同**,chunk 里也含该文本且指向原图 messageId
|
||||
* —— 即"索引早就建好了,只是查询路径没把它接出来"。缺口在 L4,不在 L1/L2/L3。
|
||||
*
|
||||
* 这一组测试把 L4 的契约钉死:
|
||||
* 1. 图片消息的 OCR 文本必须走 `imageOcrText` + `derivedSource=image_ocr` 独立字段;
|
||||
* 2. 证据**永远是原始图片消息**,不许为了 OCR 文本编造一条文字消息;
|
||||
* 3. `empty`(识别过没文字)与 `not_indexed`(还没索引)必须能被区分,
|
||||
* 两者都不允许模型凭想象描述图片内容。
|
||||
*/
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import type { ImageTextIndexCoverage } from '../../src/shared/image-text-index'
|
||||
import { decodeMessageRef } from '../../src/shared/local-query-api'
|
||||
|
||||
process.env.TZ = 'Asia/Shanghai'
|
||||
|
||||
const fixture = vi.hoisted(() => {
|
||||
const selfImageTime = Date.parse('2026-09-16T07:30:15+08:00')
|
||||
const otherImageTime = Date.parse('2026-09-16T08:10:00+08:00')
|
||||
return {
|
||||
selfImageTime,
|
||||
otherImageTime,
|
||||
contacts: [
|
||||
{
|
||||
m_nsUsrName: 'filehelper',
|
||||
m_nsNickName: '文件传输助手',
|
||||
md5: 'md5-filehelper',
|
||||
type: 'user' as const
|
||||
}
|
||||
],
|
||||
messages: [
|
||||
{
|
||||
id: '9001',
|
||||
localId: '9001',
|
||||
from: 'assistant',
|
||||
// 我发出的那张图:isSender = true(自我身份来自 mesDes,不是昵称)
|
||||
type: '图片',
|
||||
datetime: '2026/9/16 07:30:15',
|
||||
content: '',
|
||||
contentData: { type: 'image', md5: 'md5-self-image', datName: 'dat-self' },
|
||||
isSender: true,
|
||||
name: '我',
|
||||
createTime: Math.floor(selfImageTime / 1000)
|
||||
},
|
||||
{
|
||||
id: '9002',
|
||||
localId: '9002',
|
||||
from: 'user',
|
||||
type: '图片',
|
||||
datetime: '2026/9/16 08:10:00',
|
||||
content: '',
|
||||
contentData: { type: 'image', md5: 'md5-other-image', datName: 'dat-other' },
|
||||
isSender: false,
|
||||
name: '文件传输助手',
|
||||
createTime: Math.floor(otherImageTime / 1000)
|
||||
}
|
||||
]
|
||||
}
|
||||
})
|
||||
|
||||
vi.mock('../../src/main/services/chat-service', () => ({
|
||||
isReady: () => true,
|
||||
listContactsAsync: vi.fn(async () => fixture.contacts),
|
||||
listMessagesAsync: vi.fn(async () => fixture.messages)
|
||||
}))
|
||||
|
||||
import { LocalQueryApiService } from '../../src/main/services/local-query-api-service'
|
||||
import { createLocalQueryToolExecutor } from '../../src/main/services/local-query-tool-executor'
|
||||
import {
|
||||
QueryAgentService,
|
||||
type QueryAgentProvider
|
||||
} from '../../src/main/services/query-agent-service'
|
||||
|
||||
/** 与真实派生库同形:binding 主键 = `sourceMessageId(message)` = `local:<localId>`。 */
|
||||
const SELF_KEY = 'local:9001'
|
||||
const OTHER_KEY = 'local:9002'
|
||||
|
||||
function coverage(overrides: Partial<ImageTextIndexCoverage> = {}): ImageTextIndexCoverage {
|
||||
return {
|
||||
totalImageMessages: 2,
|
||||
processed: 2,
|
||||
indexed: 2,
|
||||
empty: 0,
|
||||
missing: 0,
|
||||
failed: 0,
|
||||
runtimeUnavailable: 0,
|
||||
pending: 0,
|
||||
established: true,
|
||||
complete: true,
|
||||
systemicFailure: false,
|
||||
countedAt: Date.parse('2026-09-16T09:00:00+08:00'),
|
||||
...overrides
|
||||
}
|
||||
}
|
||||
|
||||
type OcrFixture = Map<string, { state: string; text: string }>
|
||||
|
||||
function makeService(options: { ocr?: OcrFixture; coverage?: ImageTextIndexCoverage } = {}) {
|
||||
const knowledge = {
|
||||
search: vi.fn(async () => ({ state: 'ready', evidence: [] })),
|
||||
requestCatchUp: vi.fn(() => ({ triggered: false, inProgress: false })),
|
||||
waitForIndexingComplete: vi.fn(async () => false),
|
||||
lastPassDurationMs: vi.fn(() => 0),
|
||||
beginInteractiveQuery: vi.fn(),
|
||||
endInteractiveQuery: vi.fn()
|
||||
} as never
|
||||
const service = new LocalQueryApiService(knowledge, () => new Date('2026-09-16T09:30:00+08:00'))
|
||||
const ocr = options.ocr ?? new Map([[SELF_KEY, { state: 'indexed', text: 'ChatGPT Plus $20' }]])
|
||||
service.setImageOcrEntryProvider((_conversationId, messageId) => ocr.get(messageId))
|
||||
if (options.coverage !== undefined) {
|
||||
service.setImageTextCoverageProvider(() => options.coverage!)
|
||||
}
|
||||
return service
|
||||
}
|
||||
|
||||
const askSelfImages = {
|
||||
target: { query: '文件传输助手' },
|
||||
timeRange: { kind: 'all' },
|
||||
direction: 'to_target',
|
||||
messageTypes: ['image']
|
||||
}
|
||||
|
||||
describe('query_messages:图片消息必须携带 OCR 派生文本', () => {
|
||||
let service: LocalQueryApiService
|
||||
beforeEach(() => {
|
||||
service = makeService()
|
||||
})
|
||||
|
||||
it('我发出的图片带 OCR 文本时,走 imageOcrText + derivedSource,不混进 text', async () => {
|
||||
const result = await service.messages(askSelfImages as never)
|
||||
|
||||
expect(result.status).toBe('completed')
|
||||
expect(result.returnedCount).toBe(1)
|
||||
const message = result.messages![0]
|
||||
|
||||
// 派生文本必须单独一个字段:混进 `text` 就无法与"群友发的文字消息"区分。
|
||||
expect(message.imageOcrText).toBe('ChatGPT Plus $20')
|
||||
expect(message.derivedSource).toBe('image_ocr')
|
||||
expect(message.imageTextState).toBe('indexed')
|
||||
expect(message.text).toBeUndefined()
|
||||
expect(message.attachment).toEqual({ kind: 'image' })
|
||||
})
|
||||
|
||||
it('对方的图片不会被贴错 OCR 文本(键必须按消息身份匹配)', async () => {
|
||||
const result = await service.messages({
|
||||
...askSelfImages,
|
||||
direction: 'from_target'
|
||||
} as never)
|
||||
|
||||
expect(result.returnedCount).toBe(1)
|
||||
// OTHER_KEY 在派生库里没有绑定 → 只能是 not_indexed,绝不能借用另一条消息的文本。
|
||||
expect(result.messages![0].imageOcrText).toBeUndefined()
|
||||
expect(result.messages![0].imageTextState).toBe('not_indexed')
|
||||
})
|
||||
|
||||
it('识别过但图里没文字 → empty(已知结论),不是 not_indexed', async () => {
|
||||
const empty = makeService({ ocr: new Map([[SELF_KEY, { state: 'empty', text: '' }]]) })
|
||||
const result = await empty.messages(askSelfImages as never)
|
||||
|
||||
// `empty` 与 `not_indexed` 必须能分辨:前者是"已经知道没文字",
|
||||
// 后者是"还不知道"。把两者混起来,模型就会在没索引时断言"图里没内容"。
|
||||
expect(result.messages![0].imageTextState).toBe('empty')
|
||||
expect(result.messages![0].imageOcrText).toBeUndefined()
|
||||
})
|
||||
|
||||
it('图片文字索引未建立时,tool result 明确带上 not_built 覆盖度', async () => {
|
||||
const notBuilt = makeService({
|
||||
ocr: new Map(),
|
||||
coverage: coverage({
|
||||
processed: 0,
|
||||
indexed: 0,
|
||||
pending: 0,
|
||||
established: false,
|
||||
complete: false
|
||||
})
|
||||
})
|
||||
const result = await notBuilt.messages(askSelfImages as never)
|
||||
|
||||
expect(result.messages![0].imageTextState).toBe('not_indexed')
|
||||
expect(result.imageOcrCoverage?.state).toBe('not_built')
|
||||
expect(result.imageOcrCoverage?.summary).toContain('尚未建立')
|
||||
})
|
||||
|
||||
it('覆盖度部分完成时,summary 必须说明结果可能不完整(不许当 complete)', async () => {
|
||||
const partial = makeService({
|
||||
coverage: coverage({
|
||||
totalImageMessages: 100,
|
||||
processed: 30,
|
||||
indexed: 30,
|
||||
pending: 70,
|
||||
complete: false
|
||||
})
|
||||
})
|
||||
const result = await partial.messages(askSelfImages as never)
|
||||
|
||||
expect(result.imageOcrCoverage?.state).toBe('partial')
|
||||
expect(result.imageOcrCoverage?.summary).toContain('30')
|
||||
expect(result.imageOcrCoverage?.summary).toContain('100')
|
||||
})
|
||||
|
||||
it('普通文字消息完全不受影响(对照组)', async () => {
|
||||
const plain = makeService({ ocr: new Map() })
|
||||
const result = await plain.messages({
|
||||
target: { query: '文件传输助手' },
|
||||
timeRange: { kind: 'all' },
|
||||
direction: 'to_target',
|
||||
messageTypes: ['text']
|
||||
} as never)
|
||||
|
||||
// 图片那两条都是 image,文字查询必然是 0 条 —— 关键是**不能**因为接了 OCR 路径
|
||||
// 就凭空多出消息。
|
||||
expect(result.returnedCount).toBe(0)
|
||||
})
|
||||
})
|
||||
|
||||
describe('Query Agent:证据永远是原始图片消息', () => {
|
||||
function provider(
|
||||
responses: Array<Awaited<ReturnType<QueryAgentProvider['chatWithTools']>>>
|
||||
): QueryAgentProvider {
|
||||
return {
|
||||
getRuntimeConfig: () => ({
|
||||
configured: true,
|
||||
providerName: 'Fixture Provider',
|
||||
model: 'fixture-model',
|
||||
modelName: 'Fixture Model'
|
||||
}),
|
||||
chatWithTools: vi.fn(async () => responses.shift() || { success: true, data: 'done' })
|
||||
}
|
||||
}
|
||||
|
||||
const selfImageArgs = JSON.stringify({
|
||||
target: { query: '文件传输助手' },
|
||||
timeRange: { kind: 'all' },
|
||||
temporalBasis: { kind: 'none' },
|
||||
direction: 'to_target',
|
||||
messageTypes: ['image']
|
||||
})
|
||||
|
||||
it('模型能在 Tool Result 里读到 imageOcrText,且证据仍指向原图 messageRef', async () => {
|
||||
const configured = provider([
|
||||
{
|
||||
success: true,
|
||||
toolCalls: [{ id: 'c1', name: 'query_messages', arguments: selfImageArgs }]
|
||||
},
|
||||
{ success: true, data: '那张图片里的文字是 ChatGPT Plus $20。' }
|
||||
])
|
||||
const service = makeService()
|
||||
const result = await new QueryAgentService(
|
||||
configured,
|
||||
createLocalQueryToolExecutor(service)
|
||||
).run('我今早给文件传输助手发的那张图片里写了什么')
|
||||
|
||||
const calls = vi.mocked(configured.chatWithTools).mock.calls
|
||||
const toolResult = JSON.parse(
|
||||
String(calls[1]?.[0].find((message) => message.role === 'tool')?.content)
|
||||
) as Record<string, any>
|
||||
|
||||
// 1) 模型确实拿到了派生文本(这正是真机上缺的那一环)
|
||||
expect(toolResult.messages?.[0].imageOcrText).toBe('ChatGPT Plus $20')
|
||||
expect(toolResult.messages?.[0].derivedSource).toBe('image_ocr')
|
||||
expect(toolResult.messages?.[0].imageTextState).toBe('indexed')
|
||||
|
||||
// 2) 证据只有一条,且解出来就是**原始图片消息**(不是虚构的 OCR 文字消息)
|
||||
expect(result.evidence).toHaveLength(1)
|
||||
expect(decodeMessageRef(result.evidence![0].messageRef)).toEqual({
|
||||
conversationId: 'md5-filehelper',
|
||||
messageId: '9001'
|
||||
})
|
||||
expect(result.evidence![0].messageType).toBe('image')
|
||||
|
||||
// 3) UI 拿得到来源语义(「图片文字」标记),且 snippet 不进模型上下文之外的重复字段
|
||||
expect(result.evidence![0].derivedSource).toBe('image_ocr')
|
||||
expect(result.evidence![0].imageOcrText).toBe('ChatGPT Plus $20')
|
||||
})
|
||||
|
||||
it('系统提示词把图片文字的三态语义写死,并禁止凭空建议建立索引', async () => {
|
||||
const scripted = provider([{ success: true, data: 'ok' }])
|
||||
const service = makeService()
|
||||
void new QueryAgentService(scripted, createLocalQueryToolExecutor(service)).run(
|
||||
'我今早给文件传输助手发的那张图片里写了什么'
|
||||
)
|
||||
|
||||
const systemPrompt = String(vi.mocked(scripted.chatWithTools).mock.calls[0]?.[0]?.[0]?.content)
|
||||
expect(systemPrompt).toContain('imageOcrText')
|
||||
// 三态必须分别说清楚
|
||||
expect(systemPrompt).toContain('indexed')
|
||||
expect(systemPrompt).toContain('empty')
|
||||
expect(systemPrompt).toContain('not_indexed')
|
||||
// OCR 不是看图:empty 时不许猜画面
|
||||
expect(systemPrompt).toContain('OCR 不是看图')
|
||||
// 不许无条件建议"先建立图片文字索引再查"
|
||||
expect(systemPrompt).toContain('建立图片文字索引')
|
||||
})
|
||||
|
||||
it('索引已建好的情况下,模型不会拿到任何"还没建立"的误导信号', async () => {
|
||||
const configured = provider([
|
||||
{
|
||||
success: true,
|
||||
toolCalls: [{ id: 'c1', name: 'query_messages', arguments: selfImageArgs }]
|
||||
},
|
||||
{ success: true, data: '那张图里有 ChatGPT Plus $20。' }
|
||||
])
|
||||
const service = makeService({ coverage: coverage() })
|
||||
await new QueryAgentService(configured, createLocalQueryToolExecutor(service)).run(
|
||||
'我今早给文件传输助手发的那张图片里写了什么'
|
||||
)
|
||||
|
||||
const calls = vi.mocked(configured.chatWithTools).mock.calls
|
||||
const toolResult = JSON.parse(
|
||||
String(calls[1]?.[0].find((message) => message.role === 'tool')?.content)
|
||||
) as Record<string, any>
|
||||
|
||||
// 覆盖度是 complete 且带了派生文本 → 模型没有任何理由说"没有取得 OCR 文字"。
|
||||
expect(toolResult.imageOcrCoverage?.state).toBe('complete')
|
||||
expect(toolResult.messages?.[0].imageOcrText).toBe('ChatGPT Plus $20')
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,228 @@
|
||||
/**
|
||||
* Query Agent 的**说话人方向**语义(self sender)。
|
||||
*
|
||||
* 真机回归:问「我给文件传输助手发了什么图片」,planner 第一次用了 `from_target`
|
||||
* (= 对方发来),拿到 0 条后**回头问用户"是不是方向搞错了"**,而不是自己改向重查。
|
||||
*
|
||||
* 代码事实:direction 的词表是**以目标会话为参照**的 ——
|
||||
* `to_target` = 我发出的(self sender),`from_target` = 对方发来的。
|
||||
* 所以这不是"缺一个方向取值",而是 planner 选错了值 + 0 结果后没有利用既有重试机制。
|
||||
*/
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
process.env.TZ = 'Asia/Shanghai'
|
||||
|
||||
const fixture = vi.hoisted(() => {
|
||||
const selfImageTime = Date.parse('2026-09-10T10:00:00+08:00')
|
||||
const otherImageTime = Date.parse('2026-09-11T11:00:00+08:00')
|
||||
return {
|
||||
selfImageTime,
|
||||
otherImageTime,
|
||||
contacts: [
|
||||
{
|
||||
m_nsUsrName: 'filehelper',
|
||||
m_nsNickName: '文件传输助手',
|
||||
md5: 'md5-filehelper',
|
||||
type: 'user' as const
|
||||
}
|
||||
],
|
||||
messages: [
|
||||
{
|
||||
id: '1001',
|
||||
localId: '1001',
|
||||
from: 'assistant',
|
||||
// 我发出的图片:isSender = true(自我身份来自 mesDes,不是昵称)
|
||||
type: '图片',
|
||||
datetime: '2026/9/10 10:00:00',
|
||||
content: '',
|
||||
contentData: { type: 'image', md5: 'md5-self-image', datName: 'dat-self' },
|
||||
isSender: true,
|
||||
name: '我',
|
||||
createTime: Math.floor(selfImageTime / 1000)
|
||||
},
|
||||
{
|
||||
id: '1002',
|
||||
localId: '1002',
|
||||
from: 'user',
|
||||
// 对方发来的图片
|
||||
type: '图片',
|
||||
datetime: '2026/9/11 11:00:00',
|
||||
content: '',
|
||||
contentData: { type: 'image', md5: 'md5-other-image', datName: 'dat-other' },
|
||||
isSender: false,
|
||||
name: '文件传输助手',
|
||||
createTime: Math.floor(otherImageTime / 1000)
|
||||
}
|
||||
]
|
||||
}
|
||||
})
|
||||
|
||||
vi.mock('../../src/main/services/chat-service', () => ({
|
||||
isReady: () => true,
|
||||
listContactsAsync: vi.fn(async () => fixture.contacts),
|
||||
listMessagesAsync: vi.fn(async () => fixture.messages)
|
||||
}))
|
||||
|
||||
import { LocalQueryApiService } from '../../src/main/services/local-query-api-service'
|
||||
import { createLocalQueryToolExecutor } from '../../src/main/services/local-query-tool-executor'
|
||||
import {
|
||||
QueryAgentService,
|
||||
type QueryAgentProvider
|
||||
} from '../../src/main/services/query-agent-service'
|
||||
|
||||
function makeKnowledge() {
|
||||
return {
|
||||
search: vi.fn(async () => ({ state: 'ready', evidence: [] })),
|
||||
requestCatchUp: vi.fn(() => ({ triggered: false, inProgress: false })),
|
||||
waitForIndexingComplete: vi.fn(async () => false),
|
||||
lastPassDurationMs: vi.fn(() => 0),
|
||||
beginInteractiveQuery: vi.fn(),
|
||||
endInteractiveQuery: vi.fn()
|
||||
} as never
|
||||
}
|
||||
|
||||
function makeService(): LocalQueryApiService {
|
||||
return new LocalQueryApiService(makeKnowledge(), () => new Date('2026-09-16T09:00:00+08:00'))
|
||||
}
|
||||
|
||||
const queryArgs = (direction: string): string =>
|
||||
JSON.stringify({
|
||||
target: { query: '文件传输助手' },
|
||||
timeRange: { kind: 'all' },
|
||||
temporalBasis: { kind: 'none' },
|
||||
direction,
|
||||
messageTypes: ['image']
|
||||
})
|
||||
|
||||
describe('direction 语义:to_target = 我发出的(self sender)', () => {
|
||||
let service: LocalQueryApiService
|
||||
beforeEach(() => {
|
||||
service = makeService()
|
||||
})
|
||||
|
||||
it('to_target 只返回我发出的图片,排除对方发来的', async () => {
|
||||
const result = await service.messages({
|
||||
target: { query: '文件传输助手' },
|
||||
timeRange: { kind: 'all' },
|
||||
direction: 'to_target',
|
||||
messageTypes: ['image']
|
||||
} as never)
|
||||
|
||||
expect(result.status).toBe('completed')
|
||||
expect(result.messages?.map((message) => message.messageRef)).toHaveLength(1)
|
||||
expect(result.messages?.[0].direction).toBe('to_target')
|
||||
expect(result.messages?.[0].sender).toBe('我')
|
||||
expect(result.query?.direction).toBe('to_target')
|
||||
})
|
||||
|
||||
it('from_target 只返回对方发来的图片', async () => {
|
||||
const result = await service.messages({
|
||||
target: { query: '文件传输助手' },
|
||||
timeRange: { kind: 'all' },
|
||||
direction: 'from_target',
|
||||
messageTypes: ['image']
|
||||
} as never)
|
||||
|
||||
expect(result.messages).toHaveLength(1)
|
||||
expect(result.messages?.[0].direction).toBe('from_target')
|
||||
expect(result.messages?.[0].sender).toBe('文件传输助手')
|
||||
})
|
||||
})
|
||||
|
||||
describe('Query Agent:方向选反后必须自己改向重查,而不是问用户', () => {
|
||||
function provider(
|
||||
responses: Array<Awaited<ReturnType<QueryAgentProvider['chatWithTools']>>>
|
||||
): QueryAgentProvider {
|
||||
return {
|
||||
getRuntimeConfig: () => ({
|
||||
configured: true,
|
||||
providerName: 'Fixture Provider',
|
||||
model: 'fixture-model',
|
||||
modelName: 'Fixture Model'
|
||||
}),
|
||||
chatWithTools: vi.fn(async () => responses.shift() || { success: true, data: 'done' })
|
||||
}
|
||||
}
|
||||
|
||||
it('系统提示词把「我给 X 发」明确映射到 to_target,并禁止因此反问用户', () => {
|
||||
const scripted = provider([{ success: true, data: 'ok' }])
|
||||
const service = makeService()
|
||||
const executor = createLocalQueryToolExecutor(service)
|
||||
void new QueryAgentService(scripted, executor).run('我给文件传输助手发了什么图片')
|
||||
|
||||
const systemPrompt = String(vi.mocked(scripted.chatWithTools).mock.calls[0]?.[0]?.[0]?.content)
|
||||
expect(systemPrompt).toContain('说话人是我 → to_target')
|
||||
expect(systemPrompt).toContain('说话人是对方 → from_target')
|
||||
// 提示词里明确禁止"因为方向可能错就反问用户"
|
||||
expect(systemPrompt).toContain('要不要换个方向')
|
||||
})
|
||||
|
||||
it('第一次用错方向得到 0 条 → tool result 明确要求改向重查;第二次查对 → 命中我发出的图片', async () => {
|
||||
/**
|
||||
* 只保留"我发出的"那一张:这样用错方向(from_target = 对方发来)必然 0 条,
|
||||
* 才能真实复现"第一次查反了"的场景。
|
||||
*/
|
||||
const originalMessages = fixture.messages
|
||||
fixture.messages = [originalMessages[0]] as typeof fixture.messages
|
||||
const configured = provider([
|
||||
// 第一次:方向选反(对方发来)
|
||||
{ success: true, toolCalls: [{ id: 'c1', name: 'query_messages', arguments: queryArgs('from_target') }] },
|
||||
// 第二次:改向(我发出的)
|
||||
{ success: true, toolCalls: [{ id: 'c2', name: 'query_messages', arguments: queryArgs('to_target') }] },
|
||||
{ success: true, data: '你给文件传输助手发过 1 张图片。' }
|
||||
])
|
||||
const service = makeService()
|
||||
const executor = createLocalQueryToolExecutor(service)
|
||||
const result = await new QueryAgentService(configured, executor).run(
|
||||
'我给文件传输助手发了什么图片'
|
||||
)
|
||||
|
||||
const calls = vi.mocked(configured.chatWithTools).mock.calls
|
||||
const firstToolResult = JSON.parse(
|
||||
String(calls[1]?.[0].find((message) => message.role === 'tool')?.content)
|
||||
) as Record<string, any>
|
||||
|
||||
// 0 条确实发生了(说明 fixture 的方向过滤是真的在起作用)
|
||||
expect(firstToolResult.returnedCount).toBe(0)
|
||||
// 重试提示必须点明"方向选反"这件事,并且**禁止**反问用户
|
||||
expect(String(firstToolResult._agent?.note)).toContain('方向选反')
|
||||
expect(String(firstToolResult._agent?.note)).toContain('不要问用户')
|
||||
// 重试通道仍然开放(这正是既有 zero-result retry 机制)
|
||||
expect(firstToolResult._agent?.note).toContain('to_target')
|
||||
|
||||
// 取**最后一条** tool 消息:第三次调用的上下文里已经有两次 tool result。
|
||||
const secondToolResult = JSON.parse(
|
||||
String(
|
||||
calls[2]?.[0].filter((message) => message.role === 'tool').at(-1)?.content
|
||||
)
|
||||
) as Record<string, any>
|
||||
expect(secondToolResult.returnedCount).toBe(1)
|
||||
expect(secondToolResult.messages?.[0].direction).toBe('to_target')
|
||||
|
||||
// 最终答案基于第二次(正确方向)的结果
|
||||
expect(result.answer).toContain('发过')
|
||||
expect(result.toolCallCount).toBe(2)
|
||||
|
||||
fixture.messages = originalMessages
|
||||
})
|
||||
|
||||
it('正常情况下第一次就查对:一次 tool call 命中,不需要重试', async () => {
|
||||
const configured = provider([
|
||||
{ success: true, toolCalls: [{ id: 'c1', name: 'query_messages', arguments: queryArgs('to_target') }] },
|
||||
{ success: true, data: '你给文件传输助手发过 1 张图片。' }
|
||||
])
|
||||
const service = makeService()
|
||||
const executor = createLocalQueryToolExecutor(service)
|
||||
const result = await new QueryAgentService(configured, executor).run(
|
||||
'我给文件传输助手发了什么图片'
|
||||
)
|
||||
|
||||
expect(result.toolCallCount).toBe(1)
|
||||
const calls = vi.mocked(configured.chatWithTools).mock.calls
|
||||
const toolResult = JSON.parse(
|
||||
String(calls[1]?.[0].find((message) => message.role === 'tool')?.content)
|
||||
) as Record<string, any>
|
||||
expect(toolResult.returnedCount).toBe(1)
|
||||
expect(toolResult.messages?.[0].direction).toBe('to_target')
|
||||
})
|
||||
})
|
||||
Reference in New Issue
Block a user