feat: 新增微信图片文字索引与问问微信图片检索能力

This commit is contained in:
电摇小子
2026-09-16 10:42:52 +08:00
parent 24399f1d70
commit b8f08d54c8
46 changed files with 6938 additions and 75 deletions
+48 -12
View File
@@ -377,7 +377,7 @@ describe('AskWechatService — 桌面问问微信主路径', () => {
expect(result.diagnostics.outcome).toBe('invalid_question')
})
it('生产日志只记录形态字段,不含回答内容', async () => {
it('本地日志必须包含问题与模型回答原文,方便回放排查', async () => {
const logs: AskWechatLogRecord[] = []
const { provider } = providerFactory([answer('BOBO 最近在准备搬家,提到了房租和押金。')])
const service = new AskWechatService(new QueryAgentService(provider, vi.fn()), {
@@ -388,17 +388,53 @@ describe('AskWechatService — 桌面问问微信主路径', () => {
await service.ask(request('BOBO 最近在忙什么'))
expect(logs).toHaveLength(1)
expect(Object.keys(logs[0].details ?? {}).sort()).toEqual([
'entry',
'model',
'modelCallCount',
'outcome',
'provider',
'toolCallCount',
'tools',
'totalMs'
])
expect(JSON.stringify(logs)).not.toContain('准备搬家')
// 形态字段仍然一个不少(排查时既要知道"问了什么",也要知道"跑了什么工具、多久")。
for (const key of ['entry', 'model', 'modelCallCount', 'outcome', 'provider', 'toolCallCount', 'tools', 'totalMs']) {
expect(logs[0].details).toHaveProperty(key)
}
// 措辞回归:真机上曾出现"图片已识别出文字却答'没有取得 OCR 文字'",
// 当时日志里只有工具名与次数,无法判断是索引没建还是链路没接上。
// 现在问答原文进入**本机**日志(不上传、不进遥测),可以直接回放。
expect(logs[0].details?.question).toBe('BOBO 最近在忙什么')
expect(String(logs[0].details?.answer)).toContain('准备搬家')
})
it('图片 OCR 的两条结构化事实进入诊断(不重复正文)', async () => {
const logs: AskWechatLogRecord[] = []
const { provider } = providerFactory([answer('那张图里有价格文字。')])
const runtime = new QueryAgentService(provider, vi.fn())
// 直接在 Runtime 结果里放一条带图片 OCR 的 trace,验证聚合口径。
vi.spyOn(runtime, 'run').mockResolvedValue({
question: '图里写了什么',
provider: 'Fixture',
model: 'fixture',
modelCallCount: 1,
toolCallCount: 1,
toolTotalMs: 5,
totalMs: 10,
modelDurationsMs: [],
modelDiagnostics: [],
answer: '那张图里有价格文字。',
traces: [
{
toolName: 'query_messages',
input: {},
durationMs: 5,
status: 'completed',
imageOcrTextCount: 3,
imageOcrCoverageState: 'partial'
}
]
} as never)
const service = new AskWechatService(runtime, {
entry: 'desktop',
log: (record) => logs.push(record)
})
await service.ask(request('图里写了什么'))
expect(logs[0].details?.imageOcrTextCount).toBe(3)
expect(logs[0].details?.imageOcrCoverageState).toBe('partial')
})
it('forgetConversation 清掉指定会话的澄清上下文', async () => {
@@ -0,0 +1,137 @@
/**
* §6 / §7 / §18:图片文字索引覆盖度在 Query Agent 这一层的语义。
*
* 这里能确定性验证的是"覆盖度**真的进到了模型上下文**",而不是只做了个 UI 数字:
* - 系统提示词把 imageOcrCoverage 定义成**独立于文字索引**的维度,并禁止凭零结果下"没有";
* - tool result 透传里真的带着 imageOcrCoverage(模型能看见比例与结论句)。
*
* 真模型最终怎么说话不在本文件断言范围内(那需要真实模型与真实数据)。
*/
import { describe, expect, it, vi } from 'vitest'
import {
QueryAgentService,
type QueryAgentProvider
} from '../../src/main/services/query-agent-service'
function provider(
responses: Array<Awaited<ReturnType<QueryAgentProvider['chatWithTools']>>>,
configured = true
): QueryAgentProvider {
return {
getRuntimeConfig: () => ({
configured,
providerName: 'Fixture Provider',
model: 'fixture-model',
modelName: 'Fixture Model'
}),
chatWithTools: vi.fn(async () => responses.shift() || { success: true, data: 'done' })
}
}
/** 与 `buildImageOcrCoverage` 在 partial 时产出的结论句保持一致。 */
const PARTIAL_SUMMARY =
'图片文字索引只完成 30%(30 / 100 条图片消息),当前图片搜索结果可能不完整。涉及图片、截图、海报里的文字的问题,当前不能因为没搜到就回答"没有"。'
function searchOnce(execute: ReturnType<typeof vi.fn>, finalAnswer: string): QueryAgentProvider {
return provider([
{
success: true,
toolCalls: [
{
id: 'call-1',
name: 'search_messages',
arguments: JSON.stringify({
target: { query: '技术交流群' },
timeRange: { kind: 'all' },
queries: ['ChatGPT 价格']
})
}
]
},
{ success: true, data: finalAnswer }
])
}
describe('Query Agent 的图片文字索引覆盖度语义', () => {
it('系统提示词把图片覆盖度声明为独立维度,并禁止凭零结果断言"没有"', async () => {
const execute = vi.fn(async () => ({ status: 'completed', evidenceCount: 0, evidence: [] }))
const configured = searchOnce(execute, 'ok')
await new QueryAgentService(configured, execute).run('图片文字索引相关的问题')
const systemPrompt = String(vi.mocked(configured.chatWithTools).mock.calls[0]?.[0]?.[0]?.content)
expect(systemPrompt).toContain('imageOcrCoverage')
expect(systemPrompt).toContain('独立于文字索引')
expect(systemPrompt).toContain('not_built')
// 零结果诚实性必须写死在提示词里,不能指望模型自己想到。
expect(systemPrompt).toContain('绝不能')
})
it('partial 覆盖度随 tool result 进入模型上下文,而不是只留在 UI 里', async () => {
const execute = vi.fn(async () => ({
status: 'completed',
coverage: { state: 'partial' },
evidenceCount: 0,
evidence: [],
imageOcrCoverage: {
state: 'partial',
totalImageMessages: 100,
processed: 30,
indexed: 28,
empty: 2,
missing: 0,
failed: 0,
pending: 70,
countedAtLabel: '09-15 20:13',
summary: PARTIAL_SUMMARY
}
}))
const configured = searchOnce(execute, '图片文字索引只完成 30%,暂时无法确认。')
await new QueryAgentService(configured, execute).run(
'技术交流群之前是不是发过 ChatGPT 价格的图片?'
)
const toolMessage = vi
.mocked(configured.chatWithTools)
.mock.calls[1]?.[0].find((message) => message.role === 'tool')
const presented = JSON.parse(String(toolMessage?.content)) as Record<string, any>
expect(presented.imageOcrCoverage).toMatchObject({
state: 'partial',
totalImageMessages: 100,
processed: 30,
indexed: 28,
empty: 2,
pending: 70
})
expect(presented.imageOcrCoverage.summary).toContain('不能因为没搜到就回答')
})
it('complete 覆盖度不下发零结果约束(避免模型机械附加警告)', async () => {
const execute = vi.fn(async () => ({
status: 'completed',
coverage: { state: 'complete' },
evidenceCount: 1,
evidence: [{ messageRef: 'opaque', timestamp: 1, sender: '张三', sourceKind: 'image', text: 'ChatGPT Plus $20' }],
imageOcrCoverage: {
state: 'complete',
totalImageMessages: 100,
processed: 100,
indexed: 90,
empty: 10,
missing: 0,
failed: 0,
pending: 0,
summary: '图片文字索引已覆盖所统计的全部 100 条图片消息。'
}
}))
const configured = searchOnce(execute, '找到了。')
await new QueryAgentService(configured, execute).run('技术交流群发过 ChatGPT 价格的图片吗')
const toolMessage = vi
.mocked(configured.chatWithTools)
.mock.calls[1]?.[0].find((message) => message.role === 'tool')
const presented = JSON.parse(String(toolMessage?.content)) as Record<string, any>
expect(presented.imageOcrCoverage.state).toBe('complete')
expect(presented.imageOcrCoverage.summary).not.toContain('不能因为没搜到就回答')
})
})
+165
View File
@@ -0,0 +1,165 @@
/**
* 图片消息计数 / 增量的**标识符与列名**回归测试。
*
* 真机反馈「检测到图片消息为 0 → 无法统计(未找到该会话的消息表)」,根因有两个,
* 都在这里钉死:
*
* 1. **标识符混淆**:`contact.md5` 是 `md5(wxid)` 的哈希,而原生接口
* (`wcdbGetMessageTableStats` / `wcdbGetMessages`)要的是**原始 username**。
* 把 md5 直接当 username 传,原生侧匹配不到任何表 —— 既有读消息路径一直有做这层转换
* (`wechat-db.chatMd5ToUsername`、`chat-service` 的 `getUsernameByMd5`),统计路径漏了。
* 2. **类型列名硬编码**:不同微信版本的列名不统一(项目读行时用的是别名列表),
* 把 `local_type` 写进 WHERE 会在列名不同的库上抛错。
*/
import { createHash } from 'node:crypto'
import { describe, expect, it, vi } from 'vitest'
import { Wcdb4Client } from '../../src/main/wcdb4-client'
const USERNAME = 'wxid_real_user'
const ROOM = '12345678@chatroom'
const MD5_OF_USERNAME = createHash('md5').update(USERNAME).digest('hex')
const MD5_OF_ROOM = createHash('md5').update(ROOM).digest('hex')
type ClientOptions = {
sessions?: { username: string }[]
chatTables?: { name: string; db_number: string }[] | null
columns?: string[]
count?: number
maxLocalId?: number
identifierLog?: string[]
}
function makeClient(options: ClientOptions): Wcdb4Client {
const identifierLog = options.identifierLog ?? []
return Object.assign(Object.create(Wcdb4Client.prototype), {
cachedSessions: (options.sessions ?? []).map((session) => ({ ...session })),
cachedChatTables: options.chatTables ?? null,
messageTypeColumnCache: new Map<string, string | null>(),
messageTypeColumnCandidates: [
'local_type',
'localType',
'msg_type',
'msgType',
'message_type',
'messageType',
'type',
'WCDB_CT_local_type'
],
// 断言点:走到消息表查找时用的标识符必须是 username,不是 md5。
listMessageStoresAsync: vi.fn(async (identifier: string) => {
identifierLog.push(identifier)
return [{ tableName: 'Msg_fixture', dbPath: 'C:/fixture/message_0.db' }]
}),
// 聚合查询的返回(真实实现由 pickValue 解析)。
callJsonAsync: vi.fn(async () => [
{ image_count: options.count ?? 0, image_max_local_id: options.maxLocalId ?? 0 }
]),
readMessageColumns: vi.fn(() =>
(options.columns ?? []).map((name) => ({ name, declaration: 'INTEGER' }))
),
wcdbGetMessageTableStats: vi.fn(async () => 0),
wcdbExecQuery: vi.fn(async () => 0)
}) as Wcdb4Client
}
describe('图片消息统计:会话标识符必须是 username 而不是 md5', () => {
it('从 session 能把 md5 反解成 username,并把它交给消息表查找', async () => {
const identifierLog: string[] = []
const client = makeClient({
sessions: [{ username: USERNAME }],
columns: ['local_type'],
count: 7,
identifierLog
})
const result = await client.countImageMessagesAsync(MD5_OF_USERNAME)
expect(result).toEqual({ count: 7, typeColumn: 'local_type' })
// ★ 核心回归断言:绝不能把 md5 当 username 传下去。
expect(identifierLog).toEqual([USERNAME])
expect(identifierLog).not.toContain(MD5_OF_USERNAME)
})
it('群聊同样走 md5 → username', async () => {
const identifierLog: string[] = []
const client = makeClient({
sessions: [{ username: ROOM }],
columns: ['local_type'],
count: 2,
identifierLog
})
await client.countImageMessagesAsync(MD5_OF_ROOM)
expect(identifierLog).toEqual([ROOM])
})
it('不在 session 列表、只以 Chat_<md5> 表存在的会话也能反解', async () => {
const identifierLog: string[] = []
const client = makeClient({
sessions: [],
chatTables: [{ name: `Chat_${MD5_OF_ROOM}`, db_number: ROOM }],
columns: ['local_type'],
count: 3,
identifierLog
})
const result = await client.countImageMessagesAsync(MD5_OF_ROOM)
expect(result.count).toBe(3)
expect(identifierLog).toEqual([ROOM])
})
it('增量水位使用同一个标识符(否则水位永远拿不到、退化成每轮重扫)', async () => {
const identifierLog: string[] = []
const client = makeClient({
sessions: [{ username: USERNAME }],
columns: ['local_type'],
count: 7,
maxLocalId: 42,
identifierLog
})
const watermark = await client.imageConversationWatermarkAsync(MD5_OF_USERNAME)
expect(watermark).toEqual({ count: 7, maxLocalId: 42 })
expect(identifierLog).toEqual([USERNAME])
})
})
describe('图片消息统计:类型列名随版本变化', () => {
it('列名不是 local_type 时照样能统计,并把真实列名透出', async () => {
const client = makeClient({
sessions: [{ username: USERNAME }],
columns: ['msg_type'],
count: 5
})
const result = await client.countImageMessagesAsync(MD5_OF_USERNAME)
expect(result).toEqual({ count: 5, typeColumn: 'msg_type' })
})
it('探测不到类型列时明确失败,而不是静默返回 0', async () => {
const client = makeClient({ sessions: [{ username: USERNAME }], columns: [] })
const result = await client.countImageMessagesAsync(MD5_OF_USERNAME)
// "数不出来"必须是 null + 原因,不能是 0。
expect(result.count).toBeNull()
expect(result.error).toBe('消息表缺少可识别的消息类型列')
})
it('一张表都没匹配到时给出可诊断的原因', async () => {
const client = makeClient({ sessions: [{ username: USERNAME }], columns: ['local_type'] })
Object.assign(client, { listMessageStoresAsync: vi.fn(async () => []) })
const result = await client.countImageMessagesAsync(MD5_OF_USERNAME)
expect(result.count).toBeNull()
expect(result.error).toBe('未找到该会话的消息表')
})
it('原生统计通道不可用时说明是环境问题,与 0 张区分开', async () => {
const client = makeClient({ sessions: [{ username: USERNAME }], columns: ['local_type'] })
Object.assign(client, { wcdbExecQuery: null })
const result = await client.countImageMessagesAsync(MD5_OF_USERNAME)
expect(result.count).toBeNull()
expect(result.error).toBe('当前数据服务不支持消息表统计')
})
})