feat: 新增微信图片文字索引与问问微信图片检索能力

This commit is contained in:
电摇小子
2026-09-16 10:42:52 +08:00
parent 24399f1d70
commit b8f08d54c8
46 changed files with 6938 additions and 75 deletions
+144 -2
View File
@@ -150,6 +150,8 @@ function nextGetMessagesRequestId(): string {
import type { AppLogEntry } from '../shared/app-log'
import { appUpdateService } from './services/app-update-service'
import { clearCache, getCacheSummary, openKnowledgeDirectory } from './services/cache-service'
import { imageTextIndexService } from './services/image-text-index-service'
import type { ImageTextIndexStartOptions } from '../shared/image-text-index'
import type { CacheClearScope } from './services/cache-service'
import { configureRecallArchive, RecallArchiveMonitor } from './services/recall-archive-service'
import { VideoAssetService } from './video-asset-service'
@@ -637,8 +639,94 @@ app.whenReady().then(async () => {
voiceRecognition.onTranscriptUpdate((update) =>
knowledgeSearchService?.indexVoiceTranscript(update)
)
/**
* 确保图片解密服务可用(按需创建,与 `db:getImage` 冷路径同一套构造方式)。
*
* 提取成显式入口是因为原来它只存在于 `db:getImage` 的闭包里,
* 别的需要解密的路径(图片文字索引回填)拿不到、只能拿到 `null`。
*/
function ensureImageDecryptService(): ImageDecryptService | null {
if (imageDecryptService) return imageDecryptService
const { xorKey, aesKey } = getConfiguredImageKeys()
if (!aesKey) return null
imageDecryptService = new ImageDecryptService(
xorKey,
aesKey,
chat.getChatDb()?.getWcdb4Client(),
loadSettings().dbRoot
)
return imageDecryptService
}
// 图片文字索引(本地 System OCR 派生文本)。
// 与语音转写完全同构:派生文本在 main 进程解析后贴到消息上,Knowledge 侧只消费结果。
knowledgeSearchService.setImageOcrResolver((conversationId, messageId) =>
imageTextIndexService.getConversationOcr(conversationId).get(messageId)
)
/**
* 图片文字索引需要解密图片。
*
* 原先这个依赖直接读 `imageDecryptService`,而它**只在 `db:getImage`(用户点开某张图)
* 里才懒加载** —— 于是全量回填在用户没点开过任何图片时拿到 `null`,
* 45,479 张图片全部被记成 `decrypt_failed`(见事故报告)。
* 这里改成显式的"按需确保",凡是需要解密的路径都能自己把它建起来。
*/
imageTextIndexService.bind({
databaseRoot: join(app.getPath('userData'), 'image-text-index'),
resolveAccountId: () =>
chat.isReady()
? String(chat.getSelfAccountInfo()?.wxid || chat.getCurrentAccountRoot() || '')
: '',
resolveAccountRoot: () => chat.getCurrentAccountRoot() || loadSettings().dbRoot || '',
listContacts: async () => {
const contacts = await chat.listContactsAsync()
return contacts.map((contact) => ({
md5: contact.md5,
m_nsUsrName: contact.m_nsUsrName,
type: contact.type
}))
},
listMessages: (conversationId) => chat.listMessagesAsync(conversationId),
countConversationImages: (conversationId, sinceMs) =>
chat.countImageMessagesAsync(conversationId, sinceMs),
imageWatermark: (conversationId, sinceMs) =>
chat.imageConversationWatermarkAsync(conversationId, sinceMs),
decryptService: () => ensureImageDecryptService(),
capability: () => systemOcrService.getCapability(),
recognize: async (imageDataUrl) => {
const result = await systemOcrService.recognize({ imageDataUrl })
return {
success: result.success,
text: result.text,
language: result.language,
...(result.errorCode ? { errorCode: result.errorCode } : {})
}
},
// 会话图片全部处理完 → 重建该会话索引,OCR 文本才可被 search_messages 检索。
onConversationIndexed: (conversationId) =>
knowledgeSearchService?.indexImageOcr(conversationId) ?? Promise.resolve()
})
aiSearchPipelineService = new AiSearchPipelineService(knowledgeSearchService, aiProviderService)
localQueryApiService = new LocalQueryApiService(knowledgeSearchService)
// 图片文字索引覆盖度是**独立覆盖维度**:接到 search_messages 的 tool result 上,
// 让 Query Agent 在图片索引没做完时不能凭 0 条证据断言"没有"。
localQueryApiService.setImageTextCoverageProvider(() =>
imageTextIndexService.getCoverageSnapshot()
)
/**
* 单条图片消息的 OCR 派生文本也要接到精确读消息路径上。
*
* 与覆盖度是**两件不同的事**:覆盖度回答"索引建了多少",这里回答
* "这一条图片已经识别出的文字是什么"。只接前者的话,图片索引建好了模型也读不到正文,
* 只能看到一个空的 `attachment` —— 真机上就是这么把"图片里有 ChatGPT 价格"
* 答成"没有取得 OCR 文字"的。
*
* 只读派生库,**不触发 OCR / 解密 / 读原图**。
*/
localQueryApiService.setImageOcrEntryProvider((conversationId, messageId) =>
imageTextIndexService.getConversationOcr(conversationId).get(messageId)
)
setLocalQueryApiService(localQueryApiService)
// Query Agent:生产 Runtime 只在这里实例化一次,桌面问问微信与 Agent Hub 共用同一个实例。
queryAgentService = new QueryAgentService(
@@ -660,6 +748,11 @@ app.whenReady().then(async () => {
if (!window.isDestroyed()) window.webContents.send('knowledge:status', status)
}
})
imageTextIndexService.onStatusChange((status) => {
for (const window of BrowserWindow.getAllWindows()) {
if (!window.isDestroyed()) window.webContents.send('image-text-index:status', status)
}
})
voiceRecognition.modelManager.setProgressListener((status) => {
for (const window of BrowserWindow.getAllWindows()) {
if (!window.isDestroyed()) window.webContents.send('voice:modelProgress', status)
@@ -748,12 +841,25 @@ app.whenReady().then(async () => {
ipcMain.handle('cache:getSummary', () => getCacheSummary())
ipcMain.handle('cache:openKnowledgeDirectory', () => openKnowledgeDirectory())
ipcMain.handle('cache:clear', async (_, scope: CacheClearScope) => {
const allowedScopes: CacheClearScope[] = ['bootstrap', 'electron', 'knowledge', 'all']
const allowedScopes: CacheClearScope[] = [
'bootstrap',
'electron',
'knowledge',
'image-text-index',
'all'
]
if (!allowedScopes.includes(scope)) return getCacheSummary()
imageDecryptService = null
// 这里刻意**不再**提前 resetAccount():清理钩子需要先读到派生库里的
// "哪些会话有 OCR 派生文本",才能把这些会话的 Knowledge 索引一起失效。
// 句柄由 beforeClearImageTextIndex 内部的 clear() 自己关闭(删文件前)。
return clearCache(scope, {
beforeClearKnowledge: () =>
knowledgeSearchService?.prepareForCacheClear() || Promise.resolve()
knowledgeSearchService?.prepareForCacheClear() || Promise.resolve(),
beforeClearImageTextIndex: async () => {
await imageTextIndexService.prepareForCacheClear()
imageTextIndexService.resetAccount()
}
})
})
@@ -841,6 +947,8 @@ app.whenReady().then(async () => {
.catch((error) => console.warn('[WCDB4] message cursor warmup failed:', error))
}
imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
console.log(
`[WCDB4] db:init ready sessions=${sessions.length} monitoring=${monitoring} cost=${Date.now() - startedAt}ms`
)
@@ -1020,6 +1128,8 @@ app.whenReady().then(async () => {
aesKey: result.aesKey
})
if (saved.success) imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
return {
...result,
success: saved.success,
@@ -1040,6 +1150,8 @@ app.whenReady().then(async () => {
ipcMain.handle('image:saveConfig', (_, request: SaveImageKeyRequest) => {
const result = imageKeyConfigService.save(request)
if (result.success) imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
return result
})
@@ -1050,6 +1162,8 @@ app.whenReady().then(async () => {
ipcMain.handle('image:clearConfig', () => {
const result = imageKeyConfigService.clear()
if (result.success) imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
return result
})
@@ -1338,6 +1452,32 @@ app.whenReady().then(async () => {
if (!knowledgeSearchService) throw new Error('本地知识库服务尚未初始化')
return knowledgeSearchService.cancelCurrentAccountIndex()
})
// ---- 图片文字索引(本地 System OCR 派生文本,非 AI Provider)----
ipcMain.handle('image-text-index:getStatus', () => imageTextIndexService.getStatus())
/**
* 点击索引前的快速统计:纯 SQL COUNT,**不解密任何图片**。
* 这是「先告诉用户有多少张图片再决定是否开始」能足够快的前提。
*/
ipcMain.handle('image-text-index:count', (_, sinceMs?: number) =>
imageTextIndexService.countImageMessages(sinceMs)
)
ipcMain.handle(
'image-text-index:start',
(_, options?: ImageTextIndexStartOptions) => imageTextIndexService.startPass(options ?? {})
)
ipcMain.handle('image-text-index:pause', () => imageTextIndexService.pause())
ipcMain.handle(
'image-text-index:resume',
(_, options?: ImageTextIndexStartOptions) => imageTextIndexService.resume(options ?? {})
)
ipcMain.handle('image-text-index:cancel', () => imageTextIndexService.cancel())
ipcMain.handle('image-text-index:clear', () => imageTextIndexService.clear())
// 只重置失败记录(成功记录与其它数据一律不动),供"修好代码后重跑"使用。
ipcMain.handle('image-text-index:resetFailures', () =>
imageTextIndexService.resetRetriableFailures()
)
// 派生索引修复:只重建 Knowledge 里的图片派生条目(L3),**不重新 OCR**(L1 不动)。
ipcMain.handle('image-text-index:repair', () => imageTextIndexService.repairKnowledgeIndex())
ipcMain.handle('ai-search:run', (event, request: AiSearchPipelineRequest) => {
if (!aiSearchPipelineService) throw new Error('本地搜索服务尚未初始化')
return aiSearchPipelineService.run(request, (progress) => {
@@ -1963,6 +2103,8 @@ app.whenReady().then(async () => {
if (aesKey) imageKeyConfigService.save({ resourceRoot, xorKey, aesKey })
else imageKeyConfigService.clear()
imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
}
if ('recallProtectionEnabled' in patch && chat.isReady()) {
const currentDb = chat.getChatDb()
+80 -10
View File
@@ -1,6 +1,7 @@
import { monitorEventLoopDelay } from 'perf_hooks'
import * as chat from '../services/chat-service'
import type {
KnowledgeImageOcrState,
KnowledgeAttachmentMetadata,
KnowledgeEvidence,
KnowledgeMessageKind,
@@ -24,6 +25,7 @@ import {
emptyKnowledgeSearchTimings
} from '../../shared/knowledge'
import { KnowledgeService } from './knowledge-service'
import { sourceMessageId } from './message-identity'
import {
voiceAccountIdentity,
voiceMessageIdentity
@@ -118,11 +120,6 @@ function groupMemberDisplayName(member: chat.GroupSnapshot['members'][number]):
)
}
function sourceMessageId(message: chat.FormattedMessage): string {
if (message.localId) return `local:${message.localId}`
if (message.id) return String(message.id)
return `${message.createTime || 0}:${message.serverId || message.content}`
}
function sourceKind(message: chat.FormattedMessage): KnowledgeMessageKind {
if (message.voiceTranscript || message.type === '语音') return 'voice'
@@ -130,6 +127,16 @@ function sourceKind(message: chat.FormattedMessage): KnowledgeMessageKind {
return message.exportMediaType
}
if (message.exportMediaType === 'file') return 'file'
// 索引路径上 `exportMediaType` **不会被赋值**(只有 export-service 会设它),
// 所以图片/视频/表情包必须从 contentData.type 判定,否则图片会静默落成 'other',
// 进而让"图片文字索引"的 Evidence 丢掉真正的来源类型。
if (
message.contentData?.type === 'image' ||
message.contentData?.type === 'video' ||
message.contentData?.type === 'sticker'
) {
return message.contentData.type
}
if (message.contentData?.type === 'share' || message.contentData?.type === 'miniProgram') {
return message.contentData.type === 'share' && message.contentData.typeVal === '6'
? 'file'
@@ -201,12 +208,16 @@ function toSourceMessage(
accountId: string,
conversationId: string,
message: chat.FormattedMessage,
transcriptOverride?: string
transcriptOverride?: string,
imageOcr?: { state: KnowledgeImageOcrState; text: string }
): KnowledgeSourceMessage | null {
if (!message.createTime) return null
const extracted = sourceTextAndAttachment(message)
const voiceTranscript = transcriptOverride?.trim() || message.voiceTranscript?.trim() || undefined
if (!extracted.text && !extracted.attachment && !voiceTranscript) return null
// 图片 OCR 文本走与语音转写完全相同的派生通道:有文本才入库,
// 没有文字的图片(表情包/风景)不会污染索引。
const imageOcrText = imageOcr?.text?.trim() || undefined
if (!extracted.text && !extracted.attachment && !voiceTranscript && !imageOcrText) return null
return {
accountId,
conversationId,
@@ -218,7 +229,9 @@ function toSourceMessage(
kind: sourceKind(message),
text: extracted.text,
attachment: extracted.attachment,
voiceTranscript
voiceTranscript,
...(imageOcrText ? { imageOcrText } : {}),
...(imageOcrText && imageOcr?.state ? { imageOcrState: imageOcr.state } : {})
}
}
@@ -256,6 +269,14 @@ export class KnowledgeSearchService {
private interactiveIdleResolve: (() => void) | null = null
private wcdbQueueMsTotal = 0
private wcdbExecutionMsTotal = 0
/**
* 图片 OCR 文本解析器(由 main 注入)。
*
* 与语音同构:派生文本在**主进程**解析后贴到消息上,派生库不进 worker。
*/
private imageOcrResolver:
| ((conversationId: string, messageId: string) => { state: KnowledgeImageOcrState; text: string } | undefined)
| undefined
private voiceTranscriptResolver:
| ((reference: VoiceMessageReference) => VoiceTranscriptSnapshot)
| undefined
@@ -372,6 +393,15 @@ export class KnowledgeSearchService {
this.voiceTranscriptResolver = resolver
}
/** 注入图片 OCR 文本解析器(本地 System OCR 的派生结果)。 */
setImageOcrResolver(
resolver:
| ((conversationId: string, messageId: string) => { state: KnowledgeImageOcrState; text: string } | undefined)
| undefined
): void {
this.imageOcrResolver = resolver
}
/**
* A successful recognition updates its source conversation. Consecutive
* updates for the same conversation are coalesced because a complete
@@ -1074,8 +1104,16 @@ export class KnowledgeSearchService {
const reference = this.voiceReferenceFromMessage(message)
const snapshot = reference ? this.voiceTranscriptResolver?.(reference) : undefined
const hydrated = this.withVoiceTranscript(message)
const source = toSourceMessage(accountId, conversationId, hydrated, transcriptOverride)
if (!source || source.kind !== 'voice') return source
const imageOcr = this.imageOcrResolver?.(conversationId, sourceMessageId(message))
const source = toSourceMessage(
accountId,
conversationId,
hydrated,
transcriptOverride,
imageOcr
)
if (!source) return source
if (source.kind !== 'image' && source.kind !== 'voice') return source
return {
...source,
voiceTranscriptState:
@@ -1105,6 +1143,38 @@ export class KnowledgeSearchService {
}
}
/**
* 某个会话的图片 OCR 处理完成 → 重建该会话的索引。
*
* 与"语音转写完成后单会话重索引"完全同构:整会话重读 + completeSnapshot 重建,
* 让 OCR 派生文本进入 chunks/FTS,从而可被 search_messages 检索。
* 原图片消息仍然是 authoritative source —— 这里只是让它多了一段派生文本,
* 不产生任何"OCR 消息"。
*/
async indexImageOcr(conversationId: string): Promise<void> {
if (!chat.isReady()) return
const accountId = this.currentAccountId()
if (!accountId) return
const activeIndex = this.indexing.get(accountId)
if (activeIndex) await activeIndex
const contacts = await this.listContacts()
const contact = contacts.find((item) => item.md5 === conversationId)
if (!contact) return
const messages = await this.listMessages(contact.md5, undefined, undefined, 'background')
const sourceMessages = messages
.map((message) => this.toSourceMessage(accountId, contact.md5, message))
.filter((message): message is KnowledgeSourceMessage => Boolean(message))
await this.service.index({
accountId,
conversations: [
{ conversationId: contact.md5, completeSnapshot: true, messages: sourceMessages }
],
chunker: DEFAULT_KNOWLEDGE_CHUNKER,
fts: DEFAULT_KNOWLEDGE_FTS_CONFIG
})
await this.refreshStatus(accountId)
}
private async indexVoiceTranscriptNow(update: VoiceTranscriptUpdate): Promise<void> {
if (!chat.isReady()) return
if (update.state === 'transcribed' && !update.transcript?.trim()) return
+26 -6
View File
@@ -19,7 +19,11 @@ import type {
KnowledgeSearchTimings,
KnowledgeSearchResult
} from '../../shared/knowledge'
import { emptyKnowledgeSearchTimings, KNOWLEDGE_SCHEMA_VERSION } from '../../shared/knowledge'
import {
emptyKnowledgeSearchTimings,
KNOWLEDGE_SCHEMA_VERSION,
toEvidenceDisplayText
} from '../../shared/knowledge'
import { chunkConversation } from './chunker'
import { normalizeKnowledgeMessage } from './normalizer'
@@ -789,7 +793,11 @@ export class KnowledgeStore {
timestamp: Number(row.create_time),
messageIds: chunk ? chunk.map((item) => String(item.message_id)) : [messageId],
sourceKind: String(row.kind) as KnowledgeEvidence['sourceKind'],
text: String(row.searchable_text),
// 内部前缀(`图片文字:`)绝不能进 Evidence:面向用户与模型的是可读文本,
// 来源信息由下面的结构化字段表达。
text: toEvidenceDisplayText(String(row.searchable_text)),
...(row.image_ocr_text ? { imageOcrText: String(row.image_ocr_text) } : {}),
...(row.image_ocr_text ? { derivedSource: 'image_ocr' as const } : {}),
score: String(row.kind) === 'system' ? 1 : 0
}
}
@@ -899,6 +907,7 @@ export class KnowledgeStore {
attachment_json TEXT,
voice_transcript TEXT,
voice_transcript_state TEXT,
image_ocr_text TEXT,
PRIMARY KEY (conversation_id, message_id)
) STRICT;
CREATE INDEX IF NOT EXISTS knowledge_messages_conversation_time
@@ -957,6 +966,14 @@ export class KnowledgeStore {
if (!messageColumns.has('voice_transcript_state')) {
this.database.exec('ALTER TABLE knowledge_messages ADD COLUMN voice_transcript_state TEXT')
}
// 图片 OCR 派生文本单独留一列(不只是埋进 searchable_text)。
//
// 为什么必须落列而不是从 searchable_text 里截字符串:Evidence 需要回答
// "这条结果是不是来自图片里的文字",并按此给出来源标记与 OCR 片段。
// 靠解析前缀来判来源,一旦前缀格式调整就会静默失效。
if (!messageColumns.has('image_ocr_text')) {
this.database.exec('ALTER TABLE knowledge_messages ADD COLUMN image_ocr_text TEXT')
}
this.writeMetaIfMissing('schema_version', String(KNOWLEDGE_SCHEMA_VERSION))
const storedAccount = this.readMeta('account_id')
if (storedAccount && storedAccount !== this.accountId) {
@@ -1137,8 +1154,9 @@ export class KnowledgeStore {
const upsert = this.database.prepare(
`INSERT INTO knowledge_messages (
account_id, conversation_id, message_id, create_time, content_hash, searchable_text,
kind, sender_id, sender_name, attachment_json, voice_transcript, voice_transcript_state
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
kind, sender_id, sender_name, attachment_json, voice_transcript, voice_transcript_state,
image_ocr_text
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(conversation_id, message_id) DO UPDATE SET
create_time = excluded.create_time,
content_hash = excluded.content_hash,
@@ -1148,7 +1166,8 @@ export class KnowledgeStore {
sender_name = excluded.sender_name,
attachment_json = excluded.attachment_json,
voice_transcript = excluded.voice_transcript,
voice_transcript_state = excluded.voice_transcript_state`
voice_transcript_state = excluded.voice_transcript_state,
image_ocr_text = excluded.image_ocr_text`
)
for (let index = 0; index < messages.length; index += 1) {
this.assertNotAborted(signal)
@@ -1165,7 +1184,8 @@ export class KnowledgeStore {
message.senderName ?? null,
message.attachment ? encodedJson(message.attachment) : null,
message.voiceTranscript ?? null,
message.voiceTranscriptState ?? null
message.voiceTranscriptState ?? null,
message.imageOcrText ?? null
)
if (index % YIELD_EVERY === 0) {
onProgress(index + 1, 0)
+24
View File
@@ -0,0 +1,24 @@
/**
* 消息身份的**唯一真源**。
*
* 这个规则同时被三处需要:
* - Knowledge 索引写入 `knowledge_messages.message_id`
* - 图片文字索引的 binding(必须与 Knowledge 里的 message_id 完全一致,否则 OCR 文本贴不到消息上)
* - Evidence → 档案跳转的 messageRef
*
* 任何一处各自复制一份,都会在 `local:` 前缀上静默失配(项目里已经有这个坑的历史注释),
* 所以抽成一个模块,谁都不许再抄。
*/
import type * as chat from '../services/chat-service'
/**
* 源消息 → 稳定消息 id。
*
* 降级顺序刻意保守:`localId` 是 WCDB 行内最稳的本地 id;其次用消息自带 id;
* 最后才退化成「时间 + 服务端 id / 内容」的组合(仅在极端缺字段时命中)。
*/
export function sourceMessageId(message: chat.FormattedMessage): string {
if (message.localId) return `local:${message.localId}`
if (message.id) return String(message.id)
return `${message.createTime || 0}:${message.serverId || message.content}`
}
+5
View File
@@ -24,6 +24,10 @@ export function normalizeKnowledgeMessage(
const transcript = compact(source.voiceTranscript)
if (transcript) sections.push(`语音转写:${transcript}`)
// 图片 OCR 文本:与语音同样的"固定前缀"约定,让检索与展示都能识别这是派生内容。
const imageText = compact(source.imageOcrText)
if (imageText) sections.push(`图片文字:${imageText}`)
const attachmentName = compact(source.attachment?.name)
if (attachmentName) {
const label = source.attachment?.kind === 'link' ? '链接' : '附件'
@@ -45,6 +49,7 @@ export function normalizeKnowledgeMessage(
senderId: source.senderId || '',
kind: source.kind,
voiceTranscriptState: source.voiceTranscriptState || '',
imageOcrState: source.imageOcrState || '',
searchableText
})
)
+31 -9
View File
@@ -86,7 +86,8 @@ export class AskWechatService {
const diagnostics = this.diagnostics(
{ provider: '', model: '', modelCallCount: 0, toolCallCount: 0, traces: [] },
startedAt,
'runtime_error'
'runtime_error',
{ question }
)
this.writeLog('error', `Query Agent Runtime 异常(${this.options.entry})`, diagnostics)
return this.fallback(request, 'runtime_error', diagnostics)
@@ -97,13 +98,13 @@ export class AskWechatService {
engine: 'query-agent',
status: 'error',
message: EMPTY_QUESTION_MESSAGE,
diagnostics: this.diagnostics(result, startedAt, 'invalid_question')
diagnostics: this.diagnostics(result, startedAt, 'invalid_question', { question })
}
}
if (result.errorKind === 'provider_unavailable' || result.errorKind === 'provider_failure') {
const outcome: AskWechatOutcome = result.errorKind
const diagnostics = this.diagnostics(result, startedAt, outcome)
const diagnostics = this.diagnostics(result, startedAt, outcome, { question })
this.writeLog('warn', `查询 Provider 不可用(${this.options.entry})`, diagnostics)
return {
engine: 'query-agent',
@@ -114,13 +115,13 @@ export class AskWechatService {
}
if (result.errorKind === 'tool_limit') {
const diagnostics = this.diagnostics(result, startedAt, 'tool_limit')
const diagnostics = this.diagnostics(result, startedAt, 'tool_limit', { question })
this.writeLog('warn', `查询超出工具调用上限(${this.options.entry})`, diagnostics)
return this.fallback(request, 'runtime_error', diagnostics)
}
if (!result.answer?.trim()) {
const diagnostics = this.diagnostics(result, startedAt, 'runtime_error')
const diagnostics = this.diagnostics(result, startedAt, 'runtime_error', { question })
this.writeLog('warn', `Query Agent 未返回回答(${this.options.entry})`, diagnostics)
return this.fallback(request, 'runtime_error', diagnostics)
}
@@ -128,7 +129,7 @@ export class AskWechatService {
const answer = result.answer.trim()
// 澄清回答也记录:下一句("是 BOBO")需要接得上上文。
this.memory.record(conversationKey, question, answer)
const diagnostics = this.diagnostics(result, startedAt, 'answered')
const diagnostics = this.diagnostics(result, startedAt, 'answered', { question, answer })
this.writeLog('info', `Query Agent 回答完成(${this.options.entry})`, diagnostics)
return {
engine: 'query-agent',
@@ -187,17 +188,38 @@ export class AskWechatService {
> &
Partial<Pick<QueryAgentResult, 'totalMs'>>,
startedAt: number,
outcome: AskWechatOutcome
outcome: AskWechatOutcome,
/**
* 问答原文(可选)。只在本地应用日志里用,不上传、不进遥测。
*
* 排查这类"同一问题时对时错"的故障,光有工具名与次数是不够的 ——
* 必须能对着"问题 + 模型回答"回放,否则无法判断是理解错了、链路断了,还是索引没建。
*/
content?: { question?: string; answer?: string }
): QueryAgentDiagnostics {
const traces = result.traces || []
// 图片 OCR 的两条结构化事实:不回读正文,只统计"取到了几条"与"当时覆盖度是多少"。
const imageOcrTextCount = traces.reduce(
(sum, trace) => sum + (trace.imageOcrTextCount || 0),
0
)
const coverageState = traces
.map((trace) => trace.imageOcrCoverageState)
.filter((value): value is string => typeof value === 'string')
.at(-1)
return {
entry: this.options.entry,
provider: result.provider,
model: result.model,
modelCallCount: result.modelCallCount,
toolCallCount: result.toolCallCount,
tools: (result.traces || []).map((trace) => trace.toolName),
tools: traces.map((trace) => trace.toolName),
totalMs: result.totalMs || Date.now() - startedAt,
outcome
outcome,
...(imageOcrTextCount > 0 ? { imageOcrTextCount } : {}),
...(coverageState ? { imageOcrCoverageState: coverageState } : {}),
...(content?.question ? { question: content.question } : {}),
...(content?.answer ? { answer: content.answer } : {})
}
}
+16
View File
@@ -8,9 +8,12 @@ export type { CacheClearScope } from '../../shared/cache'
const BOOTSTRAP_CACHE_DIR = path.join(app.getPath('userData'), 'cache', 'bootstrap')
const KNOWLEDGE_CACHE_DIR = path.join(app.getPath('userData'), 'knowledge')
const IMAGE_TEXT_INDEX_CACHE_DIR = path.join(app.getPath('userData'), 'image-text-index')
export interface CacheClearOptions {
beforeClearKnowledge?: () => Promise<void>
/** 清理图片文字索引前调用:停任务 + 关闭派生库句柄。 */
beforeClearImageTextIndex?: () => Promise<void>
}
function inspectDirectory(directory: string): { sizeBytes: number; fileCount: number } {
@@ -46,6 +49,7 @@ export function getCacheSummary(): CacheSummary {
const bootstrap = inspectDirectory(BOOTSTRAP_CACHE_DIR)
const electron = inspectDirectory(path.join(app.getPath('userData'), 'Cache'))
const knowledge = inspectDirectory(KNOWLEDGE_CACHE_DIR)
const imageTextIndex = inspectDirectory(IMAGE_TEXT_INDEX_CACHE_DIR)
const items: CacheSummaryItem[] = [
{
id: 'bootstrap',
@@ -65,6 +69,13 @@ export function getCacheSummary(): CacheSummary {
description:
'为问问微信建立的所有账号本地检索索引。清理后需手动重新建立,不影响微信原始数据。',
...knowledge
},
{
id: 'image-text-index',
label: '图片文字索引',
description:
'本机从微信图片里识别出的文字及其检索索引。清理后无法搜索图片中的文字,可重新建立;不影响微信原始图片与聊天记录。',
...imageTextIndex
}
]
return {
@@ -89,6 +100,11 @@ export async function clearCache(
await options.beforeClearKnowledge?.()
await fs.remove(KNOWLEDGE_CACHE_DIR)
}
if (scope === 'image-text-index' || scope === 'all') {
// 先停下任务再删库,避免"边写边删"。
await options.beforeClearImageTextIndex?.()
await fs.remove(IMAGE_TEXT_INDEX_CACHE_DIR)
}
return getCacheSummary()
}
+29
View File
@@ -16,6 +16,7 @@ import {
} from '../../shared/windows-runtime'
import { mergeRecallArchiveMessages, recordRecallArchiveMessages } from './recall-archive-service'
import type { ExportImageQuality } from '../../shared/image-quality'
import type { ImageMessageCountProbe } from '../../shared/image-text-index'
import { wcdbDebugLog } from '../wcdb-debug'
import {
buildContactSearchIndex,
@@ -717,6 +718,34 @@ export async function listMessagesForExport(
* batch-selection view, where loading every conversation would make opening
* Settings noticeably slow.
*/
/**
* 图片消息计数探针(SQL 统计,不解密)。
*
* 返回 `count: null` 表示**统计失败**,不是 0 张。调用方必须区分这两件事 ——
* 否则"数不出来"会被显示成"账号里没有图片",用户会因此放弃建立索引。
*/
export async function countImageMessagesAsync(
userMd5: string,
sinceMs?: number
): Promise<ImageMessageCountProbe> {
if (!dbRef) return { count: null, typeColumn: null, error: '微信数据库尚未就绪' }
return dbRef.getWcdb4Client().countImageMessagesAsync(userMd5, sinceMs)
}
/**
* 图片消息的增量水位(条数 + 最大插入序)。
*
* 增量索引**不能只比条数**:召回一张旧图的同时新增一张新图,条数不变但集合变了。
* 返回 null = 当前数据库不支持该统计(调用方须退化成"每轮重扫",宁可慢也不可漏)。
*/
export async function imageConversationWatermarkAsync(
userMd5: string,
sinceMs?: number
): Promise<{ count: number; maxLocalId: number } | null> {
if (!dbRef) return null
return dbRef.getWcdb4Client().imageConversationWatermarkAsync(userMd5, sinceMs)
}
export async function countVoiceMessagesAsync(
userMd5: string,
startTime?: number,
@@ -0,0 +1,970 @@
/**
* 图片文字索引编排服务。
*
* 职责边界(刻意保持单一):
* - 快速统计图片消息数(SQL,**绝不解密**)
* - 按会话 + 批次驱动 OCR;**严格串行(concurrency = 1)**:循环体内只有一次
* `await`,不存在 Promise.all 扇出,且每批之间让出 event loop
* - 维护 checkpoint(可暂停 / 继续 / 取消 / 重启后恢复)
* - 把结果写进派生库,并在**会话完成时**回调,让 Knowledge 重建该会话的索引
*
* 明确不做:
* - 不修改 WCDB / 不写回原始消息 / 不产生任何"OCR 消息"
* - 不实现图片搜索 Agent(检索继续走既有 Query Agent + Knowledge)
* - 不在日志里写 OCR 正文 / 真实图片路径 / wxid / 群名
*/
import { createHash } from 'node:crypto'
import { existsSync } from 'node:fs'
import {
IMAGE_TEXT_INDEX_BATCH_SIZE,
IMAGE_TEXT_INDEX_ENGINE,
IMAGE_OCR_RETRIABLE_FAILURE_STATES,
buildImageOcrArtifactKey,
imageTextProcessedPercent,
isTerminalImageOcrState,
type ImageMessageCountProbe,
type ImageMessageWatermark,
type ImageOcrPersistedState,
type ImageOcrProvenance,
type ImageTextIndexCountResult,
type ImageTextIndexCoverage,
type ImageTextIndexProgress,
type ImageTextIndexRepairResult,
type ImageTextIndexRunState,
type ImageTextIndexStartOptions,
type ImageTextIndexStatus,
type ImageTextIndexStorageStats
} from '../../shared/image-text-index'
import { detectSystemOcrImageFormat, type SystemOcrCapability } from '../../shared/system-ocr'
import type * as chat from './chat-service'
/**
* 消息 id 必须与 Knowledge 写入的 `knowledge_messages.message_id` 完全一致,
* 否则 OCR 文本贴不到消息上、Evidence 也回不到原图。真源见 knowledge/message-identity。
*/
import { sourceMessageId } from '../knowledge/message-identity'
import type { ImageDecryptService } from '../image-decrypt-service'
import {
ImageTextIndexStore,
getImageTextIndexDatabasePath,
removeImageTextIndexDatabase,
type ConversationImageOcrEntry
} from './image-text-index-store'
/** 图片消息的数据 URL 前缀。 */
const MIME_BY_FORMAT: Record<string, string> = {
png: 'image/png',
jpeg: 'image/jpeg',
gif: 'image/gif',
bmp: 'image/bmp',
webp: 'image/webp',
tiff: 'image/tiff'
}
export interface ImageTextIndexServiceDeps {
/** `<userData>/image-text-index`。 */
databaseRoot?: string
/** 当前账号(wxid 优先,退回 accountRoot)。空串 = 微信未就绪。 */
resolveAccountId?: () => string
/** 当前微信数据根目录。 */
resolveAccountRoot?: () => string
listContacts?: () => Promise<Array<{ md5: string; m_nsUsrName: string; type: 'user' | 'group' }>>
listMessages?: (conversationId: string) => Promise<chat.FormattedMessage[]>
/**
* 单个会话的图片消息计数探针(SQL 统计,不解密)。
*
* `count: null` = 统计失败,**不等于 0 张**;调用方必须区分。
*/
countConversationImages?: (
conversationId: string,
sinceMs?: number
) => Promise<ImageMessageCountProbe>
/**
* 单个会话的图片消息增量水位(条数 + 最大插入序),SQL 聚合,不解密。
*
* 返回 null = 当前数据库不支持(调用方必须退化成"每轮重扫",宁可慢也不可漏)。
*/
imageWatermark?: (conversationId: string, sinceMs?: number) => Promise<ImageMessageWatermark | null>
decryptService?: () => ImageDecryptService | null
/** 本地 OCR。 */
recognize?: (imageDataUrl: string) => Promise<{
success: boolean
text: string
language: string | null
errorCode?: string
}>
capability?: () => Promise<SystemOcrCapability>
/** 会话图片全部处理完后回调,用于把 OCR 文本灌进 Knowledge 索引。 */
onConversationIndexed?: (conversationId: string) => Promise<void>
/** 交互查询让路钩子。 */
interactiveIdle?: () => Promise<void>
now?: () => number
}
function sha256Short(value: Uint8Array | string): string {
return createHash('sha256').update(value).digest('hex').slice(0, 32)
}
/** 图片消息判定:与 chat-service 的 `contentData.type === 'image'` 对齐。 */
export function isImageMessage(message: chat.FormattedMessage): boolean {
return message.contentData?.type === 'image'
}
export class ImageTextIndexService {
private deps: ImageTextIndexServiceDeps = {}
private store: ImageTextIndexStore | null = null
private storeAccountId = ''
private accountKey = ''
private running = false
private cancelRequested = false
private pauseRequested = false
private passPromise: Promise<void> | null = null
private counting = false
private listeners = new Set<(status: ImageTextIndexStatus) => void>()
private lastError: string | undefined
private startedAt: number | undefined
/** 上一次清理实际重建(失效)了多少个会话的 Knowledge 索引;用于诊断与测试。 */
lastInvalidatedConversations = 0
/**
* 会话级 OCR 文本缓存。
*
* Knowledge 重建一个会话时会对每条消息问一次 resolver;不加缓存就是每条消息一次 SQL。
* 写入 binding 时精确失效该会话,保证不会读到旧结果。
*/
private conversationOcrCache = new Map<string, Map<string, ConversationImageOcrEntry>>()
/** 本轮 pass 的计数器(内存态;真实来源始终是派生库)。 */
private counters = {
totalImageMessages: 0,
processedThisPass: 0,
indexed: 0,
empty: 0,
missing: 0,
failed: 0
}
private runState: ImageTextIndexRunState = 'idle'
bind(deps: ImageTextIndexServiceDeps): void {
this.deps = { ...this.deps, ...deps }
}
private now(): number {
return this.deps.now ? this.deps.now() : Date.now()
}
// ------------------------------------------------------------- store 生命周期
private resolveAccountId(): string {
return this.deps.resolveAccountId?.() || ''
}
private ensureStore(): ImageTextIndexStore | null {
const root = this.deps.databaseRoot
const accountId = this.resolveAccountId()
if (!root || !accountId) return null
if (this.store && this.storeAccountId === accountId) return this.store
this.store?.close()
const key = getImageTextIndexDatabasePath(root, accountId)
this.store = new ImageTextIndexStore(key, accountId)
this.storeAccountId = accountId
this.accountKey = key
return this.store
}
/**
* 账号切换 / 数据库切换时丢弃句柄。
*
* 派生库按 accountId 分目录,句柄必须跟着换;否则会把 A 账号的 OCR
* 写到 B 账号,或让新库读到旧账号的 coverage。
*/
resetAccount(): void {
this.conversationOcrCache.clear()
this.store?.close()
this.store = null
this.storeAccountId = ''
this.accountKey = ''
this.counters = {
totalImageMessages: 0,
processedThisPass: 0,
indexed: 0,
empty: 0,
missing: 0,
failed: 0
}
this.runState = 'idle'
this.cancelRequested = false
this.pauseRequested = false
this.lastError = undefined
}
// ------------------------------------------------------------------- 只读接口
/** Knowledge 索引时用:把某会话的 OCR 文本贴到消息上(与语音 resolver 同构)。 */
getConversationOcr(conversationId: string): Map<string, ConversationImageOcrEntry> {
const cached = this.conversationOcrCache.get(conversationId)
if (cached) return cached
const store = this.ensureStore()
if (!store) return new Map()
const result = store.getConversationOcr(conversationId)
this.conversationOcrCache.set(conversationId, result)
return result
}
/**
* 覆盖度。
*
* **分母只能来自落盘的 SQL 统计**,不能从派生库自己推:派生库只知道自己处理过什么。
* 如果按 `processed + pending` 反推 total,应用重启后 pending 无处可来,
* total 就会退化成 processed —— 30% 的部分索引会被谎报成"已覆盖全部"。
* 这正是 §18 禁止的"把 partial coverage 当 complete"。
*/
private coverageFromCounts(counts: Record<string, number>): ImageTextIndexCoverage {
const indexed = counts['indexed'] ?? 0
const empty = counts['empty'] ?? 0
const missing = (counts['image_missing'] ?? 0) + (counts['metadata_missing'] ?? 0)
const failed =
(counts['decrypt_failed'] ?? 0) +
(counts['decode_failed'] ?? 0) +
(counts['ocr_failed'] ?? 0) +
(counts['cancelled'] ?? 0)
const runtimeUnavailable = counts['decrypt_unavailable'] ?? 0
// 运行时不可用**不计入 processed**:它不是"这条图片已经处理过了"。
const processed = indexed + empty + missing + failed
const counted = this.store?.readCountedTotal() ?? null
// 内存计数器只在本轮 pass 内比落盘值更新(刚统计完、尚未落盘的窗口)。
const total =
this.counters.totalImageMessages || counted?.total || processed + runtimeUnavailable
/**
* 系统性失败:处理过一批,但一条都没能给出确定结果。
*
* 这正是本次事故的形态(45,479 张全部失败,成功 / 无文字 / 缺失都是 0)。
* 它必须阻断 `complete` —— 否则 Query Agent 会拿着"覆盖完整"去回答"没有"。
*/
const systemicFailure = processed > 0 && indexed === 0 && empty === 0 && missing === 0
return {
totalImageMessages: total,
processed,
indexed,
empty,
missing,
failed,
runtimeUnavailable,
pending: Math.max(0, total - processed - runtimeUnavailable),
// 从未统计过总数 → 不算"已建立":不知道分母就不允许声称覆盖。
established: counted !== null && (processed > 0 || runtimeUnavailable > 0),
// 分母不完整、有 pending、有运行时不可用、或"全军覆没" → 都不算 complete。
complete:
counted !== null &&
counted.complete &&
total > 0 &&
runtimeUnavailable === 0 &&
!systemicFailure &&
processed >= total,
systemicFailure,
countedAt: counted?.countedAt ?? null
}
}
/**
* 覆盖度快照(只读、同步),供 Query Agent 在工具结果里携带图片覆盖度。
*
* 刻意**不建库**:只因为用户问了一句话就凭空创建一个派生库是没道理的。
* 库不存在 = 从未建立过索引 = `not_built`。
*/
getCoverageSnapshot(): ImageTextIndexCoverage | null {
const root = this.deps.databaseRoot
const accountId = this.resolveAccountId()
if (!root || !accountId) return null
if (!this.store && !existsSync(getImageTextIndexDatabasePath(root, accountId))) {
return null
}
const store = this.ensureStore()
if (!store) return null
return this.coverageFromCounts(store.countByState())
}
private progressFromCounts(counts: Record<string, number>): ImageTextIndexProgress {
const coverage = this.coverageFromCounts(counts)
const total = coverage.totalImageMessages
const percent = imageTextProcessedPercent(coverage.processed, total)
return {
state: this.runState,
totalImageMessages: total,
processed: coverage.processed,
indexed: coverage.indexed,
empty: coverage.empty,
missing: coverage.missing,
failed: coverage.failed,
runtimeUnavailable: coverage.runtimeUnavailable,
systemicFailure: coverage.systemicFailure,
pending: coverage.pending,
percent,
processedPercent: percent,
...(this.startedAt ? { startedAt: this.startedAt } : {}),
updatedAt: this.now(),
cancellable: this.running,
paused: this.runState === 'paused',
...(this.lastError ? { lastError: this.lastError } : {})
}
}
private emptyStorage(): ImageTextIndexStorageStats {
return { indexedImages: 0, ocrTextCount: 0, totalBytes: 0, updatedAt: null }
}
async getStatus(): Promise<ImageTextIndexStatus> {
const store = this.ensureStore()
if (!store) {
return {
progress: {
state: this.runState,
totalImageMessages: this.counters.totalImageMessages,
processed: 0,
indexed: 0,
empty: 0,
missing: 0,
failed: 0,
runtimeUnavailable: 0,
systemicFailure: false,
pending: 0,
percent: 0,
processedPercent: 0,
updatedAt: this.now(),
cancellable: false,
paused: false
},
coverage: {
totalImageMessages: 0,
processed: 0,
indexed: 0,
empty: 0,
missing: 0,
failed: 0,
runtimeUnavailable: 0,
pending: 0,
established: false,
complete: false,
systemicFailure: false,
countedAt: null
},
storage: this.emptyStorage(),
counting: this.counting
}
}
const counts = store.countByState()
return {
progress: this.progressFromCounts(counts),
coverage: this.coverageFromCounts(counts),
storage: store.storageStats(),
counting: this.counting
}
}
onStatusChange(listener: (status: ImageTextIndexStatus) => void): () => void {
this.listeners.add(listener)
return () => this.listeners.delete(listener)
}
private async emit(): Promise<void> {
if (!this.listeners.size) return
const status = await this.getStatus()
for (const listener of this.listeners) {
try {
listener(status)
} catch {
// 监听器异常不得影响索引。
}
}
}
// --------------------------------------------------------------------- 统计
/**
* 快速统计当前账号的图片消息数。
*
* 走 SQL COUNT(`local_type & 65535 = 3`),**不解密任何图片** —— 这是
* 「点击索引前先告诉用户有多少张」能够足够快的前提。
*/
async countImageMessages(sinceMs?: number): Promise<ImageTextIndexCountResult> {
const startedAt = this.now()
const contacts = await (this.deps.listContacts?.() ?? Promise.resolve([]))
let total = 0
let scanned = 0
let failed = 0
let typeColumn: string | null = null
let firstError: string | undefined
for (const contact of contacts) {
const probe = await (this.deps.countConversationImages?.(contact.md5, sinceMs) ??
Promise.resolve<ImageMessageCountProbe>({
count: null,
typeColumn: null,
error: '未接入图片消息统计能力'
}))
if (probe.typeColumn && !typeColumn) typeColumn = probe.typeColumn
if (probe.count === null) {
// **统计失败不是 0 张**:必须单独计数,否则 UI 会把"数不出来"说成"没有图片"。
failed += 1
if (!firstError) firstError = probe.error
continue
}
scanned += 1
total += probe.count
}
this.counters.totalImageMessages = total
// 落盘:coverage 的分母必须能被重启后读到(见 coverageFromCounts)。
// 只要有一个会话没数上,分母就是偏小的 → 标记为不完整,coverage 拿不到 complete。
this.ensureStore()?.writeCountedTotal({
total,
countedAt: this.now(),
complete: contacts.length > 0 && failed === 0
})
return {
totalImageMessages: total,
scannedConversations: scanned,
failedConversations: failed,
typeColumn,
...(firstError ? { error: firstError } : {}),
durationMs: this.now() - startedAt
}
}
// ------------------------------------------------------------------ 单张处理
private async processOne(
message: chat.FormattedMessage,
conversationId: string,
provenance: ImageOcrProvenance
): Promise<{ state: ImageOcrPersistedState; text: string; imageIdentity: string | null }> {
const imageContent =
message.contentData?.type === 'image'
? (message.contentData as { md5?: string; datName?: string })
: undefined
const decrypt = this.deps.decryptService?.() ?? null
/**
* 解密服务缺失是**运行时**问题,不是这张图片的问题。
*
* 本次事故就是它:`imageDecryptService` 只在用户点开某张图时才懒加载,
* 于是全量回填 45,479 张全部落成 `decrypt_failed` —— 数字看着像"图片坏了",
* 实际是流水线前置依赖没接上。这里必须用独立状态,绝不能与真正的解密失败混为一谈。
*/
if (!decrypt) return { state: 'decrypt_unavailable', text: '', imageIdentity: null }
// 图片消息缺少定位字段:连"去哪找文件"都不知道,属消息侧缺失而非 OCR 失败。
if (!imageContent?.md5 && !imageContent?.datName) {
return { state: 'metadata_missing', text: '', imageIdentity: null }
}
let datPath: string | null = null
try {
datPath = decrypt.findImageFile(imageContent?.md5, imageContent?.datName, {
accountDir: this.deps.resolveAccountRoot?.() || undefined,
sessionMd5: conversationId,
createTime: message.createTime,
allowThumbnail: true,
preferThumbnail: true
})
} catch {
datPath = null
}
// 微信清理过原图与缩略图 —— 这是正常情况,不是任务级错误。
if (!datPath) return { state: 'image_missing', text: '', imageIdentity: null }
let bytes: Buffer | null = null
try {
bytes = decrypt.decryptImage(datPath)
} catch {
bytes = null
}
if (!bytes || bytes.length === 0) {
return { state: 'decrypt_failed', text: '', imageIdentity: null }
}
const imageIdentity = `sha256:${sha256Short(bytes)}`
const format = detectSystemOcrImageFormat(bytes)
// 解密"没抛错"但产出不是图片 → 解码失败,不是 OCR 失败。
if (!format) return { state: 'decode_failed', text: '', imageIdentity }
const store = this.ensureStore()
const artifactKey = buildImageOcrArtifactKey({ imageIdentity, provenance })
// 同一张图(可能被转发到多个会话)已经算过 → 直接复用,绝不重复 OCR。
const cached = store?.getArtifact(artifactKey) ?? null
if (cached && isTerminalImageOcrState(cached.state)) {
return { state: cached.state, text: cached.text, imageIdentity }
}
const mime = MIME_BY_FORMAT[format] ?? 'image/png'
let state: ImageOcrPersistedState = 'ocr_failed'
let text = ''
let errorCode: string | undefined
try {
const result = await (this.deps.recognize?.(`data:${mime};base64,${bytes.toString('base64')}`) ??
Promise.resolve({ success: false, text: '', language: null, errorCode: 'OCR_FAILED' }))
if (result.success && result.text.trim()) {
state = 'indexed'
text = result.text
} else if (result.success || result.errorCode === 'OCR_EMPTY_RESULT') {
// 表情包 / 风景 / 头像 —— 没有文字是**正常终态**,不重试。
state = 'empty'
} else {
state = 'ocr_failed'
errorCode = result.errorCode
}
} catch {
state = 'ocr_failed'
}
const now = this.now()
if (store) {
store.putArtifact({
accountId: this.storeAccountId,
artifactKey,
imageIdentity,
state,
text,
charCount: text.length,
engine: provenance.engine,
platform: provenance.platform,
runtimeVersion: provenance.runtimeVersion,
language: provenance.language,
...(errorCode ? { errorCode } : {}),
createdAt: now,
updatedAt: now
})
}
return { state, text, imageIdentity }
}
// --------------------------------------------------------------------- pass
/**
* 启动一次 pass。
*
* 「暂停 / 继续」刻意实现为「停止 + 重新跑一次 pass」而不是原地挂起:
* - checkpoint(scan_state)与 artifact 缓存都在库里,重跑会跳过已完成会话、
* 并且命中 artifact 缓存不再重复 OCR,所以恢复成本很低;
* - 与项目既有的「中断后重跑、靠 checkpoint 续做」语义一致,不引入新的挂起状态机。
*/
startPass(options: ImageTextIndexStartOptions = {}): { started: boolean; state: ImageTextIndexRunState } {
if (this.running) return { started: false, state: this.runState }
this.cancelRequested = false
this.pauseRequested = false
this.lastError = undefined
this.startedAt = this.now()
this.runState = 'running'
this.passPromise = this.runPass(options)
.catch((error) => {
this.lastError = error instanceof Error ? error.message : String(error)
this.runState = 'error'
})
.finally(() => {
this.running = false
this.passPromise = null
void this.emit()
})
void this.emit()
return { started: true, state: this.runState }
}
private async runPass(options: ImageTextIndexStartOptions): Promise<void> {
const store = this.ensureStore()
if (!store) {
this.lastError = '微信数据尚未就绪'
this.runState = 'error'
return
}
this.running = true
const capability = (await this.deps.capability?.()) ?? null
if (capability && !capability.available) {
this.lastError = '当前系统不支持本地图片文字识别'
this.runState = 'error'
return
}
/**
* **前置依赖自检(本次事故的根因防线)**:解密服务必须可用。
*
* 没有它,每张图片都会在 `processOne` 的第一步失败。原实现会把"整条流水线
* 根本跑不起来"这件事落成 45,479 条 `decrypt_failed` —— 既污染派生库,
* 又让用户以为自己的图片坏了,还让 coverage 看起来"都处理完了"。
*
* 所以必须在**写任何一条记录之前**停下来:宁可一次都不跑,也不要写一堆假失败。
*/
if (!this.deps.decryptService?.()) {
this.lastError = '图片解密服务尚未就绪,无法读取微信图片;本次未写入任何记录。'
this.runState = 'error'
return
}
const provenance: ImageOcrProvenance = {
engine: capability?.engine ?? IMAGE_TEXT_INDEX_ENGINE,
platform: capability?.platform ?? process.platform,
runtimeVersion: capability?.runtimeVersion ?? null,
language: capability?.language ?? null
}
// 统计一次总数(SQL),进度百分比才有真实分母。
this.counting = true
try {
await this.countImageMessages(options.sinceMs)
} finally {
this.counting = false
}
let contacts = await (this.deps.listContacts?.() ?? Promise.resolve([]))
if (options.conversationLimit && options.conversationLimit > 0) {
contacts = contacts.slice(0, options.conversationLimit)
}
const scanState = store.readScanState()
let budget = options.messageLimit && options.messageLimit > 0 ? options.messageLimit : Infinity
for (const contact of contacts) {
if (this.cancelRequested || this.pauseRequested) break
if (budget <= 0) break
const conversationId = contact.md5
/**
* 是否只处理一个时间窗口(用于小样本验证)。
*
* 带窗口时**不做增量跳过**:checkpoint 是围绕全量集合建立的,
* 窗口内的图片可能从未被处理过,继续按"该会话已完成"跳过会让窗口形同虚设。
*/
const windowed = Boolean(options.sinceMs && options.sinceMs > 0)
// 增量水位 = 条数 + 最大插入序(§2)。只比条数会漏掉「撤回一张旧图 +
// 新增一张新图」这种总数不变、集合却变了的会话。
const watermark = await (this.deps.imageWatermark?.(conversationId, options.sinceMs) ??
Promise.resolve(null))
const imageTotal =
watermark?.count ??
(await this.deps.countConversationImages?.(conversationId, options.sinceMs))?.count ??
0
if (imageTotal === 0) {
store.writeScanState({
conversationId,
state: 'done',
imageTotal: 0,
imageProcessed: 0,
maxLocalId: watermark?.maxLocalId ?? 0
})
continue
}
// 增量:会话已完成且**水位完全未变** → 不读 WCDB、不 OCR。
// 水位不可用时(数据库不支持该聚合)一律重扫:宁可慢,不可漏。
const previous = scanState.get(conversationId)
if (
!windowed &&
watermark &&
previous &&
previous.state === 'done' &&
previous.imageTotal === watermark.count &&
previous.maxLocalId === watermark.maxLocalId
) {
continue
}
await this.deps.interactiveIdle?.()
let messages: chat.FormattedMessage[] = []
try {
messages = await (this.deps.listMessages?.(conversationId) ?? Promise.resolve([]))
} catch {
messages = []
}
const imageMessages = messages
.filter(isImageMessage)
// 时间窗过滤:小样本验证时只看窗口内的图片,不然还是在跑全量。
.filter((message) =>
windowed ? (message.createTime || 0) * 1000 >= (options.sinceMs as number) : true
)
if (!imageMessages.length) {
store.writeScanState({
conversationId,
state: 'done',
imageTotal: 0,
imageProcessed: 0,
maxLocalId: 0
})
continue
}
// 水位取**实际读到的**消息里最大的 local_id,而不是源侧水位:
// 万一在我们查水位之后、读消息之前又落了一条新图,用观测值会让下一轮
// 发现"源水位更高"从而重扫(安全);用源侧水位则会把它永久跳过(漏索引)。
const observedMaxLocalId = imageMessages.reduce(
(max, message) => Math.max(max, Number(message.localId) || 0),
0
)
const ocrByMessage = store.getConversationOcr(conversationId)
let processedInConversation = 0
let interrupted = false
for (let index = 0; index < imageMessages.length; index += IMAGE_TEXT_INDEX_BATCH_SIZE) {
if (this.cancelRequested || this.pauseRequested) {
interrupted = true
break
}
const batch = imageMessages.slice(index, index + IMAGE_TEXT_INDEX_BATCH_SIZE)
for (const message of batch) {
if (budget <= 0) break
const messageId = sourceMessageId(message)
// 派生库里已有终态结果 → 复用(含"无文字"/"图片缺失"),不重复劳动。
const known = ocrByMessage.get(messageId)
if (known && isTerminalImageOcrState(known.state)) {
processedInConversation += 1
continue
}
const outcome = await this.processOne(message, conversationId, provenance)
const now = this.now()
store.putBinding({
accountId: this.storeAccountId,
conversationId,
messageId,
createTime: (message.createTime || 0) * 1000,
...(message.senderId || message.from ? { senderId: message.senderId || message.from } : {}),
...(message.isSender ? { senderName: '我' } : message.name ? { senderName: message.name } : {}),
imageIdentity: outcome.imageIdentity ?? '',
artifactKey: outcome.imageIdentity
? buildImageOcrArtifactKey({ imageIdentity: outcome.imageIdentity, provenance })
: buildImageOcrArtifactKey({ imageIdentity: 'unavailable', provenance }),
state: outcome.state,
updatedAt: now
})
this.conversationOcrCache.delete(conversationId)
processedInConversation += 1
this.counters.processedThisPass += 1
if (outcome.state === 'indexed') this.counters.indexed += 1
else if (outcome.state === 'empty') this.counters.empty += 1
else if (outcome.state === 'image_missing') this.counters.missing += 1
else this.counters.failed += 1
budget -= 1
}
// 批次之间让出 event loop:交互查询 / UI 永远优先于后台历史 OCR。
await new Promise<void>((resolve) => setImmediate(resolve))
await this.emit()
}
if (interrupted) {
store.writeScanState({
conversationId,
state: 'partial',
imageTotal: imageMessages.length,
imageProcessed: processedInConversation,
maxLocalId: observedMaxLocalId
})
break
}
store.writeScanState({
conversationId,
state: 'done',
imageTotal: imageMessages.length,
imageProcessed: processedInConversation,
maxLocalId: observedMaxLocalId
})
// 会话的图片都处理完了 → 让 Knowledge 重建这个会话,OCR 文本才可被搜索。
try {
await this.deps.onConversationIndexed?.(conversationId)
} catch {
// 索引回调失败不应中断 OCR:派生文本已经落库,下一遍还会再灌。
}
await this.emit()
}
if (this.cancelRequested) this.runState = 'cancelled'
else if (this.pauseRequested) this.runState = 'paused'
else this.runState = 'completed'
this.running = false
await this.emit()
}
// --------------------------------------------------------------- 控制接口
pause(): { paused: boolean; state: ImageTextIndexRunState } {
if (!this.running) return { paused: false, state: this.runState }
this.pauseRequested = true
return { paused: true, state: 'paused' }
}
resume(options: ImageTextIndexStartOptions = {}): { started: boolean; state: ImageTextIndexRunState } {
if (this.running) return { started: false, state: this.runState }
return this.startPass(options)
}
async cancel(): Promise<{ cancellable: boolean; cancelled: boolean }> {
if (!this.running) return { cancellable: false, cancelled: false }
this.cancelRequested = true
const pending = this.passPromise
if (pending) await pending.catch(() => undefined)
return { cancellable: true, cancelled: true }
}
isRunning(): boolean {
return this.running
}
// ------------------------------------------------------------------- 清理
/**
* 清理「图片文字索引能力」的全部派生数据。
*
* 只删本能力自己生成的东西:artifact(OCR 文本)、binding、checkpoint、库文件。
* 明确**不碰**:WCDB、微信图片、图片解密密钥、普通文字知识库、语音转写、聊天消息、Agent 配置。
*/
async clear(): Promise<{ removed: boolean; removedBytes: number }> {
await this.cancel()
this.conversationOcrCache.clear()
const store = this.ensureStore()
/**
* **必须在删之前**记下受影响会话。
*
* OCR 派生文本已经通过 normalizer 进了 Knowledge 的 chunks / FTS。
* 只删派生库、不做这一步,用户执行「清理图片文字索引」之后**仍然能搜到图片里的文字** ——
* 那就等于"清理成功"是假的。硬条件:清理图片文字索引 ≠ 只删 OCR SQLite。
*/
const affectedConversations = store?.conversationIdsWithOcr() ?? []
let removedBytes = 0
if (store) {
removedBytes = store.storageStats().totalBytes
store.clearDerivedData()
// 先折 WAL 再关连接,然后才允许删文件(见 store.close / removeImageTextIndexDatabase)。
store.close()
this.store = null
this.storeAccountId = ''
}
const databasePath = this.accountKey
this.accountKey = ''
const removal = databasePath
? removeImageTextIndexDatabase(databasePath)
: { removed: true, leftovers: [] as string[] }
// 总数统计也一并作废:下次回到「未建立」时重新 COUNT(*),
// 否则 UI 会拿着一个已经没有任何派生数据支撑的旧分母。
this.counters = {
totalImageMessages: 0,
processedThisPass: 0,
indexed: 0,
empty: 0,
missing: 0,
failed: 0
}
this.runState = 'idle'
this.startedAt = undefined
await this.emit()
/**
* 逐个重建**受影响的会话**,让 OCR 派生文本从 Knowledge 里消失。
*
* 刻意不做两件更省事但更糟的事:
* - 不清空整个 Knowledge(那会连普通文字消息的索引一起丢掉);
* - 不假装"删了文件就等于清理完成"(chunks/FTS 里还留着旧文字)。
*
* rebuilt 这里已经返回空(派生库已删、resolver 拿不到 OCR),所以重建出来的
* 会话副本天然不含 OCR 文本;`completeSnapshot` 会把旧的 chunk 一起替换掉。
*/
let invalidatedConversations = 0
for (const conversationId of affectedConversations) {
try {
await this.deps.onConversationIndexed?.(conversationId)
invalidatedConversations += 1
} catch {
// 单个会话重建失败不应让清理整体失败:派生数据已经删了,
// 下一遍索引也会因为 resolver 返回空而自然收敛。
}
}
this.lastInvalidatedConversations = invalidatedConversations
/**
* 重建过程中 Knowledge 会通过 resolver 调 `getConversationOcr`,
* 那会把派生库**重新打开**(`ensureStore`)。清理完必须再收干净:
* 否则「清理成功」之后还留着一个空库句柄,Windows 上也会妨碍目录删除。
*/
this.store?.close()
this.store = null
this.storeAccountId = ''
this.accountKey = ''
// `removed: false` = 文件仍被占用没删掉,必须如实上报,不能假装清理成功。
return { removed: removal.removed, removedBytes }
}
/**
* 缓存清理前先停下任务,并**把 Knowledge 里的 OCR 派生文本一起失效**。
*
* 直接复用 `clear()` 而不是只 `cancel()`:缓存清理(含"清理全部")同样会删掉派生库,
* 如果这里不顺手重建 Knowledge,用户会得到一个自相矛盾的状态 ——
* 「问问微信」里搜得到图片文字,但派生库明明已经没了。
*/
async prepareForCacheClear(): Promise<void> {
await this.clear()
}
/**
* 派生索引修复(Derived Index Repair)。
*
* 只重建 **L3(Knowledge 派生条目 / chunks / FTS)**,数据来源是已有的
* L2 binding + L1 artifact。**绝不**读原图、解密或调用 OCR 引擎 ——
* L1 是几万张图片堆出来的昂贵产物,修一个索引问题不该让它重算一遍。
*
* `ocrExecutions: 0` 不是"期望",而是这条路径的定义:类型上写死成字面量 0,
* 任何让它变成非 0 的改动都会直接编译失败。
*/
async repairKnowledgeIndex(
options: { conversationLimit?: number } = {}
): Promise<ImageTextIndexRepairResult> {
const startedAt = this.now()
// 运行中不并发重建:pass 正在写 binding,同时重建会让 Knowledge 读到半程状态。
if (this.running) {
return { conversations: 0, ocrExecutions: 0, durationMs: 0, skipped: true }
}
const store = this.ensureStore()
if (!store) {
return { conversations: 0, ocrExecutions: 0, durationMs: this.now() - startedAt, skipped: false }
}
const limit = options.conversationLimit && options.conversationLimit > 0 ? options.conversationLimit : undefined
const conversationIds = limit
? store.conversationIdsWithIndexedOcr().slice(0, limit)
: store.conversationIdsWithIndexedOcr()
let conversations = 0
for (const conversationId of conversationIds) {
try {
await this.deps.onConversationIndexed?.(conversationId)
conversations += 1
} catch {
// 单个会话重建失败不影响其余:修索引是尽力而为的廉价操作,可重试。
}
}
return {
conversations,
ocrExecutions: 0,
durationMs: this.now() - startedAt,
skipped: false
}
}
/**
* 重置**可重试的失败记录**(代码修好之后重跑用)。
*
* 刻意不做成"清空整个派生库":那会连已经成功的 OCR 记录一起丢掉,
* 用户要为此重新跑几万张图片。这里只删失败绑定 + 它们的 checkpoint,
* `indexed` / `empty` 一条不动,下一轮 pass 自然接上。
*/
async resetRetriableFailures(): Promise<{ reset: number }> {
await this.cancel()
this.conversationOcrCache.clear()
const store = this.ensureStore()
if (!store) return { reset: 0 }
const reset = store.resetFailures([...IMAGE_OCR_RETRIABLE_FAILURE_STATES])
await this.emit()
return { reset }
}
/** 缓存清理前先停下任务,避免边删边写。 */
}
export const imageTextIndexService = new ImageTextIndexService()
+572
View File
@@ -0,0 +1,572 @@
/**
* 图片文字索引的**派生存储**(与 Knowledge 派生库物理分离)。
*
* 为什么单独一个库而不是往 knowledge.sqlite 里加表:
* - 清理语义干净:整个能力 = 三个文件(.sqlite/-wal/-shm),删掉即可,不留残渣。
* - 零迁移风险:不动已发布的 knowledge schema(§26 要求升级不破坏既有派生库)。
* - 去重语义天然:artifact 按「图片内容 + OCR 运行时指纹」唯一,binding 承担多来源。
*
* 账号隔离与 Knowledge 一致:路径按 accountId 摘要分目录 + 库内 account_id 自证。
*/
import { createHash } from 'node:crypto'
import { mkdirSync, rmSync, statSync, existsSync } from 'node:fs'
import { dirname, join, resolve } from 'node:path'
import { DatabaseSync } from 'node:sqlite'
import {
IMAGE_TEXT_INDEX_SCHEMA_VERSION,
type ImageOcrArtifact,
type ImageOcrBinding,
type ImageOcrPersistedState,
type ImageTextIndexStorageStats
} from '../../shared/image-text-index'
const MAX_SAFE_ACCOUNT_SEGMENT = /^[a-f0-9]{32}$/
function sha256Hex(value: string): string {
return createHash('sha256').update(value).digest('hex')
}
/** 派生库目录名不直接暴露 accountId。 */
export function imageTextIndexAccountKey(accountId: string): string {
return sha256Hex(`image-text-index-account-v1:${accountId}`).slice(0, 32)
}
export function getImageTextIndexDatabasePath(databaseRoot: string, accountId: string): string {
const accountKey = imageTextIndexAccountKey(accountId)
if (!MAX_SAFE_ACCOUNT_SEGMENT.test(accountKey)) {
throw new Error('Invalid image text index account key')
}
return join(resolve(databaseRoot), accountKey, 'image-text-index.sqlite')
}
/**
* 精确删除三件套(与 Knowledge 的 removeKnowledgeDatabase 同构)。
*
* **删除后必须回验**:Windows 上只要还有句柄(WAL/SHM 未关干净、别的进程打开了库),
* `rmSync` 可能不报错却没真的删掉 —— 那就是"看似清理成功,实际没删"。
* 这里把没删掉的路径返回给调用方,让上层能如实报告失败,而不是假装成功。
*/
export function removeImageTextIndexDatabase(databasePath: string): {
removed: boolean
leftovers: string[]
} {
const leftovers: string[] = []
for (const suffix of ['', '-wal', '-shm']) {
const target = `${databasePath}${suffix}`
if (!existsSync(target)) continue
try {
rmSync(target, { force: true })
} catch {
// 删除失败(典型原因是文件仍被占用)→ 由下面的回验兜住。
}
if (existsSync(target)) leftovers.push(target)
}
return { removed: leftovers.length === 0, leftovers }
}
function asRows(value: unknown): Record<string, unknown>[] {
return Array.isArray(value) ? (value as Record<string, unknown>[]) : []
}
function artifactFromRow(row: Record<string, unknown>): ImageOcrArtifact {
return {
accountId: String(row.account_id),
artifactKey: String(row.artifact_key),
imageIdentity: String(row.image_identity),
state: String(row.state) as ImageOcrPersistedState,
text: String(row.text ?? ''),
charCount: Number(row.char_count ?? 0),
engine: String(row.engine),
platform: String(row.platform),
runtimeVersion: row.runtime_version ? String(row.runtime_version) : null,
language: row.language ? String(row.language) : null,
...(row.error_code ? { errorCode: String(row.error_code) } : {}),
createdAt: Number(row.created_at),
updatedAt: Number(row.updated_at)
}
}
/** 会话内「消息 → OCR 文本」,供 Knowledge 索引时解析(与语音 resolver 同构)。 */
export interface ConversationImageOcrEntry {
state: ImageOcrPersistedState
text: string
}
export class ImageTextIndexStore {
private readonly database: DatabaseSync
constructor(
private readonly databasePath: string,
private readonly accountId: string
) {
mkdirSync(dirname(databasePath), { recursive: true })
this.database = new DatabaseSync(databasePath)
this.initialize()
}
private initialize(): void {
this.database.exec(`
PRAGMA journal_mode = WAL;
PRAGMA synchronous = NORMAL;
PRAGMA busy_timeout = 5000;
CREATE TABLE IF NOT EXISTS image_ocr_meta (
key TEXT PRIMARY KEY,
value TEXT NOT NULL
) STRICT;
CREATE TABLE IF NOT EXISTS image_ocr_artifacts (
artifact_key TEXT PRIMARY KEY,
account_id TEXT NOT NULL,
image_identity TEXT NOT NULL,
state TEXT NOT NULL,
text TEXT NOT NULL,
char_count INTEGER NOT NULL DEFAULT 0,
engine TEXT NOT NULL,
platform TEXT NOT NULL,
runtime_version TEXT,
language TEXT,
error_code TEXT,
created_at INTEGER NOT NULL,
updated_at INTEGER NOT NULL
) STRICT;
CREATE INDEX IF NOT EXISTS image_ocr_artifacts_identity
ON image_ocr_artifacts (image_identity);
CREATE TABLE IF NOT EXISTS image_ocr_bindings (
conversation_id TEXT NOT NULL,
message_id TEXT NOT NULL,
account_id TEXT NOT NULL,
create_time INTEGER NOT NULL,
sender_id TEXT,
sender_name TEXT,
image_identity TEXT NOT NULL,
artifact_key TEXT NOT NULL,
state TEXT NOT NULL,
updated_at INTEGER NOT NULL,
PRIMARY KEY (conversation_id, message_id)
) STRICT;
CREATE INDEX IF NOT EXISTS image_ocr_bindings_identity
ON image_ocr_bindings (image_identity);
CREATE INDEX IF NOT EXISTS image_ocr_bindings_state
ON image_ocr_bindings (state);
-- 每个会话的扫描 checkpoint:重启后据此跳过已完成的会话。
CREATE TABLE IF NOT EXISTS image_ocr_scan_state (
conversation_id TEXT PRIMARY KEY,
account_id TEXT NOT NULL,
state TEXT NOT NULL,
image_total INTEGER NOT NULL DEFAULT 0,
image_processed INTEGER NOT NULL DEFAULT 0,
image_max_local_id INTEGER NOT NULL DEFAULT 0,
updated_at INTEGER NOT NULL
) STRICT;
`)
// 探测式加列(与 Knowledge 一致):旧库缺列时补上,不做版本号比较。
const bindingColumns = new Set(
asRows(this.database.prepare('PRAGMA table_info(image_ocr_bindings)').all()).map((row) =>
String(row.name)
)
)
if (!bindingColumns.has('artifact_key')) {
this.database.exec(
"ALTER TABLE image_ocr_bindings ADD COLUMN artifact_key TEXT NOT NULL DEFAULT ''"
)
}
const scanColumns = new Set(
asRows(this.database.prepare('PRAGMA table_info(image_ocr_scan_state)').all()).map((row) =>
String(row.name)
)
)
if (!scanColumns.has('image_max_local_id')) {
// 旧库补列后默认 0:等于「水位未知」,下一次 pass 会重扫该会话并写入真实水位。
this.database.exec(
'ALTER TABLE image_ocr_scan_state ADD COLUMN image_max_local_id INTEGER NOT NULL DEFAULT 0'
)
}
const storedAccount = this.readMeta('account_id')
if (storedAccount && storedAccount !== this.accountId) {
throw new Error('Image text index account isolation check failed')
}
if (!storedAccount) this.writeMeta('account_id', this.accountId)
if (!this.readMeta('schema_version')) {
this.writeMeta('schema_version', String(IMAGE_TEXT_INDEX_SCHEMA_VERSION))
}
}
private readMeta(key: string): string | null {
const row = this.database
.prepare('SELECT value FROM image_ocr_meta WHERE key = ?')
.get(key) as Record<string, unknown> | undefined
return row ? String(row.value) : null
}
private writeMeta(key: string, value: string): void {
this.database
.prepare(
'INSERT INTO image_ocr_meta (key, value) VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value'
)
.run(key, value)
}
// ---------------------------------------------------------------- artifacts
getArtifact(artifactKey: string): ImageOcrArtifact | null {
const row = this.database
.prepare('SELECT * FROM image_ocr_artifacts WHERE artifact_key = ?')
.get(artifactKey) as Record<string, unknown> | undefined
return row ? artifactFromRow(row) : null
}
putArtifact(artifact: ImageOcrArtifact): void {
if (artifact.accountId !== this.accountId) {
throw new Error('Image OCR artifact account does not match database')
}
const existing = this.getArtifact(artifact.artifactKey)
this.database
.prepare(
`INSERT INTO image_ocr_artifacts (
artifact_key, account_id, image_identity, state, text, char_count,
engine, platform, runtime_version, language, error_code, created_at, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(artifact_key) DO UPDATE SET
state = excluded.state,
text = excluded.text,
char_count = excluded.char_count,
runtime_version = excluded.runtime_version,
language = excluded.language,
error_code = excluded.error_code,
updated_at = excluded.updated_at`
)
.run(
artifact.artifactKey,
artifact.accountId,
artifact.imageIdentity,
artifact.state,
artifact.text,
artifact.charCount,
artifact.engine,
artifact.platform,
artifact.runtimeVersion,
artifact.language,
artifact.errorCode ?? null,
existing?.createdAt ?? artifact.createdAt,
artifact.updatedAt
)
}
// ----------------------------------------------------------------- bindings
/** 写入绑定;同一 OCR 结果可被多个会话/消息引用。 */
putBinding(binding: ImageOcrBinding): void {
if (binding.accountId !== this.accountId) {
throw new Error('Image OCR binding account does not match database')
}
this.database
.prepare(
`INSERT INTO image_ocr_bindings (
conversation_id, message_id, account_id, create_time, sender_id, sender_name,
image_identity, artifact_key, state, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(conversation_id, message_id) DO UPDATE SET
create_time = excluded.create_time,
sender_id = excluded.sender_id,
sender_name = excluded.sender_name,
image_identity = excluded.image_identity,
artifact_key = excluded.artifact_key,
state = excluded.state,
updated_at = excluded.updated_at`
)
.run(
binding.conversationId,
binding.messageId,
binding.accountId,
binding.createTime,
binding.senderId ?? null,
binding.senderName ?? null,
binding.imageIdentity,
binding.artifactKey,
binding.state,
binding.updatedAt
)
}
/**
* 某会话的「消息 → OCR 文本」映射。
*
* 与语音的 `withVoiceTranscript` 同构:在**主进程**把派生文本贴到消息上,
* 再交给 Knowledge 索引,派生库不需要被 worker 打开。
*/
getConversationOcr(
conversationId: string
): Map<string, ConversationImageOcrEntry> {
const rows = asRows(
this.database
.prepare(
`SELECT b.message_id AS message_id, b.state AS binding_state,
a.state AS artifact_state, a.text AS text
FROM image_ocr_bindings b
LEFT JOIN image_ocr_artifacts a ON a.artifact_key = b.artifact_key
WHERE b.conversation_id = ?`
)
.all(conversationId)
)
const result = new Map<string, ConversationImageOcrEntry>()
for (const row of rows) {
result.set(String(row.message_id), {
state: String(row.artifact_state || row.binding_state) as ImageOcrPersistedState,
text: String(row.text ?? '')
})
}
return result
}
// -------------------------------------------------------------- checkpoint
/**
* 已持久化的图片消息总数统计。
*
* 必须落盘:派生库只知道自己**处理过**什么,不知道源数据里**一共**有多少图片。
* 一旦把这个 total 只放在内存里,应用重启后 coverage 就会退化成
* 「processed / processed」→ 把 30% 的部分索引谎报成 100% 完整覆盖。
*/
readCountedTotal(): { total: number; countedAt: number; complete: boolean } | null {
const total = this.readMeta('total_image_messages')
const countedAt = this.readMeta('total_image_counted_at')
if (total === null || countedAt === null) return null
const parsedTotal = Number(total)
const parsedCountedAt = Number(countedAt)
if (!Number.isFinite(parsedTotal) || !Number.isFinite(parsedCountedAt)) return null
return {
total: parsedTotal,
countedAt: parsedCountedAt,
// 统计时若有会话没数上(数据库不支持该统计),分母就是偏小的 →
// 绝不能据此声称"已覆盖全部",否则少数的那些会话会被静默算进"已覆盖"。
complete: this.readMeta('total_image_messages_complete') === '1'
}
}
writeCountedTotal(input: { total: number; countedAt: number; complete: boolean }): void {
this.writeMeta('total_image_messages', String(input.total))
this.writeMeta('total_image_counted_at', String(input.countedAt))
this.writeMeta('total_image_messages_complete', input.complete ? '1' : '0')
}
readScanState(): Map<
string,
{ state: string; imageTotal: number; processed: number; maxLocalId: number }
> {
const rows = asRows(
this.database
.prepare(
'SELECT conversation_id, state, image_total, image_processed, image_max_local_id FROM image_ocr_scan_state'
)
.all()
)
const map = new Map<
string,
{ state: string; imageTotal: number; processed: number; maxLocalId: number }
>()
for (const row of rows) {
map.set(String(row.conversation_id), {
state: String(row.state),
imageTotal: Number(row.image_total ?? 0),
processed: Number(row.image_processed ?? 0),
maxLocalId: Number(row.image_max_local_id ?? 0)
})
}
return map
}
writeScanState(input: {
conversationId: string
state: 'done' | 'partial'
imageTotal: number
imageProcessed: number
/** 本会话图片消息的最大插入序(增量水位)。 */
maxLocalId: number
}): void {
this.database
.prepare(
`INSERT INTO image_ocr_scan_state (
conversation_id, account_id, state, image_total, image_processed,
image_max_local_id, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(conversation_id) DO UPDATE SET
state = excluded.state,
image_total = excluded.image_total,
image_processed = excluded.image_processed,
image_max_local_id = excluded.image_max_local_id,
updated_at = excluded.updated_at`
)
.run(
input.conversationId,
this.accountId,
input.state,
input.imageTotal,
input.imageProcessed,
input.maxLocalId,
Date.now()
)
}
// ------------------------------------------------------------------ 统计
/**
* 有 OCR 派生绑定的会话集合。
*
* 清理时**必须**先拿到它:OCR 文本早已被灌进 Knowledge 的 chunks / FTS,
* 只删派生病不会让那些派生文字失效 —— 用户仍会从旧索引里搜到图片里的文字。
*/
conversationIdsWithOcr(): string[] {
const rows = asRows(
this.database.prepare('SELECT DISTINCT conversation_id FROM image_ocr_bindings').all()
)
return rows.map((row) => String(row.conversation_id))
}
/**
* **有 OCR 派生文本**(artifact 里 char_count > 0)的会话集合。
*
* 派生索引修复只需要重建它们:只有这些会话的 Knowledge 里"应该"存在图片派生文字。
* 全是 `empty` 的会话本来就没有派生文字可修,重建它们只是白读一遍 WCDB。
*/
conversationIdsWithIndexedOcr(): string[] {
const rows = asRows(
this.database
.prepare(
`SELECT DISTINCT b.conversation_id AS conversation_id
FROM image_ocr_bindings b
JOIN image_ocr_artifacts a ON a.artifact_key = b.artifact_key
WHERE a.char_count > 0`
)
.all()
)
return rows.map((row) => String(row.conversation_id))
}
/**
* 重置指定状态的失败记录,让它们可以被下一轮 pass 重新处理。
*
* 用途:**代码修好后**,把上一次 bug 造成的假失败(例如整批 `decrypt_failed`)
* 变成可重试状态,而不是要求用户删掉整个派生库 —— 那会连已经成功的记录一起丢掉。
*
* 三件事一起做,缺一不可:
* 1. 删掉这些失败绑定;
* 2. 删掉它们所在会话的 checkpoint —— 否则 pass 会以"该会话已完成"直接跳过,
* 表现为"点了重试但什么都没发生";
* 3. 删掉因此变成孤儿的 artifact(**没有任何绑定再引用的**才删,成功记录一条不动)。
*/
resetFailures(states: ImageOcrPersistedState[]): number {
if (!states.length) return 0
const placeholders = states.map(() => '?').join(', ')
const affected = asRows(
this.database
.prepare(
`SELECT DISTINCT conversation_id FROM image_ocr_bindings WHERE state IN (${placeholders})`
)
.all(...states)
).map((row) => String(row.conversation_id))
const artifactKeys = asRows(
this.database
.prepare(
`SELECT DISTINCT artifact_key FROM image_ocr_bindings WHERE state IN (${placeholders})`
)
.all(...states)
).map((row) => String(row.artifact_key))
const info = this.database
.prepare(`DELETE FROM image_ocr_bindings WHERE state IN (${placeholders})`)
.run(...states)
const clearScan = this.database.prepare(
'DELETE FROM image_ocr_scan_state WHERE conversation_id = ?'
)
for (const conversationId of affected) clearScan.run(conversationId)
const dropOrphan = this.database.prepare(
`DELETE FROM image_ocr_artifacts
WHERE artifact_key = ?
AND NOT EXISTS (SELECT 1 FROM image_ocr_bindings b WHERE b.artifact_key = ?)`
)
for (const artifactKey of artifactKeys) dropOrphan.run(artifactKey, artifactKey)
return Number(info.changes ?? 0)
}
/** 按状态聚合绑定数 —— 覆盖度与进度都从这里取,保证与库内真实一致。 */
countByState(): Record<string, number> { const rows = asRows(
this.database
.prepare('SELECT state, COUNT(*) AS total FROM image_ocr_bindings GROUP BY state')
.all()
)
const counts: Record<string, number> = {}
for (const row of rows) counts[String(row.state)] = Number(row.total ?? 0)
return counts
}
storageStats(): ImageTextIndexStorageStats {
const counts = this.countByState()
const textRow = this.database
.prepare(
"SELECT COUNT(*) AS total FROM image_ocr_artifacts WHERE state = 'indexed' AND length(text) > 0"
)
.get() as Record<string, unknown> | undefined
const updatedRow = this.database
.prepare('SELECT MAX(updated_at) AS latest FROM image_ocr_bindings')
.get() as Record<string, unknown> | undefined
let totalBytes = 0
for (const suffix of ['', '-wal', '-shm']) {
try {
totalBytes += statSync(`${this.databasePath}${suffix}`).size
} catch {
// 文件可能尚未创建;忽略。
}
}
const latest = updatedRow?.latest
return {
indexedImages: counts['indexed'] ?? 0,
ocrTextCount: Number(textRow?.total ?? 0),
totalBytes,
updatedAt: latest === null || latest === undefined ? null : Number(latest)
}
}
/** 清空全部派生数据(表级清空;文件级删除由 service 负责)。 */
clearAll(): void {
this.database.exec(`
DELETE FROM image_ocr_bindings;
DELETE FROM image_ocr_artifacts;
DELETE FROM image_ocr_scan_state;
DELETE FROM image_ocr_meta WHERE key IN (
'total_image_messages',
'total_image_counted_at',
'total_image_messages_complete'
);
`)
}
/**
* 清空并重置检查点。
*
* 注意:必须同时清 `scan_state`,否则清理后再次索引会因为「会话已完成」
* 而直接跳过 —— UI 会停在「未建立」但实际再也不跑。
*/
clearDerivedData(): void {
this.clearAll()
}
close(): void {
try {
// 先折叠 WAL 再关连接:否则 -wal / -shm 可能仍被持有,
// Windows 上会导致后续 rmSync 静默失败("清理成功"但文件还在)。
this.database.exec('PRAGMA wal_checkpoint(TRUNCATE);')
} catch {
// 库可能已经处于不可写状态;关闭仍然要做。
}
try {
this.database.close()
} catch {
// best effort
}
}
}
+137 -5
View File
@@ -1,5 +1,6 @@
import { listContactsAsync, listMessagesAsync, isReady, type FormattedContact, type FormattedMessage } from './chat-service'
import { resolveContact } from './contact-resolution-service'
import { sourceMessageId } from '../knowledge/message-identity'
import type { KnowledgeSearchService } from '../knowledge/knowledge-search-service'
import { inferAiSearchTimeRange } from '../../shared/ai-search'
import { KNOWLEDGE_FRESHNESS_TOLERANCE_MS } from '../../shared/knowledge'
@@ -13,6 +14,7 @@ import type {
ResolvedCorpusScope,
ResolvedTimeRange,
QueryIndexCoverage,
QueryImageTextCoverage,
QuerySearchTimings,
QueryMessagesRequest,
SearchMessagesRequest,
@@ -26,6 +28,12 @@ import {
decodeMessageRef as fromRef,
normalizeMessageIdentity
} from '../../shared/local-query-api'
import {
describeImageTextCoverage,
imageTextCoverageState,
type ImageTextIndexCoverage
} from '../../shared/image-text-index'
import { toEvidenceDisplayText } from '../../shared/knowledge'
const LIMIT_MAX = 200
const CONTEXT_MAX = 50
@@ -112,6 +120,47 @@ function buildIndexCoverage(
}
}
/**
* 图片文字索引未完成时,必须附加的零结果诚实性约束。
*
* 图片 OCR 是**独立的**覆盖维度:它可能是"未建立"或"只做了 30%"。
* 此时 0 条图片证据只是**索引缺口**,不是**事实空缺**。
*/
const IMAGE_OCR_ZERO_RESULT_CAUTION =
'涉及图片、截图、海报里的文字的问题,当前不能因为没搜到就回答"没有"。'
/**
* 图片文字索引覆盖度 → 可直接引用的结论句。
*
* 与 `buildIndexCoverage` 同思路:只给结构化数字,模型会自己换算、甚至反过来
* 宣称"覆盖完整"。这里由 Engine 给出结论句,模型只需引用。
*/
export function buildImageOcrCoverage(
coverage: ImageTextIndexCoverage | null
): QueryImageTextCoverage | undefined {
if (!coverage) return undefined
const state = imageTextCoverageState(coverage)
const countedNote = coverage.countedAt
? `(图片数量统计于 ${formatLocalMinute(coverage.countedAt)})`
: ''
const base = describeImageTextCoverage(coverage)
return {
state,
totalImageMessages: coverage.totalImageMessages,
processed: coverage.processed,
indexed: coverage.indexed,
empty: coverage.empty,
missing: coverage.missing,
failed: coverage.failed,
pending: coverage.pending,
...(coverage.countedAt ? { countedAtLabel: formatLocalMinute(coverage.countedAt) } : {}),
summary:
state === 'complete'
? `${base}${countedNote}`
: `${base}${countedNote}${IMAGE_OCR_ZERO_RESULT_CAUTION}`
}
}
const KIND_LABELS: Record<QueryMessageType, string> = {
text: '文本',
image: '图片',
@@ -149,7 +198,19 @@ function resolvedTimeRange(input: QueryTimeRange, now = new Date()): ResolvedTim
const range = inferAiSearchTimeRange(phrase[input.kind], map[input.kind], now)
return { kind: input.kind, startTime: range.startTime, endTime: range.endTime, label: range.label }
}
function toQueryMessage(conversationId: string, message: FormattedMessage, target: FormattedContact): QueryMessage {
/**
* 一条消息的展示形态。
*
* `imageOcr` 是可选的**派生文本**(来自本地图片文字索引,只读、不触发 OCR)。
* 图片消息的正文永远是空的 —— 识别出的文字必须走独立字段,
* 否则"图片里的文字"会被伪装成"群友发的文字消息"。
*/
function toQueryMessage(
conversationId: string,
message: FormattedMessage,
target: FormattedContact,
imageOcr?: { state: string; text: string }
): QueryMessage {
const kind = kindOf(message)
const content = message.contentData
const attachment =
@@ -163,7 +224,16 @@ function toQueryMessage(conversationId: string, message: FormattedMessage, targe
? { kind: 'file' as const, name: message.exportMediaName || (content?.type === 'share' ? content.title : undefined), url: content?.type === 'share' ? content.url : undefined }
: undefined
const text = message.content?.trim() || message.voiceTranscript?.trim() || undefined
return { messageRef: toRef(conversationId, message.id), timestamp: (message.createTime || 0) * 1000, datetime: message.datetime, sender: message.isSender ? '我' : (message.name || target.m_nsNickName), direction: message.isSender ? 'to_target' : 'from_target', messageType: kind, sourceKind: kind, ...(attachment ? { attachment } : {}), ...(text ? { text } : {}) }
const derived = kind === 'image' ? imageOcr : undefined
const ocrText = derived && derived.state === 'indexed' ? derived.text.trim() : ''
const imageTextState: QueryMessage['imageTextState'] = !derived
? 'not_indexed'
: ocrText
? 'indexed'
: derived.state === 'empty'
? 'empty'
: 'not_indexed'
return { messageRef: toRef(conversationId, message.id), timestamp: (message.createTime || 0) * 1000, datetime: message.datetime, sender: message.isSender ? '我' : (message.name || target.m_nsNickName), direction: message.isSender ? 'to_target' : 'from_target', messageType: kind, sourceKind: kind, ...(attachment ? { attachment } : {}), ...(text ? { text } : {}), ...(ocrText ? { imageOcrText: ocrText, derivedSource: 'image_ocr' as const } : {}), ...(kind === 'image' ? { imageTextState } : {}) }
}
/**
@@ -216,7 +286,49 @@ function outsideScopeError(scope: ResolvedCorpusScope, actual: string): { status
}
export class LocalQueryApiService {
constructor(private readonly knowledge?: KnowledgeSearchService, private readonly nowProvider: () => Date = () => new Date()) {}
/**
* 图片文字索引覆盖度提供者(同步、只读)。
*
* 刻意不在 Knowledge worker 里算:OCR 派生库(`image-text-index.sqlite`)与
* `knowledge.sqlite` 物理分离,worker 不该为了一个覆盖度数字去开它。
*/
private imageTextCoverage: () => ImageTextIndexCoverage | null = () => null
/**
* 单条图片消息的 OCR 派生文本提供者(同步、只读)。
*
* L4(查询层)**只读** L1(OCR artifact)—— 这里绝不允许触发 OCR、解密或读原图。
* 之前 `query_messages` 缺这一环,导致"图片已经识别出文字"这件事在精确读消息
* 这条路径上完全不可见:模型只拿到一个空的 `attachment`,于是把"索引缺口"
* 说成"图片里没有文字",甚至反过来建议用户去建立已经建好的索引。
*/
private imageOcrEntry:
| ((conversationId: string, messageId: string) => { state: string; text: string } | undefined)
| undefined
constructor(
private readonly knowledge?: KnowledgeSearchService,
private readonly nowProvider: () => Date = () => new Date()
) {}
/**
* 注入图片文字索引覆盖度提供者。
*
* 用 setter 而不是构造参数:避免给第二个带默认值的参数写 `undefined` 占位,
* 也让测试可以直接注入假的覆盖度。
*/
setImageTextCoverageProvider(provider: () => ImageTextIndexCoverage | null): void {
this.imageTextCoverage = provider
}
/** 注入单条图片消息的 OCR 派生文本解析器(只读;见 `imageOcrEntry` 的约束)。 */
setImageOcrEntryProvider(
provider:
| ((conversationId: string, messageId: string) => { state: string; text: string } | undefined)
| undefined
): void {
this.imageOcrEntry = provider
}
capabilities(): QueryCapabilitiesResponse {
return { version: 1, tools: { query_messages: { operation: '读取指定联系人的确定性消息', directions: ['any', 'from_target', 'to_target'], messageTypes: kinds, timeRanges: ['all', 'today', 'yesterday', 'this_week', 'last_7_days', 'this_month', 'previous_month', 'this_year', 'previous_year', 'absolute'], limitMax: LIMIT_MAX }, search_messages: { operation: '受限 Knowledge 关键词检索', timeRanges: ['all', 'today', 'yesterday', 'this_week', 'last_7_days', 'this_month', 'previous_month', 'this_year', 'previous_year', 'absolute'], limitMax: LIMIT_MAX }, message_context: { operation: '读取消息前后文', timeRanges: ['all'], limitMax: CONTEXT_MAX }, conversation_overview: { operation: '按会话时间片提取概览证据', timeRanges: ['all', 'today', 'yesterday', 'this_week', 'last_7_days', 'this_month', 'previous_month', 'this_year', 'previous_year', 'absolute'], limitMax: LIMIT_MAX } } }
}
@@ -282,7 +394,18 @@ export class LocalQueryApiService {
const raw = await listMessagesAsync(contact.md5, range.startTime, range.endTime)
const direction = request.direction || 'any'; const allowed = new Set(request.messageTypes || kinds)
const filtered = raw.filter((message) => !(request.excludeSystem !== false && kindOf(message) === 'system')).filter((message) => allowed.has(kindOf(message))).filter((message) => direction === 'any' || (direction === 'to_target' ? message.isSender : !message.isSender)).sort((a, b) => ((a.createTime || 0) - (b.createTime || 0)) * ((request.order || 'asc') === 'asc' ? 1 : -1)).slice(0, Math.min(LIMIT_MAX, Math.max(1, request.limit || 20)))
return { status: 'completed' as const, target: contactView(contact), query: { direction, messageTypes: request.messageTypes || [], order: request.order || 'asc', limit: Math.min(LIMIT_MAX, Math.max(1, request.limit || 20)), excludeSystem: request.excludeSystem !== false, resolvedTimeRange: range }, coverage: { state: 'complete' as const }, returnedCount: filtered.length, messages: filtered.map((message) => toQueryMessage(contact.md5, message, contact)), scope: corpus.scope }
// 图片 OCR 派生文本:L4 只读 L1,**不触发 OCR / 解密 / 读原图**。
// 键必须用 `sourceMessageId(message)`(binding 主键就是它),不能用裸 `message.id`,
// 否则 `local:` 前缀会让查表静默失配 —— 与 Knowledge 用同一条身份规则。
const messages = filtered.map((message) => {
const ocr =
kindOf(message) === 'image'
? this.imageOcrEntry?.(contact.md5, sourceMessageId(message))
: undefined
return toQueryMessage(contact.md5, message, contact, ocr)
})
const imageOcrCoverage = buildImageOcrCoverage(this.imageTextCoverage())
return { status: 'completed' as const, target: contactView(contact), query: { direction, messageTypes: request.messageTypes || [], order: request.order || 'asc', limit: Math.min(LIMIT_MAX, Math.max(1, request.limit || 20)), excludeSystem: request.excludeSystem !== false, resolvedTimeRange: range }, coverage: { state: 'complete' as const }, returnedCount: filtered.length, messages, scope: corpus.scope, ...(imageOcrCoverage ? { imageOcrCoverage } : {}) }
}
async search(request: SearchMessagesRequest) {
const requestStartedAt = Date.now()
@@ -359,6 +482,8 @@ export class LocalQueryApiService {
const covered = indexCovers(found.indexLatestAt, requestedEnd, found.sourceLatestAt)
const indexCoverage = buildIndexCoverage(found.indexLatestAt, found.sourceLatestAt, covered)
// 图片文字索引是**独立覆盖维度**:文字索引再完整也不代表图片里的文字搜得到。
const imageOcrCoverage = buildImageOcrCoverage(this.imageTextCoverage())
const timings: QuerySearchTimings = {
totalMs: Date.now() - requestStartedAt,
scopeMs,
@@ -381,6 +506,7 @@ export class LocalQueryApiService {
sourceLatestAt: found.sourceLatestAt,
freshness: { catchUp: freshness.catchUp },
...(indexCoverage ? { indexCoverage } : {}),
...(imageOcrCoverage ? { imageOcrCoverage } : {}),
timings
}
}
@@ -454,7 +580,11 @@ export class LocalQueryApiService {
timestamp: item.timestamp,
sender: item.sender,
sourceKind: item.sourceKind,
text: item.text,
// 兜底再剥一次:不管 Knowledge 侧哪条检索路径产出的文本,
// 面向用户与模型的都不允许出现 `图片文字:` 这类引擎内部标签。
text: toEvidenceDisplayText(item.text),
...(item.derivedSource ? { derivedSource: item.derivedSource } : {}),
...(item.imageOcrText ? { imageOcrText: item.imageOcrText } : {}),
conversationName: owner ? contactView(owner).displayName : undefined,
conversationType: owner?.type
} satisfies QueryEvidenceItem
@@ -592,6 +722,7 @@ export class LocalQueryApiService {
const truncated = raw.length > OVERVIEW_SOURCE_CAP
const evidence = selectTemporalCoverageEvidence(contact, messages, OVERVIEW_EVIDENCE_TARGET)
const state: 'complete' | 'partial' = truncated ? 'partial' : 'complete'
const imageOcrCoverage = buildImageOcrCoverage(this.imageTextCoverage())
return {
status: 'completed' as const,
target: contactView(contact),
@@ -603,6 +734,7 @@ export class LocalQueryApiService {
selection: { mode: 'temporal_coverage' as const, selectedEvidenceCount: evidence.length, sampled: truncated || evidence.length < messages.length },
evidence,
scope: corpus.scope,
...(imageOcrCoverage ? { imageOcrCoverage } : {}),
origin: 'wcdb' as const
}
}
+104 -11
View File
@@ -106,6 +106,17 @@ export interface QueryAgentTraceItem {
* 会剥离)。用途:把不透明的 Tool 总耗时拆成 scope / freshness / 每个 probe / 合并 / 证据补全。
*/
searchTimings?: QuerySearchTimings
/**
* 本次 Tool Result 里携带 OCR 派生文本的图片消息/证据条数(诊断用,不进模型上下文)。
*
* 存在的意义是让"图片已经识别出文字、但模型没拿到"这类**链路断点**可以被直接观测:
* 真机上曾经出现过 `query_messages` 返回了图片消息却只带 `attachment`、
* 模型因此回答"没有取得 OCR 文字"。当时从回答文本无法判断是"索引没建"还是"没接上",
* 因为这两件事在日志里长得一模一样。有了这个数字就能一眼分开。
*/
imageOcrTextCount?: number
/** 本次 Tool Result 里图片文字索引的覆盖度状态(`not_built` / `partial` / `complete` / `failed`)。 */
imageOcrCoverageState?: string
}
export interface QueryAgentModelCallDiagnostic {
@@ -207,6 +218,12 @@ const SYSTEM_PROMPT = `你是 TraceMemo 的本地聊天查询助手,只能使
规划原则:
- 先判断问题需要哪种证据,再调用最少的 Tool。每次收到 Tool Result 后都判断“当前 Evidence 是否已经足以给出有边界的回答”;足够就立即回答,不为追求绝对完整继续调查。
- query_messages 是精确事实查询,适用于能用联系人、时间、方向、消息类型、顺序等结构条件表达的问题。earliest/latest 等时间边界也是结构条件,必须使用 order 与 limit 精确查询,不能使用抽样 overview。每次调用都必须如实声明 temporalBasis。结果已经回答问题时,不要追加 conversation_overview。
方向(direction)必须按**说话人是谁**来定,不要按语序猜:
- direction 的参照物是“目标会话”:to_target = **我发出**的(说话人是我自己),from_target = **对方发来**的。没有 other 取值,拿不准就用 any。
- 说话人是我 → to_target:“我给张三发了什么”“我发给张三的”“我发给他的文件”“我之前给他发过什么”“我发出去的图片”“我在这个群里发过什么”。
- 说话人是对方 → from_target:“张三给我发了什么”“他之前给我的图片”“张三发给我的文件”。
- **不许**因为“我”出现在句首就选 from_target;也不要凭昵称是否叫“我”来判断说话人,自我身份以消息自身的发送者标记为准。
- 一旦某次 query_messages 返回 0 条,先回头核对 direction 是否与问题的说话人**冲突**;冲突就属于允许的 substantively different retry,必须直接换方向再查一次,**不要**问用户“是不是方向搞错了/要不要换个方向”,用户已经把话说清楚了。
- 需要绝对时间范围时,startTime/endTime 必须使用带时区偏移的 ISO-8601 字符串(例如 2026-08-01T00:00:00+08:00 或 2026-07-31T16:00:00Z)。不要传 epoch 数字,也不要传没有时区的裸本地时间。
- search_messages 是关键词检索,适用于结构条件无法确定答案的问题。queries 的每一项都是一次独立的字面检索:一项只放一个简短关键词,不要把多个近义词或整句话塞进同一项。首次最多 4 项。检索到 Evidence 后直接判断;只有本次完全没有 Evidence 时,才允许再检索一次,且每一项都必须与上一次实质不同。
- conversation_overview 只用于真正需要理解一个时间范围内整体聊了什么、主要话题或整体互动的 broad summary。它返回 temporal coverage sample,不代表完整聊天,也不是检索不足时的默认 fallback。
@@ -219,6 +236,20 @@ const SYSTEM_PROMPT = `你是 TraceMemo 的本地聊天查询助手,只能使
- indexCoverage.covered 为 false 时,说明这段时间还没进索引:此时即使结果为 0 也只能说"索引尚未覆盖这段时间,暂时无法确认",**绝不能**说成"没有"。必须如实引用结论里的索引更新时间。
- 已经检索到 Evidence 时,只有当这个覆盖边界真的会影响结论时才补一句说明,不要机械附加警告。
- 只有 coverage.state 为 complete(indexCoverage.covered 为 true)且结果为 0,才可以下"没有找到"的结论。不要自己把 partial 说成 complete。
图片文字索引:search_messages 的 imageOcrCoverage 是**独立于文字索引**的覆盖维度,只针对“图片里的文字”(截图、报价图、公告截图、海报)。规则:
- 文字消息索引完整**不代表**图片里的文字搜得到。不要把这两个维度混着说。
- 问题涉及图片里的文字、而 imageOcrCoverage.state 不是 complete 时:即使图片证据为 0,也**绝不能**回答“没有”或“没找到”。必须如实引用 imageOcrCoverage.summary,说明图片文字索引尚未完成、当前无法确认全部历史图片。
- imageOcrCoverage.state 为 not_built 时,明确告诉用户图片文字索引还没建立,图片里的文字目前搜不到,并提示可以在「问问微信」里建立。
- 只有 imageOcrCoverage.state 为 complete 且图片证据为 0,才可以下“没有找到”的结论。
图片消息的文字(query_messages 与 search_messages 都适用):
- 图片消息可能带 imageOcrText / derivedSource=image_ocr —— 那是**这张图片里识别出的文字**(本地 OCR 派生),可以直接用它回答“图片里写了什么”。问法可能是“我今早发的那张图片里写了什么”“那张 ChatGPT 价格截图是什么内容”。
- imageOcrText 是派生内容,**证据永远是那条原始图片消息**:messageRef、sender、conversation、时间都只能用原始图片消息的。描述时说“图片里的文字是…”,**不许**把它说成某人发的一条文字消息,**不许**为了它编造任何不存在的消息。
- imageTextState 是**结构化事实**,三种取值含义不同,不要互相替代:
- indexed:已识别出文字(同时有 imageOcrText)。
- empty:本地识别过,这张图里确实没有文字。此时**只能**回答图片本身,**绝不许**根据 OCR 去猜人物、场景、物体或表情包含义(OCR 不是看图,没有 Vision 能力就不要假装有)。
- not_indexed:这条图片还没进图片文字索引。**不许**把“还没索引”说成“图片里没有文字”;若 imageOcrCoverage 不是 complete,必须说明当前无法确认。
- 图片文字索引状态一律以 Tool Result 的结构化字段为准。**不要**在回答里凭空建议“可以先建立图片文字索引再查”——只有 imageOcrCoverage.state 确实是 not_built 时才可以这么说。
- 区分「图片里确实没有文字」(OCR 结果为空,属于已处理的正常终态)与「图片还没被索引」(覆盖缺口):前者是事实,后者不能当成事实。
缺少必要信息时用自然语言澄清;超出工具能力时说明不能可靠完成,并给出当前工具可以执行的替代方向。`
function toolDefinitions(): AIChatToolDefinition[] {
@@ -545,7 +576,7 @@ function retryNote(name: string, result: QueryAgentToolResult, state: ZeroResult
if (result.status !== 'completed') return undefined
const counts = resultCount(result)
if (name === 'search_messages' && !counts.evidenceCount && state.searchAttempts <= ZERO_RESULT_RETRY_LIMIT) return '本次检索没有任何 Evidence。允许再执行一次 search_messages,但每一项都必须与上一次实质不同;完全相同的检索会被拒绝。'
if (name === 'query_messages' && counts.resultCount === 0 && !result.fallbackLookup && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT) return '本次精确查询返回 0 条。允许再执行一次 query_messages,用于放宽 direction 或 messageTypes 等非时间条件;改变时间范围会被拒绝。'
if (name === 'query_messages' && counts.resultCount === 0 && !result.fallbackLookup && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT) return '本次精确查询返回 0 条。只允许放宽 direction 或 messageTypes 等非时间条件(改变时间范围会被拒绝)。**特别注意方向选反这种情况**:如果问题是“我给 X 发 / 我发给 X 的”,而本次用的是 from_target(对方发来),那是方向选反了 —— 直接改用 to_target 重查一次,这属于允许的实质不同重试。不要因为有 0 条就收尾,也不要问用户“是不是方向搞错了 / 要不要换个方向”,用户已经把说话人讲清楚了。'
return undefined
}
@@ -583,10 +614,52 @@ function messageRecordForModel(value: unknown): unknown {
return record.messageType || !record.sourceKind ? record : { ...record, messageType: record.sourceKind }
}
function toolResultForModel(name: string, result: QueryAgentToolResult, callsUsed: number, nextTools: AIChatToolDefinition[], note?: string): QueryAgentToolResult {
const visible: QueryAgentToolResult = { ...result }
/**
* 从 Tool Result 里读出图片 OCR 的两条**结构化事实**(诊断 / 日志用)。
*
* 只看字段存在与否与数量,**不读文本内容**:排查链路断点不需要正文,
* 日志里也不该多留一份聊天内容。
*/
function imageOcrDiagnostics(result: QueryAgentToolResult): {
imageOcrTextCount?: number
imageOcrCoverageState?: string
} {
let count = 0
const collect = (value: unknown): void => {
if (!Array.isArray(value)) return
for (const item of value) {
if (!item || typeof item !== 'object' || Array.isArray(item)) continue
const text = (item as Record<string, unknown>).imageOcrText
if (typeof text === 'string' && text.trim()) count += 1
}
}
collect(result.messages)
collect(result.evidence)
const coverage = result.imageOcrCoverage
const state =
coverage && typeof coverage === 'object' && !Array.isArray(coverage)
? (coverage as Record<string, unknown>).state
: undefined
return {
...(count > 0 ? { imageOcrTextCount: count } : {}),
...(typeof state === 'string' ? { imageOcrCoverageState: state } : {})
}
}
function toolResultForModel(name: string, result: QueryAgentToolResult, callsUsed: number, nextTools: AIChatToolDefinition[], note?: string): QueryAgentToolResult { const visible: QueryAgentToolResult = { ...result }
if (Array.isArray(result.messages)) visible.messages = result.messages.map(messageRecordForModel)
if (Array.isArray(result.evidence)) visible.evidence = result.evidence.map(messageRecordForModel)
if (Array.isArray(result.evidence)) {
visible.evidence = result.evidence.map((item) => {
const record = messageRecordForModel(item)
if (!record || typeof record !== 'object' || Array.isArray(record)) return record
// `imageOcrText` 是给 Evidence UI 做"命中解释"的片段;它的内容已经在 `text` 里,
// 再原样带一份进模型上下文是纯重复。模型侧保留 `derivedSource` 这个语义标记即可,
// 由此知道"这条命中的是图片里的文字"。
const trimmed = { ...(record as Record<string, unknown>) }
delete trimmed.imageOcrText
return trimmed
})
}
if (result.anchor) visible.anchor = messageRecordForModel(result.anchor)
if (Array.isArray(result.before)) visible.before = result.before.map(messageRecordForModel)
if (Array.isArray(result.after)) visible.after = result.after.map(messageRecordForModel)
@@ -616,7 +689,7 @@ function toolResultForModel(name: string, result: QueryAgentToolResult, callsUse
return visible
}
function nextToolDefinitions(name: string, result: QueryAgentToolResult, state: ZeroResultRetryState, rangeWasAll = false): AIChatToolDefinition[] {
function nextToolDefinitions(name: string, result: QueryAgentToolResult, state: ZeroResultRetryState): AIChatToolDefinition[] {
// 重复重试已被拒绝,不再开放工具,避免用有限的 tool budget 反复试同一条件。
if (result.constraint === 'duplicate_retry') return []
if (result.status === 'invalid_tool_arguments') return toolDefinition(name)
@@ -631,10 +704,24 @@ function nextToolDefinitions(name: string, result: QueryAgentToolResult, state:
if (name === 'query_messages') {
// Host 已经自动执行过一次扩大查询:不再开放 retry,避免出现第三次查询。
if (result.fallbackLookup) return []
// 已经查了全部历史且 0 结果:再换时间范围毫无意义(更窄只会更少)。
if (rangeWasAll && counts.resultCount === 0) return []
// 只有 0 结果才开放一次重试;有结果时保持原有 stopping。
return counts.resultCount === 0 && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT ? toolDefinition('query_messages') : []
/**
* 这里**不能**因为"时间范围已经是全部"就关掉重试。
*
* 原实现是 `if (rangeWasAll && resultCount === 0) return []`,依据是"时间不能再放宽了、
* 更窄只会更少"。但 0 结果的重试本来就不是为了改时间 —— 它是为了放宽
* **direction / messageTypes**:「我给 X 发了什么图片」被错判成 `from_target` 时,
* 换成 `to_target` 会从 0 条变成有结果。
*
* 这个守卫的后果正是真机那个回归:工具没发出去 → 第二次调用被
* `tool_availability` 拒掉 → 模型想改向也调不动 → 只能回头问用户"是不是方向搞错了"。
*
* 时间范围不可变由 `constraint_time_range_immutable` 单独把关,
* 完全相同的重试由 `duplicate_retry` 拦下,次数由 ZERO_RESULT_RETRY_LIMIT 限制,
* 所以这里放开是安全的。
*/
return counts.resultCount === 0 && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT
? toolDefinition('query_messages')
: []
}
return []
}
@@ -772,6 +859,12 @@ class EvidenceCollector {
? { messageType: record.sourceKind }
: {}),
...(typeof record.text === 'string' && record.text ? { text: record.text } : {}),
// 「靠图片里的文字命中」这个来源语义必须带到 UI:用户要能看出这条答案来自
// 图片 OCR,而不是群友真发了一条文字消息。messageRef 仍然指向原始图片消息。
...(record.derivedSource === 'image_ocr' ? { derivedSource: 'image_ocr' as const } : {}),
...(typeof record.imageOcrText === 'string' && record.imageOcrText
? { imageOcrText: record.imageOcrText }
: {}),
...(attachmentView && Object.keys(attachmentView).length ? { attachment: attachmentView } : {}),
source
}
@@ -967,10 +1060,10 @@ export class QueryAgentService {
rawTimings && typeof rawTimings === 'object' && !Array.isArray(rawTimings)
? (rawTimings as QuerySearchTimings)
: undefined
result.traces.push({ toolName: call.name, input: sanitizeInput(traceInput), durationMs, status: completedToolResult.status, ...counts, ...(temporalBasis ? { temporalBasis } : {}), ...(autoFallback ? { autoFallback } : {}), ...(searchTimings ? { searchTimings } : {}) })
result.traces.push({ toolName: call.name, input: sanitizeInput(traceInput), durationMs, status: completedToolResult.status, ...counts, ...imageOcrDiagnostics(completedToolResult), ...(temporalBasis ? { temporalBasis } : {}), ...(autoFallback ? { autoFallback } : {}), ...(searchTimings ? { searchTimings } : {}) })
const nextTools = completedToolResult.constraint === 'tool_availability'
? tools
: nextToolDefinitions(call.name, completedToolResult, retry, rangeKind(traceInput) === 'all')
: nextToolDefinitions(call.name, completedToolResult, retry)
const note = retryNote(call.name, completedToolResult, retry)
messages.push({ role: 'tool', tool_call_id: call.id, name: call.name, content: JSON.stringify(toolResultForModel(call.name, completedToolResult, result.toolCallCount, nextTools, note)) })
tools = nextTools
+203 -1
View File
@@ -6,6 +6,7 @@ import { createRequire } from 'module'
import { createConnection, Socket } from 'net'
import { getResourceRoots } from './resource-paths'
import { wcdbDebugLog } from './wcdb-debug'
import type { ImageMessageCountProbe } from '../shared/image-text-index'
export interface Wcdb4Session {
username: string
@@ -1232,11 +1233,15 @@ export class Wcdb4Client {
}
async countVoiceMessagesAsync(
username: string,
md5OrUsername: string,
startTime?: number,
endTime?: number
): Promise<number | null> {
if (!this.wcdbGetMessageTableStats || !this.wcdbExecQuery) return null
// 与图片计数同因的修正:同一个 md5/username 混淆在这里也存在,
// 而且它更隐蔽 —— 匹配不到表时循环不执行,函数会**返回 0 而不是报错**。
const username = this.resolveMessageUsername(md5OrUsername)
if (!username) return null
let tables: Wcdb4MessageStore[]
try {
@@ -1276,6 +1281,203 @@ export class Wcdb4Client {
return total
}
/**
* 消息类型列的可能名字(按顺序探测,命中即用)。
*
* 为什么不能直接硬编码 `"local_type"`:`pickValue(row, [...别名])` 那套别名列表只作用于
* **已经读出来的行**;一旦把列名写进 WHERE,列名不同的库会当场抛错,再被 catch 吞成
* `null` —— 表现就是"检测到 0 张图片"。所以必须先探测真实列名。
*/
private readonly messageTypeColumnCandidates = [
'local_type',
'localType',
'msg_type',
'msgType',
'message_type',
'messageType',
'type',
'WCDB_CT_local_type'
]
/** 每个消息分片的真实类型列名;探测一次即缓存,避免每个会话都跑一次 PRAGMA。 */
private readonly messageTypeColumnCache = new Map<string, string | null>()
private resolveMessageTypeColumn(store: Wcdb4MessageStore): string | null {
const cacheKey = `${store.dbPath}\u0000${store.tableName}`
const cached = this.messageTypeColumnCache.get(cacheKey)
if (cached !== undefined) return cached
let resolved: string | null = null
try {
const columns = this.readMessageColumns(store).map((column) => column.name)
for (const candidate of this.messageTypeColumnCandidates) {
const hit = columns.find((name) => name.toLowerCase() === candidate.toLowerCase())
if (hit) {
resolved = hit
break
}
}
} catch {
resolved = null
}
this.messageTypeColumnCache.set(cacheKey, resolved)
return resolved
}
/**
* 把「会话 md5」解析成原生接口真正需要的 username。
*
* `contact.md5` 是 `md5(wxid)` 的**哈希**(见 chat-service 的 `dbRef.md5(user.m_nsUsrName)`),
* 而 `wcdbGetMessageTableStats` / `wcdbGetMessages` 这些原生接口要的是**原始 username**。
* 直接把 md5 当 username 传,原生侧匹配不到任何表 —— 表现为"未找到该会话的消息表",
* 而按表统计的计数会静默变成 0。
*
* 既有读消息路径一直做了这层转换(`listSourceMessages` 里的 `getUsernameByMd5`),
* **统计/水位路径漏了**,所以这里统一补上。
*
* 解析不到时原样返回:调用方本来就传 username 的路径仍然可用。
*/
private resolveMessageUsername(md5OrUsername: string): string {
const value = String(md5OrUsername || '').trim()
if (!value) return value
const bySession = this.getUsernameByMd5(value)
if (bySession) return bySession
// 有些群只以 `Chat_<md5>` 表存在、不在 session 列表里;退回按聊天表映射解析
// (与 wechat-db 的 `chatMd5ToUsername` 同一套依据)。
try {
const byChatTable = this.getChatTables().find((table) => table.name === `Chat_${value}`)
if (byChatTable?.db_number) return byChatTable.db_number
} catch {
// 映射不可用时退回原值。
}
return value
}
/** 图片消息的 WHERE 片段;`sinceMs` 用于只统计某个时间点之后的消息(测试小窗口)。 */
private imageMessageWhere(column: string, sinceMs?: number): string {
const clauses = [`(${this.quoteSqlIdentifier(column)} & 65535) = 3`]
// 微信的 create_time 是**秒**,调用方给的是毫秒。
if (sinceMs && Number.isFinite(sinceMs) && sinceMs > 0) {
clauses.push(`"create_time" >= ${Math.floor(sinceMs / 1000)}`)
}
return clauses.join(' AND ')
}
/**
* 统计图片消息条数。
*
* 与 `countVoiceMessagesAsync` 同构:纯 SQL COUNT,**不解密任何图片** ——
* 这是「点击索引前先告诉用户有多少张图片」能足够快的前提。
*
* 与语音版本的关键差别:这里**必须区分「0 张」与「统计失败」**。
* `count: null` 表示没数成,调用方绝不能把它当成 0。
*/
async countImageMessagesAsync(
md5OrUsername: string,
sinceMs?: number
): Promise<ImageMessageCountProbe> {
if (!this.wcdbGetMessageTableStats || !this.wcdbExecQuery) {
return { count: null, typeColumn: null, error: '当前数据服务不支持消息表统计' }
}
const username = this.resolveMessageUsername(md5OrUsername)
if (!username) {
return { count: null, typeColumn: null, error: '无法解析该会话的标识' }
}
let tables: Wcdb4MessageStore[]
try {
tables = await this.listMessageStoresAsync(username)
} catch {
return { count: null, typeColumn: null, error: '读取消息分片失败' }
}
if (!tables.length) {
return { count: null, typeColumn: null, error: '未找到该会话的消息表' }
}
let total = 0
let typeColumn: string | null = null
for (const table of tables) {
const column = this.resolveMessageTypeColumn(table)
if (!column) {
return { count: null, typeColumn: null, error: '消息表缺少可识别的消息类型列' }
}
if (!typeColumn) typeColumn = column
try {
const rows = await this.callJsonAsync<Record<string, unknown>[]>(
this.wcdbExecQuery as unknown as KoffiAsyncFunction,
'message',
table.dbPath,
`SELECT COUNT(*) AS "image_count" FROM ${this.quoteSqlIdentifier(table.tableName)} WHERE ${this.imageMessageWhere(column, sinceMs)}`
)
const value = Number(this.pickValue(rows[0] || {}, ['image_count', 'count', 'COUNT(*)']))
if (Number.isFinite(value)) total += value
} catch {
return { count: null, typeColumn: null, error: '图片消息统计查询失败' }
}
}
return { count: total, typeColumn }
}
/**
* 图片消息的增量水位:`count` + `max(local_id)`。
*
* 为什么不能只靠 `countImageMessagesAsync`:
* 图片总数相同**不代表**图片集合没变。撤回一张旧图 + 新增一张新图,count 不变,
* 但新图的 `local_id` 更大。只看 count 会静默跳过该会话,新图片永远搜不到。
*
* `local_id` 是 WCDB 每张消息表内的插入序(自增),所以:
* - 任何 append → `max_local_id` 严格变大;
* - 「删旧 + 增新」且总数不变 → `max_local_id` 也变大,照样被发现;
* - 只有「删掉非最大的那张且不新增」才不变,而此时集合缩小、无需重扫。
*
* 仍然是一条 SQL 聚合,**不解密任何图片**,成本与 count 同量级。
*/
async imageConversationWatermarkAsync(
md5OrUsername: string,
sinceMs?: number
): Promise<{ count: number; maxLocalId: number } | null> {
if (!this.wcdbGetMessageTableStats || !this.wcdbExecQuery) return null
// 与计数同因:必须先把会话 md5 解析成原生接口要的 username,否则永远匹配不到消息表。
const username = this.resolveMessageUsername(md5OrUsername)
if (!username) return null
let tables: Wcdb4MessageStore[]
try {
tables = await this.listMessageStoresAsync(username)
} catch {
return null
}
let count = 0
let maxLocalId = 0
for (const table of tables) {
// 同样探测真实列名:硬编码列名会让水位查询静默失败,进而退化成"永远重扫"或"永远跳过"。
const column = this.resolveMessageTypeColumn(table)
if (!column) return null
try {
const rows = await this.callJsonAsync<Record<string, unknown>[]>(
this.wcdbExecQuery as unknown as KoffiAsyncFunction,
'message',
table.dbPath,
`SELECT COUNT(*) AS "image_count", MAX("local_id") AS "image_max_local_id" FROM ${this.quoteSqlIdentifier(table.tableName)} WHERE ${this.imageMessageWhere(column, sinceMs)}`
)
const row = rows[0] || {}
const tableCount = Number(this.pickValue(row, ['image_count', 'count', 'COUNT(*)']))
const tableMax = Number(
this.pickValue(row, ['image_max_local_id', 'max_local_id', 'MAX("local_id")'])
)
if (Number.isFinite(tableCount)) count += tableCount
if (Number.isFinite(tableMax) && tableMax > maxLocalId) maxLocalId = tableMax
} catch (error) {
console.warn(
`[WCDB4] image watermark failed username=${username} db=${table.dbPath} table=${table.tableName}:`,
error
)
return null
}
}
return { count, maxLocalId }
}
private readSessionRows(): Record<string, unknown>[] {
if (!this.wcdbGetSessions) return []
const rows = this.callJson<Record<string, unknown>[]>((handle, outJson) =>
+18 -2
View File
@@ -58,6 +58,12 @@ import type {
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import type {
ImageTextIndexCountResult,
ImageTextIndexRepairResult,
ImageTextIndexStartOptions,
ImageTextIndexStatus
} from '../shared/image-text-index'
import type { AgentHubActionResult, AgentHubLogEntry, AgentHubStatus } from '../shared/agent-hub'
import type {
PersonalWechatGeneratedTtsVoiceRequest,
@@ -95,7 +101,7 @@ import type {
AppUpdateOpenDownloadPageResult,
AppUpdateState
} from '../shared/app-update'
import type { CacheSummary } from '../shared/cache'
import type { CacheClearScope, CacheSummary } from '../shared/cache'
import type { ExportRequest, ExportJobProgress, ExportResult } from '../shared/export'
import type {
VoiceBatchPreflight,
@@ -211,7 +217,7 @@ declare global {
openAppUpdateDownloadPage: () => Promise<AppUpdateOpenDownloadPageResult>
onAppUpdateState: (callback: (state: AppUpdateState) => void) => () => void
getCacheSummary: () => Promise<CacheSummary>
clearCache: (scope: 'bootstrap' | 'electron' | 'knowledge' | 'all') => Promise<CacheSummary>
clearCache: (scope: CacheClearScope) => Promise<CacheSummary>
openKnowledgeDirectory: () => Promise<{ success: boolean; error?: string }>
initDb: (key: string, accountRoot: string) => Promise<boolean | DatabaseInitResult>
discoverAccounts: (inputPath: string) => Promise<AccountDiscoveryResult>
@@ -664,6 +670,16 @@ declare global {
// 本地图片文字识别(System OCR,本地 Runtime,非 AI Provider)
getSystemOcrCapability: () => Promise<SystemOcrCapability>
recognizeLocalImageText: (request: SystemOcrRequest) => Promise<SystemOcrResult>
getImageTextIndexStatus: () => Promise<ImageTextIndexStatus>
countImageMessages: (sinceMs?: number) => Promise<ImageTextIndexCountResult>
startImageTextIndex: (options?: ImageTextIndexStartOptions) => Promise<{ started: boolean; state: string }>
pauseImageTextIndex: () => Promise<{ paused: boolean; state: string }>
resumeImageTextIndex: (options?: ImageTextIndexStartOptions) => Promise<{ started: boolean; state: string }>
cancelImageTextIndex: () => Promise<{ cancellable: boolean; cancelled: boolean }>
clearImageTextIndex: () => Promise<{ removed: boolean; removedBytes: number }>
resetImageTextIndexFailures: () => Promise<{ reset: number }>
repairImageTextIndex: () => Promise<ImageTextIndexRepairResult>
onImageTextIndexStatus: (callback: (status: ImageTextIndexStatus) => void) => () => void
getPersonalWechatSenderStatus: () => Promise<PersonalWechatSenderStatus>
getPersonalWechatSendCapability: () => Promise<PersonalWechatSendCapability>
getPersonalWechatKeepOneBotProcess: () => Promise<boolean>
+41 -2
View File
@@ -31,6 +31,12 @@ import type {
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import type {
ImageTextIndexCountResult,
ImageTextIndexRepairResult,
ImageTextIndexStartOptions,
ImageTextIndexStatus
} from '../shared/image-text-index'
import type { AgentHubLogEntry, AgentHubStatus } from '../shared/agent-hub'
import type {
PersonalWechatGeneratedTtsVoiceRequest,
@@ -63,7 +69,7 @@ import type { AppLogEntry } from '../shared/app-log'
import type { AppUpdateState } from '../shared/app-update'
import type { GroupExitMonitorState } from '../shared/group-exit-monitor'
import type { ActionLogEntry } from '../shared/action-log'
import type { CacheSummary } from '../shared/cache'
import type { CacheClearScope, CacheSummary } from '../shared/cache'
import type { ExportRequest, ExportJobProgress } from '../shared/export'
import type { ImageDecoderSelectionResult, ImageDecoderStatus } from '../shared/image-decryption'
import type { AccountDiscoveryResult } from '../shared/database-key'
@@ -123,7 +129,7 @@ const api = {
return () => ipcRenderer.removeListener('app-update:state', listener)
},
getCacheSummary: (): Promise<CacheSummary> => ipcRenderer.invoke('cache:getSummary'),
clearCache: (scope: 'bootstrap' | 'electron' | 'knowledge' | 'all'): Promise<CacheSummary> =>
clearCache: (scope: CacheClearScope): Promise<CacheSummary> =>
ipcRenderer.invoke('cache:clear', scope),
openKnowledgeDirectory: (): Promise<{ success: boolean; error?: string }> =>
ipcRenderer.invoke('cache:openKnowledgeDirectory'),
@@ -468,6 +474,39 @@ const api = {
ipcRenderer.invoke('system-ocr:getCapability'),
recognizeLocalImageText: (request: SystemOcrRequest): Promise<SystemOcrResult> =>
ipcRenderer.invoke('system-ocr:recognize', request),
// 图片文字索引(微信图片 → 本地解密 → System OCR → 派生文本 → Knowledge)
getImageTextIndexStatus: (): Promise<ImageTextIndexStatus> =>
ipcRenderer.invoke('image-text-index:getStatus'),
/** 点击索引前的快速统计(SQL COUNT,不解密图片)。 */
countImageMessages: (sinceMs?: number): Promise<ImageTextIndexCountResult> =>
ipcRenderer.invoke('image-text-index:count', sinceMs),
startImageTextIndex: (options?: ImageTextIndexStartOptions): Promise<{ started: boolean; state: string }> =>
ipcRenderer.invoke('image-text-index:start', options),
pauseImageTextIndex: (): Promise<{ paused: boolean; state: string }> =>
ipcRenderer.invoke('image-text-index:pause'),
resumeImageTextIndex: (options?: ImageTextIndexStartOptions): Promise<{ started: boolean; state: string }> =>
ipcRenderer.invoke('image-text-index:resume', options),
cancelImageTextIndex: (): Promise<{ cancellable: boolean; cancelled: boolean }> =>
ipcRenderer.invoke('image-text-index:cancel'),
clearImageTextIndex: (): Promise<{ removed: boolean; removedBytes: number }> =>
ipcRenderer.invoke('image-text-index:clear'),
/** 只重置失败记录(成功记录与其它数据不动),供"修好代码后重跑"。 */
resetImageTextIndexFailures: (): Promise<{ reset: number }> =>
ipcRenderer.invoke('image-text-index:resetFailures'),
/**
* 派生索引修复:只重建 Knowledge 里的图片派生条目与 FTS。
*
* 已有的 OCR 结果(L1)一条都不动 —— 修复索引问题永远不该让几万张图片重算。
*/
repairImageTextIndex: (): Promise<ImageTextIndexRepairResult> =>
ipcRenderer.invoke('image-text-index:repair'),
onImageTextIndexStatus: (callback: (status: ImageTextIndexStatus) => void) => {
const listener = (_event: Electron.IpcRendererEvent, status: ImageTextIndexStatus): void =>
callback(status)
ipcRenderer.on('image-text-index:status', listener)
return () => ipcRenderer.removeListener('image-text-index:status', listener)
},
getPersonalWechatSenderStatus: (): Promise<PersonalWechatSenderStatus> =>
ipcRenderer.invoke('wechat-personal:getStatus'),
getPersonalWechatSendCapability: (): Promise<PersonalWechatSendCapability> =>
@@ -77,9 +77,27 @@ export function AISearchEvidencePanel({
{item.sourceKind === 'voice' && (
<span className="block text-[11px] font-semibold text-primary">语音转写</span>
)}
{item.derivedSource === 'image_ocr' && (
<span
className="mt-0.5 inline-block rounded-sm bg-accent px-1.5 py-0.5 text-[10px] font-semibold text-primary"
data-testid="evidence-image-ocr-badge"
>
图片文字
</span>
)}
<span className="mt-[7px] block overflow-hidden text-[11px] leading-[17px] text-muted-foreground [display:-webkit-box] [-webkit-box-orient:vertical] [-webkit-line-clamp:3]">
{messageText(item.message)}
</span>
{/* 命中解释:明确告诉用户"命中的是图里的这段文字",
避免被读成群友真的发过一条这样的文字消息。 */}
{item.derivedSource === 'image_ocr' && item.imageOcrText && (
<span
className="mt-1 block overflow-hidden text-[11px] leading-[17px] text-foreground [display:-webkit-box] [-webkit-box-orient:vertical] [-webkit-line-clamp:3]"
data-testid="evidence-image-ocr-snippet"
>
“{item.imageOcrText}”
</span>
)}
<Button
variant="link"
size="sm"
@@ -43,6 +43,7 @@ import { ensureAiSearchDataConsent } from './services/aiSearchProviderConsent'
import { ExternalProviderConsentDialog } from './ExternalProviderConsentDialog'
import { AISearchComposer } from './AISearchComposer'
import { AISearchEvidencePanel } from './AISearchEvidencePanel'
import { ImageTextIndexCard } from './ImageTextIndexCard'
import {
forgetAskWechatConversation,
requestAskWechatQuery,
@@ -1394,6 +1395,9 @@ export function AISearchWorkspace({
<p>索引独立保存,不会删除或修改微信原始数据库。</p>
</details>
</section>
{/* 图片文字索引:与 Knowledge 卡片平级、但**独立的一维能力**。
文字消息索引完整不代表图片里的文字搜得到,所以两个入口必须并列可见。 */}
<ImageTextIndexCard dbReady={dbReady} onNotice={onNotice} />
</aside>
<main className="ai-search-main">
<div className="ai-search-main-scroll">
@@ -0,0 +1,427 @@
import { useEffect, useMemo, useState, type ReactElement } from 'react'
import {
AlertDialog,
AlertDialogAction,
AlertDialogCancel,
AlertDialogContent,
AlertDialogFooter,
AlertDialogHeader,
AlertDialogTitle,
Button,
Select,
SelectContent,
SelectItem,
SelectTrigger,
SelectValue
} from '../ui'
import {
describeImageTextCoverage,
imageTextCoverageState,
imageTextProcessedPercent
} from '../../../../shared/image-text-index'
import { useImageTextIndexStatus } from './hooks/useImageTextIndexStatus'
type ImageTextIndexCardProps = {
/** 微信数据是否就绪。 */
dbReady: boolean
onNotice: (message: string) => void
}
/**
* 「图片文字索引」卡片。
*
* 与 Knowledge 卡片**平级并列**(同一组索引入口),但刻意是**独立的一维能力**:
* 文字消息索引完整不代表图片里的文字搜得到。
*
* 文案遵从严禁混淆的语义(§9):这里做的是「识别图片中文字」,不是
* 「本地识图模型 / 本地 Vision / AI OCR」,也不能暗示能理解场景或表情包。
*/
export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProps): ReactElement {
const {
status,
count,
counting,
pending,
running,
paused,
established,
refreshCount,
start,
pause,
resume,
cancel,
resetFailures,
repair
} = useImageTextIndexStatus({ dbReady, onNotice })
const [confirming, setConfirming] = useState(false)
const [confirmCount, setConfirmCount] = useState<number | null>(null)
/**
* 处理时间范围(天)。
*
* 存在的意义是**可验证性**:几万张图片的全量回填没法拿来排查问题,
* 先跑"最近 1 天"这种小窗口才能证明链路真的通了。`0` = 全部历史。
*/
const [rangeDays, setRangeDays] = useState('0')
const sinceMs = useMemo(() => {
const days = Number(rangeDays)
return Number.isFinite(days) && days > 0 ? Date.now() - days * 24 * 60 * 60 * 1000 : undefined
}, [rangeDays])
// 进页面 / 切换范围时统计一次(数字必须与当前窗口一致,否则确认弹窗会说谎)。
useEffect(() => {
if (!dbReady) return
void refreshCount(sinceMs)
}, [dbReady, sinceMs, refreshCount])
const coverage = status?.coverage ?? null
const progress = status?.progress ?? null
const coverageState = coverage ? imageTextCoverageState(coverage) : 'not_built'
/**
* 处理进度百分比。
*
* 刻意不在这里做 `Math.round(x * 100)` —— `45479 / 45707` 会被四舍五入成 `100`,
* 于是出现了"已建立 · 仅完成 100%"这种自相矛盾的显示。未完成时封顶 99.9%。
*/
const percent = coverage
? imageTextProcessedPercent(coverage.processed, coverage.totalImageMessages)
: 0
const systemicFailure = coverage?.systemicFailure === true
const visualState =
progress?.state === 'error' || coverageState === 'failed'
? 'error'
: running
? 'syncing'
: paused
? 'cancelled'
: !established
? 'unavailable'
: coverageState === 'complete'
? 'ready'
: 'building'
const detectedImages = count?.totalImageMessages ?? coverage?.totalImageMessages ?? null
const countFailed = count !== null && count.failedConversations > 0
/**
* 一个会话都没数成。
*
* 这时**绝不能显示 0** —— 那会让用户以为账号里没有图片,从而放弃建立索引。
* 数不出来和确实没有是两件事。
*/
const nothingCounted =
count !== null && count.scannedConversations === 0 && count.failedConversations > 0
const stateLabel = (() => {
if (progress?.state === 'error') return '建立失败'
if (running) return `建立中 · ${percent}%`
if (paused) return `已暂停 · ${percent}%`
if (!established) return '未建立'
// 「已建立」不能等于「全失败」:处理过但一条都没成功时必须叫异常。
if (coverageState === 'failed') return '图片文字索引异常'
if (coverageState === 'complete') return '已完成'
return `部分完成 · ${percent}%`
})()
/**
* 点「建立图片文字索引」:**先重新统计、再弹确认**。
*
* 确认弹窗里的数字必须新鲜——用户可能刚在微信里收了一批图片。
* 统计是纯 SQL COUNT,不解密任何图片,所以这一步够快。
*/
const requestStart = async (): Promise<void> => {
const fresh = await refreshCount(sinceMs)
setConfirmCount(fresh?.totalImageMessages ?? detectedImages)
setConfirming(true)
}
const confirmStart = async (): Promise<void> => {
setConfirming(false)
await start({ ...(sinceMs ? { sinceMs } : {}) })
}
return (
<>
<section
className={`ai-search-knowledge-card ${visualState}`}
aria-label="图片文字索引状态"
>
<div className="ai-search-knowledge-heading">
<div className="ai-search-knowledge-heading-text">
<span className="ai-search-knowledge-kicker">IMAGE TEXT INDEX</span>
<strong className="ai-search-knowledge-state" data-testid="image-text-index-state">
{stateLabel}
</strong>
</div>
<span className="ai-search-knowledge-dot" aria-hidden />
</div>
<p className="ai-search-knowledge-description">
让「问问微信」也能搜索微信图片中的文字(截图、报价图、公告截图等)。
识别在本机进行,原始图片无需发送给 AI Provider。
</p>
{/* 未建立:先告诉用户这个账号大概有多少图片,再让他决定要不要跑。 */}
{!established && !running && (
<>
<div className="ai-search-knowledge-rows">
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">检测到的图片消息</span>
<strong className="ai-search-knowledge-value" data-testid="image-text-index-count">
{counting
? '统计中…'
: nothingCounted
? '无法统计'
: detectedImages === null
? '—'
: detectedImages.toLocaleString()}
</strong>
</div>
</div>
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">处理范围</span>
<Select value={rangeDays} onValueChange={setRangeDays}>
<SelectTrigger data-testid="image-text-index-range" className="h-6 w-[86px] text-[10px]">
<SelectValue />
</SelectTrigger>
<SelectContent>
<SelectItem value="0">全量</SelectItem>
<SelectItem value="1">近 1 天</SelectItem>
<SelectItem value="7">近 7 天</SelectItem>
<SelectItem value="30">近 30 天</SelectItem>
</SelectContent>
</Select>
</div>
{countFailed && (
<p className="ai-search-knowledge-error" data-testid="image-text-index-count-error">
{nothingCounted
? `无法统计本账号的图片消息(${count?.error || '读取消息表失败'})。这不代表账号里没有图片,可以点「重新统计」再试一次。`
: `有 ${count?.failedConversations.toLocaleString()} 个会话未能统计,上面的数字可能偏小。`}
</p>
)}
</>
)}
{/* 进度:只给真实数字,绝不显示 native handle / hash / HRESULT。 */}
{(running || paused) && progress && (
<div className="ai-search-knowledge-pass">
<div className="ai-search-knowledge-rows">
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">已处理</span>
<strong
className="ai-search-knowledge-value"
data-testid="image-text-index-progress"
>
{`${progress.processed.toLocaleString()} / ${progress.totalImageMessages.toLocaleString()}`}
</strong>
</div>
</div>
<div className="ai-search-sync-progress-track">
<span
style={{
width: `${Math.min(100, Math.max(0, progress.percent))}%`,
...(progress.totalImageMessages > 0
? {}
: { animation: 'ai-search-indeterminate 1.4s ease-in-out infinite' })
}}
/>
</div>
<p className="ai-search-knowledge-pass-line">
{`${progress.percent}% · 识别出文字 ${progress.indexed.toLocaleString()} · 没有文字 ${progress.empty.toLocaleString()} · 图片已清理 ${progress.missing.toLocaleString()} · 失败 ${progress.failed.toLocaleString()}`}
</p>
</div>
)}
{/* 已建立:给一份可核对的明细。 */}
{established && !running && !paused && coverage && (
<div className="ai-search-knowledge-rows">
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">已识别出文字</span>
<strong className="ai-search-knowledge-value">
{coverage.indexed.toLocaleString()}
</strong>
</div>
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">没有文字</span>
<strong className="ai-search-knowledge-value">{coverage.empty.toLocaleString()}</strong>
</div>
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">图片已清理</span>
<strong className="ai-search-knowledge-value">
{coverage.missing.toLocaleString()}
</strong>
</div>
{coverage.failed > 0 && (
<div className="ai-search-knowledge-row">
<span className="ai-search-knowledge-label">识别失败</span>
<strong className="ai-search-knowledge-value">
{coverage.failed.toLocaleString()}
</strong>
</div>
)}
<p className="ai-search-knowledge-pass-line">{describeImageTextCoverage(coverage)}</p>
</div>
)}
{progress?.state === 'error' && (
<p className="ai-search-knowledge-error">
{progress.lastError || '图片文字索引建立失败,可以稍后重试。'}
</p>
)}
{/* 「处理过但一条都没成功」= 索引异常,绝不能显示成"已建立"。 */}
{systemicFailure && (
<p
className="ai-search-knowledge-error"
data-testid="image-text-index-systemic-failure"
>
{`${(coverage?.failed ?? 0).toLocaleString()} 条处理失败,成功识别 0 条 —— 当前无法搜索图片中的文字。`}
</p>
)}
{paused && (
<p className="ai-search-knowledge-error">
已暂停。已经识别出的结果都保留了,点「继续」会从断点接着做,不会从第一张重新开始。
</p>
)}
{!dbReady && (
<p className="ai-search-knowledge-error">请先连接微信数据,然后再建立图片文字索引。</p>
)}
<div className="ai-search-knowledge-actions">
{!running && !paused && (
<Button
size="sm"
className="ai-search-knowledge-primary"
data-testid="image-text-index-start"
disabled={!dbReady || pending !== null || counting}
onClick={() => void requestStart()}
>
{established ? '更新图片文字索引' : '建立图片文字索引'}
</Button>
)}
{!running && !paused && countFailed && (
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-recount"
disabled={pending !== null || counting}
onClick={() => void refreshCount()}
>
{counting ? '统计中…' : '重新统计'}
</Button>
)}
{/* 修好之后重跑:只重置失败记录,成功记录与其它数据一律不动。 */}
{!running && !paused && systemicFailure && (
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-reset-failures"
disabled={pending !== null}
onClick={() => void resetFailures()}
>
{pending === 'reset' ? '处理中…' : '重试失败的图片'}
</Button>
)}
{/* 派生索引修复:只重建 Knowledge 里的图片搜索索引,**不重新识别任何图片**。
存在的意义就是"别为修一个索引问题重跑几万张图"。 */}
{!running && !paused && established && (
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-repair"
disabled={pending !== null}
onClick={() => void repair()}
>
{pending === 'repair' ? '修复中…' : '修复图片搜索索引'}
</Button>
)}
{running && (
<>
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-pause"
disabled={pending !== null}
onClick={() => void pause()}
>
{pending === 'pause' ? '暂停中…' : '暂停'}
</Button>
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-cancel"
disabled={pending !== null}
onClick={() => void cancel()}
>
{pending === 'cancel' ? '取消中…' : '取消'}
</Button>
</>
)}
{paused && (
<>
<Button
size="sm"
className="ai-search-knowledge-primary"
data-testid="image-text-index-resume"
disabled={pending !== null}
onClick={() => void resume()}
>
{pending === 'resume' ? '继续中…' : '继续'}
</Button>
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-cancel"
disabled={pending !== null}
onClick={() => void cancel()}
>
取消
</Button>
</>
)}
</div>
</section>
<AlertDialog open={confirming} onOpenChange={setConfirming}>
<AlertDialogContent>
<AlertDialogHeader>
<AlertDialogTitle>建立图片文字索引</AlertDialogTitle>
</AlertDialogHeader>
<div className="ai-search-knowledge-confirm">
<p>
当前账号检测到约{' '}
<strong>
{confirmCount === null ? '未知数量' : confirmCount.toLocaleString()} 条图片消息
</strong>
。
</p>
<p>
建立后,TraceMemo 会在本机读取这些图片中的文字,以后可以在「问问微信」里搜索截图、
报价图、公告截图等图片里的文字,并按结果回到对应的原始图片消息。
</p>
<p>识别过程:</p>
<ul>
<li>仅在本机进行识别,原始图片不会因为本地识别而自动上传</li>
<li>可能需要较长时间,可以暂停并稍后继续</li>
<li>图片已被微信清理或无法解密时会自动跳过</li>
<li>实际可识别的数量取决于本地图片文件是否仍然存在</li>
</ul>
<p>不会修改或删除微信原始图片与聊天记录。</p>
</div>
<AlertDialogFooter>
<AlertDialogCancel>取消</AlertDialogCancel>
<AlertDialogAction
data-testid="image-text-index-confirm"
onClick={() => void confirmStart()}
>
开始索引
</AlertDialogAction>
</AlertDialogFooter>
</AlertDialogContent>
</AlertDialog>
</>
)
}
@@ -52,6 +52,10 @@ export function mapAskWechatEvidence(items: AskWechatEvidenceItem[]): EvidenceIt
return {
evidenceId: `E${index + 1}`,
sourceKind: item.messageType as EvidenceItem['sourceKind'],
// 「靠图片里的文字命中」是来源语义,必须原样带到 UI;
// 但 authoritative source 仍然是原始图片消息(messageRef 已指向它)。
...(item.derivedSource ? { derivedSource: item.derivedSource } : {}),
...(item.imageOcrText ? { imageOcrText: item.imageOcrText } : {}),
contact: evidenceContact(item, anchor),
messageRef: item.messageRef,
message: {
@@ -0,0 +1,235 @@
import { useCallback, useEffect, useState } from 'react'
import type {
ImageTextIndexCountResult,
ImageTextIndexStartOptions,
ImageTextIndexStatus
} from '../../../../../shared/image-text-index'
type UseImageTextIndexStatusOptions = {
/** 微信数据是否已就绪。未就绪时既不统计也不允许建立索引。 */
dbReady: boolean
onNotice: (message: string) => void
}
export type ImageTextIndexAction = 'start' | 'pause' | 'resume' | 'cancel' | 'reset' | 'repair'
/**
* 「图片文字索引」的 renderer 侧状态。
*
* 三条不能省的语义:
* 1. **重启后进度是真的**:进度与覆盖度全部来自主进程的派生库快照,
* renderer 不自己累加、也不缓存百分比。应用重启后重新拉一次即可恢复真实进度。
* 2. **数量统计是显式动作**:COUNT(*) 要走一遍会话列表,不在每次渲染时触发;
* 只在「未建立」时拉一次、以及点击建立前重新拉一次(确认弹窗里的数字必须新鲜)。
* 3. **暂停 / 继续 / 取消都是待确认操作**:主进程返回 started/paused/cancelled
* 才提示成功;例如 `started: false` 表示已经有任务在跑,此时说"已开始"是假话。
*/
export function useImageTextIndexStatus({
dbReady,
onNotice
}: UseImageTextIndexStatusOptions): {
status: ImageTextIndexStatus | null
count: ImageTextIndexCountResult | null
counting: boolean
pending: ImageTextIndexAction | null
running: boolean
paused: boolean
established: boolean
refreshCount: (sinceMs?: number) => Promise<ImageTextIndexCountResult | null>
start: (options?: ImageTextIndexStartOptions) => Promise<void>
pause: () => Promise<void>
resume: () => Promise<void>
cancel: () => Promise<void>
resetFailures: () => Promise<void>
repair: () => Promise<void>
} {
const [status, setStatus] = useState<ImageTextIndexStatus | null>(null)
const [count, setCount] = useState<ImageTextIndexCountResult | null>(null)
const [counting, setCounting] = useState(false)
const [pending, setPending] = useState<ImageTextIndexAction | null>(null)
useEffect(() => {
// 这是一个**次要侧栏能力**:桥接缺失(旧 preload / 测试里手写的 window.api)
// 或推送异常,都不允许把整个「问问微信」拖垮。缺少桥接时按「未建立」降级即可。
const bridge = window.api as unknown as {
getImageTextIndexStatus?: () => Promise<ImageTextIndexStatus>
onImageTextIndexStatus?: (
callback: (status: ImageTextIndexStatus) => void
) => (() => void) | undefined
}
const loadStatus = bridge.getImageTextIndexStatus
const subscribe = bridge.onImageTextIndexStatus
if (typeof loadStatus !== 'function' || typeof subscribe !== 'function') return
let active = true
void loadStatus
.call(bridge)
.then((snapshot) => {
if (active) setStatus(snapshot)
})
.catch(() => undefined)
const unsubscribe = subscribe((snapshot) => {
if (active) setStatus(snapshot)
})
return () => {
active = false
if (typeof unsubscribe === 'function') unsubscribe()
}
}, [])
const refreshCount = useCallback(
async (sinceMs?: number): Promise<ImageTextIndexCountResult | null> => {
if (!dbReady) return null
setCounting(true)
try {
const result = await window.api.countImageMessages(sinceMs)
setCount(result)
return result
} catch (error) {
onNotice(error instanceof Error ? error.message : '统计图片消息数量失败')
return null
} finally {
setCounting(false)
}
},
[dbReady, onNotice]
)
const established = status?.coverage.established ?? false
void established
const start = useCallback(
async (options?: ImageTextIndexStartOptions): Promise<void> => {
if (!dbReady) {
onNotice('请先连接微信数据后再建立图片文字索引')
return
}
setPending('start')
try {
const result = await window.api.startImageTextIndex(options)
if (!result.started) {
onNotice('图片文字索引已经在进行中')
return
}
onNotice('已开始建立图片文字索引,可以继续使用软件')
} catch (error) {
onNotice(error instanceof Error ? error.message : '启动图片文字索引失败')
} finally {
setPending(null)
}
},
[dbReady, onNotice]
)
const pause = useCallback(async (): Promise<void> => {
setPending('pause')
try {
const result = await window.api.pauseImageTextIndex()
onNotice(result.paused ? '已暂停,已完成的识别结果会保留' : '当前没有正在进行的索引')
} catch (error) {
onNotice(error instanceof Error ? error.message : '暂停失败')
} finally {
setPending(null)
}
}, [onNotice])
const resume = useCallback(
async (options?: ImageTextIndexStartOptions): Promise<void> => {
setPending('resume')
try {
const result = await window.api.resumeImageTextIndex(options)
onNotice(result.started ? '已继续建立图片文字索引' : '索引已经在进行中')
} catch (error) {
onNotice(error instanceof Error ? error.message : '继续失败')
} finally {
setPending(null)
}
},
[onNotice]
)
const cancel = useCallback(async (): Promise<void> => {
setPending('cancel')
try {
const result = await window.api.cancelImageTextIndex()
if (!result.cancellable) {
onNotice('当前没有正在进行的索引')
return
}
if (!result.cancelled) {
onNotice('索引刚刚已经结束,无需取消')
return
}
onNotice('已取消,已识别的结果会保留,下次可从中断处继续')
} catch (error) {
onNotice(error instanceof Error ? error.message : '取消失败')
} finally {
setPending(null)
}
}, [onNotice])
/**
* 重置失败记录(代码修好后重跑)。
*
* 只说"已重置 N 条"是不够的 —— 必须同时讲清楚**成功记录没有被删**,
* 否则用户会以为刚才把已经跑好的结果也清掉了。
*/
const resetFailures = useCallback(async (): Promise<void> => {
setPending('reset')
try {
const result = await window.api.resetImageTextIndexFailures()
onNotice(
result.reset > 0
? `已把 ${result.reset.toLocaleString()} 条失败记录重置为待处理;已成功识别的记录保持不变。可以点「更新图片文字索引」重新处理这些图片`
: '没有需要重置的失败记录'
)
} catch (error) {
onNotice(error instanceof Error ? error.message : '重置失败记录失败')
} finally {
setPending(null)
}
}, [onNotice])
/**
* 派生索引修复:只重建 Knowledge 里的图片派生条目(L3),**不重新 OCR**(L1 不动)。
*
* 措辞必须讲清楚"没有重新识别":否则用户会以为又要等一小时,
* 从而不敢点这个按钮 —— 而这个按钮存在的全部意义就是"别重跑几万张图"。
*/
const repair = useCallback(async (): Promise<void> => {
setPending('repair')
try {
const result = await window.api.repairImageTextIndex()
if (result.skipped) {
onNotice('索引任务正在进行中,请等它结束后再修复搜索索引')
return
}
onNotice(
result.conversations > 0
? `已重建 ${result.conversations} 个会话的图片搜索索引;没有重新识别任何图片(已识别结果全部复用)`
: '没有需要重建的图片搜索索引'
)
} catch (error) {
onNotice(error instanceof Error ? error.message : '修复图片搜索索引失败')
} finally {
setPending(null)
}
}, [onNotice])
return {
status,
count,
counting,
pending,
running: status?.progress.state === 'running',
paused: status?.progress.state === 'paused',
established,
refreshCount,
start,
pause,
resume,
cancel,
resetFailures,
repair
}
}
@@ -24,6 +24,10 @@ export const mapPipelineEvidenceItem = (
return {
evidenceId: item.id,
sourceKind: item.sourceKind,
// 「靠图片里的文字命中」的来源语义与 OCR 片段同样要带到 UI,
// 否则 Legacy 检索路径下用户看不到「图片文字」标记(两条路径表现会不一致)。
...(item.derivedSource ? { derivedSource: item.derivedSource } : {}),
...(item.imageOcrText ? { imageOcrText: item.imageOcrText } : {}),
contact,
// 这条路径本来就同时知道真实会话 id 与消息 id,顺手补上稳定引用,
// 让 Legacy / ai-search 证据也能被精确定位(而不是只有 Query Agent 路径能跳准)。
@@ -34,6 +34,15 @@ export interface EvidenceItem {
/** Program-owned Final Evidence ID. Cached legacy records may omit it. */
evidenceId?: string
sourceKind?: KnowledgeMessageKind
/**
* 命中所依赖的派生来源。
*
* `image_ocr` = 这条结果靠**图片里的文字**命中,而不是群友真的发了一条文字消息。
* 有值时 Evidence 卡片显示轻量来源标记(「图片文字」)。
*/
derivedSource?: 'image_ocr'
/** 「从图片里读出来的文字」片段,只作命中解释。 */
imageOcrText?: string
contact: Contact
message: Message
/**
@@ -1,6 +1,16 @@
import { useCallback, useEffect, useState } from 'react'
import type { CacheSummary } from '../../../../../shared/cache'
import { Button } from '../../../components/ui'
import type { CacheSummary, CacheClearScope } from '../../../../../shared/cache'
import {
AlertDialog,
AlertDialogAction,
AlertDialogCancel,
AlertDialogContent,
AlertDialogDescription,
AlertDialogFooter,
AlertDialogHeader,
AlertDialogTitle,
Button
} from '../../../components/ui'
const SEARCH_CACHE_KEYS = [
'wxe_ai_search_cache_v8',
@@ -25,9 +35,9 @@ export function CacheCleanupPage({
onNotice: (message: string) => void
}): React.ReactElement {
const [summary, setSummary] = useState<CacheSummary | null>(null)
const [busyScope, setBusyScope] = useState<
'bootstrap' | 'electron' | 'knowledge' | 'knowledge-directory' | 'all' | 'local' | null
>(null)
const [busyScope, setBusyScope] = useState<CacheClearScope | 'knowledge-directory' | 'local' | null>(null)
/** 需要二次确认的清理范围(目前只有图片文字索引)。 */
const [confirmingScope, setConfirmingScope] = useState<CacheClearScope | null>(null)
const refresh = useCallback(async (): Promise<void> => {
setSummary(await window.api.getCacheSummary())
@@ -44,7 +54,7 @@ export function CacheCleanupPage({
onNotice('已清理检索和导出本地缓存')
}
const clear = async (scope: 'bootstrap' | 'electron' | 'knowledge' | 'all'): Promise<void> => {
const clear = async (scope: CacheClearScope): Promise<void> => {
setBusyScope(scope)
try {
setSummary(await window.api.clearCache(scope))
@@ -54,9 +64,11 @@ export function CacheCleanupPage({
onNotice(
scope === 'knowledge'
? '已清理所有账号的本地知识库索引,需要时可在问问微信中重新建立'
: scope === 'all'
? '已清理全部可恢复缓存和检索记录'
: '缓存已清理'
: scope === 'image-text-index'
? '已清理图片文字索引,微信原始图片与聊天记录未受影响;需要时可在问问微信中重新建立'
: scope === 'all'
? '已清理全部可恢复缓存和检索记录'
: '缓存已清理'
)
} catch (error) {
onNotice(error instanceof Error ? error.message : '清理缓存失败')
@@ -65,6 +77,33 @@ export function CacheCleanupPage({
}
}
/**
* 清理图片文字索引。两步各司其职,不能省成一步:
*
* 1. `clearImageTextIndex()` —— 主进程先停任务、折叠 WAL、关连接、删三件套,
* 并**回验文件是否真的删掉**(Windows 上文件被占用时 rmSync 会静默失败)。
* 2. `clearCache('image-text-index')` —— 再扫掉整个派生目录(含其它账号的派生库),
* 并返回刷新后的占用摘要。
*
* 只要第 1 步回验失败,就必须如实报告,不能说"已清理"。
*/
const clearImageTextIndex = async (): Promise<void> => {
setBusyScope('image-text-index')
try {
const result = await window.api.clearImageTextIndex()
setSummary(await window.api.clearCache('image-text-index'))
onNotice(
result.removed
? '已清理图片文字索引;微信原始图片、聊天记录和普通文字知识库都未受影响。需要时可在「问问微信」里重新建立'
: '图片文字索引的数据文件仍被占用,没能完全删除。请重启 TraceMemo 后再试一次'
)
} catch (error) {
onNotice(error instanceof Error ? error.message : '清理图片文字索引失败')
} finally {
setBusyScope(null)
}
}
const openKnowledge = async (): Promise<void> => {
setBusyScope('knowledge-directory')
try {
@@ -134,9 +173,14 @@ export function CacheCleanupPage({
<Button
variant="outline"
size="sm"
data-testid={`cache-clear-${item.id}`}
disabled={busyScope !== null}
aria-busy={busyScope === item.id}
onClick={() => void clear(item.id)}
onClick={() =>
item.id === 'image-text-index'
? setConfirmingScope('image-text-index')
: void clear(item.id)
}
>
{busyScope === item.id ? '清理中...' : '清理'}
</Button>
@@ -169,6 +213,41 @@ export function CacheCleanupPage({
</div>
</div>
</div>
{/* 图片文字索引是「重新建立成本很高」的派生数据,必须二次确认并写清不可逆的范围。 */}
<AlertDialog
open={confirmingScope === 'image-text-index'}
onOpenChange={(open) => setConfirmingScope(open ? 'image-text-index' : null)}
>
<AlertDialogContent>
<AlertDialogHeader>
<AlertDialogTitle>清理图片文字索引?</AlertDialogTitle>
<AlertDialogDescription>
将删除 TraceMemo 本地生成的图片 OCR 文本和对应搜索索引。
</AlertDialogDescription>
</AlertDialogHeader>
<div className="settings-confirm-detail">
<p>不会删除:</p>
<ul>
<li>微信原始图片</li>
<li>微信聊天记录</li>
<li>普通文字知识库</li>
<li>微信数据库</li>
</ul>
<p>清理后,「问问微信」将无法搜索图片中的文字;之后可以重新建立。</p>
</div>
<AlertDialogFooter>
<AlertDialogCancel>取消</AlertDialogCancel>
<AlertDialogAction
data-testid="cache-clear-image-text-index-confirm"
className="bg-destructive text-destructive-foreground hover:bg-destructive/90"
onClick={() => void clearImageTextIndex()}
>
确认清理
</AlertDialogAction>
</AlertDialogFooter>
</AlertDialogContent>
</AlertDialog>
</div>
)
}
+27
View File
@@ -278,6 +278,33 @@
line-height: 15px;
}
/* 「建立图片文字索引」确认弹窗的正文:侧栏卡片用的 10px 在弹窗里太挤,
这里单独给一档更大的字号,并保持与卡片一致的次要文字色。 */
.ai-search-knowledge-confirm {
display: flex;
flex-direction: column;
gap: 8px;
color: var(--wxex-text-muted);
font-size: 12px;
line-height: 19px;
p {
margin: 0;
}
strong {
color: var(--wxex-text-primary);
}
ul {
margin: 0;
padding-left: 18px;
display: flex;
flex-direction: column;
gap: 3px;
}
}
.ai-search-knowledge-error {
color: var(--wxex-warning);
}
@@ -79,6 +79,29 @@
}
}
/* 清理类确认弹窗的正文(「不会删除……」清单)。
侧栏卡片那种 10px 在弹窗里太小,这里单独给一档。 */
.settings-confirm-detail {
display: flex;
flex-direction: column;
gap: 8px;
color: var(--wxex-text-muted);
font-size: 12px;
line-height: 19px;
p {
margin: 0;
}
ul {
margin: 0;
padding-left: 18px;
display: flex;
flex-direction: column;
gap: 3px;
}
}
.voice-runtime-card dl {
display: grid;
grid-template-columns: repeat(3, minmax(0, 1fr));
+7 -2
View File
@@ -1,7 +1,12 @@
export type CacheClearScope = 'bootstrap' | 'electron' | 'knowledge' | 'all'
export type CacheClearScope =
| 'bootstrap'
| 'electron'
| 'knowledge'
| 'image-text-index'
| 'all'
export interface CacheSummaryItem {
id: 'bootstrap' | 'electron' | 'knowledge'
id: 'bootstrap' | 'electron' | 'knowledge' | 'image-text-index'
label: string
description: string
sizeBytes: number
+427
View File
@@ -0,0 +1,427 @@
/**
* 图片文字索引(Image OCR Derived Text)契约。
*
* 硬规则(与语音转写同源的设计约束):
* - 微信图片消息是 **authoritative source**,OCR 文本是 **derived content**。
* - OCR 文本绝不写回原始消息、绝不修改 WCDB、绝不伪装成用户发送的文字消息。
* - OCR 命中时 Evidence 必须回到**原始图片消息**,而不是一条虚构的 OCR 消息。
*
* 因此这里刻意分成两层:
* 1. `ImageOcrArtifact` —— 按「图片内容 + OCR 运行时指纹」去重的派生文本(可能一张图被转发到多个会话)。
* 2. `ImageOcrBinding` —— 「某个会话里的某条图片消息 → 某个 artifact」的绑定,保证去重不丢来源。
*/
/** 派生文本的引擎标识;与 System OCR 的引擎常量保持一致。 */
export const IMAGE_TEXT_INDEX_ENGINE = 'windows-system-ocr'
/** 派生库自身的 schema 版本(与 Knowledge 的 schema 相互独立)。 */
export const IMAGE_TEXT_INDEX_SCHEMA_VERSION = 1
/**
* OCR 并发上限。
*
* 当前实现**严格串行**(循环体内只有一次 await,无 Promise.all 扇出),等价于 1。
* 这个常量是后续调高的唯一入口:Windows OCR 是进程内 WinRT 调用,实测单张
* 20–40ms,串行已足够;调高只会和 Query Agent 抢 CPU。
*/
export const DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY = 1
/** 每个批次的图片条数;批间让出 event loop,保证 UI / 查询不被卡住。 */
export const IMAGE_TEXT_INDEX_BATCH_SIZE = 12
/** 已完成一批之后、回到会话循环前的让出时间。 */
export const IMAGE_TEXT_INDEX_YIELD_MS = 0
/** 单张图片的 OCR 结果状态。 */
export type ImageOcrState =
/** 尚未处理 */
| 'pending'
/** 正在处理(进程中断后会回到 pending) */
| 'processing'
/** 成功识别出文字 */
| 'indexed'
/** 成功识别,但图片里没有文字(表情包 / 风景 / 头像…)——这是**正常终态**,不重试 */
| 'empty'
/** 图片消息本身缺少定位字段(md5 / datName),无法找到文件 */
| 'metadata_missing'
/** 图片文件已不存在(微信清理过原图与缩略图)——**正常终态**,不是 OCR 失败 */
| 'image_missing'
/**
* 解密服务不可用(运行时环境问题)。
*
* **这不是单张图片的失败** —— 它意味着整条流水线的前置依赖缺失。
* 它的存在会阻断 `complete`,并且正常流程应该在 preflight 就拦下、根本不写这种状态。
*/
| 'decrypt_unavailable'
/** 找到了文件,但解密失败(密钥/账号上下文不对,或文件损坏) */
| 'decrypt_failed'
/** 解密产出无法识别为图片格式(解码失败) */
| 'decode_failed'
/** 解码成功,但 OCR 执行失败 */
| 'ocr_failed'
/** 用户取消时正在处理 */
| 'cancelled'
/**
* 可以落库的状态。
*
* `processing` 是瞬态的(只存在于一次 pass 的内存里):进程崩溃后它没有任何意义,
* 而且它绝不允许进入 Knowledge 索引 —— Knowledge 只应该看到"已定态"。
*/
export type ImageOcrPersistedState = Exclude<ImageOcrState, 'processing'>
/** 终态集合:落在这里的状态不会在下次 pass 被自动重试。 */
export const IMAGE_OCR_TERMINAL_STATES: readonly ImageOcrState[] = [
'indexed',
'empty',
'metadata_missing',
'image_missing',
'decrypt_failed',
'decode_failed',
'ocr_failed',
'cancelled'
]
/**
* 运行时不可用态:**不是**单张图片的终态。
*
* 它与终态分开,是为了让「这 4.5 万张都失败了」永远不能被当成"该条已处理"。
*/
export const IMAGE_OCR_RUNTIME_UNAVAILABLE_STATES: readonly ImageOcrState[] = [
'decrypt_unavailable'
]
export function isTerminalImageOcrState(state: ImageOcrState): boolean {
return IMAGE_OCR_TERMINAL_STATES.includes(state)
}
export function isRuntimeUnavailableImageOcrState(state: ImageOcrState): boolean {
return IMAGE_OCR_RUNTIME_UNAVAILABLE_STATES.includes(state)
}
/**
* **可重试的失败态**。
*
* 代码修好之后,这些状态的记录可以安全地重跑 —— 它们要么是运行时依赖缺失,
* 要么是"当时环境不对"造成的失败。重置只删这些绑定与它们的 checkpoint,
* 成功记录(indexed / empty)一条都不动。
*/
export const IMAGE_OCR_RETRIABLE_FAILURE_STATES: readonly ImageOcrPersistedState[] = [
'decrypt_unavailable',
'decrypt_failed',
'decode_failed',
'ocr_failed'
]
/**
* OCR 运行时指纹。
*
* 缓存身份**不能只是图片 hash**:换 OCR 引擎 / 升级运行时 / 换语言配置之后
* 必须允许重新识别,否则用户会永远拿到旧引擎的结果。
*/
export interface ImageOcrProvenance {
engine: string
platform: string
runtimeVersion: string | null
/** 实际使用的 OCR 语言标签;null 表示由系统用户语言决定。 */
language: string | null
}
/**
* artifact 去重键 = 图片内容身份 + OCR 运行时指纹。
*
* 刻意不包含 conversationId / messageId —— 同一张图片被转发到多个会话时,
* OCR 只算一次,但会有多条 binding 指向同一个 artifact。
*/
export function buildImageOcrArtifactKey(input: {
imageIdentity: string
provenance: ImageOcrProvenance
}): string {
const { imageIdentity, provenance } = input
return [
imageIdentity,
provenance.engine,
provenance.platform,
provenance.runtimeVersion ?? 'unknown',
provenance.language ?? 'auto'
].join('|')
}
/** 派生文本记录(按 artifact key 唯一)。 */
export interface ImageOcrArtifact {
accountId: string
artifactKey: string
imageIdentity: string
state: ImageOcrPersistedState
/** OCR 正文;`empty` 状态为空串。 */
text: string
charCount: number
engine: string
platform: string
runtimeVersion: string | null
language: string | null
/** 只在失败时写入;用于诊断,绝不包含 OCR 正文。 */
errorCode?: string
createdAt: number
updatedAt: number
}
/** 「某会话的某条图片消息」到 artifact 的绑定。 */
export interface ImageOcrBinding {
accountId: string
conversationId: string
messageId: string
/** Unix epoch **毫秒**(Knowledge 契约统一用毫秒)。 */
createTime: number
senderId?: string
senderName?: string
/** 图片内容身份(去重维度 1)。 */
imageIdentity: string
/**
* 指向的 artifact(内容身份 + OCR 运行时指纹)。
*
* 必须携带完整 artifact key 而不是只存 imageIdentity:换了 OCR 引擎/运行时之后
* 同一张图会有多个 artifact,绑定必须能精确指到"这次用哪个指纹算出来的文本"。
*/
artifactKey: string
state: ImageOcrPersistedState
updatedAt: number
}
/** 索引任务运行态。 */
export type ImageTextIndexRunState =
| 'idle'
| 'counting'
| 'running'
| 'paused'
| 'completed'
| 'cancelled'
| 'error'
/** 进度(面向 UI;只含数字与状态,绝不含 OCR 正文 / 路径 / wxid)。 */
export interface ImageTextIndexProgress {
state: ImageTextIndexRunState
/** 检测到的图片消息总数(SQL 统计,未解密)。 */
totalImageMessages: number
processed: number
indexed: number
empty: number
missing: number
failed: number
/** 运行时不可用(如解密服务缺失);不计入 processed,且会阻断 complete。 */
runtimeUnavailable: number
/** 系统性失败(处理过但一条都没成功)——UI 必须显示"异常"而不是"已建立"。 */
systemicFailure: boolean
pending: number
/** 0–100,保留 1 位小数;未完成时封顶 99.9。 */
percent: number
/** 处理进度百分比(与 percent 同源,语义化别名)。 */
processedPercent: number
startedAt?: number
updatedAt: number
cancellable: boolean
paused: boolean
lastError?: string
}
/**
* 图片文字索引的覆盖度 —— **独立的覆盖维度**。
*
* 文字消息索引 100% 不代表图片文字可用;Query Agent 必须能单独看到这一维。
*/
export interface ImageTextIndexCoverage {
totalImageMessages: number
/** 已进入**非运行时**终态的条数(indexed + empty + missing + failed)。 */
processed: number
indexed: number
empty: number
missing: number
failed: number
/**
* 运行时不可用(如解密服务缺失)的条数。
*
* 单独一列、**不计入 processed**:它代表"流水线前置依赖缺失",
* 绝不能与"这条图片已经处理过了"混为一谈。
*/
runtimeUnavailable: number
pending: number
/** 是否建立过(有落盘统计且处理过)。 */
established: boolean
/** 是否**真正**覆盖完整(分母可信 + 无 pending + 无运行时不可用 + 不是"全军覆没")。 */
complete: boolean
/**
* 系统性失败:处理过一批,但 indexed / empty / missing 全为 0、失败却不为 0。
*
* 这就是"4.5 万张全部失败、却告诉用户已建立"那种情况的判据 ——
* 它必须阻断 `complete`,并让 UI 显示"异常"。
*/
systemicFailure: boolean
/**
* `totalImageMessages` 的统计时刻(epoch ms);null = 从未统计过。
*
* 必须有这个时间戳:total 是**某一时刻**的 SQL 统计,之后微信里新增的图片
* 还没进索引。只说"已覆盖全部 N 条"而不给统计时刻,就是在把「当时完整」
* 冒充成「现在完整」。
*/
countedAt: number | null
}
/** 覆盖度状态(外加"未建立")。UI 与 Query Agent 共用同一判据,避免两处各推一套口径漂移。 */
export type ImageTextCoverageState = 'not_built' | 'partial' | 'complete' | 'failed'
export function imageTextCoverageState(coverage: ImageTextIndexCoverage): ImageTextCoverageState {
if (!coverage.established) return 'not_built'
if (coverage.systemicFailure) return 'failed'
return coverage.complete ? 'complete' : 'partial'
}
/**
* 处理进度百分比。
*
* 保留 1 位小数,且**未完成时封顶 99.9%**:
* `Math.round(45479 / 45707 * 100)` 会得到 `100`,于是出现了"已建立 · 仅完成 100%"
* 这种自相矛盾的显示。进度条可以近似,结论句不行。
*/
export function imageTextProcessedPercent(processed: number, total: number): number {
if (!(total > 0)) return 0
const raw = (processed / total) * 100
if (raw >= 100) return 100
return Math.min(99.9, Math.round(raw * 10) / 10)
}
/** 覆盖度的人话结论,供 Query Agent / UI 直接引用。 */
export function describeImageTextCoverage(coverage: ImageTextIndexCoverage): string {
if (!coverage.established) {
return '图片文字索引尚未建立:目前只能搜索文字消息,图片里的文字还搜不到。'
}
if (coverage.systemicFailure) {
return `图片文字索引当前异常:已处理的 ${coverage.processed.toLocaleString()} 条图片消息全部失败(成功识别 0 条、无文字 0 条、图片缺失 0 条)。当前无法搜索图片中的文字。`
}
if (coverage.complete) {
return `图片文字索引已覆盖全部 ${coverage.totalImageMessages.toLocaleString()} 条图片消息。`
}
const percent = imageTextProcessedPercent(coverage.processed, coverage.totalImageMessages)
return `图片文字索引只完成 ${percent}%(${coverage.processed.toLocaleString()} / ${coverage.totalImageMessages.toLocaleString()} 条图片消息),当前图片搜索结果可能不完整。`
}
/** 快速统计结果(不含解密)。 */
export interface ImageTextIndexCountResult {
totalImageMessages: number
scannedConversations: number
/**
* 统计失败(拿不到数)的会话数。
*
* 必须与 `totalImageMessages = 0` 区分开:**"一张图片都没有"和"根本没数成"是两件事**。
* 把后者显示成 0 会让用户以为账号里没有图片,从而放弃建立索引 —— 这正是本功能
* 一直在避免的那类谎话。
*/
failedConversations: number
/** 实际用于判定"这是图片消息"的列名;null = 一个会话都没探测到。 */
typeColumn: string | null
/** 失败原因摘要(仅供诊断,不含用户数据)。 */
error?: string
durationMs: number
}
/**
* 单个会话的图片消息计数探针。
*
* `count: null` = **统计失败**,不等于 0 张。调用方必须区分处理。
*/
export interface ImageMessageCountProbe {
count: number | null
/** 实际用于判定图片消息的类型列名。 */
typeColumn: string | null
/** 失败原因摘要(不含任何用户内容)。 */
error?: string
}
/**
* 会话级增量水位。
*
* 刻意用 **两个** 判据而不是只比 count:
* - `count` 能发现大多数增删;
* - `maxLocalId`(消息插入序的最大值)能发现「总数相同但集合变了」——
* 例如撤回一张旧图的同时新增一张新图,count 不变但新图的 local_id 更大。
*
* 只用 count 会静默漏掉新图片;只用 create_time 会被「后到的旧时间消息」
* (网络延迟 / 消息恢复 / 合并转发回填)骗过。`local_id` 是 WCDB 行内单调的
* 插入序,对 append 与「等量替换」两种情况都成立。
*/
export interface ImageMessageWatermark {
count: number
/** 该会话图片消息的最大插入序;没有图片时为 0。 */
maxLocalId: number
}
/**
* 派生索引修复的结果。
*
* 分层前提(任何一层都不许越界去动上一层):
* - L1 Image OCR Artifact —— 昂贵,持久化,**尽量永不重复计算**
* - L2 Message Binding —— 便宜,可修复
* - L3 Knowledge Derived Entry / FTS —— 便宜,可重建
* - L4 Query Agent / Evidence —— 查询层,只读
*
* 修 L2/L3/L4 **绝不能**自动清 L1。`ocrExecutions` 因此被写死成字面量 `0`:
* 修复路径一旦开始调 OCR,类型就不再成立,编译期就会拦下来。
*/
export interface ImageTextIndexRepairResult {
/** 实际重建了派生索引的会话数。 */
conversations: number
/** 永远是 0 —— 修复路径禁止触发 OCR(这一条是契约,不是观察值)。 */
ocrExecutions: 0
durationMs: number
/** 索引任务正在运行时拒绝并发修复(避免读到半程 binding)。 */
skipped: boolean
}
/** 派生数据占用(设置 → 缓存与清理)。 */
export interface ImageTextIndexStorageStats {
indexedImages: number
ocrTextCount: number
totalBytes: number
updatedAt: number | null
}
/** 索引过程中用于写入派生库的单条结果。 */
export interface ImageOcrWriteInput {
accountId: string
conversationId: string
messageId: string
createTime: number
senderId?: string
senderName?: string
/** 已解密的图片内容身份;取不到图片时为 null。 */
imageIdentity: string | null
state: ImageOcrPersistedState
text: string
provenance: ImageOcrProvenance
errorCode?: string
}
/** 索引服务的启动参数。 */
export interface ImageTextIndexStartOptions {
/** 只处理前 N 个会话,用于受控 smoke;不传 = 全量。 */
conversationLimit?: number
/** 只处理前 N 条图片消息,用于受控 smoke。 */
messageLimit?: number
/**
* 只处理这个时刻(epoch ms)**之后**的图片消息;不传 = 全部历史。
*
* 存在的意义是**可验证性**:几万张图片的全量回填没法用来排查问题,
* 先跑"最近一天"这种小窗口才能证明链路是通的。
* 带窗口运行时会**跳过增量跳过逻辑**(每次都重扫窗口内的消息),
* 因为 checkpoint 是围绕全量集合建立的,混用会让"跳过"变得不可解释。
*/
sinceMs?: number
}
/** 对外状态快照(问问微信卡片 / 设置清理页共用同一份)。 */
export interface ImageTextIndexStatus {
progress: ImageTextIndexProgress
coverage: ImageTextIndexCoverage
storage: ImageTextIndexStorageStats
/** 正在做「检测到多少条图片消息」的 SQL 统计。 */
counting: boolean
}
+51
View File
@@ -42,8 +42,30 @@ export interface KnowledgeSourceMessage {
voiceTranscript?: string
/** Local coverage state only. Error text is never copied into the index. */
voiceTranscriptState?: 'pending' | 'transcribed' | 'failed'
/**
* 图片里的文字(本地 System OCR 的派生结果)。
*
* 与 voiceTranscript 同构:这是 **derived content**,原图片消息仍然是
* authoritative source。它绝不写回 message.body,也绝不产生"OCR 消息"。
*/
imageOcrText?: string
/** 图片 OCR 的本地状态;与 voiceTranscriptState 一样不含错误正文。 */
imageOcrState?: KnowledgeImageOcrState
}
/** 图片 OCR 的本地覆盖状态(错误详情绝不进索引)。 */
export type KnowledgeImageOcrState =
| 'pending'
| 'indexed'
| 'empty'
| 'metadata_missing'
| 'image_missing'
| 'decrypt_unavailable'
| 'decrypt_failed'
| 'decode_failed'
| 'ocr_failed'
| 'cancelled'
export interface KnowledgeNormalizedMessage extends KnowledgeSourceMessage {
searchableText: string
contentHash: string
@@ -194,9 +216,38 @@ export interface KnowledgeEvidence {
/** The source type belongs to the original message, not the retrieval method. */
sourceKind: KnowledgeMessageKind
text: string
/**
* 这条证据里「从图片里读出来的文字」(本地 System OCR 的派生结果)。
*
* 只用于**来源解释**:让用户/模型知道这段内容来自图片,而不是群友真的发了一条文字消息。
* authoritative source 始终是原始图片消息 —— 这里不产生任何"OCR 消息"。
*/
imageOcrText?: string
/**
* 命中所依赖的**派生来源**。
*
* 有值 = 这条结果依赖本地派生内容才能命中(而不是原始消息本身的文字)。
* 与 `sourceKind` 正交:`sourceKind` 说的是原始消息是什么,这里说的是"靠什么搜到的"。
*/
derivedSource?: 'image_ocr'
score?: number
}
/**
* 证据文本面向用户 / 模型时的可读化处理。
*
* `searchableText` 里的 `图片文字:` 只是索引期用来区分派生内容的内部标签,
* 它**绝不能出现在 Evidence 里**:用户不该看到引擎内部前缀,
* 而且"这段文字来自图片"应该由结构化的来源标记表达,而不是靠一个冒号前缀。
*/
export function toEvidenceDisplayText(searchableText: string): string {
return searchableText
.split('\n')
.map((line) => line.replace(/^\s*(?:图片文字|OCR|system-ocr)\s*[::]\s*/i, ''))
.join('\n')
.trim()
}
export interface KnowledgeVoiceCoverage {
voiceMessageCount: number
transcribedVoiceCount: number
+86
View File
@@ -112,6 +112,21 @@ export interface QueryEvidenceItem
/** 该证据所属会话的展示名(群名 / 联系人名)。 */
conversationName?: string
conversationType?: 'user' | 'group'
/**
* 命中所依赖的**派生来源**(与 `sourceKind` 正交)。
*
* 有值时 Evidence UI 加一个轻量来源标记(如「图片文字」),
* 让用户知道这段内容来自**图片里的文字**,而不是群友真的发了一条文字消息。
* authoritative source 仍然是原始图片消息,`messageRef` 也仍然指向原图。
*/
derivedSource?: 'image_ocr'
/**
* 「从图片里读出来的文字」片段,只用作命中解释。
*
* 刻意与 `text` 分开:`text` 是这条消息的内容,这里只回答"命中是因为图里的哪段文字"。
* 普通文字消息不会有这个字段。
*/
imageOcrText?: string
}
/**
@@ -217,6 +232,28 @@ export interface QueryMessage {
url?: string
sizeBytes?: number
}
/**
* 图片 OCR 派生文本(**仅**图片消息、且本地已识别出文字时存在)。
*
* 它是 derived content,不是消息正文:这条消息的正文仍然是空的"图片附件",
* authoritative evidence 也仍然是**原始图片消息**(`messageRef` 指向它)。
* 之所以必须单独一个字段而不是塞进 `text`:一旦混进去,模型与 UI 就无法区分
* "群友发了一段文字"和"图片里识别出这段文字",而这正是本功能的诚实性前提。
*/
imageOcrText?: string
/** 派生来源语义:`image_ocr` = 这段文字来自图片识别,而不是原始文字消息。 */
derivedSource?: 'image_ocr'
/**
* 这条图片消息在本地图片文字索引里的状态。
*
* - `indexed`:识别过且有文字(此时 `imageOcrText` 有值)
* - `empty`:识别过,但图里确实没有文字 —— 这是**已知结论**,不是"没索引"
* - `not_indexed`:尚未进入索引(未建立 / 还没处理到 / 已被清理)
*
* 区分这三者是硬要求:`not_indexed` 不允许被当成"图里没内容",
* `empty` 也不允许被当成"可以凭画面猜内容"(OCR 不是 Vision)。
*/
imageTextState?: 'indexed' | 'empty' | 'not_indexed'
}
export interface QueryMessagesResponse {
status: string
@@ -228,6 +265,14 @@ export interface QueryMessagesResponse {
candidates?: Array<{ displayName: string; type: 'user' | 'group' }>
/** 本次实际使用的语料边界。 */
scope?: ResolvedCorpusScope
/**
* 图片文字索引的覆盖度。
*
* 与 `search_messages` 同源同口径 —— 精确读消息这条路径同样必须知道
* "图片里的文字到底索引了多少",否则模型在图片文字尚未索引时
* 只能看到一个光秃秃的 `attachment`,进而把"索引缺口"说成"图片没有文字"。
*/
imageOcrCoverage?: QueryImageTextCoverage
}
export interface SearchMessagesRequest {
target: QueryTarget
@@ -279,6 +324,13 @@ export interface SearchMessagesResponse {
* **本地时间**与结论,模型只需引用,不需要自己判断,也不需要输出 epoch 数字。
*/
indexCoverage?: QueryIndexCoverage
/**
* 图片文字索引覆盖度(**独立于**文字索引的维度)。
*
* `state !== 'complete'` 时,涉及图片/截图/海报的问题**不允许**因为 0 条证据
* 就回答"没有"——必须说明图片文字索引尚未完成、当前结果无法覆盖全部图片。
*/
imageOcrCoverage?: QueryImageTextCoverage
/** 本次检索的真实耗时分解(ADDITIVE,用于诊断与 UI 展示;不进入模型上下文)。 */
timings?: QuerySearchTimings
}
@@ -294,6 +346,33 @@ export interface QueryIndexCoverage {
summary: string
}
/**
* 图片文字索引(本地 OCR 派生文本)的覆盖度 —— 与文字索引覆盖度**互相独立**。
*
* 为什么必须单独一个维度:文字消息索引 100% 不代表图片里的文字可被搜索。
* 图片 OCR 是用户确认后才建立的重活,可能"未建立",也可能"只做了 30%"。
* 这时如果模型因为 0 条证据就回答"没有",就是把**索引缺口**说成了**事实空缺**。
*/
export interface QueryImageTextCoverage {
/**
* `failed` = 索引**当前异常**(处理过一批但一条都没成功,或运行时依赖缺失)。
*
* 它与 `partial` 都必须让 Query Agent 拒绝凭零结果下"没有"的结论。
*/
state: 'not_built' | 'partial' | 'complete' | 'failed'
totalImageMessages: number
processed: number
indexed: number
empty: number
missing: number
failed: number
pending: number
/** 图片数量统计时刻(本地时间 `MM-DD HH:mm`);从未统计时为 undefined。 */
countedAtLabel?: string
/** 可直接引用的结论句;模型只引用,不要自己换算或推断。 */
summary: string
}
/**
* `search_messages` 的真实耗时分解(ADDITIVE 诊断字段)。
*
@@ -346,6 +425,13 @@ export interface ConversationOverviewResponse {
evidence?: QueryEvidenceItem[]
candidates?: Array<{ displayName: string; type: 'user' | 'group' }>
scope?: ResolvedCorpusScope
/**
* 图片文字索引覆盖度(**独立维度**,与 `voiceCoverage` 平级)。
*
* 会话概览以源数据为准,所以能如实反映"这段时间聊了什么";但"图片里的文字"
* 只存在于本地 OCR 派生索引里,概览的完整性**不覆盖**这一维。
*/
imageOcrCoverage?: QueryImageTextCoverage
/**
* 证据来源:`wcdb` = 直接读源数据(会话概览的事实来源);`knowledge` = 派生索引。
* 派生索引可能滞后,故概览以源数据为准。
+27
View File
@@ -33,6 +33,15 @@ export interface AskWechatEvidenceItem {
timestamp?: number
messageType?: string
text?: string
/**
* 命中所依赖的派生来源(与 `messageType` 正交)。
*
* `image_ocr` = 这条结果靠**图片里的文字**命中,而不是群友真的发了一条文字消息。
* Evidence UI 会据此显示轻量来源标记。authoritative source 仍是原始图片消息。
*/
derivedSource?: 'image_ocr'
/** 「从图片里读出来的文字」片段,只作命中解释(普通文字消息不会有)。 */
imageOcrText?: string
attachment?: { kind?: string; name?: string; url?: string; sizeBytes?: number }
/** 产生这条证据的 Tool(诊断 / 分组)。 */
source: string
@@ -133,6 +142,24 @@ export interface QueryAgentDiagnostics {
tools: string[]
totalMs: number
outcome: AskWechatOutcome
/**
* 本次查询里**实际取到 OCR 派生文本**的图片消息/证据条数(诊断,不含正文)。
*
* `0` 配合 `tools` 就能区分两种完全不同的故障:
* 图片文字索引没建(索引问题),还是建好了但查询路径没接上(链路问题)。
*/
imageOcrTextCount?: number
/** 本次查询里图片文字索引的覆盖度状态(`not_built` / `partial` / `complete` / `failed`)。 */
imageOcrCoverageState?: string
/**
* 用户问题原文。
*
* 这一条**刻意**包含聊天内容:排查"同一个问题为什么这次答对上次答错"必须知道问的是什么。
* 日志只写在用户本机的应用日志目录(设置 → 检索诊断里可查看 / 清空),不上传、不进遥测。
*/
question?: string
/** 模型最终回答原文(同上,仅本地日志,用于排查)。 */
answer?: string
}
export type AskWechatOutcome =