feat: 新增mac ocr转文字,增加到问一问微信 图片索引优化速度

- 图片文字索引性能与进度诚实化
- 问问微信:证据卡区分「消息类型」与「派生来源」,派生命中内容自报来源
- 问问微信:回答规则禁止未真实执行的多轮承诺
- 本地图片文字识别:支持 macOS 系统 OCR(Apple Vision)
This commit is contained in:
Wxw-Gu
2026-09-17 13:11:20 +08:00
parent b8f08d54c8
commit 1337bcb7de
53 changed files with 4985 additions and 624 deletions
+62 -33
View File
@@ -32,10 +32,7 @@ import {
resolveFfmpegExecutable,
type DecodedImage
} from './image-decrypt-service'
import {
exportGroupReportSnapshot,
extractGroupReportRenderSnapshot
} from './group-report-service'
import { exportGroupReportSnapshot, extractGroupReportRenderSnapshot } from './group-report-service'
import {
deleteGeneratedReport,
listGeneratedReports,
@@ -45,9 +42,7 @@ import {
} from './report-history-service'
import { reportTemplateService } from './report-template-service'
import { registerReportTemplateIpc } from './report-template-ipc'
import type {
GroupReportRenderSnapshotExportRequest
} from '../shared/group-report'
import type { GroupReportRenderSnapshotExportRequest } from '../shared/group-report'
import type {
SaveGeneratedReportRequest,
PrepareGeneratedReportTemplateSwitchRequest,
@@ -667,10 +662,8 @@ app.whenReady().then(async () => {
/**
* 图片文字索引需要解密图片。
*
* 原先这个依赖直接读 `imageDecryptService`,而它**只在 `db:getImage`(用户点开某张图)
* 里才懒加载** —— 于是全量回填在用户没点开过任何图片时拿到 `null`,
* 45,479 张图片全部被记成 `decrypt_failed`(见事故报告)。
* 这里改成显式的"按需确保",凡是需要解密的路径都能自己把它建起来。
* 解密服务原本只在 `db:getImage`(用户点开某张图)里才懒加载,于是没点开过图片时
* 全量回填会拿到 `null`。这里改成显式"按需确保",凡是需要解密的路径都能自己建起来。
*/
imageTextIndexService.bind({
databaseRoot: join(app.getPath('userData'), 'image-text-index'),
@@ -687,13 +680,46 @@ app.whenReady().then(async () => {
type: contact.type
}))
},
listMessages: (conversationId) => chat.listMessagesAsync(conversationId),
listMessages: (conversationId) =>
chat.listMessagesAsync(
conversationId,
undefined,
undefined,
undefined,
undefined,
'image-text-index'
),
/**
* 图片索引走**专用查询**:只读图片消息,不读整个会话。
*
* 全量读取一个 20 万条消息的会话实测要 15s 以上,而其中 99% 以上的行
* 图片索引根本不看 —— 那是数据边界错了,不是 OCR 慢。
*/
listImageMessages: (conversationId) =>
chat.listImageMessagesAsync(conversationId, undefined, 'image-text-index'),
countConversationImages: (conversationId, sinceMs) =>
chat.countImageMessagesAsync(conversationId, sinceMs),
imageWatermark: (conversationId, sinceMs) =>
chat.imageConversationWatermarkAsync(conversationId, sinceMs),
decryptService: () => ensureImageDecryptService(),
capability: () => systemOcrService.getCapability(),
/**
* OCR 并发度的运行时覆盖;不设置则走 `DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY`。
*
* 同一份二进制、同一批图片只改这一个数,才能把并发度当作对照变量来比较。
* 非法值会被 `resolveImageTextOcrConcurrency` 收敛掉。
*/
...(process.env.TRACEMEMO_OCR_CONCURRENCY
? { ocrConcurrency: Number(process.env.TRACEMEMO_OCR_CONCURRENCY) }
: {}),
/** 低频性能画像:只写性能数字,不含图片内容 / 路径 / 会话标识。 */
logStageProfile: (profile) =>
appLogger.write({
level: 'info',
scope: 'image-text-index',
message: '图片文字索引性能画像',
details: { ...profile }
}),
recognize: async (imageDataUrl) => {
const result = await systemOcrService.recognize({ imageDataUrl })
return {
@@ -719,8 +745,7 @@ app.whenReady().then(async () => {
*
* 与覆盖度是**两件不同的事**:覆盖度回答"索引建了多少",这里回答
* "这一条图片已经识别出的文字是什么"。只接前者的话,图片索引建好了模型也读不到正文,
* 只能看到一个空的 `attachment` —— 真机上就是这么把"图片里有 ChatGPT 价格"
* 答成"没有取得 OCR 文字"的。
* 只能看到一个空的 `attachment`。
*
* 只读派生库,**不触发 OCR / 解密 / 读原图**。
*/
@@ -740,7 +765,13 @@ app.whenReady().then(async () => {
if (!aiSearchPipelineService) throw new Error('本地搜索服务尚未初始化')
return aiSearchPipelineService.run(request, () => undefined)
},
log: (record) => appLogger.write({ level: record.level, scope: 'query-agent', message: record.message, details: record.details })
log: (record) =>
appLogger.write({
level: record.level,
scope: 'query-agent',
message: record.message,
details: record.details
})
})
agentHubService.setQueryAgentService(queryAgentService)
knowledgeSearchService.onStatusChange((status) => {
@@ -1128,8 +1159,8 @@ app.whenReady().then(async () => {
aesKey: result.aesKey
})
if (saved.success) imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
return {
...result,
success: saved.success,
@@ -1150,8 +1181,8 @@ app.whenReady().then(async () => {
ipcMain.handle('image:saveConfig', (_, request: SaveImageKeyRequest) => {
const result = imageKeyConfigService.save(request)
if (result.success) imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
return result
})
@@ -1162,8 +1193,8 @@ app.whenReady().then(async () => {
ipcMain.handle('image:clearConfig', () => {
const result = imageKeyConfigService.clear()
if (result.success) imageDecryptService = null
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
// 派生库按 accountId 分目录,切账号必须换句柄,否则会串账号。
imageTextIndexService.resetAccount()
return result
})
@@ -1288,7 +1319,7 @@ app.whenReady().then(async () => {
const requestId = nextGetMessagesRequestId()
const startedAt = Date.now()
wcdbDebugLog(
`[${requestId}] IPC db:getMessages start userMd5=${userMd5} start=${startTime || 0} end=${endTime || 0} limit=${options?.limit || 0}`
`[${requestId}] IPC db:getMessages start start=${startTime || 0} end=${endTime || 0} limit=${options?.limit || 0}`
)
try {
const messages = await chat.listMessagesAsync(
@@ -1349,7 +1380,7 @@ app.whenReady().then(async () => {
return { messages: [], found: false, radiusSeconds: 0, truncated: false }
}
wcdbDebugLog(
`[${requestId}] IPC db:getMessagesAround start userMd5=${userMd5} messageId=${target.messageId} anchor=${anchorSeconds || 0}`
`[${requestId}] IPC db:getMessagesAround start messageId=${target.messageId} anchor=${anchorSeconds || 0}`
)
for (const radius of radii) {
const start = Math.max(0, (anchorSeconds as number) - radius)
@@ -1461,14 +1492,12 @@ app.whenReady().then(async () => {
ipcMain.handle('image-text-index:count', (_, sinceMs?: number) =>
imageTextIndexService.countImageMessages(sinceMs)
)
ipcMain.handle(
'image-text-index:start',
(_, options?: ImageTextIndexStartOptions) => imageTextIndexService.startPass(options ?? {})
ipcMain.handle('image-text-index:start', (_, options?: ImageTextIndexStartOptions) =>
imageTextIndexService.startPass(options ?? {})
)
ipcMain.handle('image-text-index:pause', () => imageTextIndexService.pause())
ipcMain.handle(
'image-text-index:resume',
(_, options?: ImageTextIndexStartOptions) => imageTextIndexService.resume(options ?? {})
ipcMain.handle('image-text-index:resume', (_, options?: ImageTextIndexStartOptions) =>
imageTextIndexService.resume(options ?? {})
)
ipcMain.handle('image-text-index:cancel', () => imageTextIndexService.cancel())
ipcMain.handle('image-text-index:clear', () => imageTextIndexService.clear())
@@ -2036,12 +2065,12 @@ app.whenReady().then(async () => {
)
// ============================================================
// 本地图片文字识别(System OCR / Windows System OCR Runtime)
// 本地图片文字识别(System OCR Runtime:Windows 系统 OCR / macOS 系统 OCR)
// ============================================================
// 这是本地 Runtime,不是 AI Vision Provider:
// - 不联网、不上传原图;
// - 不读写 AI Provider / Vision 模型配置;
// - 结果不落库(派生内容,本轮只做内存级闭环)。
// - 本 IPC 只返回识别文本、不落库:派生文本的持久化由图片文字索引负责。
ipcMain.handle('system-ocr:getCapability', async (): Promise<SystemOcrCapability> => {
return imageInsightService.getSystemOcrCapability()
})
@@ -2049,8 +2078,8 @@ app.whenReady().then(async () => {
ipcMain.handle(
'system-ocr:recognize',
async (_, request: SystemOcrRequest): Promise<SystemOcrResult> => {
// 日志只记录结构性信息,不记录 base64、不记录识别正文。
console.log('[IPC] system-ocr:recognize hash=%s', request?.imageHash || 'auto')
// 单图识别是用户主动触发的一次操作,结果里已经带了 text / durationMs / errorCode,
// 调用方直接用返回值判断即可,这里不再打日志(尤其不打稳定的图片标识)。
return imageInsightService.extractLocalText(request)
}
)
@@ -1081,7 +1081,7 @@ export class KnowledgeSearchService {
lane: WcdbReadLane = 'interactive'
): ReturnType<typeof chat.listMessagesAsync> {
return this.enqueueWcdbRead(
() => chat.listMessagesAsync(conversationId, startTime, endTime),
() => chat.listMessagesAsync(conversationId, startTime, endTime, undefined, undefined, 'knowledge'),
lane
)
}
+14 -2
View File
@@ -765,7 +765,8 @@ export class KnowledgeStore {
evidence: asRows(
this.database
.prepare(
`SELECT m.conversation_id, m.message_id, m.create_time, m.searchable_text, m.kind, m.sender_id, m.sender_name
`SELECT m.conversation_id, m.message_id, m.create_time, m.searchable_text, m.kind, m.sender_id, m.sender_name,
m.image_ocr_text, m.voice_transcript
FROM knowledge_messages m
WHERE ${clauses.join(' AND ')}
ORDER BY m.create_time DESC
@@ -797,7 +798,18 @@ export class KnowledgeStore {
// 来源信息由下面的结构化字段表达。
text: toEvidenceDisplayText(String(row.searchable_text)),
...(row.image_ocr_text ? { imageOcrText: String(row.image_ocr_text) } : {}),
...(row.image_ocr_text ? { derivedSource: 'image_ocr' as const } : {}),
/*
* 来源标记按"这条消息带什么派生内容"判定,与 `sourceKind` 正交:
* `image_ocr` = 靠图片里的文字命中,`voice_transcript` = 靠语音转写命中。
*
* 两者都有时以图片 OCR 为先 —— 图片消息不会同时带语音转写,这里只是取确定值,
* 实际不会出现需要二选一的数据。
*/
...(row.image_ocr_text
? { derivedSource: 'image_ocr' as const }
: String(row.voice_transcript || '').trim()
? { derivedSource: 'voice_transcript' as const }
: {}),
score: String(row.kind) === 'system' ? 1 : 0
}
}
+275 -68
View File
@@ -24,6 +24,87 @@ import {
type ContactSearchIndex
} from '../../shared/contact-search'
/**
* 谁在读消息。
*
* 只允许下面这几个固定标签 —— 日志里**不能**出现会话 md5 / session id / wxid /
* 群名 / 联系人 / 路径,所以调用方身份只能靠标签表达。
* 落在集合外的调用点一律记 `unknown`。
*/
export type ListMessagesCaller =
| 'image-text-index'
| 'group-monitor'
| 'archive'
| 'knowledge'
| 'unknown'
/** 进程生命周期内单调递增的读取序号,用来把"同一段时间的几次调用"关联起来(不是稳定标识)。 */
let listMessagesRequestSeq = 0
export function nextListMessagesRequestId(): string {
listMessagesRequestSeq += 1
return `request-${listMessagesRequestSeq}`
}
/**
* 一次 `listMessages` 的性能拆解。
*
* 存在的意义:大会话的全量读取会把主进程卡住数秒,而原来只有一行 `totalMs`,
* 无法判断时间花在 **WCDB 查询**、**JS 逐条格式化**,还是 **内容解析**上。
*
* 覆盖面:`totalMs` 是外层入口的整段耗时;`formatMs` 包含 `contentParseMs` 与
* `dateFormatMs`(后两者是它的子集,不可与 `formatMs` 相加)。
*/
export interface ListMessagesPerf {
caller: ListMessagesCaller
requestId: string
/** WCDB 返回的原始行数(异步路径里含原生查询时间)。 */
rawRows: number
formattedRows: number
totalMs: number
/** 取原始行:同步 `getUserMessages` 或 `await getUserMessagesAsync`。 */
rawReadMs: number
/** 逐条构造 `FormattedMessage`(整个 `map`)。 */
formatMs: number
/** └ 其中:日期格式化(`toLocaleString`)。 */
dateFormatMs: number
/** └ 其中:内容解析(`parseMessageContent` / `parseStickerMessageFromRow`)。 */
contentParseMs: number
/** 召回归档合并与排序。 */
sortMs: number
/** `totalMs` 减去上面已计部分。 */
otherMs: number
}
function emptyPerf(caller: ListMessagesCaller, requestId: string): ListMessagesPerf {
return {
caller,
requestId,
rawRows: 0,
formattedRows: 0,
totalMs: 0,
rawReadMs: 0,
formatMs: 0,
dateFormatMs: 0,
contentParseMs: 0,
sortMs: 0,
otherMs: 0
}
}
/**
* 只在**值得看**的时候打一行:大会话、或者总耗时已经明显影响交互。
* 单行、可 grep、无任何会话标识。
*/
function logListMessagesPerf(perf: ListMessagesPerf): void {
perf.otherMs = Math.max(0, perf.totalMs - perf.rawReadMs - perf.formatMs - perf.sortMs)
const noteworthy = perf.formattedRows >= 20_000 || perf.totalMs >= 1_000
if (!noteworthy) return
console.log(
`[ChatServicePerf] caller=${perf.caller} request=${perf.requestId} rows=${perf.formattedRows} rawRows=${perf.rawRows} totalMs=${perf.totalMs} rawReadMs=${perf.rawReadMs} formatMs=${perf.formatMs} dateFormatMs=${perf.dateFormatMs} contentParseMs=${perf.contentParseMs} sortMs=${perf.sortMs} otherMs=${perf.otherMs}`
)
}
export function getCurrentKey(): string {
if (!dbRef) return ''
try {
@@ -199,7 +280,8 @@ export function isReady(): boolean {
/** Session 行的时间字段可能是秒,也可能是毫秒;1e11 以下按秒换算。 */
function sessionTimeToEpochMs(value: unknown): number | null {
const numeric = typeof value === 'number' ? value : typeof value === 'string' ? Number(value) : NaN
const numeric =
typeof value === 'number' ? value : typeof value === 'string' ? Number(value) : NaN
if (!Number.isFinite(numeric) || numeric <= 0) return null
return Math.round(numeric < 1e11 ? numeric * 1000 : numeric)
}
@@ -379,28 +461,52 @@ function listSourceMessages(
endTime?: number,
options?: { limit?: number },
rawMessagesOverride?: WechatMessage[],
requestId = 'NO-REQUEST'
requestId = 'NO-REQUEST',
perf?: ListMessagesPerf
): FormattedMessage[] {
if (!dbRef) return []
const startedAt = Date.now()
const wcdb4Client = dbRef.getWcdb4Client()
const username = wcdb4Client.getUsernameByMd5(userMd5)
const isGroupChat = Boolean(username?.endsWith('@chatroom'))
wcdbDebugLog(
`[${requestId}] ChatService listSourceMessages start md5=${userMd5} username=${username || ''} start=${startTime || 0} end=${endTime || 0} limit=${options?.limit || 0}`
`[${requestId}] ChatService listSourceMessages start start=${startTime || 0} end=${endTime || 0} limit=${options?.limit || 0} hasOverride=${rawMessagesOverride ? 1 : 0}`
)
const rawReadStartedAt = Date.now()
const rawMessages =
rawMessagesOverride ?? dbRef.getUserMessages(userMd5, startTime, endTime, options)
if (perf) {
perf.rawReadMs += Date.now() - rawReadStartedAt
perf.rawRows += rawMessages.length
}
wcdbDebugLog(
`[${requestId}] ChatService raw snapshot ready raw=${rawMessages.length} cost=${Date.now() - startedAt}ms`
`[${requestId}] ChatService raw snapshot ready raw=${rawMessages.length} cost=${Date.now() - rawReadStartedAt}ms`
)
/** 把"内容解析"单独计时,才能区分"消息多"和"每条都在做解析"。 */
const timedParse = <T>(fn: () => T): T => {
if (!perf) return fn()
const startedAt = Date.now()
try {
return fn()
} finally {
perf.contentParseMs += Date.now() - startedAt
}
}
const formatStartedAt = Date.now()
const formatted = rawMessages.map((msg: WechatMessage) => {
const rawMsgType = parseInt(msg.messageType)
const msgType = normalizeMsgType(msg.messageType)
const createTime = parseInt(msg.msgCreateTime)
const date = new Date(createTime * 1000)
/**
* 逐条 `toLocaleString` 每次都会新建一个 ICU formatter —— 大会话里这是主要成本,
* 所以单独计时,避免它被笼统算进"格式化耗时"。
*/
const dateFormatStartedAt = perf ? Date.now() : 0
const datetimeText = date.toLocaleString('zh-CN', { hour12: false })
if (perf) perf.dateFormatMs += Date.now() - dateFormatStartedAt
const isMine = msg.mesDes !== 1
const localId = parseInt(msg.mesLocalID) || 0
@@ -434,7 +540,7 @@ function listSourceMessages(
/<patinfo\b|<type>\s*62\s*<\/type>/i.test(rawContent) ||
([10000, 10002].includes(msgType) && /拍了拍/i.test(rawContent))
if (isPatMessage) {
const system = parseMessageContent(content, 10000)
const system = timedParse(() => parseMessageContent(content, 10000))
const patContent =
system.type === 'system'
? { ...system, pat: true }
@@ -461,9 +567,9 @@ function listSourceMessages(
/<(?:emoji|sticker|emoticon)\b/i.test(content) || /<type>\s*47\s*<\/type>/i.test(content)
const rowSticker =
inferredMsgType === 47 || (inferredMsgType === 49 && !isQuotePayload && hasStickerPayload)
? parseStickerMessageFromRow(msg, content)
? timedParse(() => parseStickerMessageFromRow(msg, content))
: undefined
const parsedContent = parseMessageContent(content, inferredMsgType)
const parsedContent = timedParse(() => parseMessageContent(content, inferredMsgType))
const rowStickerUrl = rowSticker?.type === 'sticker' ? String(rowSticker.url || '') : ''
const parsedShareUrl = parsedContent.type === 'share' ? parsedContent.url : ''
const redPacketUrl = rowStickerUrl || parsedShareUrl
@@ -535,7 +641,7 @@ function listSourceMessages(
}
if (!contentData && typeof content === 'string' && /^[0-9a-fA-F]{64,}$/.test(content.trim())) {
const parsed = parseStickerMessageFromRow(msg, content)
const parsed = timedParse(() => parseStickerMessageFromRow(msg, content))
if (parsed.type === 'sticker') {
if (!parsed.url && parsed.md5) {
parsed.url = wcdb4Client.resolveEmoticonCdnUrl(parsed.md5)
@@ -616,7 +722,7 @@ function listSourceMessages(
from: contentData?.type === 'system' ? 'system' : isMine ? 'assistant' : 'user',
isSender: isMine,
type: displayType,
datetime: date.toLocaleString('zh-CN', { hour12: false }),
datetime: datetimeText,
content,
img,
name,
@@ -634,9 +740,10 @@ function listSourceMessages(
}
})
console.log(
`[ChatService] listMessages end md5=${userMd5} formatted=${formatted.length} cost=${Date.now() - startedAt}ms`
)
if (perf) {
perf.formatMs += Date.now() - formatStartedAt
perf.formattedRows += formatted.length
}
return formatted
}
@@ -644,13 +751,38 @@ export function listMessages(
userMd5: string,
startTime?: number,
endTime?: number,
options?: { limit?: number }
options?: { limit?: number },
caller: ListMessagesCaller = 'unknown'
): FormattedMessage[] {
const sourceMessages = listSourceMessages(userMd5, startTime, endTime, options)
if (!dbRef) return sourceMessages
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
recordRecallArchiveMessages(userMd5, username, sourceMessages)
return mergeRecallArchiveMessages(userMd5, sourceMessages, startTime, endTime, options?.limit)
const perf = emptyPerf(caller, nextListMessagesRequestId())
const totalStartedAt = Date.now()
try {
const sourceMessages = listSourceMessages(
userMd5,
startTime,
endTime,
options,
undefined,
perf.requestId,
perf
)
if (!dbRef) return sourceMessages
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
const recallStartedAt = Date.now()
recordRecallArchiveMessages(userMd5, username, sourceMessages)
const result = mergeRecallArchiveMessages(
userMd5,
sourceMessages,
startTime,
endTime,
options?.limit
)
perf.sortMs += Date.now() - recallStartedAt
return result
} finally {
perf.totalMs = Date.now() - totalStartedAt
logListMessagesPerf(perf)
}
}
export async function listMessagesAsync(
@@ -658,42 +790,106 @@ export async function listMessagesAsync(
startTime?: number,
endTime?: number,
options?: { limit?: number },
requestId = 'NO-REQUEST'
requestId = '',
caller: ListMessagesCaller = 'unknown'
): Promise<FormattedMessage[]> {
if (!dbRef) return []
const startedAt = Date.now()
wcdbDebugLog(`[${requestId}] ChatService listMessagesAsync start md5=${userMd5}`)
const rawMessages = await dbRef.getUserMessagesAsync(
userMd5,
startTime,
endTime,
options,
requestId
)
wcdbDebugLog(
`[${requestId}] ChatService getUserMessagesAsync end raw=${rawMessages.length} cost=${Date.now() - startedAt}ms`
)
const sourceMessages = listSourceMessages(
userMd5,
startTime,
endTime,
options,
rawMessages,
requestId
)
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
recordRecallArchiveMessages(userMd5, username, sourceMessages)
const result = mergeRecallArchiveMessages(
userMd5,
sourceMessages,
startTime,
endTime,
options?.limit
)
wcdbDebugLog(
`[${requestId}] ChatService listMessagesAsync end formatted=${result.length} cost=${Date.now() - startedAt}ms`
)
return result
const perf = emptyPerf(caller, requestId || nextListMessagesRequestId())
const totalStartedAt = Date.now()
try {
wcdbDebugLog(`[${perf.requestId}] ChatService listMessagesAsync start`)
const rawReadStartedAt = Date.now()
const rawMessages = await dbRef.getUserMessagesAsync(
userMd5,
startTime,
endTime,
options,
perf.requestId
)
perf.rawReadMs += Date.now() - rawReadStartedAt
wcdbDebugLog(
`[${perf.requestId}] ChatService getUserMessagesAsync end raw=${rawMessages.length} cost=${Date.now() - rawReadStartedAt}ms`
)
const sourceMessages = listSourceMessages(
userMd5,
startTime,
endTime,
options,
rawMessages,
perf.requestId,
perf
)
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
const recallStartedAt = Date.now()
recordRecallArchiveMessages(userMd5, username, sourceMessages)
const result = mergeRecallArchiveMessages(
userMd5,
sourceMessages,
startTime,
endTime,
options?.limit
)
perf.sortMs += Date.now() - recallStartedAt
wcdbDebugLog(
`[${perf.requestId}] ChatService listMessagesAsync end formatted=${result.length} cost=${Date.now() - totalStartedAt}ms`
)
return result
} finally {
perf.totalMs = Date.now() - totalStartedAt
logListMessagesPerf(perf)
}
}
/**
* 只取**图片消息**(图片文字索引专用)。
*
* 与 `listMessagesAsync` 的唯一差别是"读哪些行":由 WCDB 在 SQL 层按消息类型过滤,
* 而不是把整个会话读进来再在 JS 里筛。格式化和消息身份走的是**同一套代码**
* (同一个 `listSourceMessages`),所以 `messageId` / `contentData` / 派生键完全不变。
*
* 存在的理由:大会话(十几万到二十几万条消息)全量读一次要 15s 以上,
* 而图片索引只关心图片;这是数据边界错了,不是性能调优问题。
*/
export async function listImageMessagesAsync(
userMd5: string,
requestId = '',
caller: ListMessagesCaller = 'unknown'
): Promise<FormattedMessage[]> {
if (!dbRef) return []
const perf = emptyPerf(caller, requestId || nextListMessagesRequestId())
const totalStartedAt = Date.now()
try {
const rawReadStartedAt = Date.now()
const rawMessages = await dbRef
.getWcdb4Client()
.listImageMessagesAsync(userMd5, { requestId: perf.requestId })
perf.rawReadMs += Date.now() - rawReadStartedAt
const sourceMessages = listSourceMessages(
userMd5,
undefined,
undefined,
undefined,
rawMessages,
perf.requestId,
perf
)
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
const recallStartedAt = Date.now()
recordRecallArchiveMessages(userMd5, username, sourceMessages)
// 召回归档里可能还留着已被撤回的图片;跳过合并会漏索引,所以照旧合并。
const result = mergeRecallArchiveMessages(
userMd5,
sourceMessages,
undefined,
undefined,
undefined
)
perf.sortMs += Date.now() - recallStartedAt
return result
} finally {
perf.totalMs = Date.now() - totalStartedAt
logListMessagesPerf(perf)
}
}
export async function listMessagesForExport(
@@ -702,24 +898,35 @@ export async function listMessagesForExport(
endTime?: number
): Promise<FormattedMessage[]> {
if (!dbRef) return []
const rawMessages = await dbRef.getUserMessagesForExport(userMd5, startTime, endTime)
const sourceMessages = listSourceMessages(userMd5, startTime, endTime, undefined, rawMessages)
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
recordRecallArchiveMessages(userMd5, username, sourceMessages)
const mergedMessages = mergeRecallArchiveMessages(userMd5, sourceMessages, startTime, endTime)
console.log(
`[ChatService] listMessagesForExport end md5=${userMd5} source=${sourceMessages.length} merged=${mergedMessages.length}`
)
return mergedMessages
const perf = emptyPerf('archive', nextListMessagesRequestId())
const totalStartedAt = Date.now()
try {
const rawReadStartedAt = Date.now()
const rawMessages = await dbRef.getUserMessagesForExport(userMd5, startTime, endTime)
perf.rawReadMs += Date.now() - rawReadStartedAt
const sourceMessages = listSourceMessages(
userMd5,
startTime,
endTime,
undefined,
rawMessages,
perf.requestId,
perf
)
const username = dbRef.getWcdb4Client().getUsernameByMd5(userMd5) || ''
const recallStartedAt = Date.now()
recordRecallArchiveMessages(userMd5, username, sourceMessages)
const mergedMessages = mergeRecallArchiveMessages(userMd5, sourceMessages, startTime, endTime)
perf.sortMs += Date.now() - recallStartedAt
return mergedMessages
} finally {
perf.totalMs = Date.now() - totalStartedAt
logListMessagesPerf(perf)
}
}
/**
* Count voice rows without hydrating message content. This is used by the
* batch-selection view, where loading every conversation would make opening
* Settings noticeably slow.
*/
/**
* 图片消息计数探针(SQL 统计,不解密)。
* 图片消息计数(SQL 统计,不解密)。
*
* 返回 `count: null` 表示**统计失败**,不是 0 张。调用方必须区分这两件事 ——
* 否则"数不出来"会被显示成"账号里没有图片",用户会因此放弃建立索引。
+3 -3
View File
@@ -322,16 +322,16 @@ class ImageInsightService {
// 与 Vision 路径的关系:
// ImageInsightService 是统一编排入口,下面挂两条互不干扰的运行时——
// - Vision Model Runtime(AIProviderService,走 AI Provider,可能联网)
// - Windows System OCR Runtime(SystemOcrService,纯本地,不联网)
// - System OCR Runtime(SystemOcrService,纯本地,不联网)
//
// 边界与约束:
// 1. 本地 OCR 结果属于 **派生内容**,原始消息始终是权威来源;
// 本轮不落库、不写 Knowledge、不做历史图片 backfill。
// 本服务只返回识别文本,不做持久化 —— 落库与 Knowledge 回填在图片文字索引侧。
// 2. 本地 OCR 结果 **不会** 写入 image-insights.json——那是 Vision 结果的缓存,
// 两者的缓存键空间也不同(见 buildSystemOcrCacheKey)。
// 3. 这里不读取也绝不修改 AI Vision Provider / 模型配置。
/** 本机是否支持本地图片文字识别(Windows System OCR)。 */
/** 本机是否支持本地图片文字识别(System OCR,引擎按平台决定)。 */
getSystemOcrCapability(): Promise<SystemOcrCapability> {
return systemOcrService.getCapability()
}
File diff suppressed because it is too large Load Diff
+24 -1
View File
@@ -3,7 +3,7 @@
*
* 为什么单独一个库而不是往 knowledge.sqlite 里加表:
* - 清理语义干净:整个能力 = 三个文件(.sqlite/-wal/-shm),删掉即可,不留残渣。
* - 零迁移风险:不动已发布的 knowledge schema(§26 要求升级不破坏既有派生库)。
* - 零迁移风险:不动已发布的 knowledge schema(升级不得破坏既有派生库)。
* - 去重语义天然:artifact 按「图片内容 + OCR 运行时指纹」唯一,binding 承担多来源。
*
* 账号隔离与 Knowledge 一致:路径按 accountId 摘要分目录 + 库内 account_id 自证。
@@ -345,6 +345,29 @@ export class ImageTextIndexStore {
}
}
/**
* 扫描进度:所有会话累计「应扫多少张图片消息」与「实际扫过多少张」。
*
* 与上面的 `readCountedTotal()` 分工必须分清:
* - `readCountedTotal()` 是 `countImageMessages()` 给出的**预估**分母(遍历消息表数出来的,
* 会随新消息变动,且与「归档合并后流水线真正拿到的消息集合」并不完全一致);
* - 这里是流水线**真实走过**的集合。
*
* 进度必须用后者。拿预估值当分母,进度会永远差最后几个百分点,
* 让已经跑完的索引一直显示成"部分完成"。
*/
readScanProgress(): { total: number; processed: number } {
const row = this.database
.prepare(
'SELECT SUM(image_total) AS total, SUM(image_processed) AS processed FROM image_ocr_scan_state'
)
.get() as Record<string, unknown> | undefined
return {
total: Number(row?.total ?? 0) || 0,
processed: Number(row?.processed ?? 0) || 0
}
}
writeCountedTotal(input: { total: number; countedAt: number; complete: boolean }): void {
this.writeMeta('total_image_messages', String(input.total))
this.writeMeta('total_image_counted_at', String(input.countedAt))
+33 -15
View File
@@ -109,10 +109,8 @@ export interface QueryAgentTraceItem {
/**
* 本次 Tool Result 里携带 OCR 派生文本的图片消息/证据条数(诊断用,不进模型上下文)。
*
* 存在的意义是让"图片已经识别出文字、但模型没拿到"这类**链路断点**可以被直接观测:
* 真机上曾经出现过 `query_messages` 返回了图片消息却只带 `attachment`、
* 模型因此回答"没有取得 OCR 文字"。当时从回答文本无法判断是"索引没建"还是"没接上",
* 因为这两件事在日志里长得一模一样。有了这个数字就能一眼分开。
* 用来区分"图片索引没建"与"索引建了但没接到 tool result 上"这两类链路断点 ——
* 没有这个数字时,两者在回答文本里长得一样。
*/
imageOcrTextCount?: number
/** 本次 Tool Result 里图片文字索引的覆盖度状态(`not_built` / `partial` / `complete` / `failed`)。 */
@@ -213,6 +211,26 @@ export interface QueryAgentEvidenceItem {
const MAX_EVIDENCE_ITEMS = 40
/**
* 回答格式与单轮语义的硬规则。
*
* 单独抽出来是为了让它可被测试直接断言 —— 这几条是产品契约,不是措辞偏好:
* 换行、改写都可以,但三条实质约束不能丢。
*/
export const ANSWER_RULES = `
回答结构(检索型结果):
- **不要用 Markdown 表格**承载多条命中结果 —— 结果栏很窄,表格列宽会错位、长字段换行后难读。改用编号列表:先给一句结论,再逐条列出(发送者 / 时间 / 会话 / 类型 / 内容),最后按需说明与范围。
- 逐条里的内容若来自本地派生(图片 OCR、语音转写),要写明它来自派生内容,不要说成群友发过的一条这样的文字消息。
单轮语义(重要):
- 当前是**单次检索回答**:一次提问、一次检索、一次回答。没有自动连续的多轮工具执行。
- **禁止**任何"下一步还能帮你继续"的邀约,包括但不限于:"如果你需要,我可以…""要不要我继续…""我还可以帮你进一步…""需要的话我再查…""我可以再帮你分析…"。除非该动作在**本轮已经真实执行过**。
- 需要收口时,用陈述句说明范围(如"以上为当前检索范围内的结果"),或者直接结束。
事实与推断:
- 没有内容哈希 / artifact 同一性这些直接证据时,不要写"就是同一张图转发了三次"这类确定说法,只能写"内容高度相似,可能是同一张或同系列"。
- 范围说明只在确有必要时给:索引覆盖不完整、内容属于本地派生、时间或检索范围受限、有已知未覆盖数据。不要在每次回答末尾机械复读同一句。`
const SYSTEM_PROMPT = `你是 TraceMemo 的本地聊天查询助手,只能使用提供的四个 Query Tool 获取事实,最终回答只基于 Tool Result。
规划原则:
@@ -250,7 +268,8 @@ const SYSTEM_PROMPT = `你是 TraceMemo 的本地聊天查询助手,只能使
- not_indexed:这条图片还没进图片文字索引。**不许**把“还没索引”说成“图片里没有文字”;若 imageOcrCoverage 不是 complete,必须说明当前无法确认。
- 图片文字索引状态一律以 Tool Result 的结构化字段为准。**不要**在回答里凭空建议“可以先建立图片文字索引再查”——只有 imageOcrCoverage.state 确实是 not_built 时才可以这么说。
- 区分「图片里确实没有文字」(OCR 结果为空,属于已处理的正常终态)与「图片还没被索引」(覆盖缺口):前者是事实,后者不能当成事实。
缺少必要信息时用自然语言澄清;超出工具能力时说明不能可靠完成,并给出当前工具可以执行的替代方向。`
缺少必要信息时用自然语言澄清;超出工具能力时说明不能可靠完成,并给出当前工具可以执行的替代方向。
${ANSWER_RULES}`
function toolDefinitions(): AIChatToolDefinition[] {
return LOCAL_QUERY_TOOL_DEFINITIONS.map((tool) => ({
@@ -707,13 +726,9 @@ function nextToolDefinitions(name: string, result: QueryAgentToolResult, state:
/**
* 这里**不能**因为"时间范围已经是全部"就关掉重试。
*
* 原实现是 `if (rangeWasAll && resultCount === 0) return []`,依据是"时间不能再放宽了、
* 更窄只会更少"。但 0 结果的重试本来就不是为了改时间 —— 它是为了放宽
* **direction / messageTypes**:「我给 X 发了什么图片」被错判成 `from_target` 时,
* 换成 `to_target` 会从 0 条变成有结果。
*
* 这个守卫的后果正是真机那个回归:工具没发出去 → 第二次调用被
* `tool_availability` 拒掉 → 模型想改向也调不动 → 只能回头问用户"是不是方向搞错了"。
* 0 结果的重试不是为了让时间更宽 —— 它的作用是放宽 **direction / messageTypes**:
* 「我给 X 发了什么图片」被判成 `from_target` 时,换成 `to_target` 会从 0 条变成有结果。
* 一旦在这里关掉,模型想改向也调不动,只能回头问用户"是不是方向搞错了"。
*
* 时间范围不可变由 `constraint_time_range_immutable` 单独把关,
* 完全相同的重试由 `duplicate_retry` 拦下,次数由 ZERO_RESULT_RETRY_LIMIT 限制,
@@ -859,9 +874,12 @@ class EvidenceCollector {
? { messageType: record.sourceKind }
: {}),
...(typeof record.text === 'string' && record.text ? { text: record.text } : {}),
// 「靠图片里的文字命中」这个来源语义必须带到 UI:用户要能看出这条答案来自
// 图片 OCR,而不是群友真发了一条文字消息。messageRef 仍然指向原始图片消息。
...(record.derivedSource === 'image_ocr' ? { derivedSource: 'image_ocr' as const } : {}),
// 派生来源原样透传到 UI(取值集合由 `KnowledgeDerivedSource` 约束):
// 用户要能看出这条答案来自图片 OCR / 语音转写,而不是群友真发了一条文字消息。
// messageRef 始终指向原始消息,authoritative source 不变。
...(record.derivedSource === 'image_ocr' || record.derivedSource === 'voice_transcript'
? { derivedSource: record.derivedSource }
: {}),
...(typeof record.imageOcrText === 'string' && record.imageOcrText
? { imageOcrText: record.imageOcrText }
: {}),
+131 -52
View File
@@ -24,22 +24,34 @@
// 抛出的错误是 `Windows error 操作成功完成。 (0x00000000)`(HRESULT 为 S_OK)。
// - 空白图不会报错,返回空文本 → 映射成 OCR_EMPTY_RESULT。
// - CJK 字符之间会被引擎插入空格,结果里做归一化。
//
// macOS 后端:Apple Vision(同一 native 包,darwin binding)。
// 已实测的引擎行为(1.2.0 / macOS 15.7.7 / arm64):
// - Buffer 输入 PNG / JPEG / WEBP / GIF / BMP / TIFF **全部直接可用**,
// 所以 macOS 不做任何归一化,原始字节直通(不落盘、不起 ffmpeg 子进程)。
// - preferredLangs 对识别结果没有可观测影响(Vision 自行决定识别语言),
// 因此默认不传语言提示;显式指定 language 时仍然透传。
// - 畸形图片抛普通 Error:`CRImage Reader Detector was given zero-dimensioned image (0 x 0)`;
// 任一边 ≤2px 抛 `The image is too small in at least one dimension ...` → 都映射成 IMAGE_DECODE_FAILED。
// - macOS 没有"语言包缺失"这一失败模式。
import crypto from 'node:crypto'
import { spawn } from 'node:child_process'
import {
SYSTEM_OCR_CACHE_TTL_MS,
SYSTEM_OCR_ENGINE,
SYSTEM_OCR_PROBE_PNG_BASE64,
buildSystemOcrCacheKey,
detectSystemOcrImageFormat,
isSystemOcrPlatform,
mapSystemOcrNativeError,
normalizeSystemOcrText,
parseImageDataUrl,
resolveSystemOcrEngine,
resolveSystemOcrLanguageTag
} from '../../shared/system-ocr'
import type {
SystemOcrCapability,
SystemOcrEngine,
SystemOcrErrorCode,
SystemOcrImageFormat,
SystemOcrLine,
@@ -71,13 +83,18 @@ interface NativeRuntime {
export interface SystemOcrServiceDeps {
/** 加载 native 运行时;不可用时返回 null(不允许抛) */
loadRuntime?: () => NativeRuntime | null
/** 把输入图片转成 PNG 字节;失败返回 null */
/**
* 把输入图片转成引擎可接受的字节;失败返回 null。
*
* Windows 后端只吃 PNG,必须走这一步;macOS 的 Vision 直接接受
* PNG / JPEG / WEBP / GIF / BMP / TIFF,默认实现直接透传原始字节。
*/
toPngBytes?: (input: {
buffer: Buffer
format: SystemOcrImageFormat
}) => Promise<Buffer | null>
/**
* ffmpeg 可执行文件解析器。只用于 GIF/BMP/WebP/TIFF → PNG 的兜底归一化。
* ffmpeg 可执行文件解析器。只用于 Windows 上 GIF/BMP/WebP/TIFF → PNG 的兜底归一化。
* main/index.ts 会注入项目统一的解析逻辑(与图片解密共用一套候选路径)。
*/
resolveFfmpegExecutable?: () => string
@@ -87,6 +104,22 @@ export interface SystemOcrServiceDeps {
locale?: () => string
}
const failure = (
engine: SystemOcrEngine,
errorCode: SystemOcrErrorCode,
error: string,
startedAt: number
): SystemOcrResult => ({
success: false,
text: '',
lines: [],
language: null,
engine,
durationMs: Date.now() - startedAt,
errorCode,
error
})
/** 未被显式注入时的兜底:环境变量 → 打包内 ffmpeg-static → PATH。 */
const defaultResolveFfmpegExecutable = (): string => {
const fromEnvironment = String(process.env['FFMPEG_BIN'] || '').trim()
@@ -114,22 +147,7 @@ const toLines = (lines: NativeLine[] | undefined): SystemOcrLine[] =>
}))
: []
const failure = (
errorCode: SystemOcrErrorCode,
error: string,
startedAt: number
): SystemOcrResult => ({
success: false,
text: '',
lines: [],
language: null,
engine: SYSTEM_OCR_ENGINE,
durationMs: Date.now() - startedAt,
errorCode,
error
})
/** 把任意容器(gif/bmp/webp/tiff)用 ffmpeg 走内存管道转成 PNG。不落盘。 */
/** 把任意容器(gif/bmp/webp/tiff)用 ffmpeg 走内存管道转成 PNG。不落盘。仅 Windows 归一化路径会用到。 */
const convertWithFfmpeg = (buffer: Buffer, executable: string): Promise<Buffer | null> =>
new Promise((resolve) => {
let settled = false
@@ -222,6 +240,16 @@ class SystemOcrService {
return this.deps.arch ?? process.arch
}
/** 该平台对应的引擎标识。进 artifact 指纹与缓存 key,不要硬编码。 */
private get engine(): SystemOcrEngine {
return resolveSystemOcrEngine(this.platform)
}
/** 本平台是否为 macOS 后端(决定是否跳过图片归一化)。 */
private get isMacBackend(): boolean {
return this.platform === 'darwin'
}
private get locale(): string {
if (this.deps.locale) {
try {
@@ -245,7 +273,7 @@ class SystemOcrService {
this.runtime = this.deps.loadRuntime()
return this.runtime
}
if (this.platform !== 'win32') {
if (!isSystemOcrPlatform(this.platform)) {
this.runtime = null
return this.runtime
}
@@ -267,7 +295,7 @@ class SystemOcrService {
} catch (error) {
console.warn(
'[SystemOcrService] native runtime unavailable engine=%s platform=%s reason=%s',
SYSTEM_OCR_ENGINE,
this.engine,
this.platform,
error instanceof Error ? error.message.split('\n')[0] : String(error)
)
@@ -281,6 +309,11 @@ class SystemOcrService {
format: SystemOcrImageFormat
): Promise<Buffer | null> {
if (this.deps.toPngBytes) return this.deps.toPngBytes({ buffer, format })
/*
* macOS:Vision 后端直接接受 PNG / JPEG / WEBP / GIF / BMP / TIFF(已实测),
* 归一化没有收益,只会白白多一次转码或一个 ffmpeg 子进程 —— 原字节直通。
*/
if (this.isMacBackend) return buffer
if (format === 'png') return buffer
if (format === 'jpeg') {
// 项目内已有的进程内解码能力,优先于 ffmpeg(更快、无子进程)。
@@ -322,18 +355,18 @@ class SystemOcrService {
SystemOcrCapability,
'engine' | 'platform' | 'arch' | 'runtimeVersion' | 'language'
> = {
engine: SYSTEM_OCR_ENGINE,
engine: this.engine,
platform: this.platform,
arch: this.arch,
runtimeVersion: null,
language: null
}
if (this.platform !== 'win32') {
if (!isSystemOcrPlatform(this.platform)) {
return {
...base,
available: false,
reason: 'UNSUPPORTED_PLATFORM',
message: '本地图片文字识别目前仅支持 Windows。'
message: '本地图片文字识别目前支持 Windows 与 macOS。'
}
}
const runtime = this.loadRuntime()
@@ -355,20 +388,26 @@ class SystemOcrService {
message: probed.message
}
}
const engineLabel = this.isMacBackend ? 'macOS 系统 OCR' : 'Windows 系统 OCR'
return {
...base,
runtimeVersion: runtime.version,
available: true,
language: probed.language,
message: probed.language
? `本地图片文字识别可用(Windows 系统 OCR,${probed.language})。`
: '本地图片文字识别可用(Windows 系统 OCR,跟随系统语言)。'
? `本地图片文字识别可用(${engineLabel},${probed.language})。`
: `本地图片文字识别可用(${engineLabel},跟随系统语言)。`
}
}
/**
* 用一个 64x32 纯白 PNG 探测语言可用性:引擎能创建即说明语言包可用。
* 首选「系统 locale 推导出的标签」,失败再退回「系统用户语言配置」。
* 用一个 64x32 纯白 PNG 探测引擎是否真的能跑。
*
* Windows:引擎创建依赖语言包,首选「系统 locale 推导出的标签」,
* 失败再退回「系统用户语言配置」,并据此区分 LANGUAGE_UNAVAILABLE。
*
* macOS:Vision 自行决定识别语言,**没有语言包缺失这一失败模式**,
* 所以不传语言提示,探测失败只可能是引擎本身起不来。
*/
private async probeLanguage(
runtime: NativeRuntime
@@ -377,7 +416,29 @@ class SystemOcrService {
| { language: null; reason: 'LANGUAGE_UNAVAILABLE' | 'NATIVE_MODULE_MISSING'; message: string }
> {
const probeBuffer = Buffer.from(SYSTEM_OCR_PROBE_PNG_BASE64, 'base64')
const preferred = resolveSystemOcrLanguageTag(this.locale)
if (this.isMacBackend) {
try {
await runtime.recognize(probeBuffer, undefined, undefined)
return { language: null }
} catch (error) {
/*
* 探测图是纯白图。Vision 对"图里没有文字"是**抛错**(`No text recognized`),
* 而抛这个错恰恰证明识别器跑通了 —— 不能当成引擎故障。
*/
if (
mapSystemOcrNativeError(error instanceof Error ? error.message : String(error)) ===
'OCR_EMPTY_RESULT'
) {
return { language: null }
}
return {
language: null,
reason: 'NATIVE_MODULE_MISSING',
message: '本地文字识别引擎初始化失败,请重启 TraceMemo 或重新安装。'
}
}
}
const preferred = resolveSystemOcrLanguageTag(this.locale, this.platform)
const candidates: Array<string | null> = preferred ? [preferred, null] : [null]
let lastCode: SystemOcrErrorCode = 'OCR_FAILED'
for (const candidate of candidates) {
@@ -433,22 +494,28 @@ class SystemOcrService {
*/
async recognize(request: SystemOcrRequest): Promise<SystemOcrResult> {
const startedAt = Date.now()
const engine = this.engine
const parsed = parseImageDataUrl(request.imageDataUrl)
if (!parsed) {
return failure('UNSUPPORTED_IMAGE', '仅支持 PNG、JPG、JPEG、WebP、GIF、BMP 图片。', startedAt)
return failure(
engine,
'UNSUPPORTED_IMAGE',
'仅支持 PNG、JPG、JPEG、WebP、GIF、BMP 图片。',
startedAt
)
}
let sourceBuffer: Buffer
try {
sourceBuffer = Buffer.from(parsed.base64, 'base64')
} catch {
return failure('IMAGE_DECODE_FAILED', '图片数据无法解码。', startedAt)
return failure(engine, 'IMAGE_DECODE_FAILED', '图片数据无法解码。', startedAt)
}
if (sourceBuffer.length === 0) {
return failure('IMAGE_DECODE_FAILED', '图片数据为空。', startedAt)
return failure(engine, 'IMAGE_DECODE_FAILED', '图片数据为空。', startedAt)
}
const format = detectSystemOcrImageFormat(sourceBuffer)
if (!format) {
return failure('UNSUPPORTED_IMAGE', '无法识别的图片格式。', startedAt)
return failure(engine, 'UNSUPPORTED_IMAGE', '无法识别的图片格式。', startedAt)
}
const imageHash =
@@ -464,7 +531,7 @@ class SystemOcrService {
: capability.reason === 'LANGUAGE_UNAVAILABLE'
? 'OCR_LANGUAGE_UNAVAILABLE'
: 'SYSTEM_OCR_UNAVAILABLE'
return failure(errorCode, capability.message, startedAt)
return failure(engine, errorCode, capability.message, startedAt)
}
const languageForCache = requestedLanguage ?? capability.language
@@ -472,7 +539,8 @@ class SystemOcrService {
imageHash,
language: languageForCache,
runtimeVersion: capability.runtimeVersion,
platform: capability.platform
platform: capability.platform,
engine: capability.engine
})
if (requestedLanguage === null) {
const cached = this.readCache(cacheKey, startedAt)
@@ -481,12 +549,12 @@ class SystemOcrService {
const runtime = this.loadRuntime()
if (!runtime) {
return failure('SYSTEM_OCR_UNAVAILABLE', '本地文字识别组件不可用。', startedAt)
return failure(engine, 'SYSTEM_OCR_UNAVAILABLE', '本地文字识别组件不可用。', startedAt)
}
const png = await this.toPngBytes(sourceBuffer, format)
if (!png || png.length === 0 || !detectSystemOcrImageFormat(png)) {
return failure('IMAGE_DECODE_FAILED', '图片解码失败,无法读取这张图片。', startedAt)
const prepared = await this.toPngBytes(sourceBuffer, format)
if (!prepared || prepared.length === 0 || !detectSystemOcrImageFormat(prepared)) {
return failure(engine, 'IMAGE_DECODE_FAILED', '图片解码失败,无法读取这张图片。', startedAt)
}
const candidates: Array<string | null> = requestedLanguage
@@ -499,8 +567,11 @@ class SystemOcrService {
let usedLanguage: string | null = null
for (const candidate of candidates) {
try {
// accuracy 传 undefined = 用 native 默认值,而该默认是 `Accurate`
// (见 @napi-rs/system-ocr 的 recognize 文档)。**不要改成 Fast**:
// 低精度档在中文上会明显掉字。Windows 忽略该参数。
const result = await runtime.recognize(
png,
prepared,
undefined,
candidate ? [candidate] : undefined
)
@@ -508,30 +579,37 @@ class SystemOcrService {
const lines = toLines(result?.lines)
usedLanguage = candidate
if (!text) {
return failure('OCR_EMPTY_RESULT', '没有在这张图片里识别到文字。', startedAt)
return failure(engine, 'OCR_EMPTY_RESULT', '没有在这张图片里识别到文字。', startedAt)
}
const succeeded: SystemOcrResult = {
success: true,
text,
lines,
language: usedLanguage,
engine: SYSTEM_OCR_ENGINE,
engine,
durationMs: Date.now() - startedAt
}
// 生产日志只记录 error code / engine / platform / duration,绝不记录识别正文。
console.log(
'[SystemOcrService] ok engine=%s platform=%s language=%s chars=%d durationMs=%d',
SYSTEM_OCR_ENGINE,
capability.platform,
usedLanguage ?? 'system-default',
text.length,
succeeded.durationMs
)
/**
* 成功路径**刻意不逐张打日志**。
*
* 后台回填会连续识别几万张图片,逐张一条成功日志既是没有信息量的噪声,
* 又会把日志刷爆。逐张耗时由 `ImageTextIndexService` 的阶段画像低频汇总,
* 单张的 `durationMs` / 字数依然在**返回值**里(设置页的单图诊断就是用它)。
* 只有失败才值得在默认输出里留痕 —— 见下面的 `failed`。
*/
this.writeCache(cacheKey, succeeded)
return succeeded
} catch (error) {
lastErrorMessage = error instanceof Error ? error.message : String(error)
lastErrorCode = mapSystemOcrNativeError(lastErrorMessage)
/*
* macOS 的 Vision 在"图里没有文字"时是抛错(`No text recognized`)而不是返回空文本。
* 它必须走 empty 语义:表情包 / 风景 / 头像都是**正常终态**,不是 OCR 失败 ——
* 否则这些图片会落成可重试失败,被反复重算,覆盖率也会说谎。
*/
if (lastErrorCode === 'OCR_EMPTY_RESULT') {
return failure(engine, 'OCR_EMPTY_RESULT', '没有在这张图片里识别到文字。', startedAt)
}
// 语言不可用才值得换下一个候选;其它错误直接结束,避免无意义重试。
if (lastErrorCode !== 'OCR_LANGUAGE_UNAVAILABLE') break
}
@@ -539,12 +617,13 @@ class SystemOcrService {
console.warn(
'[SystemOcrService] failed engine=%s platform=%s errorCode=%s durationMs=%d',
SYSTEM_OCR_ENGINE,
engine,
capability.platform,
lastErrorCode,
Date.now() - startedAt
)
return failure(
engine,
lastErrorCode,
lastErrorCode === 'OCR_LANGUAGE_UNAVAILABLE'
? '当前 Windows 未安装可用的 OCR 语言支持。'
+78
View File
@@ -1531,6 +1531,84 @@ export class Wcdb4Client {
return this.finalizeMessages(username, allRows, startTime, endTime, limit)
}
/**
* 只读**图片消息**,供图片文字索引使用。
*
* 为什么需要它:`getMessagesAsync` 会把整个会话的消息都读出来,
* 一个 20 万条消息的会话要花十几秒(实测 `rawReadMs≈15s`),
* 而图片索引只关心其中的图片 —— 那是错误的数据边界。
*
* 实现上刻意**复用 `finalizeMessages`**(即 `normalizeMessage` + 群昵称解析 + 排序),
* 这样产出的 `Wcdb4Message` 与全量路径**逐字段同构**,`messageId` / `contentData`
* 语义完全一致 —— 否则 artifact / binding / checkpoint 的键会全变。
* 变化的只有"读哪些行":靠 `imageMessageWhere` 在 SQL 层过滤。
*
* 这里仍是一次性读完该会话的图片(**未分页**):行数由图片数量决定而不是消息数量,
* 已经比全量小一到两个数量级。超过 `limit` 会被截断并告警,调用方应改成分页。
*/
async listImageMessagesAsync(
md5OrUsername: string,
options: { sinceMs?: number; limit?: number; requestId?: string } = {}
): Promise<Wcdb4Message[]> {
if (!this.wcdbExecQuery) return []
const requestId = options.requestId ?? 'NO-REQUEST'
const username = this.resolveMessageUsername(md5OrUsername)
if (!username) return []
const startedAt = Date.now()
let tables: Wcdb4MessageStore[] = []
try {
tables = await this.listMessageStoresAsync(username)
} catch (error) {
console.warn('[WCDB4] image message table stats failed:', error)
return []
}
if (!tables.length) return []
const limit = Math.max(1, options.limit ?? 200_000)
const allRows: Record<string, unknown>[] = []
let successfulTables = 0
for (const table of tables) {
// 真实类型列名必须逐表探测:硬编码会让过滤静默失效,把全量消息当图片读回来。
const column = this.resolveMessageTypeColumn(table)
if (!column) continue
try {
const where = this.imageMessageWhere(column, options.sinceMs)
// `local_id` 参与排序:`create_time` 同秒的消息需要一个稳定次序,
// 否则多次读取的行序可能不同,调用方无法做稳定游标。
const sql = `SELECT * FROM ${this.quoteSqlIdentifier(table.tableName)} WHERE ${where} ORDER BY "create_time" ASC, "local_id" ASC LIMIT ${limit}`
const queryStartedAt = Date.now()
const rows = await this.callJsonAsync<Record<string, unknown>[]>(
this.wcdbExecQuery as unknown as KoffiAsyncFunction,
'message',
table.dbPath,
sql
)
successfulTables += 1
if (Array.isArray(rows)) allRows.push(...rows)
wcdbDebugLog(
`[${requestId}] WCDB image messages table=${table.tableName} rows=${Array.isArray(rows) ? rows.length : 0} cost=${Date.now() - queryStartedAt}ms`
)
} catch (error) {
console.warn(
`[WCDB4] image message scan failed table=${table.tableName}:`,
error
)
}
}
if (tables.length > 0 && successfulTables === 0) return []
const messages = this.finalizeMessages(username, allRows)
if (allRows.length >= limit) {
// 不静默丢数据:截断会让该会话被标成"处理完了",下一遍靠水位修正。
console.warn(`[WCDB4] image message scan truncated rows=${allRows.length} limit=${limit}`)
}
wcdbDebugLog(
`[${requestId}] WCDB image messages end rows=${messages.length} cost=${Date.now() - startedAt}ms`
)
return messages
}
private async getMessagesByTableScanAsync(
username: string,
startTime?: number,
@@ -1,5 +1,6 @@
import * as React from 'react'
import { Button, EmptyState } from '../ui'
import { derivedSnippetPrefix, evidenceSourceBadges } from './evidenceSourceLabels'
import type { EvidenceItem } from './searchTypes'
import { formatMessageTime, messageIdentity, messageText, senderName } from './searchUtils'
@@ -47,6 +48,7 @@ export function AISearchEvidencePanel({
const flashing = evidenceFlash.index === index
const evidenceLabel = item.evidenceId || `E${index + 1}`
const evidenceSender = senderName(item.message, item.contact, senderNames)
const badges = evidenceSourceBadges(item)
return (
<article
key={`${messageIdentity(item.message)}-${index}-${flashing ? evidenceFlash.nonce : 0}`}
@@ -74,18 +76,30 @@ export function AISearchEvidencePanel({
<span className="mt-0.5 block text-[10px] leading-[15px] text-primary">
{item.contact.m_nsNickName}
</span>
{item.sourceKind === 'voice' && (
<span className="block text-[11px] font-semibold text-primary">语音转写</span>
)}
{item.derivedSource === 'image_ocr' && (
<span
className="mt-0.5 inline-block rounded-sm bg-accent px-1.5 py-0.5 text-[10px] font-semibold text-primary"
data-testid="evidence-image-ocr-badge"
>
图片文字
{/*
来源标签:消息类型 + 派生来源。
两者正交,最多两个;不做 tooltip,避免把右栏撑成说明文档。
*/}
{badges.length > 0 && (
<span className="mt-1 flex flex-wrap items-center gap-1">
{badges.map((badge) => (
<span
key={badge.key}
data-testid={`evidence-badge-${badge.key}`}
className="rounded-sm bg-accent px-1.5 py-0.5 text-[10px] font-semibold leading-[14px] text-primary"
>
{badge.label}
</span>
))}
</span>
)}
<span className="mt-[7px] block overflow-hidden text-[11px] leading-[17px] text-muted-foreground [display:-webkit-box] [-webkit-box-orient:vertical] [-webkit-line-clamp:3]">
{/* 派生命中内容必须自报来源,不能被读成群友真发过这段文字。 */}
{derivedSnippetPrefix(item.derivedSource) && (
<span className="font-semibold text-primary">
{derivedSnippetPrefix(item.derivedSource)}
</span>
)}
{messageText(item.message)}
</span>
{/* 命中解释:明确告诉用户"命中的是图里的这段文字",
@@ -8,6 +8,10 @@ import {
AlertDialogHeader,
AlertDialogTitle,
Button,
DropdownMenu,
DropdownMenuContent,
DropdownMenuItem,
DropdownMenuTrigger,
Select,
SelectContent,
SelectItem,
@@ -33,7 +37,7 @@ type ImageTextIndexCardProps = {
* 与 Knowledge 卡片**平级并列**(同一组索引入口),但刻意是**独立的一维能力**:
* 文字消息索引完整不代表图片里的文字搜得到。
*
* 文案遵从严禁混淆的语义(§9):这里做的是「识别图片中文字」,不是
* 文案遵从严禁混淆的语义:这里做的是「识别图片中文字」,不是
* 「本地识图模型 / 本地 Vision / AI OCR」,也不能暗示能理解场景或表情包。
*/
export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProps): ReactElement {
@@ -79,8 +83,8 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
/**
* 处理进度百分比。
*
* 刻意不在这里做 `Math.round(x * 100)` —— `45479 / 45707` 会被四舍五入成 `100`,
* 于是出现了"已建立 · 仅完成 100%"这种自相矛盾的显示。未完成时封顶 99.9%。
* 刻意不在这里做 `Math.round(x * 100)` —— 那会把 99.5% 显示成 100%,
* 于是出现"已建立 · 仅完成 100%"这种自相矛盾的显示。未完成时封顶 99.9%。
*/
const percent = coverage
? imageTextProcessedPercent(coverage.processed, coverage.totalImageMessages)
@@ -110,10 +114,25 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
const nothingCounted =
count !== null && count.scannedConversations === 0 && count.failedConversations > 0
/** 识别失败的图片数(派生库的真实统计),决定「更多」里有没有重试入口。 */
const failureCount = coverage?.failed ?? 0
/**
* 中断但**可续做**。
*
* `paused` 和 `cancelled` 都能靠 checkpoint 从断点接上(`startPass` 会跳过已完成会话、
* 命中已有 artifact 不再重复 OCR),所以两者必须给**同一个**「继续」入口。
* 只认 `paused` 的后果真实发生过:点过「取消」之后卡片只剩「更新图片文字索引」,
* 状态还被显示成「部分完成 · 2.1%」—— 用户既看不出自己中断过,也找不到继续的地方。
*/
const interrupted = paused || progress?.state === 'cancelled'
const stateLabel = (() => {
if (progress?.state === 'error') return '建立失败'
if (running) return `建立中 · ${percent}%`
if (paused) return `已暂停 · ${percent}%`
// 取消 ≠ 部分完成:进度是保留的,但"被打断过"这件事必须说出来。
if (progress?.state === 'cancelled') return `已取消 · ${percent}%`
if (!established) return '未建立'
// 「已建立」不能等于「全失败」:处理过但一条都没成功时必须叫异常。
if (coverageState === 'failed') return '图片文字索引异常'
@@ -227,6 +246,21 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
<p className="ai-search-knowledge-pass-line">
{`${progress.percent}% · 识别出文字 ${progress.indexed.toLocaleString()} · 没有文字 ${progress.empty.toLocaleString()} · 图片已清理 ${progress.missing.toLocaleString()} · 失败 ${progress.failed.toLocaleString()}`}
</p>
{/*
速度用最近窗口的实测值(Main 给的就是窗口速度,不是全程平均)。
样本还不足时如实说"计算中",不要编一个数 —— 全量回填要跑几小时,
一个假 ETA 比没有 ETA 更糟。
*/}
<p
className="ai-search-knowledge-pass-line"
data-testid="image-text-index-rate"
>
{`当前速度:${
typeof progress.speedPerSec === 'number' && progress.speedPerSec > 0
? `约 ${progress.speedPerSec.toFixed(1)} 张/秒`
: '计算中'
} · 预计剩余:${formatEta(progress.etaMs)}`}
</p>
</div>
)}
@@ -284,8 +318,9 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
<p className="ai-search-knowledge-error">请先连接微信数据,然后再建立图片文字索引。</p>
)}
<div className="ai-search-knowledge-actions">
{!running && !paused && (
{/* 这张卡最多并列 3 个操作,横向排会撑破窄侧栏;修饰类把它改成单列堆叠。 */}
<div className="ai-search-knowledge-actions ai-search-image-index-actions">
{!running && !interrupted && (
<Button
size="sm"
className="ai-search-knowledge-primary"
@@ -296,44 +331,50 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
{established ? '更新图片文字索引' : '建立图片文字索引'}
</Button>
)}
{!running && !paused && countFailed && (
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-recount"
disabled={pending !== null || counting}
onClick={() => void refreshCount()}
>
{counting ? '统计中…' : '重新统计'}
</Button>
)}
{/* 修好之后重跑:只重置失败记录,成功记录与其它数据一律不动。 */}
{!running && !paused && systemicFailure && (
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-reset-failures"
disabled={pending !== null}
onClick={() => void resetFailures()}
>
{pending === 'reset' ? '处理中…' : '重试失败的图片'}
</Button>
)}
{/* 派生索引修复:只重建 Knowledge 里的图片搜索索引,**不重新识别任何图片**。
存在的意义就是"别为修一个索引问题重跑几万张图"。 */}
{!running && !paused && established && (
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-repair"
disabled={pending !== null}
onClick={() => void repair()}
>
{pending === 'repair' ? '修复中…' : '修复图片搜索索引'}
</Button>
{/*
修复类操作收进「更多」。
它们各自只在很窄的情况下才有用(搜索索引不一致 / 有识别失败的图片),
而主路径永远只有一个:更新索引。平铺出来时,用户看到的是四个都在说
「索引」的按钮,只能靠猜哪个该点。
*/}
{!running && (established || failureCount > 0) && (
<DropdownMenu>
<DropdownMenuTrigger asChild>
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-more"
disabled={pending !== null}
>
更多
</Button>
</DropdownMenuTrigger>
<DropdownMenuContent align="end">
{/* 派生索引修复:只重建 Knowledge 里的图片搜索索引,**不重新识别任何图片**。
存在的意义就是"别为修一个索引问题重跑几万张图"。 */}
{established && (
<DropdownMenuItem
data-testid="image-text-index-repair"
disabled={pending !== null}
onSelect={() => void repair()}
>
图片内容搜不到?修复搜索索引
</DropdownMenuItem>
)}
{/* 修好之后重跑:只重置失败记录,成功记录与其它数据一律不动。 */}
{failureCount > 0 && (
<DropdownMenuItem
data-testid="image-text-index-reset-failures"
disabled={pending !== null}
onSelect={() => void resetFailures()}
>
{`重试识别失败的图片(${failureCount.toLocaleString()} 张)`}
</DropdownMenuItem>
)}
</DropdownMenuContent>
</DropdownMenu>
)}
{running && (
<>
@@ -359,28 +400,35 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
</Button>
</>
)}
{/*
中断(暂停 / 取消)之后必须能找到「继续」。
两种状态的 checkpoint 都是保留的,继续 = 从断点接上,
所以这里刻意合并成一个入口 —— 否则「取消」过的索引会只剩
「更新图片文字索引」,用户根本看不出还能接着做。
*/}
{interrupted && (
<Button
size="sm"
className="ai-search-knowledge-primary"
data-testid="image-text-index-resume"
disabled={pending !== null}
onClick={() => void resume()}
>
{pending === 'resume' ? '继续中…' : '继续'}
</Button>
)}
{/* 只有真的处在"暂停中"才有东西可取消:已取消的状态再点取消没有意义。 */}
{paused && (
<>
<Button
size="sm"
className="ai-search-knowledge-primary"
data-testid="image-text-index-resume"
disabled={pending !== null}
onClick={() => void resume()}
>
{pending === 'resume' ? '继续中…' : '继续'}
</Button>
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-cancel"
disabled={pending !== null}
onClick={() => void cancel()}
>
取消
</Button>
</>
<Button
size="sm"
variant="outline"
className="ai-search-knowledge-cancel"
data-testid="image-text-index-cancel"
disabled={pending !== null}
onClick={() => void cancel()}
>
取消
</Button>
)}
</div>
</section>
@@ -425,3 +473,18 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp
</>
)
}
/**
* 剩余时间文案。
*
* `null` = 分母不可信或速度样本还不足 —— 如实说"计算中"。
* 刻意不显示 p50 / p95 这类开发指标:这是用户界面,不是性能面板。
*/
function formatEta(etaMs: number | null | undefined): string {
if (typeof etaMs !== 'number' || !Number.isFinite(etaMs) || etaMs <= 0) return '计算中'
const totalMinutes = Math.round(etaMs / 60_000)
if (totalMinutes < 1) return '不到 1 分钟'
const hours = Math.floor(totalMinutes / 60)
const minutes = totalMinutes % 60
return hours > 0 ? `${hours} 小时 ${minutes} 分` : `${minutes} 分`
}
@@ -0,0 +1,63 @@
import type { KnowledgeDerivedSource, KnowledgeMessageKind } from '../../../../shared/knowledge'
/**
* 证据卡的来源标签。
*
* 两个维度**正交**,必须分开表达,不能混成一个标签:
* - 消息类型(`sourceKind`):原始消息本身是什么;
* - 派生来源(`derivedSource`):这条结果是**靠什么命中**的(本地派生的 OCR / 转写)。
*
* 文案面向用户:不出现 `image_ocr` 这类工程词。每条证据最多两个标签,
* 顺序固定为「消息类型 + 派生来源」。
*/
/** 只列用户看得懂的类型;`other` 之类没有信息量的取值不给标签。 */
const MESSAGE_TYPE_LABELS: Record<string, string> = {
text: '文本消息',
image: '图片消息',
voice: '语音消息',
video: '视频消息',
file: '文件消息',
link: '链接消息',
sticker: '表情消息',
system: '系统消息'
}
const DERIVED_SOURCE_LABELS: Record<KnowledgeDerivedSource, string> = {
image_ocr: 'OCR命中',
voice_transcript: '转写命中'
}
/**
* 派生命中内容的 snippet 前缀。
*
* 目的只有一个:**不能让派生文本看起来像原始聊天内容**。
* 普通文本消息不加前缀,原样展示。
*/
const DERIVED_SNIPPET_PREFIXES: Record<KnowledgeDerivedSource, string> = {
image_ocr: 'OCR摘录:',
voice_transcript: '转写摘录:'
}
export interface EvidenceSourceBadge {
/** 稳定的 DOM key,不用下标 —— 标签顺序可能随数据变化。 */
key: 'messageType' | 'derivedSource'
label: string
}
export function evidenceSourceBadges(item: {
sourceKind?: KnowledgeMessageKind | string
derivedSource?: KnowledgeDerivedSource
}): EvidenceSourceBadge[] {
const badges: EvidenceSourceBadge[] = []
const messageLabel = item.sourceKind ? MESSAGE_TYPE_LABELS[item.sourceKind] : undefined
if (messageLabel) badges.push({ key: 'messageType', label: messageLabel })
const derivedLabel = item.derivedSource ? DERIVED_SOURCE_LABELS[item.derivedSource] : undefined
if (derivedLabel) badges.push({ key: 'derivedSource', label: derivedLabel })
return badges
}
/** 派生内容的 snippet 前缀;普通文本消息返回空串(原样展示)。 */
export function derivedSnippetPrefix(source?: KnowledgeDerivedSource): string {
return source ? DERIVED_SNIPPET_PREFIXES[source] : ''
}
@@ -94,5 +94,33 @@ export const renderMarkdown = (value: string, options: MarkdownOptions = {}): Re
</div>
)
}
/*
* Markdown 表格行。
*
* 结果栏很窄,真表格在这里只会挤成一团(列宽错位、长字段换行难读)。提示词已经
* 禁止模型为检索结果产表格,但历史回答与其它入口仍可能出现,所以这里把它降级成
* **逐行的键值列表**:内容读得出来,且永远不会横向溢出容器。
*/
if (/^\s*\|.*\|\s*$/.test(line)) {
const cells = line
.trim()
.replace(/^\||\|$/g, '')
.split('|')
.map((cell) => cell.trim())
.filter((cell) => cell.length > 0)
// `|---|---|` 这类分隔行没有信息,当作空行处理。
if (!cells.length || cells.every((cell) => /^:?-{2,}:?$/.test(cell))) {
return <div key={key} className="ai-search-markdown-spacer" />
}
return (
<div key={key} className="ai-search-markdown-table-row">
{cells.map((cell, cellIndex) => (
<span key={`${key}-${cellIndex}`} className="ai-search-markdown-table-cell">
{inlineMarkdown(cell, `${key}-${cellIndex}`, options)}
</span>
))}
</div>
)
}
return <p key={key}>{inlineMarkdown(line, key, options)}</p>
})
@@ -6,7 +6,11 @@ import type {
AiSearchProgressEvent,
AiSearchProgressStage
} from '../../../../shared/ai-search'
import type { KnowledgeMessageKind, KnowledgeVoiceCoverage } from '../../../../shared/knowledge'
import type {
KnowledgeDerivedSource,
KnowledgeMessageKind,
KnowledgeVoiceCoverage
} from '../../../../shared/knowledge'
import type { Contact, Message } from '../../../../shared/types'
export type SearchStage = 'idle' | 'loading' | 'result' | 'partial' | 'insufficient'
@@ -37,10 +41,11 @@ export interface EvidenceItem {
/**
* 命中所依赖的派生来源。
*
* `image_ocr` = 这条结果靠**图片里的文字**命中,而不是群友真的发了一条文字消息。
* 有值时 Evidence 卡片显示轻量来源标记(「图片文字」)。
* 有值 = 这条结果靠**本地派生内容**命中,而不是原始消息本身的文字
* (`image_ocr` = 图片里的文字,`voice_transcript` = 语音转写)。
* authoritative source 始终是原始消息 —— 这里只用来多挂一个来源标记。
*/
derivedSource?: 'image_ocr'
derivedSource?: KnowledgeDerivedSource
/** 「从图片里读出来的文字」片段,只作命中解释。 */
imageOcrText?: string
contact: Contact
@@ -1,11 +1,10 @@
import * as React from 'react'
import * as AlertDialogPrimitive from '@radix-ui/react-alert-dialog'
import { cn } from '../../lib/cn'
import { buttonVariants } from './button'
const AlertDialog = AlertDialogPrimitive.Root
const AlertDialogTrigger = AlertDialogPrimitive.Trigger
const AlertDialogCancel = AlertDialogPrimitive.Cancel
const AlertDialogAction = AlertDialogPrimitive.Action
const AlertDialogContent = React.forwardRef<
React.ElementRef<typeof AlertDialogPrimitive.Content>,
@@ -70,6 +69,39 @@ const AlertDialogDescription = React.forwardRef<
))
AlertDialogDescription.displayName = AlertDialogPrimitive.Description.displayName
/**
* 取消:次要动作,走 `outline`。
*
* 这两个组件必须**显式**挂上 `buttonVariants`。直接 `export const X = Primitive.X`
* 会把 Radix 原始 primitive 原样抛出去,渲染成浏览器默认按钮(黑白方角),
* 跟产品主题完全不搭 —— 这类"忘了挂样式"的 primitive 是默认样式的常见来源。
*/
const AlertDialogCancel = React.forwardRef<
React.ElementRef<typeof AlertDialogPrimitive.Cancel>,
React.ComponentPropsWithoutRef<typeof AlertDialogPrimitive.Cancel>
>(({ className, ...props }, ref) => (
<AlertDialogPrimitive.Cancel
ref={ref}
className={cn(buttonVariants({ variant: 'outline' }), className)}
{...props}
/>
))
AlertDialogCancel.displayName = AlertDialogPrimitive.Cancel.displayName
/**
* 确认:主要动作,走 `default`(主题色)。
*
* 危险动作(删除、清空等)由调用方传 `className` 覆盖成 destructive ——
* `cn` 走的是 tailwind-merge,同族类会被后者替换,不必在这里开新的分支。
*/
const AlertDialogAction = React.forwardRef<
React.ElementRef<typeof AlertDialogPrimitive.Action>,
React.ComponentPropsWithoutRef<typeof AlertDialogPrimitive.Action>
>(({ className, ...props }, ref) => (
<AlertDialogPrimitive.Action ref={ref} className={cn(buttonVariants(), className)} {...props} />
))
AlertDialogAction.displayName = AlertDialogPrimitive.Action.displayName
export {
AlertDialog,
AlertDialogTrigger,
@@ -1,5 +1,9 @@
import { useCallback, useEffect, useState } from 'react'
import type { SystemOcrCapability, SystemOcrResult } from '../../../../../shared/system-ocr'
import type {
SystemOcrCapability,
SystemOcrEngine,
SystemOcrResult
} from '../../../../../shared/system-ocr'
import { Button } from '../../../components/ui'
const MAX_FILE_BYTES = 10 * 1024 * 1024
@@ -18,12 +22,15 @@ interface LocalOcrState {
}
/**
* 本地图片文字识别(Windows 系统 OCR)。
* 本地图片文字识别(系统 OCR)。
*
* 这是**本地 Runtime**,不是 AI 图片理解:
* - 只把图片里的文字读出来;不描述画面、人物、场景,也不做视觉推理;
* - 原始图片不会因为这一步发给任何 AI Provider;
* - 结果只是派生内容,不会写进本地知识库。
*
* 引擎由平台决定(Windows 系统 OCR / macOS 系统 OCR),UI 一律从 capability 派生文案,
* 不硬编码平台名。
*/
export function LocalImageTextRecognition(): React.ReactElement {
const [capability, setCapability] = useState<SystemOcrCapability | null>(null)
@@ -120,13 +127,14 @@ export function LocalImageTextRecognition(): React.ReactElement {
const running = state.status === 'running'
const result = state.result
const engineLabel = systemOcrEngineLabel(capability?.engine)
return (
<section className="settings-card local-ocr-test">
<header>
<div>
<h2>本地图片文字识别</h2>
<p>使用 Windows 系统 OCR 在本机读取图片中的文字,原始图片无需发送给 AI Provider。</p>
<p>使用{engineLabel}在本机读取图片中的文字,原始图片无需发送给 AI Provider。</p>
</div>
<span className={`local-ocr-capability ${capability?.available ? 'supported' : ''}`}>
{capability ? (capability.available ? '本机可用' : '本机不可用') : '检测中…'}
@@ -138,7 +146,7 @@ export function LocalImageTextRecognition(): React.ReactElement {
) : null}
{capability?.available ? (
<p className="local-ocr-runtime">
引擎:Windows 系统 OCR
引擎:{engineLabel}
{capability.runtimeVersion ? ` · 组件 ${capability.runtimeVersion}` : ''}
{capability.language ? ` · 语言 ${capability.language}` : ' · 语言跟随系统'}
</p>
@@ -183,7 +191,7 @@ export function LocalImageTextRecognition(): React.ReactElement {
<dl>
<div>
<dt>引擎</dt>
<dd>Windows 系统 OCR</dd>
<dd>{systemOcrEngineLabel(result.engine)}</dd>
</div>
<div>
<dt>语言</dt>
@@ -218,10 +226,22 @@ export function LocalImageTextRecognition(): React.ReactElement {
)
}
/** 引擎标识 → 展示名。UI 不硬编码平台,一律从 capability / result 派生。 */
function systemOcrEngineLabel(engine: SystemOcrEngine | undefined): string {
switch (engine) {
case 'macos-system-ocr':
return 'macOS 系统 OCR'
case 'windows-system-ocr':
return 'Windows 系统 OCR'
default:
return '系统 OCR'
}
}
function localOcrErrorMessage(result: SystemOcrResult): string {
switch (result.errorCode) {
case 'UNSUPPORTED_PLATFORM':
return '本地图片文字识别目前仅支持 Windows。'
return '本地图片文字识别目前支持 Windows 与 macOS。'
case 'SYSTEM_OCR_UNAVAILABLE':
return '本地文字识别组件不可用,请重新安装 TraceMemo。'
case 'OCR_LANGUAGE_UNAVAILABLE':
@@ -1,5 +1,33 @@
import { useState } from 'react'
import type { ImageDecryptionState } from './types'
import { Input } from '../../../components/ui'
import { Button, Input } from '../../../components/ui'
/**
* 密钥显示切换图标。
*
* 项目里没有现成的眼睛图标(`LineIcon` 只有 database / shield 之类),
* 这里就地画一个 16px 线框图标,避免为一处 UI 引入图标依赖。
*/
function EyeIcon({ crossed }: { crossed: boolean }): React.ReactElement {
return (
<svg
width="16"
height="16"
viewBox="0 0 24 24"
fill="none"
stroke="currentColor"
strokeWidth="1.7"
strokeLinecap="round"
strokeLinejoin="round"
aria-hidden
focusable="false"
>
<path d="M2.5 12S6 5.75 12 5.75 21.5 12 21.5 12 18 18.25 12 18.25 2.5 12 2.5 12Z" />
<circle cx="12" cy="12" r="2.6" />
{crossed ? <path d="M4.5 4.5l15 15" /> : null}
</svg>
)
}
export function ImageKeyConfiguration({
state,
@@ -10,6 +38,13 @@ export function ImageKeyConfiguration({
disabled: boolean
onEdit: (field: 'xorKey' | 'aesKey', value: string) => void
}): React.ReactElement {
/**
* 只控制**本机的显示方式**,不影响任何存储、校验或解密行为:
* 密钥仍然以 password 语义渲染(浏览器/密码管理器照旧),
* 切换只是把 input 的 type 换成 text,让用户能核对自己填的 16 位密钥。
*/
const [revealed, setRevealed] = useState(false)
return (
<section className="settings-card image-key-editor">
<div className="image-key-grid">
@@ -23,14 +58,29 @@ export function ImageKeyConfiguration({
</label>
<label>
<span>AES Key</span>
<Input
type="password"
value={state.aesKey}
disabled={disabled}
autoComplete="off"
placeholder="输入 16 位图片密钥"
onChange={(event) => onEdit('aesKey', event.target.value)}
/>
<div className="image-key-secret">
<Input
type={revealed ? 'text' : 'password'}
value={state.aesKey}
disabled={disabled}
autoComplete="off"
placeholder="输入 16 位图片密钥"
onChange={(event) => onEdit('aesKey', event.target.value)}
/>
<Button
size="icon"
variant="ghost"
className="image-key-secret-toggle"
data-testid="image-key-reveal"
aria-label={revealed ? '隐藏图片密钥' : '显示图片密钥'}
aria-pressed={revealed}
title={revealed ? '隐藏图片密钥' : '显示图片密钥'}
disabled={disabled}
onClick={() => setRevealed((current) => !current)}
>
<EyeIcon crossed={revealed} />
</Button>
</div>
</label>
</div>
<p>修改后请先选择会话完成图片解析测试,再确认保存。</p>
+54
View File
@@ -420,6 +420,34 @@
min-width: 0;
}
/*
* 图片文字索引卡片的操作区:每个按钮独占一行。
*
* 这张卡最多并列 3 个操作(更新图片文字索引 / 更多 / 继续或取消),
* 而上面那套两列栅格是按「1 主 + 1 次」设计的:`auto` 列不可收缩,
* 窄侧栏下第 2、3 个按钮会把卡片撑出横向溢出。
* 改成纵向堆叠后按钮宽度只跟随容器,结构上不可能溢出。
*/
.ai-search-knowledge-actions.ai-search-image-index-actions {
display: flex;
flex-direction: column;
gap: 8px;
width: 100%;
max-width: 100%;
min-width: 0;
}
/*
* 宽度严格跟随容器:`min-width: 0` 覆盖按钮自身的 68px 下限
* (单列下那个下限已经没有意义,反而会阻止收缩)。
* 标签最长 8 个汉字,最窄侧栏(190px)下仍有余量,所以保持单行不换行。
*/
.ai-search-knowledge-actions.ai-search-image-index-actions > * {
width: 100%;
max-width: 100%;
min-width: 0;
}
.ai-search-knowledge-primary {
min-width: 0;
}
@@ -1082,6 +1110,32 @@
font-weight: 700;
}
/*
* Markdown 表格的降级渲染。
*
* 结果栏很窄,真表格塞进来只会列宽错位、长字段换行难读。渲染层把表格行转成
* 逐行的键值列表(见 searchMarkdown.tsx),这里只保证两件事:
* 读得出来,且**永远不会横向溢出容器**。所以用 flex + wrap,不用 table。
*/
.ai-search-markdown-table-row {
display: flex;
flex-wrap: wrap;
gap: 2px 10px;
margin: 3px 0;
min-width: 0;
}
.ai-search-markdown-table-cell {
min-width: 0;
overflow-wrap: anywhere;
color: var(--wxex-text-secondary);
}
.ai-search-markdown-table-cell:first-child {
color: var(--wxex-text-primary);
font-weight: 600;
}
@media (max-width: 760px) {
.ai-search-header-actions {
align-items: flex-end;
+19
View File
@@ -944,6 +944,25 @@
grid-template-columns: 180px minmax(0, 1fr);
gap: 14px;
}
/*
* AES 密钥字段 +「显示/隐藏」开关。
*
* 用 flex 让按钮做**同级的兄弟节点**,而不是浮在输入框上:
* 这样不需要绝对定位、不需要给输入框留 padding,值很长时也不会被图标压住。
*/
.image-key-secret {
display: flex;
align-items: center;
gap: 6px;
}
.image-key-secret > input {
/* Input 自身是 w-full(width: 100%),flex 里必须放开 min-width 才能收缩。 */
flex: 1 1 auto;
min-width: 0;
}
.image-key-secret-toggle {
flex: 0 0 auto;
}
.image-key-editor > p {
margin: 0;
color: #66706b;
+146 -15
View File
@@ -11,24 +11,62 @@
* 2. `ImageOcrBinding` —— 「某个会话里的某条图片消息 → 某个 artifact」的绑定,保证去重不丢来源。
*/
/** 派生文本的引擎标识;与 System OCR 的引擎常量保持一致。 */
export const IMAGE_TEXT_INDEX_ENGINE = 'windows-system-ocr'
/**
* 派生文本的引擎标识**不在这里定义**:它是 System OCR 运行时按平台决定的
* (`resolveSystemOcrEngine`),并随 `ImageOcrProvenance` 一起进入 artifact 指纹。
* 本模块只消费该值,不再持有任何单一平台的引擎常量。
*/
/** 派生库自身的 schema 版本(与 Knowledge 的 schema 相互独立)。 */
export const IMAGE_TEXT_INDEX_SCHEMA_VERSION = 1
/**
* OCR 并发上限。
*
* 当前实现**严格串行**(循环体内只有一次 await,无 Promise.all 扇出),等价于 1。
* 这个常量是后续调高的唯一入口:Windows OCR 是进程内 WinRT 调用,实测单张
* 20–40ms,串行已足够;调高只会和 Query Agent 抢 CPU。
*/
export const DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY = 1
/** 每个批次的图片条数;批间让出 event loop,保证 UI / 查询不被卡住。 */
export const IMAGE_TEXT_INDEX_BATCH_SIZE = 12
/**
* 正常运行态下,向 Renderer 推送进度的最小间隔。
*
* 后台仍然按 `IMAGE_TEXT_INDEX_BATCH_SIZE` 推进(batch / checkpoint / 并发都不受影响),
* 但**UI 不该感知 batch 大小** —— 每批都推会让计数以「+12」的粒度跳动。
* 所以这里只节流**通知**:状态变化(开始/暂停/继续/取消/失败/完成/清理)一律立即推送。
*/
export const IMAGE_TEXT_INDEX_PROGRESS_INTERVAL_MS = 5000
/** 速度统计窗口:取最近这段时间的增量,而不是整个任务的平均。 */
export const IMAGE_TEXT_INDEX_RATE_WINDOW_MS = 60_000
/** 窗口内至少要有这么长的跨度才给出速度,否则显示「计算中」。 */
export const IMAGE_TEXT_INDEX_RATE_MIN_SPAN_MS = 20_000
/**
* OCR 并发度**硬上限**。
*
* `@napi-rs/system-ocr` 的 `recognize()` 是 napi AsyncTask,实际执行会占用
* libuv **共享**线程池(fs / zlib / dns 等 native 异步工作也在用同一个池)。
* 开得太高不会让单个识别更快,只会挤占同一进程里其它 native 异步工作。
*/
export const MAX_IMAGE_TEXT_OCR_CONCURRENCY = 4
/**
* 默认 OCR 并发度(生产值)。
*
* 取值规则:在"吞吐明显更高、且 CPU / UI 交互代价可接受"的前提下取**最低**并发。
* 超过 2 之后单次识别耗时会明显劣化(多个识别互相争抢 CPU),
* 属于"多出来的并发全花在争抢上"。
*
* **这个值是待复测的**:原取舍依据来自一次现已修复的固定开销存在时的对照,
* 而那个开销不随并发变化、会压扁并发收益。需要重新做锁定输入的对照后再决定;
* 在那之前保持 2,不要按"池子多大就用多大"去推。
*/
export const DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY = 2
/** 解析并发度:只接受 1..MAX 的整数,其余一律回落到默认值。 */
export function resolveImageTextOcrConcurrency(raw?: string | number | null): number {
const value = typeof raw === 'number' ? raw : Number.parseInt(String(raw ?? ''), 10)
if (!Number.isFinite(value) || value < 1) return DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY
return Math.min(MAX_IMAGE_TEXT_OCR_CONCURRENCY, Math.floor(value))
}
/** 已完成一批之后、回到会话循环前的让出时间。 */
export const IMAGE_TEXT_INDEX_YIELD_MS = 0
@@ -222,6 +260,15 @@ export interface ImageTextIndexProgress {
cancellable: boolean
paused: boolean
lastError?: string
/**
* 最近窗口(`IMAGE_TEXT_INDEX_RATE_WINDOW_MS`)的实测速度,单位 张/秒。
*
* 刻意用**滑动窗口**而不是整个任务的平均:全量回填要跑几小时,
* 历史平均会把"现在到底快不快"完全糊掉。样本跨度不足时为 null(UI 显示"计算中")。
*/
speedPerSec?: number | null
/** 按当前窗口速度估算的剩余时间(毫秒);速度不可用或分母不可信时为 null。 */
etaMs?: number | null
}
/**
@@ -278,9 +325,8 @@ export function imageTextCoverageState(coverage: ImageTextIndexCoverage): ImageT
/**
* 处理进度百分比。
*
* 保留 1 位小数,且**未完成时封顶 99.9%**:
* `Math.round(45479 / 45707 * 100)` 会得到 `100`,于是出现了"已建立 · 仅完成 100%"
* 这种自相矛盾的显示。进度条可以近似,结论句不行。
* 保留 1 位小数,且**未完成时封顶 99.9%**:直接四舍五入会把 99.5% 显示成 100%,
* 于是出现"已建立 · 仅完成 100%"这种自相矛盾的显示。进度条可以近似,结论句不行。
*/
export function imageTextProcessedPercent(processed: number, total: number): number {
if (!(total > 0)) return 0
@@ -324,7 +370,7 @@ export interface ImageTextIndexCountResult {
}
/**
* 单个会话的图片消息计数探针。
* 单个会话的图片消息计数结果。
*
* `count: null` = **统计失败**,不等于 0 张。调用方必须区分处理。
*/
@@ -417,6 +463,89 @@ export interface ImageTextIndexStartOptions {
sinceMs?: number
}
/** 单个阶段的耗时聚合。**只含性能数字**,不含任何图片内容 / 路径 / 标识。 */
export interface ImageTextIndexStageStat {
count: number
mean: number
p50: number
p95: number
max: number
}
/**
* backfill 的**只读性能画像**,用来回答"时间花在哪一段"。
*
* 只写性能数字,不含图片内容 / 路径 / 会话标识;UI 不渲染,仅落到 app log。
* 刻意不暴露单张图片的耗时序列:那会把"哪张图慢"变成可推断的信息。
*/
export interface ImageTextIndexStageTimings {
/** 本遍累计计数(与 UI 进度同源)。让这一行日志自洽,不必再去别处对数。 */
counters: {
processed: number
indexed: number
empty: number
missing: number
failed: number
}
/** 最近窗口的实测速度(张/秒);样本不足或分母不可信时为 null。 */
ratePerSec: number | null
/** 本遍实际执行过的 OCR 次数(命中已有 artifact 而跳过的不计)。 */
ocrExecutions: number
/** 本遍生效的 OCR 并发度。 */
ocrConcurrency: number
/** 找图片文件(同步,占主线程)。 */
locate: ImageTextIndexStageStat
/** 解密(同步 + CPU,占主线程)。 */
decrypt: ImageTextIndexStageStat
/** 构造可识别输入(base64 编码;Windows 还包含转 PNG)。 */
normalize: ImageTextIndexStageStat
/** 识别(异步)。 */
ocr: ImageTextIndexStageStat
/** 写 artifact + binding(SQLite,单 writer)。 */
persist: ImageTextIndexStageStat
/**
* 单张图片在流水线里的净耗时。
*
* 分母只含真正进入流水线的图片,所以这个值可以直接与上面五段之和对照;
* **不要用"整遍耗时 ÷ 处理张数"**,那会把 `preLoop` 的一次性成本摊进每张图片。
*/
perImageMs: number
/**
* 本遍因为"没有可搜索内容变化"而**跳过** Knowledge 重建的会话数。
*
* 与 `preLoop.onConversationIndexedMs` 配套看:跳过越多、那段时间越小,
* 说明门控在起作用。它同时是"到底有没有白做"的直接证据。
*/
knowledgeIndexSkipped: number
/** 进入流水线**之前**的一次性成本(不按图片数摊)。 */
preLoop: ImageTextIndexPreLoopCost
}
/**
* 流水线**之外**的成本(单位毫秒),用来解释"单张成本很低、整遍却很慢"。
*
* 这个结构的每一个字段都是"有理由不属于单张成本"的量:
* 一次性的、每会话一次的、以及**别的模块**的。它们必须单独可见 ——
* 否则 `perImageMs` 会看起来很好,而墙钟吞吐差好几倍,且无从归因。
*/
export interface ImageTextIndexPreLoopCost {
/** 一遍 pass 开始前的一次性成本(能力探测 + 全账号图片统计 + 会话列表)。 */
startupMs: number
/** └ 其中:统计图片消息总数(遍历全部会话的 SQL)。 */
countImageMessagesMs: number
/** 每个会话进入流水线前的准备累计(水位 / 计数 / `listMessages`)。 */
conversationSetupMs: number
/** └ 其中:读取并格式化会话消息累计。**已知的大头之一**。 */
listMessagesMs: number
/**
* 会话完成后等待 Knowledge 重建(`onConversationIndexed`)的累计。
*
* 这是**别的模块**的成本:Knowledge 侧会对同一个会话再全量读一遍消息并整篇写索引,
* 而且如果此时有索引在跑还会先等它。它不在 batch 循环里,所以 `perImageMs` 看不到它。
*/
onConversationIndexedMs: number
}
/** 对外状态快照(问问微信卡片 / 设置清理页共用同一份)。 */
export interface ImageTextIndexStatus {
progress: ImageTextIndexProgress
@@ -424,4 +553,6 @@ export interface ImageTextIndexStatus {
storage: ImageTextIndexStorageStats
/** 正在做「检测到多少条图片消息」的 SQL 统计。 */
counting: boolean
/** 各阶段耗时画像(可选的附加诊断字段,UI 不渲染)。 */
stageTimings?: ImageTextIndexStageTimings
}
+10 -1
View File
@@ -229,10 +229,19 @@ export interface KnowledgeEvidence {
* 有值 = 这条结果依赖本地派生内容才能命中(而不是原始消息本身的文字)。
* 与 `sourceKind` 正交:`sourceKind` 说的是原始消息是什么,这里说的是"靠什么搜到的"。
*/
derivedSource?: 'image_ocr'
derivedSource?: KnowledgeDerivedSource
score?: number
}
/**
* 派生来源的种类。
*
* 用 union 而不是 `isOcr: boolean`:以后接视频字幕 / 文件解析时只需要加一个成员,
* 不必给每个消费方再添一个布尔字段。UI 侧的展示文案集中在
* `renderer/src/components/search/evidenceSourceLabels.ts`,不在这里。
*/
export type KnowledgeDerivedSource = 'image_ocr' | 'voice_transcript'
/**
* 证据文本面向用户 / 模型时的可读化处理。
*
+3 -3
View File
@@ -1,4 +1,4 @@
import type { KnowledgeEvidence, KnowledgeVoiceCoverage } from './knowledge'
import type { KnowledgeDerivedSource, KnowledgeEvidence, KnowledgeVoiceCoverage } from './knowledge'
export type QueryDirection = 'any' | 'from_target' | 'to_target'
export type QueryOrder = 'asc' | 'desc'
@@ -119,7 +119,7 @@ export interface QueryEvidenceItem
* 让用户知道这段内容来自**图片里的文字**,而不是群友真的发了一条文字消息。
* authoritative source 仍然是原始图片消息,`messageRef` 也仍然指向原图。
*/
derivedSource?: 'image_ocr'
derivedSource?: KnowledgeDerivedSource
/**
* 「从图片里读出来的文字」片段,只用作命中解释。
*
@@ -242,7 +242,7 @@ export interface QueryMessage {
*/
imageOcrText?: string
/** 派生来源语义:`image_ocr` = 这段文字来自图片识别,而不是原始文字消息。 */
derivedSource?: 'image_ocr'
derivedSource?: KnowledgeDerivedSource
/**
* 这条图片消息在本地图片文字索引里的状态。
*
+5 -3
View File
@@ -1,4 +1,5 @@
import type { AiSearchPipelineRequest, AiSearchPipelineResult } from './ai-search'
import type { KnowledgeDerivedSource } from './knowledge'
import type { QueryCorpusScope } from './local-query-api'
/**
@@ -36,10 +37,11 @@ export interface AskWechatEvidenceItem {
/**
* 命中所依赖的派生来源(与 `messageType` 正交)。
*
* `image_ocr` = 这条结果靠**图片里的文字**命中,而不是群友真的发了一条文字消息。
* Evidence UI 会据此显示轻量来源标记。authoritative source 仍是原始图片消息。
* 有值 = 这条结果靠**本地派生内容**命中,而不是原始消息本身的文字
* (`image_ocr` = 图片里的文字,`voice_transcript` = 语音转写)。
* Evidence UI 会据此多挂一个来源标记;authoritative source 仍是原始消息。
*/
derivedSource?: 'image_ocr'
derivedSource?: KnowledgeDerivedSource
/** 「从图片里读出来的文字」片段,只作命中解释(普通文字消息不会有)。 */
imageOcrText?: string
attachment?: { kind?: string; name?: string; url?: string; sizeBytes?: number }
+137 -25
View File
@@ -7,15 +7,38 @@
// 它不占用 AIVisionRuntimeConfig.source,也不产生任何网络请求。
// - 能力边界:只把图片里的文字读出来。它不等于「理解人物 / 理解场景 /
// 描述照片 / 理解表情包语义 / 视觉推理」——那些仍然属于 Vision Model。
// - Windows 后端为 Windows.Media.Ocr.OcrEngine(经 @napi-rs/system-ocr 调用)。
// macOS 本轮只保留架构位置,未实现;Linux 不支持。
// - 后端按平台选择(统一经 @napi-rs/system-ocr 调用):
// Windows → Windows.Media.Ocr.OcrEngine
// macOS → Apple Vision(VNRecognizeTextRequest / RecognizeDocumentsRequest)
// Linux 不支持。
// - 引擎标识会进入 artifact 指纹与缓存 key,两个平台的结果**不得互相复用**。
//
// 数据边界(本轮不做):
// - 不做历史图片全量 OCR、不做 Knowledge 回填、不把 OCR 文字伪装成原始聊天文字。
// 原始消息始终是权威来源,OCR 文字只是派生内容(本轮仅存在于内存)。
// 数据边界:
// - OCR 文字始终是**派生内容**,会把原图定位回去(artifact + binding),
// 但绝不写回 WCDB、也绝不伪装成原始聊天文字;原始消息始终是权威来源。
// - 历史图片回填与 Knowledge 回填由 image-text-index 负责,本模块只提供识别能力。
/** Windows 引擎标识(Windows.Media.Ocr.OcrEngine)。 */
export const SYSTEM_OCR_ENGINE_WINDOWS = 'windows-system-ocr'
/** macOS 引擎标识(Apple Vision)。 */
export const SYSTEM_OCR_ENGINE_MACOS = 'macos-system-ocr'
/** System OCR 引擎标识。这是本地 Runtime,不是 provider id。 */
export const SYSTEM_OCR_ENGINE = 'windows-system-ocr'
export type SystemOcrEngine = typeof SYSTEM_OCR_ENGINE_WINDOWS | typeof SYSTEM_OCR_ENGINE_MACOS
/** 支持 System OCR 的平台。Linux 明确不支持。 */
export const isSystemOcrPlatform = (platform: string): boolean =>
platform === 'win32' || platform === 'darwin'
/**
* 平台 → 引擎标识。
*
* 不要把引擎串硬编码成某一个平台:它同时是 artifact 指纹的一部分,
* 一旦写死,跨平台结果就会互相复用。
*/
export const resolveSystemOcrEngine = (platform: string): SystemOcrEngine =>
platform === 'darwin' ? SYSTEM_OCR_ENGINE_MACOS : SYSTEM_OCR_ENGINE_WINDOWS
/** 本地 OCR 结果在内存中的缓存时长。 */
export const SYSTEM_OCR_CACHE_TTL_MS = 10 * 60 * 1000
@@ -27,13 +50,13 @@ export const SYSTEM_OCR_CACHE_TTL_MS = 10 * 60 * 1000
export type SystemOcrErrorCode =
/** 运行时不可用(native binding 缺失 / 加载失败) */
| 'SYSTEM_OCR_UNAVAILABLE'
/** 当前平台不支持(Linux,或非 Windows 平台) */
/** 当前平台不支持(Linux,或非 Windows / macOS 平台) */
| 'UNSUPPORTED_PLATFORM'
/** 图片格式不在支持范围内 */
| 'UNSUPPORTED_IMAGE'
/** 图片解码失败(格式可识别但内容损坏或无法转成 PNG) */
/** 图片解码失败(格式可识别但内容损坏或无法转成可识别图像) */
| 'IMAGE_DECODE_FAILED'
/** 当前 Windows 未安装对应的 OCR 语言支持 */
/** 当前 Windows 未安装对应的 OCR 语言支持(macOS 由 Vision 自行决定,不会出现) */
| 'OCR_LANGUAGE_UNAVAILABLE'
/** 引擎执行失败 */
| 'OCR_FAILED'
@@ -49,12 +72,17 @@ export type SystemOcrUnavailableReason =
export interface SystemOcrCapability {
/** 本机当前是否真的可以识别图片文字 */
available: boolean
engine: typeof SYSTEM_OCR_ENGINE
engine: SystemOcrEngine
platform: NodeJS.Platform
arch: string
/** @napi-rs/system-ocr 运行时版本;无法读取时为 null */
runtimeVersion: string | null
/** 实际可用的 OCR 语言标签(对应 Windows 语言包);null 表示走系统用户语言 */
/**
* 实际使用的 OCR 语言标签。
*
* Windows 为系统语言包对应的标签(如 zh-Hans-CN);macOS 由 Vision 自行决定识别语言,
* 这里恒为 null(对应 UI 的「跟随系统语言」)。null 也表示走系统语言。
*/
language: string | null
reason?: SystemOcrUnavailableReason
/** 面向用户的中文说明,可直接展示 */
@@ -71,20 +99,20 @@ export interface SystemOcrBoundingBox {
export interface SystemOcrLine {
text: string
/** Windows 恒为 1.0 */
/** Windows 恒为 1.0;macOS 为 Vision 返回的逐行平均置信度 */
confidence: number
boundingBox: SystemOcrBoundingBox
}
/** 本地 OCR 结果。不包含任何 Windows handle / native 内部对象。 */
/** 本地 OCR 结果。不包含任何平台 handle / native 内部对象。 */
export interface SystemOcrResult {
success: boolean
/** 归一化后的文本(去掉 CJK 字符之间的引擎伪空格) */
text: string
lines: SystemOcrLine[]
/** 实际使用的 OCR 语言标签;null 表示由系统用户语言决定 */
/** 实际使用的 OCR 语言标签;null 表示由系统决定识别语言 */
language: string | null
engine: typeof SYSTEM_OCR_ENGINE
engine: SystemOcrEngine
durationMs: number
/** 命中内存缓存时为 true */
fromCache?: boolean
@@ -104,21 +132,24 @@ export interface SystemOcrRequest {
/**
* 缓存 key 组合。刻意与 ImageInsight 的 `imageHash` 保持不同的键空间,
* 保证远端 Vision 的旧结果永远不会被当成"本地 OCR 结果"复用,
* 也保证 System OCR 运行时升级后不会永远命中旧结果。
* 也保证 System OCR 运行时升级 / 切换平台后不会永远命中旧结果。
*/
export const buildSystemOcrCacheKey = (input: {
imageHash: string
language: string | null
runtimeVersion: string | null
platform?: string
}): string =>
[
engine?: SystemOcrEngine
}): string => {
const platform = input.platform ?? 'unknown'
return [
input.imageHash,
SYSTEM_OCR_ENGINE,
input.platform ?? 'unknown',
input.engine ?? resolveSystemOcrEngine(platform),
platform,
input.language ?? 'auto',
input.runtimeVersion ?? 'unknown'
].join('|')
}
const CJK_CHAR =
/[\u3000-\u303f\u3040-\u30ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff00-\uffef\uac00-\ud7af]/
@@ -126,6 +157,9 @@ const CJK_CHAR =
/**
* Windows OCR 会在每个 CJK 字符之间插入空格("本 地 图 片")。
* 这里只删除 **两侧都是 CJK** 的空格,保留 "TraceMemo 本地图片文字识别" 里的真实分隔。
*
* macOS(Vision)本就输出连续中文,这条规则对它恒等;保留是为了两个平台共用一条
* 归一化路径,而不是给 macOS 加特例。
*/
export const normalizeSystemOcrText = (value: string): string => {
const source = String(value ?? '')
@@ -149,8 +183,11 @@ export const normalizeSystemOcrText = (value: string): string => {
return result.trim()
}
/** 把系统 locale(如 zh-CN / en-US)映射成 Windows OCR 语言标签。 */
const LANGUAGE_TAG_BY_LOCALE: Record<string, string> = {
/**
* 把系统 locale(如 zh-CN / en-US)映射成 **Windows OCR 语言标签**
* (即 Windows 语言包里注册的 BCP-47 标签,中文带 region 子标签)。
*/
const WINDOWS_LANGUAGE_TAG_BY_LOCALE: Record<string, string> = {
zh: 'zh-Hans-CN',
'zh-cn': 'zh-Hans-CN',
'zh-hans': 'zh-Hans-CN',
@@ -186,7 +223,51 @@ const LANGUAGE_TAG_BY_LOCALE: Record<string, string> = {
'ru-ru': 'ru-RU'
}
export const resolveSystemOcrLanguageTag = (
/**
* 把系统 locale 映射成 **Apple Vision 语言标签**。
*
* 与 Windows 表刻意分开:Vision 只认脚本级子标签(`zh-Hans` / `zh-Hant`),
* 不认 `zh-Hans-CN` 这类 region 组合;港台繁体统一收敛到 `zh-Hant`。
*/
const MACOS_LANGUAGE_TAG_BY_LOCALE: Record<string, string> = {
zh: 'zh-Hans',
'zh-cn': 'zh-Hans',
'zh-sg': 'zh-Hans',
'zh-hans': 'zh-Hans',
'zh-hans-cn': 'zh-Hans',
'zh-hans-sg': 'zh-Hans',
'zh-tw': 'zh-Hant',
'zh-hk': 'zh-Hant',
'zh-mo': 'zh-Hant',
'zh-hant': 'zh-Hant',
'zh-hant-tw': 'zh-Hant',
'zh-hant-hk': 'zh-Hant',
'zh-hant-mo': 'zh-Hant',
en: 'en-US',
'en-us': 'en-US',
'en-gb': 'en-GB',
'en-au': 'en-AU',
'en-ca': 'en-CA',
ja: 'ja-JP',
'ja-jp': 'ja-JP',
ko: 'ko-KR',
'ko-kr': 'ko-KR',
fr: 'fr-FR',
'fr-fr': 'fr-FR',
de: 'de-DE',
'de-de': 'de-DE',
es: 'es-ES',
'es-es': 'es-ES',
it: 'it-IT',
'it-it': 'it-IT',
pt: 'pt-BR',
'pt-br': 'pt-BR',
ru: 'ru-RU',
'ru-ru': 'ru-RU'
}
const lookupLanguageTag = (
table: Record<string, string>,
locale: string | null | undefined
): string | null => {
const normalized = String(locale ?? '')
@@ -194,11 +275,26 @@ export const resolveSystemOcrLanguageTag = (
.toLowerCase()
.replace(/_/g, '-')
if (!normalized) return null
if (LANGUAGE_TAG_BY_LOCALE[normalized]) return LANGUAGE_TAG_BY_LOCALE[normalized]
if (table[normalized]) return table[normalized]
const primary = normalized.split('-')[0]
return LANGUAGE_TAG_BY_LOCALE[primary] ?? null
return table[primary] ?? null
}
/** 系统 locale → Windows OCR 语言标签。 */
export const resolveWindowsOcrLanguageTag = (locale: string | null | undefined): string | null =>
lookupLanguageTag(WINDOWS_LANGUAGE_TAG_BY_LOCALE, locale)
/** 系统 locale → Apple Vision 语言标签。 */
export const resolveMacOcrLanguageTag = (locale: string | null | undefined): string | null =>
lookupLanguageTag(MACOS_LANGUAGE_TAG_BY_LOCALE, locale)
/** 按平台把系统 locale 映射成该平台 OCR 引擎接受的语言标签。 */
export const resolveSystemOcrLanguageTag = (
locale: string | null | undefined,
platform: string = 'win32'
): string | null =>
platform === 'darwin' ? resolveMacOcrLanguageTag(locale) : resolveWindowsOcrLanguageTag(locale)
/**
* 把 native 错误映射成产品级错误码。
*
@@ -206,6 +302,16 @@ export const resolveSystemOcrLanguageTag = (
* - 语言包缺失 / 引擎无法创建:`Windows error 操作成功完成。 (0x00000000)`
* —— TryCreateFromLanguage 返回 null 引擎但 HRESULT 是 S_OK,非常容易误判。
* - 送给解码器的字节不是可识别的图片:`Windows error Could not recognize file (0x80070005)`
*
* 已确认的 macOS 行为(1.2.0 / Vision):
* - 图片无法解码成 CGImage(截断、伪造魔数、维度非法):
* `CRImage Reader Detector was given zero-dimensioned image (0 x 0)`
* - 图片任一边不超过 2px:`The image is too small in at least one dimension 2 x 2 ...`
* - **图片没有文字时是抛错而不是返回空文本**:`No text recognized`
* —— 它必须映射成 `OCR_EMPTY_RESULT`(正常终态)。映射成失败会让表情包 /
* 风景图 / 头像全部变成"可重试失败",既污染派生库也会被反复重试。
* - macOS 没有"语言包缺失"这个概念(Vision 自行决定识别语言),
* 所以这里不会映射出 OCR_LANGUAGE_UNAVAILABLE。
*/
export const mapSystemOcrNativeError = (message: string): SystemOcrErrorCode => {
const detail = String(message ?? '')
@@ -216,6 +322,12 @@ export const mapSystemOcrNativeError = (message: string): SystemOcrErrorCode =>
if (/\(0x00000000\)/.test(detail)) return 'OCR_LANGUAGE_UNAVAILABLE'
if (/Could not recognize file/i.test(detail)) return 'IMAGE_DECODE_FAILED'
if (/Could not open file/i.test(detail)) return 'IMAGE_DECODE_FAILED'
// macOS Vision / CoreImage 解码失败。
if (/zero-dimensioned image/i.test(detail)) return 'IMAGE_DECODE_FAILED'
if (/The image is too small/i.test(detail)) return 'IMAGE_DECODE_FAILED'
if (/CRImage|CIImage|CGImage/i.test(detail)) return 'IMAGE_DECODE_FAILED'
// macOS Vision 的"图里没有文字":正常终态,不是失败。
if (/No text recognized/i.test(detail)) return 'OCR_EMPTY_RESULT'
return 'OCR_FAILED'
}