diff --git a/docs/README.md b/docs/README.md index 5a5f11d..28f73a9 100644 --- a/docs/README.md +++ b/docs/README.md @@ -49,6 +49,7 @@ Agent Hub 让微信机器人调用本机 TraceMemo;Reader Skill / Local HTTP A ## 开发文档 - [开发、测试与构建](./development/overview.md) +- [界面开发规范:按钮与主题色](./development/ui-guidelines.md) - [Query Agent POC(开发测试入口)](./development/query-agent-poc.md) - [本地启动排障](./development/local-startup-troubleshooting.md) - [macOS 数据访问说明](./platform/macos.md) diff --git a/docs/development/ui-guidelines.md b/docs/development/ui-guidelines.md new file mode 100644 index 0000000..967150d --- /dev/null +++ b/docs/development/ui-guidelines.md @@ -0,0 +1,171 @@ +# 界面开发规范:按钮与主题色 + +这份规范回答一件事:**为什么同一个产品里,有的按钮是主题色,有的还是浏览器默认的黑白方角。** + +先看一个真实案例 —— 同一屏里的两组按钮: + +``` +主界面:「更新图片文字索引」 ← 主题色(正确) +弹窗里:「取消」「开始索引」 ← 浏览器默认样式(错误) +``` + +两者渲染出来完全不同,用户会以为是两个不同的产品。根因不是"设计没定颜色", +而是**组件在导出时把样式丢了**。下面写清楚怎么避免。 + +--- + +## 1. 永远不要写裸 ` +``` + +**唯一的例外**:结构性控件(导航项、Tab、列表行、图标热区)——它们有自己 +成套的布局样式,用原生 ` +``` + +**判据**:如果这个按钮在别的界面也会以同样形态出现("取消"、"保存"、"删除"), +它就该是 `Button`;如果它只在某一个位置有意义(侧栏导航项),才考虑原生。 + +--- + +## 2. 三种角色,只有三个默认变体 + +`Button` 提供 6 个变体,但**日常只用其中 3 个**: + +| 角色 | `variant` | 长什么样 | 用在哪 | +| --- | --- | --- | --- | +| 主要 | `default` | 主题色实底 | 这一步用户唯一该做的事 | +| 次要 | `outline` / `ghost` | 描边 / 无底色 | 取消、返回、并列的辅助操作 | +| 危险 | `destructive` | 红色实底 | 删除、清空、不可恢复的操作 | + +另外两个(`secondary` / `link`)按需用;`link` 只用于正文里的行内跳转。 + +**一条硬约束:同一个界面(或同一个弹窗)里,`default` 最多出现一次。** +两个主题色实底按钮并排,等于没有主次。 + +--- + +## 3. 弹窗按钮:组件已经带样式了,不要再包一层 + +`AlertDialogCancel` 和 `AlertDialogAction` **自带**按钮样式(分别是 `outline` +和 `default`),直接写文字即可: + +```tsx + + 取消 + 开始索引 + +``` + +**不要**再套一层 `Button`: + +```tsx +// 反面写法:外层已经有样式了,再包一层只会产生重复类名 + + + +``` + +需要危险动作时,用 `className` 覆盖(`cn` 走 tailwind-merge,同族类后者生效): + +```tsx + + 删除 + +``` + +--- + +## 4. 颜色只能用语义 token,禁止硬编码 + +颜色全部走 Tailwind 的语义类,它们背后是 `--tm-*` 变量,换主题时自动跟随: + +``` +背景 bg-primary / bg-surface / bg-accent / bg-destructive +文字 text-foreground / text-primary-foreground / text-muted-foreground +描边 border-border / border-border-subtle / border-disabled-border +``` + +```tsx +// 对 + + +// 错 —— 换主题时这行不会跟着变 + +``` + +**判据**:JSX 里出现 `#` 开头的颜色、`rgb(...)`、或 Tailwind 的调色板名 +(`bg-green-600`、`text-slate-500`)—— 都是漏用 semantic token 的信号。 + +--- + +## 5. 「默认样式」的三个常见来源 + +排查界面里冒出来的黑白方角按钮时,按这个顺序找: + +**① 组件导出时把样式丢了。** 最常见。把 Radix 的 primitive 原样导出: + +```tsx +// 错:渲染出来就是浏览器默认按钮 +const AlertDialogCancel = AlertDialogPrimitive.Cancel +``` + +正确做法是 `forwardRef` 包一层,挂上 `buttonVariants`: + +```tsx +const AlertDialogCancel = React.forwardRef<...>(({ className, ...props }, ref) => ( + +)) +``` + +**判据**:`components/ui/` 里凡是导出 Radix primitive 的地方,都要确认它是 +"样式化的封装"还是"原样透传"。原样透传只对布局容器(`Root` / `Portal` / +`Group`)成立,对**可点元素**(`Close` / `Action` / `Cancel` / `Item`)不成立。 + +**② `asChild` 里重复包了一层。** 外层已经带样式、子元素又带一次,虽然因为 +同族类后生效而不会出错,但会产生冗余类名。**能去掉一层就去掉。** + +**③ 原生 ` - )} - {/* 修好之后重跑:只重置失败记录,成功记录与其它数据一律不动。 */} - {!running && !paused && systemicFailure && ( - - )} - {/* 派生索引修复:只重建 Knowledge 里的图片搜索索引,**不重新识别任何图片**。 - 存在的意义就是"别为修一个索引问题重跑几万张图"。 */} - {!running && !paused && established && ( - + {/* + 修复类操作收进「更多」。 + + 它们各自只在很窄的情况下才有用(搜索索引不一致 / 有识别失败的图片), + 而主路径永远只有一个:更新索引。平铺出来时,用户看到的是四个都在说 + 「索引」的按钮,只能靠猜哪个该点。 + */} + {!running && (established || failureCount > 0) && ( + + + + + + {/* 派生索引修复:只重建 Knowledge 里的图片搜索索引,**不重新识别任何图片**。 + 存在的意义就是"别为修一个索引问题重跑几万张图"。 */} + {established && ( + void repair()} + > + 图片内容搜不到?修复搜索索引 + + )} + {/* 修好之后重跑:只重置失败记录,成功记录与其它数据一律不动。 */} + {failureCount > 0 && ( + void resetFailures()} + > + {`重试识别失败的图片(${failureCount.toLocaleString()} 张)`} + + )} + + )} {running && ( <> @@ -359,28 +400,35 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp )} + {/* + 中断(暂停 / 取消)之后必须能找到「继续」。 + 两种状态的 checkpoint 都是保留的,继续 = 从断点接上, + 所以这里刻意合并成一个入口 —— 否则「取消」过的索引会只剩 + 「更新图片文字索引」,用户根本看不出还能接着做。 + */} + {interrupted && ( + + )} + {/* 只有真的处在"暂停中"才有东西可取消:已取消的状态再点取消没有意义。 */} {paused && ( - <> - - - + )} @@ -425,3 +473,18 @@ export function ImageTextIndexCard({ dbReady, onNotice }: ImageTextIndexCardProp ) } + +/** + * 剩余时间文案。 + * + * `null` = 分母不可信或速度样本还不足 —— 如实说"计算中"。 + * 刻意不显示 p50 / p95 这类开发指标:这是用户界面,不是性能面板。 + */ +function formatEta(etaMs: number | null | undefined): string { + if (typeof etaMs !== 'number' || !Number.isFinite(etaMs) || etaMs <= 0) return '计算中' + const totalMinutes = Math.round(etaMs / 60_000) + if (totalMinutes < 1) return '不到 1 分钟' + const hours = Math.floor(totalMinutes / 60) + const minutes = totalMinutes % 60 + return hours > 0 ? `${hours} 小时 ${minutes} 分` : `${minutes} 分` +} diff --git a/src/renderer/src/components/search/evidenceSourceLabels.ts b/src/renderer/src/components/search/evidenceSourceLabels.ts new file mode 100644 index 0000000..d2a5092 --- /dev/null +++ b/src/renderer/src/components/search/evidenceSourceLabels.ts @@ -0,0 +1,63 @@ +import type { KnowledgeDerivedSource, KnowledgeMessageKind } from '../../../../shared/knowledge' + +/** + * 证据卡的来源标签。 + * + * 两个维度**正交**,必须分开表达,不能混成一个标签: + * - 消息类型(`sourceKind`):原始消息本身是什么; + * - 派生来源(`derivedSource`):这条结果是**靠什么命中**的(本地派生的 OCR / 转写)。 + * + * 文案面向用户:不出现 `image_ocr` 这类工程词。每条证据最多两个标签, + * 顺序固定为「消息类型 + 派生来源」。 + */ + +/** 只列用户看得懂的类型;`other` 之类没有信息量的取值不给标签。 */ +const MESSAGE_TYPE_LABELS: Record = { + text: '文本消息', + image: '图片消息', + voice: '语音消息', + video: '视频消息', + file: '文件消息', + link: '链接消息', + sticker: '表情消息', + system: '系统消息' +} + +const DERIVED_SOURCE_LABELS: Record = { + image_ocr: 'OCR命中', + voice_transcript: '转写命中' +} + +/** + * 派生命中内容的 snippet 前缀。 + * + * 目的只有一个:**不能让派生文本看起来像原始聊天内容**。 + * 普通文本消息不加前缀,原样展示。 + */ +const DERIVED_SNIPPET_PREFIXES: Record = { + image_ocr: 'OCR摘录:', + voice_transcript: '转写摘录:' +} + +export interface EvidenceSourceBadge { + /** 稳定的 DOM key,不用下标 —— 标签顺序可能随数据变化。 */ + key: 'messageType' | 'derivedSource' + label: string +} + +export function evidenceSourceBadges(item: { + sourceKind?: KnowledgeMessageKind | string + derivedSource?: KnowledgeDerivedSource +}): EvidenceSourceBadge[] { + const badges: EvidenceSourceBadge[] = [] + const messageLabel = item.sourceKind ? MESSAGE_TYPE_LABELS[item.sourceKind] : undefined + if (messageLabel) badges.push({ key: 'messageType', label: messageLabel }) + const derivedLabel = item.derivedSource ? DERIVED_SOURCE_LABELS[item.derivedSource] : undefined + if (derivedLabel) badges.push({ key: 'derivedSource', label: derivedLabel }) + return badges +} + +/** 派生内容的 snippet 前缀;普通文本消息返回空串(原样展示)。 */ +export function derivedSnippetPrefix(source?: KnowledgeDerivedSource): string { + return source ? DERIVED_SNIPPET_PREFIXES[source] : '' +} diff --git a/src/renderer/src/components/search/searchMarkdown.tsx b/src/renderer/src/components/search/searchMarkdown.tsx index 7fcd58b..35eb537 100644 --- a/src/renderer/src/components/search/searchMarkdown.tsx +++ b/src/renderer/src/components/search/searchMarkdown.tsx @@ -94,5 +94,33 @@ export const renderMarkdown = (value: string, options: MarkdownOptions = {}): Re ) } + /* + * Markdown 表格行。 + * + * 结果栏很窄,真表格在这里只会挤成一团(列宽错位、长字段换行难读)。提示词已经 + * 禁止模型为检索结果产表格,但历史回答与其它入口仍可能出现,所以这里把它降级成 + * **逐行的键值列表**:内容读得出来,且永远不会横向溢出容器。 + */ + if (/^\s*\|.*\|\s*$/.test(line)) { + const cells = line + .trim() + .replace(/^\||\|$/g, '') + .split('|') + .map((cell) => cell.trim()) + .filter((cell) => cell.length > 0) + // `|---|---|` 这类分隔行没有信息,当作空行处理。 + if (!cells.length || cells.every((cell) => /^:?-{2,}:?$/.test(cell))) { + return
+ } + return ( +
+ {cells.map((cell, cellIndex) => ( + + {inlineMarkdown(cell, `${key}-${cellIndex}`, options)} + + ))} +
+ ) + } return

{inlineMarkdown(line, key, options)}

}) diff --git a/src/renderer/src/components/search/searchTypes.ts b/src/renderer/src/components/search/searchTypes.ts index 85e9247..980440a 100644 --- a/src/renderer/src/components/search/searchTypes.ts +++ b/src/renderer/src/components/search/searchTypes.ts @@ -6,7 +6,11 @@ import type { AiSearchProgressEvent, AiSearchProgressStage } from '../../../../shared/ai-search' -import type { KnowledgeMessageKind, KnowledgeVoiceCoverage } from '../../../../shared/knowledge' +import type { + KnowledgeDerivedSource, + KnowledgeMessageKind, + KnowledgeVoiceCoverage +} from '../../../../shared/knowledge' import type { Contact, Message } from '../../../../shared/types' export type SearchStage = 'idle' | 'loading' | 'result' | 'partial' | 'insufficient' @@ -37,10 +41,11 @@ export interface EvidenceItem { /** * 命中所依赖的派生来源。 * - * `image_ocr` = 这条结果靠**图片里的文字**命中,而不是群友真的发了一条文字消息。 - * 有值时 Evidence 卡片显示轻量来源标记(「图片文字」)。 + * 有值 = 这条结果靠**本地派生内容**命中,而不是原始消息本身的文字 + * (`image_ocr` = 图片里的文字,`voice_transcript` = 语音转写)。 + * authoritative source 始终是原始消息 —— 这里只用来多挂一个来源标记。 */ - derivedSource?: 'image_ocr' + derivedSource?: KnowledgeDerivedSource /** 「从图片里读出来的文字」片段,只作命中解释。 */ imageOcrText?: string contact: Contact diff --git a/src/renderer/src/components/ui/alert-dialog.tsx b/src/renderer/src/components/ui/alert-dialog.tsx index ac26545..1702898 100644 --- a/src/renderer/src/components/ui/alert-dialog.tsx +++ b/src/renderer/src/components/ui/alert-dialog.tsx @@ -1,11 +1,10 @@ import * as React from 'react' import * as AlertDialogPrimitive from '@radix-ui/react-alert-dialog' import { cn } from '../../lib/cn' +import { buttonVariants } from './button' const AlertDialog = AlertDialogPrimitive.Root const AlertDialogTrigger = AlertDialogPrimitive.Trigger -const AlertDialogCancel = AlertDialogPrimitive.Cancel -const AlertDialogAction = AlertDialogPrimitive.Action const AlertDialogContent = React.forwardRef< React.ElementRef, @@ -70,6 +69,39 @@ const AlertDialogDescription = React.forwardRef< )) AlertDialogDescription.displayName = AlertDialogPrimitive.Description.displayName +/** + * 取消:次要动作,走 `outline`。 + * + * 这两个组件必须**显式**挂上 `buttonVariants`。直接 `export const X = Primitive.X` + * 会把 Radix 原始 primitive 原样抛出去,渲染成浏览器默认按钮(黑白方角), + * 跟产品主题完全不搭 —— 这类"忘了挂样式"的 primitive 是默认样式的常见来源。 + */ +const AlertDialogCancel = React.forwardRef< + React.ElementRef, + React.ComponentPropsWithoutRef +>(({ className, ...props }, ref) => ( + +)) +AlertDialogCancel.displayName = AlertDialogPrimitive.Cancel.displayName + +/** + * 确认:主要动作,走 `default`(主题色)。 + * + * 危险动作(删除、清空等)由调用方传 `className` 覆盖成 destructive —— + * `cn` 走的是 tailwind-merge,同族类会被后者替换,不必在这里开新的分支。 + */ +const AlertDialogAction = React.forwardRef< + React.ElementRef, + React.ComponentPropsWithoutRef +>(({ className, ...props }, ref) => ( + +)) +AlertDialogAction.displayName = AlertDialogPrimitive.Action.displayName + export { AlertDialog, AlertDialogTrigger, diff --git a/src/renderer/src/features/settings/ai-model/LocalImageTextRecognition.tsx b/src/renderer/src/features/settings/ai-model/LocalImageTextRecognition.tsx index c832b7e..e60252f 100644 --- a/src/renderer/src/features/settings/ai-model/LocalImageTextRecognition.tsx +++ b/src/renderer/src/features/settings/ai-model/LocalImageTextRecognition.tsx @@ -1,5 +1,9 @@ import { useCallback, useEffect, useState } from 'react' -import type { SystemOcrCapability, SystemOcrResult } from '../../../../../shared/system-ocr' +import type { + SystemOcrCapability, + SystemOcrEngine, + SystemOcrResult +} from '../../../../../shared/system-ocr' import { Button } from '../../../components/ui' const MAX_FILE_BYTES = 10 * 1024 * 1024 @@ -18,12 +22,15 @@ interface LocalOcrState { } /** - * 本地图片文字识别(Windows 系统 OCR)。 + * 本地图片文字识别(系统 OCR)。 * * 这是**本地 Runtime**,不是 AI 图片理解: * - 只把图片里的文字读出来;不描述画面、人物、场景,也不做视觉推理; * - 原始图片不会因为这一步发给任何 AI Provider; * - 结果只是派生内容,不会写进本地知识库。 + * + * 引擎由平台决定(Windows 系统 OCR / macOS 系统 OCR),UI 一律从 capability 派生文案, + * 不硬编码平台名。 */ export function LocalImageTextRecognition(): React.ReactElement { const [capability, setCapability] = useState(null) @@ -120,13 +127,14 @@ export function LocalImageTextRecognition(): React.ReactElement { const running = state.status === 'running' const result = state.result + const engineLabel = systemOcrEngineLabel(capability?.engine) return (

本地图片文字识别

-

使用 Windows 系统 OCR 在本机读取图片中的文字,原始图片无需发送给 AI Provider。

+

使用{engineLabel}在本机读取图片中的文字,原始图片无需发送给 AI Provider。

{capability ? (capability.available ? '本机可用' : '本机不可用') : '检测中…'} @@ -138,7 +146,7 @@ export function LocalImageTextRecognition(): React.ReactElement { ) : null} {capability?.available ? (

- 引擎:Windows 系统 OCR + 引擎:{engineLabel} {capability.runtimeVersion ? ` · 组件 ${capability.runtimeVersion}` : ''} {capability.language ? ` · 语言 ${capability.language}` : ' · 语言跟随系统'}

@@ -183,7 +191,7 @@ export function LocalImageTextRecognition(): React.ReactElement {
引擎
-
Windows 系统 OCR
+
{systemOcrEngineLabel(result.engine)}
语言
@@ -218,10 +226,22 @@ export function LocalImageTextRecognition(): React.ReactElement { ) } +/** 引擎标识 → 展示名。UI 不硬编码平台,一律从 capability / result 派生。 */ +function systemOcrEngineLabel(engine: SystemOcrEngine | undefined): string { + switch (engine) { + case 'macos-system-ocr': + return 'macOS 系统 OCR' + case 'windows-system-ocr': + return 'Windows 系统 OCR' + default: + return '系统 OCR' + } +} + function localOcrErrorMessage(result: SystemOcrResult): string { switch (result.errorCode) { case 'UNSUPPORTED_PLATFORM': - return '本地图片文字识别目前仅支持 Windows。' + return '本地图片文字识别目前支持 Windows 与 macOS。' case 'SYSTEM_OCR_UNAVAILABLE': return '本地文字识别组件不可用,请重新安装 TraceMemo。' case 'OCR_LANGUAGE_UNAVAILABLE': diff --git a/src/renderer/src/features/settings/image-decryption/ImageKeyConfiguration.tsx b/src/renderer/src/features/settings/image-decryption/ImageKeyConfiguration.tsx index 962f9c9..9bd47a9 100644 --- a/src/renderer/src/features/settings/image-decryption/ImageKeyConfiguration.tsx +++ b/src/renderer/src/features/settings/image-decryption/ImageKeyConfiguration.tsx @@ -1,5 +1,33 @@ +import { useState } from 'react' import type { ImageDecryptionState } from './types' -import { Input } from '../../../components/ui' +import { Button, Input } from '../../../components/ui' + +/** + * 密钥显示切换图标。 + * + * 项目里没有现成的眼睛图标(`LineIcon` 只有 database / shield 之类), + * 这里就地画一个 16px 线框图标,避免为一处 UI 引入图标依赖。 + */ +function EyeIcon({ crossed }: { crossed: boolean }): React.ReactElement { + return ( + + + + {crossed ? : null} + + ) +} export function ImageKeyConfiguration({ state, @@ -10,6 +38,13 @@ export function ImageKeyConfiguration({ disabled: boolean onEdit: (field: 'xorKey' | 'aesKey', value: string) => void }): React.ReactElement { + /** + * 只控制**本机的显示方式**,不影响任何存储、校验或解密行为: + * 密钥仍然以 password 语义渲染(浏览器/密码管理器照旧), + * 切换只是把 input 的 type 换成 text,让用户能核对自己填的 16 位密钥。 + */ + const [revealed, setRevealed] = useState(false) + return (
@@ -23,14 +58,29 @@ export function ImageKeyConfiguration({

修改后请先选择会话完成图片解析测试,再确认保存。

diff --git a/src/renderer/src/styles/search.scss b/src/renderer/src/styles/search.scss index f1f8e93..ad27065 100644 --- a/src/renderer/src/styles/search.scss +++ b/src/renderer/src/styles/search.scss @@ -420,6 +420,34 @@ min-width: 0; } +/* + * 图片文字索引卡片的操作区:每个按钮独占一行。 + * + * 这张卡最多并列 3 个操作(更新图片文字索引 / 更多 / 继续或取消), + * 而上面那套两列栅格是按「1 主 + 1 次」设计的:`auto` 列不可收缩, + * 窄侧栏下第 2、3 个按钮会把卡片撑出横向溢出。 + * 改成纵向堆叠后按钮宽度只跟随容器,结构上不可能溢出。 + */ +.ai-search-knowledge-actions.ai-search-image-index-actions { + display: flex; + flex-direction: column; + gap: 8px; + width: 100%; + max-width: 100%; + min-width: 0; +} + +/* + * 宽度严格跟随容器:`min-width: 0` 覆盖按钮自身的 68px 下限 + * (单列下那个下限已经没有意义,反而会阻止收缩)。 + * 标签最长 8 个汉字,最窄侧栏(190px)下仍有余量,所以保持单行不换行。 + */ +.ai-search-knowledge-actions.ai-search-image-index-actions > * { + width: 100%; + max-width: 100%; + min-width: 0; +} + .ai-search-knowledge-primary { min-width: 0; } @@ -1082,6 +1110,32 @@ font-weight: 700; } +/* + * Markdown 表格的降级渲染。 + * + * 结果栏很窄,真表格塞进来只会列宽错位、长字段换行难读。渲染层把表格行转成 + * 逐行的键值列表(见 searchMarkdown.tsx),这里只保证两件事: + * 读得出来,且**永远不会横向溢出容器**。所以用 flex + wrap,不用 table。 + */ +.ai-search-markdown-table-row { + display: flex; + flex-wrap: wrap; + gap: 2px 10px; + margin: 3px 0; + min-width: 0; +} + +.ai-search-markdown-table-cell { + min-width: 0; + overflow-wrap: anywhere; + color: var(--wxex-text-secondary); +} + +.ai-search-markdown-table-cell:first-child { + color: var(--wxex-text-primary); + font-weight: 600; +} + @media (max-width: 760px) { .ai-search-header-actions { align-items: flex-end; diff --git a/src/renderer/src/styles/settings.scss b/src/renderer/src/styles/settings.scss index d1a11f1..610dd8a 100644 --- a/src/renderer/src/styles/settings.scss +++ b/src/renderer/src/styles/settings.scss @@ -944,6 +944,25 @@ grid-template-columns: 180px minmax(0, 1fr); gap: 14px; } +/* + * AES 密钥字段 +「显示/隐藏」开关。 + * + * 用 flex 让按钮做**同级的兄弟节点**,而不是浮在输入框上: + * 这样不需要绝对定位、不需要给输入框留 padding,值很长时也不会被图标压住。 + */ +.image-key-secret { + display: flex; + align-items: center; + gap: 6px; +} +.image-key-secret > input { + /* Input 自身是 w-full(width: 100%),flex 里必须放开 min-width 才能收缩。 */ + flex: 1 1 auto; + min-width: 0; +} +.image-key-secret-toggle { + flex: 0 0 auto; +} .image-key-editor > p { margin: 0; color: #66706b; diff --git a/src/shared/image-text-index.ts b/src/shared/image-text-index.ts index 7b1ef39..2738863 100644 --- a/src/shared/image-text-index.ts +++ b/src/shared/image-text-index.ts @@ -11,24 +11,62 @@ * 2. `ImageOcrBinding` —— 「某个会话里的某条图片消息 → 某个 artifact」的绑定,保证去重不丢来源。 */ -/** 派生文本的引擎标识;与 System OCR 的引擎常量保持一致。 */ -export const IMAGE_TEXT_INDEX_ENGINE = 'windows-system-ocr' +/** + * 派生文本的引擎标识**不在这里定义**:它是 System OCR 运行时按平台决定的 + * (`resolveSystemOcrEngine`),并随 `ImageOcrProvenance` 一起进入 artifact 指纹。 + * 本模块只消费该值,不再持有任何单一平台的引擎常量。 + */ /** 派生库自身的 schema 版本(与 Knowledge 的 schema 相互独立)。 */ export const IMAGE_TEXT_INDEX_SCHEMA_VERSION = 1 -/** - * OCR 并发上限。 - * - * 当前实现**严格串行**(循环体内只有一次 await,无 Promise.all 扇出),等价于 1。 - * 这个常量是后续调高的唯一入口:Windows OCR 是进程内 WinRT 调用,实测单张 - * 20–40ms,串行已足够;调高只会和 Query Agent 抢 CPU。 - */ -export const DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY = 1 - /** 每个批次的图片条数;批间让出 event loop,保证 UI / 查询不被卡住。 */ export const IMAGE_TEXT_INDEX_BATCH_SIZE = 12 +/** + * 正常运行态下,向 Renderer 推送进度的最小间隔。 + * + * 后台仍然按 `IMAGE_TEXT_INDEX_BATCH_SIZE` 推进(batch / checkpoint / 并发都不受影响), + * 但**UI 不该感知 batch 大小** —— 每批都推会让计数以「+12」的粒度跳动。 + * 所以这里只节流**通知**:状态变化(开始/暂停/继续/取消/失败/完成/清理)一律立即推送。 + */ +export const IMAGE_TEXT_INDEX_PROGRESS_INTERVAL_MS = 5000 + +/** 速度统计窗口:取最近这段时间的增量,而不是整个任务的平均。 */ +export const IMAGE_TEXT_INDEX_RATE_WINDOW_MS = 60_000 + +/** 窗口内至少要有这么长的跨度才给出速度,否则显示「计算中」。 */ +export const IMAGE_TEXT_INDEX_RATE_MIN_SPAN_MS = 20_000 + +/** + * OCR 并发度**硬上限**。 + * + * `@napi-rs/system-ocr` 的 `recognize()` 是 napi AsyncTask,实际执行会占用 + * libuv **共享**线程池(fs / zlib / dns 等 native 异步工作也在用同一个池)。 + * 开得太高不会让单个识别更快,只会挤占同一进程里其它 native 异步工作。 + */ +export const MAX_IMAGE_TEXT_OCR_CONCURRENCY = 4 + +/** + * 默认 OCR 并发度(生产值)。 + * + * 取值规则:在"吞吐明显更高、且 CPU / UI 交互代价可接受"的前提下取**最低**并发。 + * 超过 2 之后单次识别耗时会明显劣化(多个识别互相争抢 CPU), + * 属于"多出来的并发全花在争抢上"。 + * + * **这个值是待复测的**:原取舍依据来自一次现已修复的固定开销存在时的对照, + * 而那个开销不随并发变化、会压扁并发收益。需要重新做锁定输入的对照后再决定; + * 在那之前保持 2,不要按"池子多大就用多大"去推。 + */ +export const DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY = 2 + +/** 解析并发度:只接受 1..MAX 的整数,其余一律回落到默认值。 */ +export function resolveImageTextOcrConcurrency(raw?: string | number | null): number { + const value = typeof raw === 'number' ? raw : Number.parseInt(String(raw ?? ''), 10) + if (!Number.isFinite(value) || value < 1) return DEFAULT_IMAGE_TEXT_OCR_CONCURRENCY + return Math.min(MAX_IMAGE_TEXT_OCR_CONCURRENCY, Math.floor(value)) +} + /** 已完成一批之后、回到会话循环前的让出时间。 */ export const IMAGE_TEXT_INDEX_YIELD_MS = 0 @@ -222,6 +260,15 @@ export interface ImageTextIndexProgress { cancellable: boolean paused: boolean lastError?: string + /** + * 最近窗口(`IMAGE_TEXT_INDEX_RATE_WINDOW_MS`)的实测速度,单位 张/秒。 + * + * 刻意用**滑动窗口**而不是整个任务的平均:全量回填要跑几小时, + * 历史平均会把"现在到底快不快"完全糊掉。样本跨度不足时为 null(UI 显示"计算中")。 + */ + speedPerSec?: number | null + /** 按当前窗口速度估算的剩余时间(毫秒);速度不可用或分母不可信时为 null。 */ + etaMs?: number | null } /** @@ -278,9 +325,8 @@ export function imageTextCoverageState(coverage: ImageTextIndexCoverage): ImageT /** * 处理进度百分比。 * - * 保留 1 位小数,且**未完成时封顶 99.9%**: - * `Math.round(45479 / 45707 * 100)` 会得到 `100`,于是出现了"已建立 · 仅完成 100%" - * 这种自相矛盾的显示。进度条可以近似,结论句不行。 + * 保留 1 位小数,且**未完成时封顶 99.9%**:直接四舍五入会把 99.5% 显示成 100%, + * 于是出现"已建立 · 仅完成 100%"这种自相矛盾的显示。进度条可以近似,结论句不行。 */ export function imageTextProcessedPercent(processed: number, total: number): number { if (!(total > 0)) return 0 @@ -324,7 +370,7 @@ export interface ImageTextIndexCountResult { } /** - * 单个会话的图片消息计数探针。 + * 单个会话的图片消息计数结果。 * * `count: null` = **统计失败**,不等于 0 张。调用方必须区分处理。 */ @@ -417,6 +463,89 @@ export interface ImageTextIndexStartOptions { sinceMs?: number } +/** 单个阶段的耗时聚合。**只含性能数字**,不含任何图片内容 / 路径 / 标识。 */ +export interface ImageTextIndexStageStat { + count: number + mean: number + p50: number + p95: number + max: number +} + +/** + * backfill 的**只读性能画像**,用来回答"时间花在哪一段"。 + * + * 只写性能数字,不含图片内容 / 路径 / 会话标识;UI 不渲染,仅落到 app log。 + * 刻意不暴露单张图片的耗时序列:那会把"哪张图慢"变成可推断的信息。 + */ +export interface ImageTextIndexStageTimings { + /** 本遍累计计数(与 UI 进度同源)。让这一行日志自洽,不必再去别处对数。 */ + counters: { + processed: number + indexed: number + empty: number + missing: number + failed: number + } + /** 最近窗口的实测速度(张/秒);样本不足或分母不可信时为 null。 */ + ratePerSec: number | null + /** 本遍实际执行过的 OCR 次数(命中已有 artifact 而跳过的不计)。 */ + ocrExecutions: number + /** 本遍生效的 OCR 并发度。 */ + ocrConcurrency: number + /** 找图片文件(同步,占主线程)。 */ + locate: ImageTextIndexStageStat + /** 解密(同步 + CPU,占主线程)。 */ + decrypt: ImageTextIndexStageStat + /** 构造可识别输入(base64 编码;Windows 还包含转 PNG)。 */ + normalize: ImageTextIndexStageStat + /** 识别(异步)。 */ + ocr: ImageTextIndexStageStat + /** 写 artifact + binding(SQLite,单 writer)。 */ + persist: ImageTextIndexStageStat + /** + * 单张图片在流水线里的净耗时。 + * + * 分母只含真正进入流水线的图片,所以这个值可以直接与上面五段之和对照; + * **不要用"整遍耗时 ÷ 处理张数"**,那会把 `preLoop` 的一次性成本摊进每张图片。 + */ + perImageMs: number + /** + * 本遍因为"没有可搜索内容变化"而**跳过** Knowledge 重建的会话数。 + * + * 与 `preLoop.onConversationIndexedMs` 配套看:跳过越多、那段时间越小, + * 说明门控在起作用。它同时是"到底有没有白做"的直接证据。 + */ + knowledgeIndexSkipped: number + /** 进入流水线**之前**的一次性成本(不按图片数摊)。 */ + preLoop: ImageTextIndexPreLoopCost +} + +/** + * 流水线**之外**的成本(单位毫秒),用来解释"单张成本很低、整遍却很慢"。 + * + * 这个结构的每一个字段都是"有理由不属于单张成本"的量: + * 一次性的、每会话一次的、以及**别的模块**的。它们必须单独可见 —— + * 否则 `perImageMs` 会看起来很好,而墙钟吞吐差好几倍,且无从归因。 + */ +export interface ImageTextIndexPreLoopCost { + /** 一遍 pass 开始前的一次性成本(能力探测 + 全账号图片统计 + 会话列表)。 */ + startupMs: number + /** └ 其中:统计图片消息总数(遍历全部会话的 SQL)。 */ + countImageMessagesMs: number + /** 每个会话进入流水线前的准备累计(水位 / 计数 / `listMessages`)。 */ + conversationSetupMs: number + /** └ 其中:读取并格式化会话消息累计。**已知的大头之一**。 */ + listMessagesMs: number + /** + * 会话完成后等待 Knowledge 重建(`onConversationIndexed`)的累计。 + * + * 这是**别的模块**的成本:Knowledge 侧会对同一个会话再全量读一遍消息并整篇写索引, + * 而且如果此时有索引在跑还会先等它。它不在 batch 循环里,所以 `perImageMs` 看不到它。 + */ + onConversationIndexedMs: number +} + /** 对外状态快照(问问微信卡片 / 设置清理页共用同一份)。 */ export interface ImageTextIndexStatus { progress: ImageTextIndexProgress @@ -424,4 +553,6 @@ export interface ImageTextIndexStatus { storage: ImageTextIndexStorageStats /** 正在做「检测到多少条图片消息」的 SQL 统计。 */ counting: boolean + /** 各阶段耗时画像(可选的附加诊断字段,UI 不渲染)。 */ + stageTimings?: ImageTextIndexStageTimings } diff --git a/src/shared/knowledge.ts b/src/shared/knowledge.ts index cfdedcf..0008d77 100644 --- a/src/shared/knowledge.ts +++ b/src/shared/knowledge.ts @@ -229,10 +229,19 @@ export interface KnowledgeEvidence { * 有值 = 这条结果依赖本地派生内容才能命中(而不是原始消息本身的文字)。 * 与 `sourceKind` 正交:`sourceKind` 说的是原始消息是什么,这里说的是"靠什么搜到的"。 */ - derivedSource?: 'image_ocr' + derivedSource?: KnowledgeDerivedSource score?: number } +/** + * 派生来源的种类。 + * + * 用 union 而不是 `isOcr: boolean`:以后接视频字幕 / 文件解析时只需要加一个成员, + * 不必给每个消费方再添一个布尔字段。UI 侧的展示文案集中在 + * `renderer/src/components/search/evidenceSourceLabels.ts`,不在这里。 + */ +export type KnowledgeDerivedSource = 'image_ocr' | 'voice_transcript' + /** * 证据文本面向用户 / 模型时的可读化处理。 * diff --git a/src/shared/local-query-api.ts b/src/shared/local-query-api.ts index 255aa0b..3ae1784 100644 --- a/src/shared/local-query-api.ts +++ b/src/shared/local-query-api.ts @@ -1,4 +1,4 @@ -import type { KnowledgeEvidence, KnowledgeVoiceCoverage } from './knowledge' +import type { KnowledgeDerivedSource, KnowledgeEvidence, KnowledgeVoiceCoverage } from './knowledge' export type QueryDirection = 'any' | 'from_target' | 'to_target' export type QueryOrder = 'asc' | 'desc' @@ -119,7 +119,7 @@ export interface QueryEvidenceItem * 让用户知道这段内容来自**图片里的文字**,而不是群友真的发了一条文字消息。 * authoritative source 仍然是原始图片消息,`messageRef` 也仍然指向原图。 */ - derivedSource?: 'image_ocr' + derivedSource?: KnowledgeDerivedSource /** * 「从图片里读出来的文字」片段,只用作命中解释。 * @@ -242,7 +242,7 @@ export interface QueryMessage { */ imageOcrText?: string /** 派生来源语义:`image_ocr` = 这段文字来自图片识别,而不是原始文字消息。 */ - derivedSource?: 'image_ocr' + derivedSource?: KnowledgeDerivedSource /** * 这条图片消息在本地图片文字索引里的状态。 * diff --git a/src/shared/query-agent.ts b/src/shared/query-agent.ts index 079b566..dae09dc 100644 --- a/src/shared/query-agent.ts +++ b/src/shared/query-agent.ts @@ -1,4 +1,5 @@ import type { AiSearchPipelineRequest, AiSearchPipelineResult } from './ai-search' +import type { KnowledgeDerivedSource } from './knowledge' import type { QueryCorpusScope } from './local-query-api' /** @@ -36,10 +37,11 @@ export interface AskWechatEvidenceItem { /** * 命中所依赖的派生来源(与 `messageType` 正交)。 * - * `image_ocr` = 这条结果靠**图片里的文字**命中,而不是群友真的发了一条文字消息。 - * Evidence UI 会据此显示轻量来源标记。authoritative source 仍是原始图片消息。 + * 有值 = 这条结果靠**本地派生内容**命中,而不是原始消息本身的文字 + * (`image_ocr` = 图片里的文字,`voice_transcript` = 语音转写)。 + * Evidence UI 会据此多挂一个来源标记;authoritative source 仍是原始消息。 */ - derivedSource?: 'image_ocr' + derivedSource?: KnowledgeDerivedSource /** 「从图片里读出来的文字」片段,只作命中解释(普通文字消息不会有)。 */ imageOcrText?: string attachment?: { kind?: string; name?: string; url?: string; sizeBytes?: number } diff --git a/src/shared/system-ocr.ts b/src/shared/system-ocr.ts index b3e28ed..7c72a6c 100644 --- a/src/shared/system-ocr.ts +++ b/src/shared/system-ocr.ts @@ -7,15 +7,38 @@ // 它不占用 AIVisionRuntimeConfig.source,也不产生任何网络请求。 // - 能力边界:只把图片里的文字读出来。它不等于「理解人物 / 理解场景 / // 描述照片 / 理解表情包语义 / 视觉推理」——那些仍然属于 Vision Model。 -// - Windows 后端为 Windows.Media.Ocr.OcrEngine(经 @napi-rs/system-ocr 调用)。 -// macOS 本轮只保留架构位置,未实现;Linux 不支持。 +// - 后端按平台选择(统一经 @napi-rs/system-ocr 调用): +// Windows → Windows.Media.Ocr.OcrEngine +// macOS → Apple Vision(VNRecognizeTextRequest / RecognizeDocumentsRequest) +// Linux 不支持。 +// - 引擎标识会进入 artifact 指纹与缓存 key,两个平台的结果**不得互相复用**。 // -// 数据边界(本轮不做): -// - 不做历史图片全量 OCR、不做 Knowledge 回填、不把 OCR 文字伪装成原始聊天文字。 -// 原始消息始终是权威来源,OCR 文字只是派生内容(本轮仅存在于内存)。 +// 数据边界: +// - OCR 文字始终是**派生内容**,会把原图定位回去(artifact + binding), +// 但绝不写回 WCDB、也绝不伪装成原始聊天文字;原始消息始终是权威来源。 +// - 历史图片回填与 Knowledge 回填由 image-text-index 负责,本模块只提供识别能力。 + +/** Windows 引擎标识(Windows.Media.Ocr.OcrEngine)。 */ +export const SYSTEM_OCR_ENGINE_WINDOWS = 'windows-system-ocr' + +/** macOS 引擎标识(Apple Vision)。 */ +export const SYSTEM_OCR_ENGINE_MACOS = 'macos-system-ocr' /** System OCR 引擎标识。这是本地 Runtime,不是 provider id。 */ -export const SYSTEM_OCR_ENGINE = 'windows-system-ocr' +export type SystemOcrEngine = typeof SYSTEM_OCR_ENGINE_WINDOWS | typeof SYSTEM_OCR_ENGINE_MACOS + +/** 支持 System OCR 的平台。Linux 明确不支持。 */ +export const isSystemOcrPlatform = (platform: string): boolean => + platform === 'win32' || platform === 'darwin' + +/** + * 平台 → 引擎标识。 + * + * 不要把引擎串硬编码成某一个平台:它同时是 artifact 指纹的一部分, + * 一旦写死,跨平台结果就会互相复用。 + */ +export const resolveSystemOcrEngine = (platform: string): SystemOcrEngine => + platform === 'darwin' ? SYSTEM_OCR_ENGINE_MACOS : SYSTEM_OCR_ENGINE_WINDOWS /** 本地 OCR 结果在内存中的缓存时长。 */ export const SYSTEM_OCR_CACHE_TTL_MS = 10 * 60 * 1000 @@ -27,13 +50,13 @@ export const SYSTEM_OCR_CACHE_TTL_MS = 10 * 60 * 1000 export type SystemOcrErrorCode = /** 运行时不可用(native binding 缺失 / 加载失败) */ | 'SYSTEM_OCR_UNAVAILABLE' - /** 当前平台不支持(Linux,或非 Windows 平台) */ + /** 当前平台不支持(Linux,或非 Windows / macOS 平台) */ | 'UNSUPPORTED_PLATFORM' /** 图片格式不在支持范围内 */ | 'UNSUPPORTED_IMAGE' - /** 图片解码失败(格式可识别但内容损坏或无法转成 PNG) */ + /** 图片解码失败(格式可识别但内容损坏或无法转成可识别图像) */ | 'IMAGE_DECODE_FAILED' - /** 当前 Windows 未安装对应的 OCR 语言支持 */ + /** 当前 Windows 未安装对应的 OCR 语言支持(macOS 由 Vision 自行决定,不会出现) */ | 'OCR_LANGUAGE_UNAVAILABLE' /** 引擎执行失败 */ | 'OCR_FAILED' @@ -49,12 +72,17 @@ export type SystemOcrUnavailableReason = export interface SystemOcrCapability { /** 本机当前是否真的可以识别图片文字 */ available: boolean - engine: typeof SYSTEM_OCR_ENGINE + engine: SystemOcrEngine platform: NodeJS.Platform arch: string /** @napi-rs/system-ocr 运行时版本;无法读取时为 null */ runtimeVersion: string | null - /** 实际可用的 OCR 语言标签(对应 Windows 语言包);null 表示走系统用户语言 */ + /** + * 实际使用的 OCR 语言标签。 + * + * Windows 为系统语言包对应的标签(如 zh-Hans-CN);macOS 由 Vision 自行决定识别语言, + * 这里恒为 null(对应 UI 的「跟随系统语言」)。null 也表示走系统语言。 + */ language: string | null reason?: SystemOcrUnavailableReason /** 面向用户的中文说明,可直接展示 */ @@ -71,20 +99,20 @@ export interface SystemOcrBoundingBox { export interface SystemOcrLine { text: string - /** Windows 恒为 1.0 */ + /** Windows 恒为 1.0;macOS 为 Vision 返回的逐行平均置信度 */ confidence: number boundingBox: SystemOcrBoundingBox } -/** 本地 OCR 结果。不包含任何 Windows handle / native 内部对象。 */ +/** 本地 OCR 结果。不包含任何平台 handle / native 内部对象。 */ export interface SystemOcrResult { success: boolean /** 归一化后的文本(去掉 CJK 字符之间的引擎伪空格) */ text: string lines: SystemOcrLine[] - /** 实际使用的 OCR 语言标签;null 表示由系统用户语言决定 */ + /** 实际使用的 OCR 语言标签;null 表示由系统决定识别语言 */ language: string | null - engine: typeof SYSTEM_OCR_ENGINE + engine: SystemOcrEngine durationMs: number /** 命中内存缓存时为 true */ fromCache?: boolean @@ -104,21 +132,24 @@ export interface SystemOcrRequest { /** * 缓存 key 组合。刻意与 ImageInsight 的 `imageHash` 保持不同的键空间, * 保证远端 Vision 的旧结果永远不会被当成"本地 OCR 结果"复用, - * 也保证 System OCR 运行时升级后不会永远命中旧结果。 + * 也保证 System OCR 运行时升级 / 切换平台后不会永远命中旧结果。 */ export const buildSystemOcrCacheKey = (input: { imageHash: string language: string | null runtimeVersion: string | null platform?: string -}): string => - [ + engine?: SystemOcrEngine +}): string => { + const platform = input.platform ?? 'unknown' + return [ input.imageHash, - SYSTEM_OCR_ENGINE, - input.platform ?? 'unknown', + input.engine ?? resolveSystemOcrEngine(platform), + platform, input.language ?? 'auto', input.runtimeVersion ?? 'unknown' ].join('|') +} const CJK_CHAR = /[\u3000-\u303f\u3040-\u30ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff00-\uffef\uac00-\ud7af]/ @@ -126,6 +157,9 @@ const CJK_CHAR = /** * Windows OCR 会在每个 CJK 字符之间插入空格("本 地 图 片")。 * 这里只删除 **两侧都是 CJK** 的空格,保留 "TraceMemo 本地图片文字识别" 里的真实分隔。 + * + * macOS(Vision)本就输出连续中文,这条规则对它恒等;保留是为了两个平台共用一条 + * 归一化路径,而不是给 macOS 加特例。 */ export const normalizeSystemOcrText = (value: string): string => { const source = String(value ?? '') @@ -149,8 +183,11 @@ export const normalizeSystemOcrText = (value: string): string => { return result.trim() } -/** 把系统 locale(如 zh-CN / en-US)映射成 Windows OCR 语言标签。 */ -const LANGUAGE_TAG_BY_LOCALE: Record = { +/** + * 把系统 locale(如 zh-CN / en-US)映射成 **Windows OCR 语言标签** + * (即 Windows 语言包里注册的 BCP-47 标签,中文带 region 子标签)。 + */ +const WINDOWS_LANGUAGE_TAG_BY_LOCALE: Record = { zh: 'zh-Hans-CN', 'zh-cn': 'zh-Hans-CN', 'zh-hans': 'zh-Hans-CN', @@ -186,7 +223,51 @@ const LANGUAGE_TAG_BY_LOCALE: Record = { 'ru-ru': 'ru-RU' } -export const resolveSystemOcrLanguageTag = ( +/** + * 把系统 locale 映射成 **Apple Vision 语言标签**。 + * + * 与 Windows 表刻意分开:Vision 只认脚本级子标签(`zh-Hans` / `zh-Hant`), + * 不认 `zh-Hans-CN` 这类 region 组合;港台繁体统一收敛到 `zh-Hant`。 + */ +const MACOS_LANGUAGE_TAG_BY_LOCALE: Record = { + zh: 'zh-Hans', + 'zh-cn': 'zh-Hans', + 'zh-sg': 'zh-Hans', + 'zh-hans': 'zh-Hans', + 'zh-hans-cn': 'zh-Hans', + 'zh-hans-sg': 'zh-Hans', + 'zh-tw': 'zh-Hant', + 'zh-hk': 'zh-Hant', + 'zh-mo': 'zh-Hant', + 'zh-hant': 'zh-Hant', + 'zh-hant-tw': 'zh-Hant', + 'zh-hant-hk': 'zh-Hant', + 'zh-hant-mo': 'zh-Hant', + en: 'en-US', + 'en-us': 'en-US', + 'en-gb': 'en-GB', + 'en-au': 'en-AU', + 'en-ca': 'en-CA', + ja: 'ja-JP', + 'ja-jp': 'ja-JP', + ko: 'ko-KR', + 'ko-kr': 'ko-KR', + fr: 'fr-FR', + 'fr-fr': 'fr-FR', + de: 'de-DE', + 'de-de': 'de-DE', + es: 'es-ES', + 'es-es': 'es-ES', + it: 'it-IT', + 'it-it': 'it-IT', + pt: 'pt-BR', + 'pt-br': 'pt-BR', + ru: 'ru-RU', + 'ru-ru': 'ru-RU' +} + +const lookupLanguageTag = ( + table: Record, locale: string | null | undefined ): string | null => { const normalized = String(locale ?? '') @@ -194,11 +275,26 @@ export const resolveSystemOcrLanguageTag = ( .toLowerCase() .replace(/_/g, '-') if (!normalized) return null - if (LANGUAGE_TAG_BY_LOCALE[normalized]) return LANGUAGE_TAG_BY_LOCALE[normalized] + if (table[normalized]) return table[normalized] const primary = normalized.split('-')[0] - return LANGUAGE_TAG_BY_LOCALE[primary] ?? null + return table[primary] ?? null } +/** 系统 locale → Windows OCR 语言标签。 */ +export const resolveWindowsOcrLanguageTag = (locale: string | null | undefined): string | null => + lookupLanguageTag(WINDOWS_LANGUAGE_TAG_BY_LOCALE, locale) + +/** 系统 locale → Apple Vision 语言标签。 */ +export const resolveMacOcrLanguageTag = (locale: string | null | undefined): string | null => + lookupLanguageTag(MACOS_LANGUAGE_TAG_BY_LOCALE, locale) + +/** 按平台把系统 locale 映射成该平台 OCR 引擎接受的语言标签。 */ +export const resolveSystemOcrLanguageTag = ( + locale: string | null | undefined, + platform: string = 'win32' +): string | null => + platform === 'darwin' ? resolveMacOcrLanguageTag(locale) : resolveWindowsOcrLanguageTag(locale) + /** * 把 native 错误映射成产品级错误码。 * @@ -206,6 +302,16 @@ export const resolveSystemOcrLanguageTag = ( * - 语言包缺失 / 引擎无法创建:`Windows error 操作成功完成。 (0x00000000)` * —— TryCreateFromLanguage 返回 null 引擎但 HRESULT 是 S_OK,非常容易误判。 * - 送给解码器的字节不是可识别的图片:`Windows error Could not recognize file (0x80070005)` + * + * 已确认的 macOS 行为(1.2.0 / Vision): + * - 图片无法解码成 CGImage(截断、伪造魔数、维度非法): + * `CRImage Reader Detector was given zero-dimensioned image (0 x 0)` + * - 图片任一边不超过 2px:`The image is too small in at least one dimension 2 x 2 ...` + * - **图片没有文字时是抛错而不是返回空文本**:`No text recognized` + * —— 它必须映射成 `OCR_EMPTY_RESULT`(正常终态)。映射成失败会让表情包 / + * 风景图 / 头像全部变成"可重试失败",既污染派生库也会被反复重试。 + * - macOS 没有"语言包缺失"这个概念(Vision 自行决定识别语言), + * 所以这里不会映射出 OCR_LANGUAGE_UNAVAILABLE。 */ export const mapSystemOcrNativeError = (message: string): SystemOcrErrorCode => { const detail = String(message ?? '') @@ -216,6 +322,12 @@ export const mapSystemOcrNativeError = (message: string): SystemOcrErrorCode => if (/\(0x00000000\)/.test(detail)) return 'OCR_LANGUAGE_UNAVAILABLE' if (/Could not recognize file/i.test(detail)) return 'IMAGE_DECODE_FAILED' if (/Could not open file/i.test(detail)) return 'IMAGE_DECODE_FAILED' + // macOS Vision / CoreImage 解码失败。 + if (/zero-dimensioned image/i.test(detail)) return 'IMAGE_DECODE_FAILED' + if (/The image is too small/i.test(detail)) return 'IMAGE_DECODE_FAILED' + if (/CRImage|CIImage|CGImage/i.test(detail)) return 'IMAGE_DECODE_FAILED' + // macOS Vision 的"图里没有文字":正常终态,不是失败。 + if (/No text recognized/i.test(detail)) return 'OCR_EMPTY_RESULT' return 'OCR_FAILED' } diff --git a/tests/component/alert-dialog-button-theme.test.tsx b/tests/component/alert-dialog-button-theme.test.tsx new file mode 100644 index 0000000..32198d9 --- /dev/null +++ b/tests/component/alert-dialog-button-theme.test.tsx @@ -0,0 +1,60 @@ +/** + * 弹窗按钮必须自带主题样式。 + * + * 这两处曾经把 Radix 的 primitive **原样导出**,渲染出来是浏览器默认的黑白方角 + * 按钮 —— 跟主界面的主题色按钮完全脱节,用户会以为是两个不同的产品。 + * + * 所以这里锁的是"类名里确实带了按钮变体",而不是某个具体颜色值。 + * 规范见 `docs/development/ui-guidelines.md`。 + */ +import { render, screen } from '@testing-library/react' +import { describe, expect, it } from 'vitest' +import { + AlertDialog, + AlertDialogAction, + AlertDialogCancel, + AlertDialogContent, + AlertDialogFooter +} from '../../src/renderer/src/components/ui/alert-dialog' + +function renderFooter(children: React.ReactNode): void { + render( + + + {children} + + + ) +} + +describe('弹窗按钮的主题样式', () => { + it('确认按钮走主题色实底', () => { + renderFooter(开始索引) + + const action = screen.getByRole('button', { name: '开始索引' }) + expect(action.className).toContain('bg-primary') + expect(action.className).toContain('text-primary-foreground') + }) + + it('取消按钮走次要描边,不抢主按钮的主题色', () => { + renderFooter(取消) + + const cancel = screen.getByRole('button', { name: '取消' }) + expect(cancel.className).toContain('border') + expect(cancel.className).not.toContain('bg-primary') + }) + + it('调用方传 className 能覆盖成危险动作', () => { + renderFooter( + + 删除 + + ) + + const action = screen.getByRole('button', { name: '删除' }) + expect(action.className).toContain('bg-destructive') + // 只断言"独立的 bg-primary 类不存在"—— `hover:bg-primary-hover` 含同样子串, + // 用裸字符串匹配会误伤。 + expect(action.className).not.toMatch(/(^|\s)bg-primary(\s|$)/) + }) +}) diff --git a/tests/component/evidence-image-ocr-source.test.tsx b/tests/component/evidence-image-ocr-source.test.tsx deleted file mode 100644 index 07760cd..0000000 --- a/tests/component/evidence-image-ocr-source.test.tsx +++ /dev/null @@ -1,106 +0,0 @@ -/** - * §1 / §17:Evidence 的「图片文字」来源语义。 - * - * 三条不能退让的约束: - * 1. 来自图片 OCR 的命中,UI 必须有轻量来源标记(「图片文字」), - * 让用户知道这段内容来自图片,而不是群友真的发了一条文字消息; - * 2. authoritative source 仍然是**原始图片消息** —— messageRef 不变,跳转目标就是原图; - * 3. 引擎内部前缀(`图片文字:` / `OCR:` / `system-ocr`)绝不允许出现在用户可见文本里; - * 4. 普通文字消息的 Evidence 完全不受影响(不该凭空多出一个标记)。 - */ -import { render, screen } from '@testing-library/react' -import { describe, expect, it, vi } from 'vitest' -import { AISearchEvidencePanel } from '../../src/renderer/src/components/search/AISearchEvidencePanel' -import { mapAskWechatEvidence } from '../../src/renderer/src/components/search/askWechatPresentation' -import type { EvidenceItem } from '../../src/renderer/src/components/search/searchTypes' -import type { AskWechatEvidenceItem } from '../../src/shared/query-agent' -import { encodeMessageRef } from '../../src/shared/local-query-api' - -const OCR_TEXT = 'OpenAI ChatGPT Plus $20 Pro $200' -const IMAGE_REF = encodeMessageRef('md5-tech-group', '9001') -const TEXT_REF = encodeMessageRef('md5-tech-group', '9002') - -/** Query Agent 交给渲染层的证据(图片 OCR 命中)。 */ -const imageOcrEvidence: AskWechatEvidenceItem = { - messageRef: IMAGE_REF, - conversationName: '技术交流群', - conversationType: 'group', - sender: '张三', - timestamp: Date.parse('2026-09-03T14:32:00+08:00'), - messageType: 'image', - // 已经由 main 侧剥掉内部前缀的可读文本 - text: OCR_TEXT, - derivedSource: 'image_ocr', - imageOcrText: OCR_TEXT, - source: 'search_messages' -} - -/** 普通文字消息证据(对照组)。 */ -const plainEvidence: AskWechatEvidenceItem = { - messageRef: TEXT_REF, - conversationName: '技术交流群', - conversationType: 'group', - sender: '张三', - timestamp: Date.parse('2026-09-03T14:30:00+08:00'), - messageType: 'text', - text: '今天正常讨论一下 API', - source: 'search_messages' -} - -function renderPanel(evidence: EvidenceItem[]) { - const props: React.ComponentProps = { - evidence, - collectionCount: evidence.length, - selectedEvidence: 0, - evidenceFlash: { index: -1, nonce: 0 }, - senderNames: {}, - hasMoreEvidence: false, - onFocusEvidence: vi.fn(), - onJumpToEvidence: vi.fn(), - onLoadMoreEvidence: vi.fn(), - setEvidenceCardRef: vi.fn() - } - render() - return props -} - -describe('图片文字 Evidence 的来源语义', () => { - it('映射层保留派生来源与 OCR 片段,且跳转目标仍是原始图片消息', () => { - const [mapped] = mapAskWechatEvidence([imageOcrEvidence]) - - expect(mapped.derivedSource).toBe('image_ocr') - expect(mapped.imageOcrText).toBe(OCR_TEXT) - // authoritative source = 原始图片消息:引用不变 - expect(mapped.messageRef).toBe(IMAGE_REF) - expect(mapped.message.id).toBe('9001') - expect(mapped.contact.m_nsNickName).toBe('技术交流群') - expect(mapped.sourceKind).toBe('image') - }) - - it('图片 OCR 命中显示「图片文字」标记与命中解释,不泄露内部前缀', () => { - const evidence = mapAskWechatEvidence([imageOcrEvidence]) - renderPanel(evidence) - - const badge = screen.getByTestId('evidence-image-ocr-badge') - expect(badge).toBeVisible() - expect(badge.textContent).toBe('图片文字') - - const snippet = screen.getByTestId('evidence-image-ocr-snippet') - expect(snippet.textContent).toContain(OCR_TEXT) - - // 内部前缀绝不能出现在用户可见文本里 - const panelText = document.body.textContent || '' - expect(panelText).not.toContain('图片文字:') - expect(panelText).not.toContain('OCR:') - expect(panelText).not.toContain('system-ocr') - }) - - it('普通文字消息的 Evidence 不受影响:没有来源标记,也没有 OCR 片段', () => { - const evidence = mapAskWechatEvidence([plainEvidence]) - renderPanel(evidence) - - expect(screen.queryByTestId('evidence-image-ocr-badge')).not.toBeInTheDocument() - expect(screen.queryByTestId('evidence-image-ocr-snippet')).not.toBeInTheDocument() - expect(screen.getByText('今天正常讨论一下 API')).toBeVisible() - }) -}) diff --git a/tests/component/evidence-source-labels.test.tsx b/tests/component/evidence-source-labels.test.tsx new file mode 100644 index 0000000..923e6c7 --- /dev/null +++ b/tests/component/evidence-source-labels.test.tsx @@ -0,0 +1,136 @@ +/** + * Evidence 的来源语义与来源标签。 + * + * 不能退让的约束: + * 1. **消息类型**与**派生来源**是两个正交维度,UI 必须分别标出来: + * 前者说"原始消息是什么"(文本 / 图片 / 语音…), + * 后者说"这条结果是靠什么命中的"(OCR命中 / 转写命中); + * 2. authoritative source 仍是原始消息 —— messageRef 不变,跳转目标就是它; + * 3. 派生命中内容必须自报来源(`OCR摘录:` / `转写摘录:`), + * 不能被读成群友真的发过这样一段文字; + * 4. 引擎内部前缀(`图片文字:` / `OCR:` / `system-ocr`)绝不出现在用户可见文本里; + * 5. 普通文字消息只标「文本消息」,不凭空多出来源标记。 + */ +import { render, screen } from '@testing-library/react' +import { describe, expect, it, vi } from 'vitest' +import { AISearchEvidencePanel } from '../../src/renderer/src/components/search/AISearchEvidencePanel' +import { mapAskWechatEvidence } from '../../src/renderer/src/components/search/askWechatPresentation' +import type { EvidenceItem } from '../../src/renderer/src/components/search/searchTypes' +import type { AskWechatEvidenceItem } from '../../src/shared/query-agent' +import { encodeMessageRef } from '../../src/shared/local-query-api' + +const OCR_TEXT = '今日特价 248 元' +const TRANSCRIPT_TEXT = '明天下午三点评审' +const IMAGE_REF = encodeMessageRef('fixture-conversation-a', '9001') +const VOICE_REF = encodeMessageRef('fixture-conversation-a', '9002') +const TEXT_REF = encodeMessageRef('fixture-conversation-a', '9003') + +const base = { + conversationName: '测试群', + conversationType: 'group' as const, + sender: '用户A', + source: 'search_messages' +} + +/** 图片 OCR 命中。 */ +const imageOcrEvidence: AskWechatEvidenceItem = { + ...base, + messageRef: IMAGE_REF, + timestamp: Date.parse('2026-09-03T14:32:00+08:00'), + messageType: 'image', + text: OCR_TEXT, + derivedSource: 'image_ocr', + imageOcrText: OCR_TEXT +} + +/** 语音转写命中。 */ +const voiceTranscriptEvidence: AskWechatEvidenceItem = { + ...base, + messageRef: VOICE_REF, + timestamp: Date.parse('2026-09-03T14:34:00+08:00'), + messageType: 'voice', + text: TRANSCRIPT_TEXT, + derivedSource: 'voice_transcript' +} + +/** 普通文字消息(对照组)。 */ +const plainEvidence: AskWechatEvidenceItem = { + ...base, + messageRef: TEXT_REF, + timestamp: Date.parse('2026-09-03T14:30:00+08:00'), + messageType: 'text', + text: '这是一条普通的文字消息' +} + +function renderPanel(evidence: EvidenceItem[]): React.ComponentProps { + const props: React.ComponentProps = { + evidence, + collectionCount: evidence.length, + selectedEvidence: 0, + evidenceFlash: { index: -1, nonce: 0 }, + senderNames: {}, + hasMoreEvidence: false, + onFocusEvidence: vi.fn(), + onJumpToEvidence: vi.fn(), + onLoadMoreEvidence: vi.fn(), + setEvidenceCardRef: vi.fn() + } + render() + return props +} + +describe('Evidence 的来源语义与来源标签', () => { + it('映射层保留消息类型与派生来源,且 authoritative source 不变', () => { + const [image] = mapAskWechatEvidence([imageOcrEvidence]) + expect(image.sourceKind).toBe('image') + expect(image.derivedSource).toBe('image_ocr') + expect(image.imageOcrText).toBe(OCR_TEXT) + expect(image.messageRef).toBe(IMAGE_REF) + + const [voice] = mapAskWechatEvidence([voiceTranscriptEvidence]) + expect(voice.sourceKind).toBe('voice') + expect(voice.derivedSource).toBe('voice_transcript') + expect(voice.messageRef).toBe(VOICE_REF) + }) + + it('图片 OCR 命中:标「图片消息 + OCR命中」,片段写明是 OCR 摘录', () => { + renderPanel(mapAskWechatEvidence([imageOcrEvidence])) + + expect(screen.getByTestId('evidence-badge-messageType').textContent).toBe('图片消息') + expect(screen.getByTestId('evidence-badge-derivedSource').textContent).toBe('OCR命中') + expect(screen.getByTestId('evidence-image-ocr-snippet').textContent).toContain(OCR_TEXT) + + const panelText = document.body.textContent || '' + expect(panelText).toContain('OCR摘录:') + // 内部前缀绝不能出现在用户可见文本里 + expect(panelText).not.toContain('图片文字:') + expect(panelText).not.toContain('OCR:') + expect(panelText).not.toContain('system-ocr') + // 不暴露工程字段名 + expect(panelText).not.toContain('image_ocr') + expect(panelText).not.toContain('derivedSource') + }) + + it('语音转写命中:标「语音消息 + 转写命中」,片段写明是转写摘录', () => { + renderPanel(mapAskWechatEvidence([voiceTranscriptEvidence])) + + expect(screen.getByTestId('evidence-badge-messageType').textContent).toBe('语音消息') + expect(screen.getByTestId('evidence-badge-derivedSource').textContent).toBe('转写命中') + + const panelText = document.body.textContent || '' + expect(panelText).toContain('转写摘录:') + expect(panelText).not.toContain('voice_transcript') + }) + + it('普通文字消息只标「文本消息」,没有派生来源标记', () => { + renderPanel(mapAskWechatEvidence([plainEvidence])) + + expect(screen.getByTestId('evidence-badge-messageType').textContent).toBe('文本消息') + expect(screen.queryByTestId('evidence-badge-derivedSource')).not.toBeInTheDocument() + expect(screen.queryByTestId('evidence-image-ocr-snippet')).not.toBeInTheDocument() + expect(screen.getByText('这是一条普通的文字消息')).toBeVisible() + + const panelText = document.body.textContent || '' + expect(panelText).not.toContain('摘录:') + }) +}) diff --git a/tests/component/image-key-reveal.test.tsx b/tests/component/image-key-reveal.test.tsx new file mode 100644 index 0000000..4730c63 --- /dev/null +++ b/tests/component/image-key-reveal.test.tsx @@ -0,0 +1,85 @@ +/** + * 图片密钥的「显示 / 隐藏」开关。 + * + * 契约(产品要求):AES 密钥默认遮蔽,用户点一下眼睛能看见自己填的值。 + * 关键边界:这只是**显示层**行为 —— 不许改动值、不许触发保存、不许影响校验。 + */ +import { render, screen } from '@testing-library/react' +import userEvent from '@testing-library/user-event' +import { describe, expect, it, vi } from 'vitest' +import { ImageKeyConfiguration } from '../../src/renderer/src/features/settings/image-decryption/ImageKeyConfiguration' +import type { ImageDecryptionState } from '../../src/renderer/src/features/settings/image-decryption/types' + +const AES_KEY = '0123456789abcdef' + +function state(overrides: Partial = {}): ImageDecryptionState { + return { + phase: 'idle', + config: null, + status: null, + contacts: [], + selectedUserMd5: '', + resourceRoot: '/tmp/fixture', + xorKey: '0x40', + aesKey: AES_KEY, + testResult: null, + autoPhase: 'idle', + autoProgress: '', + dirty: false, + ...overrides + } +} + +const aesInput = (): HTMLInputElement => { + const inputs = document.querySelectorAll('.image-key-secret > input') + if (!inputs.length) throw new Error('AES input missing') + return inputs[0] +} + +describe('图片密钥显示开关', () => { + it('默认遮蔽,且按钮自称「显示图片密钥」', () => { + render() + + expect(aesInput().type).toBe('password') + expect(aesInput().value).toBe(AES_KEY) + + const toggle = screen.getByTestId('image-key-reveal') + expect(toggle).toHaveAttribute('aria-pressed', 'false') + expect(toggle).toHaveAccessibleName('显示图片密钥') + }) + + it('点一下明文显示,再点一下回到遮蔽', async () => { + render() + const toggle = screen.getByTestId('image-key-reveal') + + await userEvent.click(toggle) + expect(aesInput().type).toBe('text') + expect(toggle).toHaveAttribute('aria-pressed', 'true') + expect(toggle).toHaveAccessibleName('隐藏图片密钥') + // 明文里能真的读到密钥本身。 + expect(aesInput().value).toBe(AES_KEY) + + await userEvent.click(toggle) + expect(aesInput().type).toBe('password') + expect(toggle).toHaveAttribute('aria-pressed', 'false') + }) + + it('只是显示层:切换不许改值、不许触发保存', async () => { + const onEdit = vi.fn() + render() + + await userEvent.click(screen.getByTestId('image-key-reveal')) + + expect(onEdit).not.toHaveBeenCalled() + expect(aesInput().value).toBe(AES_KEY) + // XOR 字段不该被顺手改成密码框 —— 它本来就不是敏感值,一直是明文。 + expect(document.querySelectorAll('input[type="password"]')).toHaveLength(0) + }) + + it('不可编辑时开关也跟着禁用,避免"能看不能改"的错觉', () => { + render() + + expect(screen.getByTestId('image-key-reveal')).toBeDisabled() + expect(aesInput()).toBeDisabled() + }) +}) diff --git a/tests/component/image-text-index-card.test.tsx b/tests/component/image-text-index-card.test.tsx index 3bfdfcd..03d6679 100644 --- a/tests/component/image-text-index-card.test.tsx +++ b/tests/component/image-text-index-card.test.tsx @@ -1,5 +1,5 @@ /** - * §3 / §4:「图片文字索引」卡片的用户可见行为。 + * 「图片文字索引」卡片的用户可见行为。 * * 这些断言对应的是产品需求里**写死的**交互契约,不是实现细节: * - 未建立时先给出检测到的图片消息数量,而不是一个空洞的按钮; @@ -7,7 +7,8 @@ * - 确认弹窗要写清本机执行、原图不会因识别而自动上传、可暂停、实际可识别数量取决于本地文件; * - 进度只给真实数字(processed/total、识别出文字、没有文字、图片已清理、失败、百分比); * - 暂停 / 继续 / 取消三个动作都在,且暂停后能继续; - * - 重启后进度来自主进程快照(这里用「首帧就是 paused 快照」模拟)。 + * - 重启后进度来自主进程快照(这里用「首帧就是 paused 快照」模拟); + * - 操作区是**单列堆叠**:这张卡最多并列 3 个操作,横向排会撑破窄侧栏。 */ import { act, render, screen } from '@testing-library/react' import userEvent from '@testing-library/user-event' @@ -85,6 +86,20 @@ const paused = status({ progress: { ...running.progress, state: 'paused', cancellable: false, paused: true } }) +/** 取消:进度全部保留,但状态是 cancelled 而不是 paused。 */ +const cancelled = status({ + ...running, + progress: { ...running.progress, state: 'cancelled', cancellable: false, paused: false }, + coverage: { ...running.coverage, established: true } +}) + +/** 已建立但未完成:这张状态下操作区最多并列 3 个按钮(更新 / 重新统计 / 修复)。 */ +const establishedPartial = status({ + ...running, + progress: { ...running.progress, state: 'idle', cancellable: false }, + coverage: { ...running.coverage, established: true } +}) + const api = { getImageTextIndexStatus: vi.fn(), countImageMessages: vi.fn(), @@ -174,6 +189,37 @@ describe('图片文字索引卡片', () => { expect(api.cancelImageTextIndex).toHaveBeenCalledTimes(1) }) + it('运行中给出窗口速度与 ETA,而不是历史平均', async () => { + api.getImageTextIndexStatus.mockResolvedValue({ + ...running, + progress: { + ...running.progress, + speedPerSec: 38, + etaMs: 2 * 60 * 60 * 1000 + 25 * 60 * 1000 + } + }) + await renderCard() + + const rate = screen.getByTestId('image-text-index-rate') + expect(rate.textContent).toContain('约 38.0 张/秒') + expect(rate.textContent).toContain('2 小时 25 分') + // 用户界面不出现开发指标。 + expect(rate.textContent).not.toMatch(/p50|p95|percentile/i) + }) + + it('速度样本不足时如实说「计算中」,不编数字', async () => { + // 默认的 running 夹具没有 speedPerSec / etaMs(主进程给 null 的情形)。 + api.getImageTextIndexStatus.mockResolvedValue({ + ...running, + progress: { ...running.progress, speedPerSec: null, etaMs: null } + }) + await renderCard() + + const rate = screen.getByTestId('image-text-index-rate') + expect(rate.textContent).toContain('当前速度:计算中') + expect(rate.textContent).toContain('预计剩余:计算中') + }) + it('暂停后可以继续,进度仍来自主进程快照', async () => { api.getImageTextIndexStatus.mockResolvedValue(paused) await renderCard() @@ -183,6 +229,27 @@ describe('图片文字索引卡片', () => { expect(api.resumeImageTextIndex).toHaveBeenCalledTimes(1) }) + /** + * 「取消」之后的入口曾经是缺失的:状态落回「部分完成」、按钮只剩「更新图片文字索引」, + * 用户既看不出自己中断过,也找不到继续的地方 —— 于是以为进度丢了。 + * 取消和暂停一样保留 checkpoint,所以必须给同样的「继续」。 + */ + it('取消之后仍然能继续:状态说「已取消」,「继续」入口还在', async () => { + api.getImageTextIndexStatus.mockResolvedValue(cancelled) + await renderCard() + + expect(screen.getByTestId('image-text-index-state').textContent).toMatch(/^已取消 · /) + const resume = screen.getByTestId('image-text-index-resume') + expect(resume.textContent).toBe('继续') + // 「更新图片文字索引」和「继续」是同一件事,不能同时抢位。 + expect(screen.queryByTestId('image-text-index-start')).toBeNull() + // 已经取消了,没有东西可再取消。 + expect(screen.queryByTestId('image-text-index-cancel')).toBeNull() + + await userEvent.click(resume) + expect(api.resumeImageTextIndex).toHaveBeenCalledTimes(1) + }) + it('主进程推送真实进度后,卡片跟着更新(重启后恢复的进度同一条路径)', async () => { await renderCard() expect(screen.getByTestId('image-text-index-state').textContent).toBe('未建立') @@ -194,7 +261,7 @@ describe('图片文字索引卡片', () => { expect(screen.getByTestId('image-text-index-progress').textContent).toBe('3,842 / 12,483') }) - it('统计失败时显示「无法统计」而不是 0,并给出原因与重新统计入口', async () => { + it('统计失败时显示「无法统计」而不是 0,并给出原因', async () => { api.countImageMessages.mockResolvedValue({ totalImageMessages: 0, scannedConversations: 0, @@ -213,7 +280,48 @@ describe('图片文字索引卡片', () => { const error = screen.getByTestId('image-text-index-count-error') expect(error.textContent).toContain('读取消息分片失败') expect(error.textContent).toContain('不代表账号里没有图片') - expect(screen.getByTestId('image-text-index-recount')).toBeVisible() + // 「重新统计」入口已收掉:进度改用流水线真实走过的集合之后, + // 重算那个预估值不再影响任何东西,留着只会多一个看不懂的按钮。 + // 断言按**用户可见文案**而不是已删除的 testid —— 对已移除 testid 断言 + // 「不存在」是恒真的,删掉按钮之后它就再也测不出任何东西。 + expect(screen.queryByText('重新统计')).not.toBeInTheDocument() + }) + + /** + * 操作区布局契约:主按钮与「更多」各占一行。 + * + * 修复类操作(修复搜索索引 / 重试失败的图片)收进了「更多」菜单 —— + * 平铺出来时,用户看到的是几个都在说「索引」的按钮,只能靠猜哪个该点。 + * + * jsdom 不做真实排版,所以这里锁的是**能推出该结果的结构**: + * 两个按钮是同一个操作容器的直接子元素,且该容器带单列堆叠修饰类 + * (共享的栅格类 + `ai-search-image-index-actions`,后者把 grid 覆盖成 flex column)。 + * 一旦有人在按钮外面套一层 wrapper、或去掉修饰类,这个测试就会失败。 + */ + it('操作区只留主按钮与「更多」:同一容器的直接子元素,顺序为 更新 / 更多', async () => { + api.getImageTextIndexStatus.mockResolvedValue(establishedPartial) + api.countImageMessages.mockResolvedValue({ + totalImageMessages: 12_483, + scannedConversations: 42, + failedConversations: 1, + typeColumn: 'local_type', + durationMs: 30 + }) + await renderCard() + + const start = screen.getByTestId('image-text-index-start') + const more = screen.getByTestId('image-text-index-more') + + expect(start.textContent).toBe('更新图片文字索引') + expect(more.textContent).toBe('更多') + + const container = start.parentElement + expect(container).toBe(more.parentElement) + expect(container?.classList.contains('ai-search-knowledge-actions')).toBe(true) + expect(container?.classList.contains('ai-search-image-index-actions')).toBe(true) + + // 直接子元素 == 独占一行;顺序断言同时锁住视觉顺序。 + expect(Array.from(container?.children ?? [])).toEqual([start, more]) }) it('部分会话统计失败时给出真实数字并提示偏小', async () => { @@ -244,7 +352,8 @@ describe('图片文字索引卡片', () => { expect(screen.getByTestId('image-text-index-count').textContent).toBe('0') expect(screen.queryByTestId('image-text-index-count-error')).not.toBeInTheDocument() - expect(screen.queryByTestId('image-text-index-recount')).not.toBeInTheDocument() + // 同上:按文案断言,避免对已删除的 testid 做恒真断言。 + expect(screen.queryByText('重新统计')).not.toBeInTheDocument() }) }) @@ -286,7 +395,7 @@ describe('图片文字索引卡片 — 修复图片搜索索引', () => { } } as Partial) - it('已建立且空闲时提供修复入口,只在点击后调用主进程', async () => { + it('已建立且空闲时,修复入口收在「更多」里,点击后才调用主进程', async () => { api.getImageTextIndexStatus.mockResolvedValue(established) api.repairImageTextIndex.mockResolvedValue({ conversations: 12, @@ -296,11 +405,15 @@ describe('图片文字索引卡片 — 修复图片搜索索引', () => { }) const { onNotice } = await renderCard() - const button = screen.getByTestId('image-text-index-repair') - expect(button.textContent).toBe('修复图片搜索索引') + // 修复类操作不再平铺在操作区,必须先展开「更多」。 + expect(screen.queryByTestId('image-text-index-repair')).not.toBeInTheDocument() + await userEvent.click(screen.getByTestId('image-text-index-more')) + + const item = await screen.findByTestId('image-text-index-repair') + expect(item.textContent).toContain('修复搜索索引') expect(api.repairImageTextIndex).not.toHaveBeenCalled() - await userEvent.click(button) + await userEvent.click(item) expect(api.repairImageTextIndex).toHaveBeenCalledTimes(1) // 提示语必须讲清楚"没有重新识别",否则用户会以为又要跑几万张图。 @@ -312,6 +425,7 @@ describe('图片文字索引卡片 — 修复图片搜索索引', () => { api.getImageTextIndexStatus.mockResolvedValue(running) await renderCard() + expect(screen.queryByTestId('image-text-index-more')).not.toBeInTheDocument() expect(screen.queryByTestId('image-text-index-repair')).not.toBeInTheDocument() }) @@ -325,7 +439,8 @@ describe('图片文字索引卡片 — 修复图片搜索索引', () => { }) const { onNotice } = await renderCard() - await userEvent.click(screen.getByTestId('image-text-index-repair')) + await userEvent.click(screen.getByTestId('image-text-index-more')) + await userEvent.click(await screen.findByTestId('image-text-index-repair')) expect(String(onNotice.mock.calls.at(-1)?.[0])).toContain('正在进行中') }) diff --git a/tests/component/search-markdown-table.test.tsx b/tests/component/search-markdown-table.test.tsx new file mode 100644 index 0000000..a9108be --- /dev/null +++ b/tests/component/search-markdown-table.test.tsx @@ -0,0 +1,56 @@ +/** + * Markdown 表格的降级渲染。 + * + * 结果栏很窄,真表格在这里只会列宽错位、长字段换行难读。提示词已经禁止模型为 + * 检索结果产表格,但**历史回答**与其它入口仍可能出现,所以渲染层必须保证它 + * 不会横向炸出容器,且内容读得出来。 + * + * 这里的做法是把它降级成逐行的键值列表(每格一个 span,靠 CSS flex-wrap 换行), + * 而不是渲染 `` —— 没有 table 就不会有列宽挤压与横向溢出。 + */ +import { render, screen } from '@testing-library/react' +import { describe, expect, it } from 'vitest' +import { renderMarkdown } from '../../src/renderer/src/components/search/searchMarkdown' + +function renderValue(value: string): HTMLElement { + const { container } = render(
{renderMarkdown(value)}
) + return container +} + +describe('Markdown 表格降级渲染', () => { + it('表格行渲染成逐格的键值单元,而不是
', () => { + const container = renderValue('| 发送人 | 时间 | 图片中的文字 |') + + expect(container.querySelector('table')).toBeNull() + const row = container.querySelector('.ai-search-markdown-table-row') + expect(row).not.toBeNull() + const cells = container.querySelectorAll('.ai-search-markdown-table-cell') + expect(cells).toHaveLength(3) + expect(cells[0].textContent).toBe('发送人') + expect(cells[2].textContent).toBe('图片中的文字') + }) + + it('分隔行(|---|---|)不产生内容', () => { + const container = renderValue('|---|---|') + + expect(container.querySelectorAll('.ai-search-markdown-table-cell')).toHaveLength(0) + expect(container.querySelector('.ai-search-markdown-spacer')).not.toBeNull() + }) + + it('超长单元格文本原样保留,不截断也不丢字', () => { + const longText = '这是一段很长的识别文本'.repeat(12) + const container = renderValue(`| 用户A | ${longText} |`) + + const cells = container.querySelectorAll('.ai-search-markdown-table-cell') + expect(cells).toHaveLength(2) + expect(cells[1].textContent).toBe(longText) + }) + + it('普通段落与列表不受影响', () => { + const container = renderValue('这是一段普通说明\n\n1. 第一条\n2. 第二条') + + expect(container.querySelector('.ai-search-markdown-table-row')).toBeNull() + expect(screen.getByText('这是一段普通说明')).toBeVisible() + expect(container.querySelectorAll('.ai-search-markdown-list-item')).toHaveLength(2) + }) +}) diff --git a/tests/integration/image-text-index-p0.test.ts b/tests/integration/image-text-index-checkpoint-and-coverage.test.ts similarity index 97% rename from tests/integration/image-text-index-p0.test.ts rename to tests/integration/image-text-index-checkpoint-and-coverage.test.ts index 6f15fe4..f8fd2d4 100644 --- a/tests/integration/image-text-index-p0.test.ts +++ b/tests/integration/image-text-index-checkpoint-and-coverage.test.ts @@ -1,5 +1,5 @@ /** - * 「图片文字索引」的 P0 语义测试。 + * 「图片文字索引」的 checkpoint(增量水位)与覆盖度契约。 * * 这里覆盖的都是**不能用 UI 数字糊过去**的硬约束: * - 覆盖度必须在重启后依然诚实(派生库只知道处理过什么,不知道源数据一共多少); @@ -93,7 +93,7 @@ function makeHarness(options: { messages?: chat.FormattedMessage[] } = {}): Harn } } -describe('§2 增量水位:只比条数会漏掉「等量替换」', () => { +describe('增量水位:只比条数会漏掉「等量替换」', () => { it('水位(条数 + 最大插入序)都没变时才跳过,不读 WCDB', async () => { const harness = makeHarness({ messages: [imageMessage(10, 1000), imageMessage(20, 2000)] }) harness.watermark.count = 2 @@ -162,7 +162,7 @@ describe('§2 增量水位:只比条数会漏掉「等量替换」', () => { }) }) -describe('§1 覆盖度诚实性', () => { +describe('覆盖度诚实性', () => { it('重启后仍是 partial:分母来自落盘统计,不会退化成 processed', async () => { const { databaseRoot, databasePath } = makeHarness() // 先按「已建立过索引」写库:总数 100,实际只处理了 30 条。 @@ -227,7 +227,7 @@ describe('§1 覆盖度诚实性', () => { }) }) -describe('§5 清理:删得掉才算成功', () => { +describe('清理:删得掉才算成功', () => { it('清理后派生库文件消失,覆盖度回到未建立', async () => { const { service, databasePath } = makeHarness() // 建一份有内容的派生数据(建库 + 写 artifact/binding/水位 + 落盘总数)。 @@ -261,7 +261,7 @@ describe('§5 清理:删得掉才算成功', () => { }) }) -describe('§1/§7 查询层:覆盖度必须是独立维度且带零结果诚实性', () => { +describe('查询层:覆盖度必须是独立维度且带零结果诚实性', () => { const coverageOf = (input: Partial): ImageTextIndexCoverage => ({ totalImageMessages: 0, processed: 0, diff --git a/tests/integration/image-text-index-concurrency.test.ts b/tests/integration/image-text-index-concurrency.test.ts new file mode 100644 index 0000000..77b176e --- /dev/null +++ b/tests/integration/image-text-index-concurrency.test.ts @@ -0,0 +1,500 @@ +/** + * 图片文字索引的**并发契约**。 + * + * 背景:backfill 从「批内严格串行」改成有界流水线(prepare 同步 → OCR 有限并行 → 单 writer 落库)。 + * 并发一旦引入,下面这些性质就不再是"显然成立",必须被测试锁住: + * + * 1. 同一张图并发派发 → OCR **最多一次**(否则白算,还违反"最多识别一次"的契约); + * 2. 不同图片并发 → 结果不许串(文本 / 状态 / 身份各归各的); + * 3. 暂停 → 在途的**安全收尾**(算了不落库等于白算),但**不再领取新任务**; + * 4. 取消 → checkpoint 正确(partial),已落库的终态一条不丢; + * 5. 某个 worker 报错 → 其它图片照常完成,整体任务不崩; + * 6. 进度语义不因并发失真:`processed` 只在终态之后 +1,且不重不漏; + * 7. 已有终态在并发下**依然一次都不重算**(并发不能把复用逻辑绕过去)。 + * + * 全部使用 synthetic 图片(合法 PNG 魔数 + 唯一尾部字节),不碰任何真实数据。 + */ +import { mkdtempSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type * as chat from '../../src/main/services/chat-service' +import { ImageTextIndexService } from '../../src/main/services/image-text-index-service' +import type { ImageTextIndexStageTimings } from '../../src/shared/image-text-index' +import { + ImageTextIndexStore, + getImageTextIndexDatabasePath +} from '../../src/main/services/image-text-index-store' + +const ACCOUNT = 'wxid_concurrency_fixture' +const CONVERSATION = 'md5-concurrency' +const roots: string[] = [] + +function makeRoot(): string { + const root = mkdtempSync(join(tmpdir(), 'tm-image-concurrency-')) + roots.push(root) + return root +} + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +/** 合法 PNG 头 + 唯一尾部:不同 seed → 不同内容身份(sha256)。 */ +function pngBytes(seed: number): Buffer { + return Buffer.from([ + 0x89, + 0x50, + 0x4e, + 0x47, + 0x0d, + 0x0a, + 0x1a, + 0x0a, + seed & 0xff, + (seed >> 8) & 0xff, + 0x00 + ]) +} + +function imageMessage(localId: number): chat.FormattedMessage { + return { + id: String(localId), + localId: String(localId), + from: 'user', + type: '图片', + content: '', + isSender: false, + name: '对方', + contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` }, + createTime: 1_700_000_000 + localId + } as unknown as chat.FormattedMessage +} + +const capability = async (): Promise<{ + available: boolean + engine: 'windows-system-ocr' + platform: 'win32' + runtimeVersion: string + language: string +}> => ({ + available: true, + engine: 'windows-system-ocr', + platform: 'win32', + runtimeVersion: '1.2.0', + language: 'zh-Hans-CN' +}) + +/** 从 data URL 里还原出这张图的 seed,用来断言"结果没有串图"。 */ +function seedFromDataUrl(dataUrl: string): number { + const base64 = dataUrl.slice(dataUrl.indexOf(',') + 1) + const bytes = Buffer.from(base64, 'base64') + return bytes[8] | (bytes[9] << 8) +} + +/** + * 从假路径里取消息序号。 + * + * 刻意锚定 `.dat` 后缀:直接用 `replace(/\D/g,'')` 会连 "md5" 里那个 **5** 一起抓进来 + * (`md5-1` → "51"),这种坑只有真跑一次才会发现。 + */ +function seedFromPath(path: string): number { + const matched = /(\d+)\.dat$/.exec(String(path)) + return matched ? Number(matched[1]) : 1 +} + +const fakeDecryptImage = (path: string): Buffer => pngBytes(seedFromPath(path)) + +const fakeFindImageFile = (md5: string): string => `C:/fake/${md5}.dat` + +interface Harness { + service: ImageTextIndexService + recognize: ReturnType + databaseRoot: string + databasePath: string + close: () => void +} + +function harness(options: { + count: number + ocrConcurrency: number + /** 自定义 decrypt:默认按消息 id 给出唯一图片。 */ + decryptImage?: (path: string) => Buffer + recognizeImpl?: ( + dataUrl: string, + ctx: { callIndex: number; service: ImageTextIndexService } + ) => Promise<{ success: boolean; text: string; language: string | null; errorCode?: string }> +}): Harness { + const databaseRoot = makeRoot() + const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT) + const messages = Array.from({ length: options.count }, (_, index) => imageMessage(index + 1)) + + const service = new ImageTextIndexService() + let callIndex = 0 + const recognize = vi.fn(async (dataUrl: string) => { + callIndex += 1 + if (options.recognizeImpl) return options.recognizeImpl(dataUrl, { callIndex, service }) + return { success: true, text: `TEXT_${seedFromDataUrl(dataUrl)}`, language: null } + }) + + const decryptImage = options.decryptImage ?? fakeDecryptImage + + service.bind({ + databaseRoot, + resolveAccountId: () => ACCOUNT, + ocrConcurrency: options.ocrConcurrency, + listContacts: async () => [ + { md5: CONVERSATION, m_nsUsrName: 'concurrency', type: 'user' as const } + ], + listMessages: async () => messages, + countConversationImages: async () => ({ count: messages.length, typeColumn: 'local_type' }), + imageWatermark: async () => ({ count: messages.length, maxLocalId: messages.length }), + capability, + decryptService: () => ({ findImageFile: fakeFindImageFile, decryptImage }) as never, + recognize + }) + + return { + service, + recognize, + databaseRoot, + databasePath, + close: () => service.resetAccount() + } +} + +const finishPass = async (service: ImageTextIndexService): Promise => { + await vi.waitFor(() => expect(service.isRunning()).toBe(false)) +} + +describe('并发契约', () => { + it('同一张图被并发派发 → OCR 只执行一次,但每个消息各自有 binding', async () => { + // 8 条消息,decrypt 全部返回**同一份字节** → 同一个内容身份 / artifact key。 + const h = harness({ + count: 8, + ocrConcurrency: 4, + decryptImage: () => pngBytes(42) + }) + + await h.service.startPass() + await finishPass(h.service) + + expect(h.recognize).toHaveBeenCalledTimes(1) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + // binding 去重不成 1 条:8 个消息各自有 binding(去重不许丢来源)。 + expect(store.countByState().indexed).toBe(8) + expect(store.getConversationOcr(CONVERSATION).size).toBe(8) + // artifact 才是被去重的那一层:8 张相同内容只产生 1 个带文本的 artifact。 + expect(store.storageStats().ocrTextCount).toBe(1) + store.close() + h.close() + }) + + it('不同图片并发 → 结果各归各的,不串图', async () => { + const h = harness({ count: 16, ocrConcurrency: 4 }) + + await h.service.startPass() + await finishPass(h.service) + + expect(h.recognize).toHaveBeenCalledTimes(16) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + const ocr = store.getConversationOcr(CONVERSATION) + expect(ocr.size).toBe(16) + // 每条消息的文本必须等于**它自己那张图**的 seed —— 串图会立刻打挂这里。 + for (let id = 1; id <= 16; id += 1) { + expect(ocr.get(`local:${id}`)?.text).toBe(`TEXT_${id}`) + } + store.close() + h.close() + }) + + it('同时在途的 OCR 不超过配置的并发度(有界,不是无界扇出)', async () => { + let inFlight = 0 + let maxInFlight = 0 + const h = harness({ + count: 40, + ocrConcurrency: 2, + recognizeImpl: async () => { + inFlight += 1 + maxInFlight = Math.max(maxInFlight, inFlight) + // 让所有任务都有机会重叠:无界实现会在这里冲到 40。 + await new Promise((resolve) => setTimeout(resolve, 5)) + inFlight -= 1 + return { success: true, text: 'TEXT', language: null } + } + }) + + await h.service.startPass() + await finishPass(h.service) + + expect(h.recognize).toHaveBeenCalledTimes(40) + expect(maxInFlight).toBe(2) + + h.close() + }) + + it('并发下进度不重不漏:processed 只统计已进入终态的图片', async () => { + const h = harness({ count: 16, ocrConcurrency: 4 }) + + await h.service.startPass() + await finishPass(h.service) + + const status = await h.service.getStatus() + const { processed, indexed, empty, missing, failed, totalImageMessages } = status.progress + expect(processed).toBe(16) + expect(indexed + empty + missing + failed).toBe(processed) + expect(totalImageMessages).toBe(16) + // 并发度必须如实反映在诊断字段上。 + expect(status.stageTimings?.ocrConcurrency).toBe(4) + expect(status.stageTimings?.ocrExecutions).toBe(16) + h.close() + }) + + it('暂停 → 在途任务安全收尾,且不再领取新任务', async () => { + const CONCURRENCY = 4 + let releaseGate: (() => void) | null = null + const gate = new Promise((resolve) => { + releaseGate = () => resolve() + }) + + const h = harness({ + count: 32, + ocrConcurrency: CONCURRENCY, + recognizeImpl: async (_dataUrl, ctx) => { + // 槽位刚好填满的那一刻按暂停:此后不应再派发任何新任务。 + if (ctx.callIndex === CONCURRENCY) ctx.service.pause() + await gate + return { success: true, text: 'TRACE_PAUSED', language: null } + } + }) + + void h.service.startPass() + // 等槽位填满(4 个在途 OCR 都卡在 gate 上)。 + await vi.waitFor(() => expect(h.recognize).toHaveBeenCalledTimes(CONCURRENCY)) + + releaseGate?.() + await finishPass(h.service) + + // 只派发过这一批:暂停之后不再领取新任务。 + expect(h.recognize).toHaveBeenCalledTimes(CONCURRENCY) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + // 在途的 4 张都**落库**了(算了不落库等于白算)。 + expect(store.countByState().indexed).toBe(CONCURRENCY) + expect(store.getConversationOcr(CONVERSATION).size).toBe(CONCURRENCY) + // 中断的会话必须留 partial checkpoint,下一遍才接得上。 + expect(store.readScanState().get(CONVERSATION)?.state).toBe('partial') + store.close() + + const status = await h.service.getStatus() + expect(status.progress.state).toBe('paused') + h.close() + }) + + it('取消 → checkpoint 正确,已落库的终态一条不丢,且不重算', async () => { + const CONCURRENCY = 4 + let releaseGate: (() => void) | null = null + const gate = new Promise((resolve) => { + releaseGate = () => resolve() + }) + + const h = harness({ + count: 32, + ocrConcurrency: CONCURRENCY, + recognizeImpl: async (_dataUrl, ctx) => { + if (ctx.callIndex === CONCURRENCY) void ctx.service.cancel() + await gate + return { success: true, text: 'TRACE_CANCELLED', language: null } + } + }) + + void h.service.startPass() + await vi.waitFor(() => expect(h.recognize).toHaveBeenCalledTimes(CONCURRENCY)) + releaseGate?.() + await finishPass(h.service) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + const persisted = store.countByState().indexed + expect(persisted).toBe(CONCURRENCY) + expect(store.readScanState().get(CONVERSATION)?.state).toBe('partial') + store.close() + + const status = await h.service.getStatus() + expect(status.progress.state).toBe('cancelled') + h.close() + }) + + it('某个 worker 报错 → 其它图片照常完成,整体任务不崩', async () => { + const h = harness({ + count: 12, + ocrConcurrency: 4, + recognizeImpl: async (dataUrl) => { + const seed = seedFromDataUrl(dataUrl) + if (seed === 5) throw new Error('native OCR blew up') + return { success: true, text: `TEXT_${seed}`, language: null } + } + }) + + await h.service.startPass() + await finishPass(h.service) + + const status = await h.service.getStatus() + // 崩掉的那张记成失败,其余全部成功 —— 不是"整批失败"。 + expect(status.progress.processed).toBe(12) + expect(status.progress.indexed).toBe(11) + expect(status.progress.failed).toBe(1) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + expect(store.getConversationOcr(CONVERSATION).get('local:6')?.text).toBe('TEXT_6') + store.close() + h.close() + }) + + it('并发不绕过复用:已有 100 条终态 + 新增 20 张 → OCR 只跑 20 次', async () => { + const h = harness({ count: 120, ocrConcurrency: 4 }) + + // 预热 1..100 为终态(与生产一致的复用语义)。 + const seed = new ImageTextIndexStore(h.databasePath, ACCOUNT) + for (let index = 1; index <= 100; index += 1) { + const artifactKey = `seeded|${index}` + seed.putArtifact({ + accountId: ACCOUNT, + artifactKey, + imageIdentity: `sha256:seeded-${index}`, + state: 'indexed', + text: `TRACE_SEEDED_${index}`, + charCount: 16, + engine: 'windows-system-ocr', + platform: 'win32', + runtimeVersion: '1.2.0', + language: 'zh-Hans-CN', + createdAt: index, + updatedAt: index + }) + seed.putBinding({ + accountId: ACCOUNT, + conversationId: CONVERSATION, + messageId: `local:${index}`, + createTime: index, + imageIdentity: `sha256:seeded-${index}`, + artifactKey, + state: 'indexed', + updatedAt: index + }) + } + seed.close() + + await h.service.startPass() + await finishPass(h.service) + + // 关键断言:20 次,不是 120 次。 + expect(h.recognize).toHaveBeenCalledTimes(20) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + // 旧终态文本原样保留,一条都没被重算覆盖。 + expect(store.getArtifact('seeded|1')?.text).toBe('TRACE_SEEDED_1') + expect(store.getArtifact('seeded|100')?.text).toBe('TRACE_SEEDED_100') + store.close() + h.close() + }) + + it('按固定张数间隔写出分阶段性能画像(供无 GUI 排查)', async () => { + const profiles: ImageTextIndexStageTimings[] = [] + const h = harness({ count: 1050, ocrConcurrency: 2 }) + h.service.bind({ logStageProfile: (profile) => profiles.push(profile) }) + + await h.service.startPass() + await finishPass(h.service) + + // 1050 张 / 每 500 张一条 → 恰好 2 条(504 与 1008)。 + expect(profiles).toHaveLength(2) + + const first = profiles[0] + expect(first.ocrConcurrency).toBe(2) + // 五个阶段都必须有样本,否则"时间花在哪一段"仍然是猜的。 + for (const stage of [first.locate, first.decrypt, first.ocr, first.persist]) { + expect(stage.count).toBeGreaterThan(0) + } + expect(first.ocr.p95).toBeGreaterThanOrEqual(first.ocr.p50) + // 计数必须自洽:日志里要能直接看出进度,不必再回 UI 对数。 + expect(first.counters.processed).toBeGreaterThan(0) + expect( + first.counters.indexed + first.counters.empty + first.counters.missing + first.counters.failed + ).toBe(first.counters.processed) + // 速度字段必须在(样本不足时允许为 null,但不能缺字段)。 + expect(first).toHaveProperty('ratePerSec') + // 单张净耗时与五段之和同量级:不能把 preLoop 的一次性成本摊进来。 + expect(first.perImageMs).toBeGreaterThan(0) + expect(first.perImageMs).toBeLessThan( + first.locate.mean + + first.decrypt.mean + + first.normalize.mean + + first.ocr.mean + + first.persist.mean + + 50 + ) + // 画像里的数字是**累计**推进的,第二条必须更大 —— 否则它就不是"随时间推移的画像"。 + expect(profiles[1].ocrExecutions).toBeGreaterThan(first.ocrExecutions) + + h.close() + }) + + it('预算(messageLimit)用完 → 写 partial,不写假的 done checkpoint', async () => { + const h = harness({ count: 30, ocrConcurrency: 2 }) + + await h.service.startPass({ messageLimit: 10 }) + await finishPass(h.service) + + expect(h.recognize).toHaveBeenCalledTimes(10) + + const store = new ImageTextIndexStore(h.databasePath, ACCOUNT) + const scan = store.readScanState().get(CONVERSATION) + // 关键:不能是 done —— 否则"已完成"在说谎,下一遍要么错误跳过(永久漏索引), + // 要么整会话重扫。真实库里曾经躺着 `done + processed=20 / total=23712`。 + expect(scan?.state).toBe('partial') + expect(scan?.processed).toBe(10) + expect(scan?.imageTotal).toBe(30) + store.close() + h.close() + }) + + it('重启后不丢终态:新实例重跑同一批 → OCR 一次都不再执行', async () => { + const h = harness({ count: 24, ocrConcurrency: 4 }) + await h.service.startPass() + await finishPass(h.service) + expect(h.recognize).toHaveBeenCalledTimes(24) + + // 换一个全新的 service 实例(模拟重启),复用同一个派生库。 + const restarted = new ImageTextIndexService() + const recognize2 = vi.fn(async () => ({ + success: true, + text: 'SHOULD_NOT_RUN', + language: null + })) + restarted.bind({ + databaseRoot: h.databaseRoot, + resolveAccountId: () => ACCOUNT, + ocrConcurrency: 4, + listContacts: async () => [ + { md5: CONVERSATION, m_nsUsrName: 'concurrency', type: 'user' as const } + ], + listMessages: async () => Array.from({ length: 24 }, (_, index) => imageMessage(index + 1)), + countConversationImages: async () => ({ count: 24, typeColumn: 'local_type' }), + imageWatermark: async () => ({ count: 24, maxLocalId: 24 }), + capability, + decryptService: () => + ({ findImageFile: fakeFindImageFile, decryptImage: fakeDecryptImage }) as never, + recognize: recognize2 + }) + + await restarted.startPass() + await vi.waitFor(() => expect(restarted.isRunning()).toBe(false)) + expect(recognize2).not.toHaveBeenCalled() + restarted.resetAccount() + + h.close() + }) +}) diff --git a/tests/integration/image-text-index-incident.test.ts b/tests/integration/image-text-index-coverage-honesty.test.ts similarity index 83% rename from tests/integration/image-text-index-incident.test.ts rename to tests/integration/image-text-index-coverage-honesty.test.ts index 1b6a2a3..21a1140 100644 --- a/tests/integration/image-text-index-incident.test.ts +++ b/tests/integration/image-text-index-coverage-honesty.test.ts @@ -1,14 +1,13 @@ /** - * 事故回归:**"4.5 万张全部失败,UI 却说已建立"** 这一整套语义。 + * 覆盖度诚实性:**派生库的统计绝不能替用户宣称"已经建好了"。** * - * 真机现场(派生库实测): - * total = 45,707 / 全部 binding = decrypt_failed 45,479 / artifacts = 0 行 - * 根因是解密服务在回填时不存在(只在 db:getImage 里懒加载),每张图都在 - * `processOne` 第一步就失败。这里把"不许再发生"的四件事钉死: - * 1. 前置依赖缺失时必须**一条记录都不写**(preflight); - * 2. 处理过但一条没成功 = **异常**,不是"已建立"; - * 3. 百分比不许四舍五入到 100(45,479 / 45,707); + * 这一组覆盖四条彼此独立的硬约束: + * 1. 前置依赖缺失时必须**一条记录都不写**(否则会写出一堆假失败); + * 2. 处理过但一条都没成功 = **异常**,不是"已建立",且必须阻断 complete; + * 3. 百分比不许四舍五入到 100(99.5% 不能显示成"全部完成"); * 4. 重置失败记录**不能**动已经成功的记录。 + * + * 判据来自 `ImageTextIndexStore.countByState()` 的落盘统计,不依赖任何内存计数器。 */ import { mkdtempSync } from 'node:fs' import { rm } from 'node:fs/promises' @@ -28,12 +27,12 @@ import { type ImageTextIndexCoverage } from '../../src/shared/image-text-index' -const ACCOUNT = 'wxid_incident_fixture' -const CONVERSATION = 'md5-incident' +const ACCOUNT = 'wxid_coverage_fixture' +const CONVERSATION = 'md5-coverage' const roots: string[] = [] function makeRoot(): string { - const root = mkdtempSync(join(tmpdir(), 'tm-image-incident-')) + const root = mkdtempSync(join(tmpdir(), 'tm-image-coverage-')) roots.push(root) return root } @@ -69,13 +68,13 @@ function coverageOf(overrides: Partial): ImageTextIndexC } } -describe('事故语义:全失败不能叫"已建立"', () => { +describe('全失败不能叫"已建立"', () => { it('indexed/empty/missing 全为 0 而 failed 不为 0 → 异常,且 complete 必为 false', () => { const coverage = coverageOf({ - totalImageMessages: 45_707, - processed: 45_479, - failed: 45_479, - pending: 228, + totalImageMessages: 2000, + processed: 1990, + failed: 1990, + pending: 10, established: true, countedAt: 1_789_516_520_246, complete: false, // 服务侧已经算出 false;这里验证状态与文案 @@ -87,12 +86,12 @@ describe('事故语义:全失败不能叫"已建立"', () => { expect(describeImageTextCoverage(coverage)).not.toContain('已覆盖全部') }) - it('45,479 / 45,707 不能显示成 100%', () => { - // Math.round(45479 / 45707 * 100) === 100 —— 这正是"仅完成 100%"的来源。 - expect(Math.round((45_479 / 45_707) * 100)).toBe(100) + it('处理好绝大多数时不能四舍五入显示成 100%', () => { + // Math.round(1990 / 2000 * 100) === 100 —— 这就是"未完成却显示 100%"的来源。 + expect(Math.round((1990 / 2000) * 100)).toBe(100) // 正确口径:保留 1 位小数,未完成时封顶 99.9。 - expect(imageTextProcessedPercent(45_479, 45_707)).toBe(99.5) - expect(imageTextProcessedPercent(45_707, 45_707)).toBe(100) + expect(imageTextProcessedPercent(1990, 2000)).toBe(99.5) + expect(imageTextProcessedPercent(2000, 2000)).toBe(100) expect(imageTextProcessedPercent(0, 0)).toBe(0) }) @@ -172,7 +171,7 @@ describe('事故语义:全失败不能叫"已建立"', () => { }) }) -describe('事故防线:前置依赖缺失时一条记录都不写', () => { +describe('前置依赖缺失时一条记录都不写', () => { it('解密服务不可用 → pass 直接报错,不写任何 binding', async () => { const databaseRoot = makeRoot() const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT) @@ -181,7 +180,7 @@ describe('事故防线:前置依赖缺失时一条记录都不写', () => { databaseRoot, resolveAccountId: () => ACCOUNT, listContacts: async () => [ - { md5: CONVERSATION, m_nsUsrName: 'incident', type: 'group' as const } + { md5: CONVERSATION, m_nsUsrName: 'coverage', type: 'group' as const } ], listMessages: async () => [imageMessage(1)], countConversationImages: async () => ({ count: 1, typeColumn: 'local_type' }), @@ -193,7 +192,7 @@ describe('事故防线:前置依赖缺失时一条记录都不写', () => { runtimeVersion: '1.2.0', language: 'zh-Hans-CN' }), - // 关键:没有解密服务(本次事故的根因形态) + // 关键:解密服务缺失是**运行时**问题,不能落成每张图的"解密失败" decryptService: () => null }) @@ -201,7 +200,7 @@ describe('事故防线:前置依赖缺失时一条记录都不写', () => { await vi.waitFor(() => expect(service.isRunning()).toBe(false)) const status = await service.getStatus() - // 这一条就是整场事故的防线:宁可一次都不跑,也不要写 45,479 条假失败。 + // 判据:宁可一次都不跑,也不要写一堆假失败把派生库和 coverage 一起污染。 expect(status.progress.state).toBe('error') expect(status.progress.lastError).toContain('解密服务') expect(status.coverage.processed).toBe(0) @@ -215,7 +214,7 @@ describe('事故防线:前置依赖缺失时一条记录都不写', () => { }) }) -describe('事故收尾:重置失败记录不能动成功记录', () => { +describe('重置失败记录不能动成功记录', () => { it('只删失败绑定与它们的 checkpoint,indexed 一条不动', async () => { const databaseRoot = makeRoot() const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT) diff --git a/tests/integration/image-text-index-image-query.test.ts b/tests/integration/image-text-index-image-query.test.ts new file mode 100644 index 0000000..fe75500 --- /dev/null +++ b/tests/integration/image-text-index-image-query.test.ts @@ -0,0 +1,190 @@ +/** + * 图片索引的**数据边界**契约。 + * + * 历史问题:图片索引为了找图片,先读整个会话(十几万到二十几万条消息)再在 JS 里筛。 + * 大会话实测单次读取 15s 以上,而那些行 99% 以上是图片索引根本不看的文本消息 —— + * 这是数据边界错了,不是 OCR 慢。 + * + * 这一组锁三件事: + * 1. 接了专用查询就必须走它,**不能再回退到全量读取**; + * 2. 专用查询产出的 `FormattedMessage` 与全量路径**逐字段同构**(否则 artifact / + * binding / checkpoint 的键会变); + * 3. 专用路径仍然按图片类型过滤(召回归档可能补进非图片的撤回消息)。 + */ +import { mkdtempSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type * as chat from '../../src/main/services/chat-service' +import { ImageTextIndexService } from '../../src/main/services/image-text-index-service' +import { + ImageTextIndexStore, + getImageTextIndexDatabasePath +} from '../../src/main/services/image-text-index-store' + +const ACCOUNT = 'wxid_boundary_fixture' +const CONVERSATION = 'md5-boundary' +const TEXT_COUNT = 40 +const IMAGE_COUNT = 12 +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +function pngBytes(seed: number): Buffer { + return Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, seed & 0xff]) +} + +/** 一半文本、一半图片:只有真图片会被索引。 */ +function mixedMessages(): chat.FormattedMessage[] { + const messages: chat.FormattedMessage[] = [] + for (let index = 0; index < TEXT_COUNT; index += 1) { + messages.push({ + id: `text-${index}`, + localId: String(index + 1), + from: 'user', + type: '文本', + content: `第 ${index} 条文本`, + isSender: false, + name: '对方', + contentData: { type: 'text', text: `第 ${index} 条文本` }, + createTime: 1_700_000_000 + index + } as unknown as chat.FormattedMessage) + } + for (let index = 0; index < IMAGE_COUNT; index += 1) { + messages.push({ + id: `image-${index}`, + localId: String(TEXT_COUNT + index + 1), + from: 'user', + type: '图片', + content: '', + isSender: false, + name: '对方', + contentData: { + type: 'image', + md5: `imgmd5-${index}`, + datName: `imgdat-${index}` + }, + createTime: 1_700_000_000 + TEXT_COUNT + index + } as unknown as chat.FormattedMessage) + } + return messages +} + +interface Harness { + service: ImageTextIndexService + databasePath: string + listMessages: ReturnType + listImageMessages: ReturnType +} + +function createHarness(): Harness { + const databaseRoot = mkdtempSync(join(tmpdir(), 'tm-boundary-')) + roots.push(databaseRoot) + const all = mixedMessages() + const imagesOnly = all.filter((message) => message.contentData?.type === 'image') + + const listMessages = vi.fn(async () => all) + const listImageMessages = vi.fn(async () => imagesOnly) + + const service = new ImageTextIndexService() + service.bind({ + databaseRoot, + progressNotifyIntervalMs: 0, + resolveAccountId: () => ACCOUNT, + ocrConcurrency: 1, + listContacts: async () => [ + { md5: CONVERSATION, m_nsUsrName: 'boundary', type: 'user' as const } + ], + countConversationImages: async () => ({ count: IMAGE_COUNT, typeColumn: 'local_type' }), + imageWatermark: async () => ({ count: IMAGE_COUNT, maxLocalId: TEXT_COUNT + IMAGE_COUNT }), + listMessages, + listImageMessages, + capability: async () => ({ + available: true, + engine: 'macos-system-ocr', + platform: 'darwin', + runtimeVersion: '1.2.0', + language: null + }), + decryptService: () => + ({ + findImageFile: (md5: string) => `C:/fake/${md5}.dat`, + decryptImage: (path: string) => pngBytes(Number(/(\d+)\.dat$/.exec(String(path))?.[1] ?? 0)) + }) as never, + recognize: async () => ({ success: true, text: 'TEXT', language: null }) + }) + + return { + service, + databasePath: getImageTextIndexDatabasePath(databaseRoot, ACCOUNT), + listMessages, + listImageMessages + } +} + +describe('图片索引的数据边界', () => { + it('uses the image-only query and never falls back to the full conversation read', async () => { + const h = createHarness() + + await h.service.startPass() + await vi.waitFor(() => expect(h.service.isRunning()).toBe(false), { timeout: 60_000 }) + + // 专用查询被调用;全量读取**一次都不许发生**。 + expect(h.listImageMessages).toHaveBeenCalledTimes(1) + expect(h.listMessages).not.toHaveBeenCalled() + // 拿到的行数必须只与图片有关,而不是整个会话。 + expect(h.listImageMessages.mock.calls[0][0]).toBe(CONVERSATION) + + const status = await h.service.getStatus() + expect(status.progress.processed).toBe(IMAGE_COUNT) + + h.service.resetAccount() + }) + + it('produces the same bindings as the full-conversation path', async () => { + const h = createHarness() + const full = createHarness() + // 对照:拿掉专用查询,强制走老的"全量读 + JS 筛"。 + full.service.bind({ listImageMessages: undefined }) + + await h.service.startPass() + await vi.waitFor(() => expect(h.service.isRunning()).toBe(false), { timeout: 60_000 }) + await full.service.startPass() + await vi.waitFor(() => expect(full.service.isRunning()).toBe(false), { timeout: 60_000 }) + + expect(h.listMessages).not.toHaveBeenCalled() + expect(full.listMessages).toHaveBeenCalledTimes(1) + + const readBindings = (databasePath: string): Array> => { + const store = new ImageTextIndexStore(databasePath, ACCOUNT) + try { + return store.getConversationOcr(CONVERSATION).size > 0 + ? [...store.getConversationOcr(CONVERSATION).entries()].map(([messageId, entry]) => ({ + messageId, + state: entry.state, + text: entry.text + })) + : [] + } finally { + store.close() + } + } + + const withImageQuery = readBindings(h.databasePath) + const withFullRead = readBindings(full.databasePath) + + expect(withImageQuery.length).toBe(IMAGE_COUNT) + /** + * 这条是本用例的核心:两条路径产出的**绑定键与结果**必须逐条相同。 + * 一旦有人改了图片专用查询里的字段映射,`messageId` 会变、键会变, + * 已有的 artifact / checkpoint 就会全部失效 —— 那正是不会报错但很贵的回归。 + */ + expect(withImageQuery).toEqual(withFullRead) + + h.service.resetAccount() + full.service.resetAccount() + }) +}) diff --git a/tests/integration/image-text-index-knowledge-invalidation.test.ts b/tests/integration/image-text-index-knowledge-invalidation.test.ts index da576a0..45e7157 100644 --- a/tests/integration/image-text-index-knowledge-invalidation.test.ts +++ b/tests/integration/image-text-index-knowledge-invalidation.test.ts @@ -1,5 +1,5 @@ /** - * §2 / §3 的硬条件:清理图片文字索引必须让 **Knowledge 里已经产生的 OCR 派生文字**一起失效。 + * 硬条件:清理图片文字索引必须让 **Knowledge 里已经产生的 OCR 派生文字**一起失效。 * * 背景:OCR 文本经 normalizer 的固定前缀 `图片文字:` 拼进 `searchableText`, * 再进 chunks / FTS。所以"清理成功"不能只等于"派生 SQLite 删掉了" —— @@ -96,9 +96,7 @@ function imageMessageWithoutOcr(caption?: string): KnowledgeSourceMessage { } function searchTokens(store: KnowledgeStore, text: string): string[] { - return store - .search({ accountId: ACCOUNT, text, limit: 20 }) - .map((item) => item.messageId) + return store.search({ accountId: ACCOUNT, text, limit: 20 }).map((item) => item.messageId) } function evidenceFor(store: KnowledgeStore, text: string) { @@ -115,7 +113,7 @@ async function indexConversation( }) } -describe('§2-A Knowledge 侧的失效机制:OCR 派生文字必须能真的消失', () => { +describe('Knowledge 侧的失效机制:OCR 派生文字必须能真的消失', () => { it('图片消息仍然存在、只是 OCR 文本没了 → 旧 OCR 文字搜不到,普通文字不受影响', async () => { const store = new KnowledgeStore(makeRoot(), ACCOUNT, fts) @@ -155,7 +153,7 @@ describe('§2-A Knowledge 侧的失效机制:OCR 派生文字必须能真的 store.close() }) - it('§3:OCR 文本变化(state 仍是 indexed)也必须让旧文本失效', async () => { + it('OCR 文本变化(state 仍是 indexed)也必须让旧文本失效', async () => { const store = new KnowledgeStore(makeRoot(), ACCOUNT, fts) await indexConversation(store, [textMessage(), imageMessageWithOcr()]) @@ -175,7 +173,7 @@ describe('§2-A Knowledge 侧的失效机制:OCR 派生文字必须能真的 }) }) -describe('§2-B 生产路径:清理必须逐会话重建 Knowledge', () => { +describe('生产路径:清理必须逐会话重建 Knowledge', () => { function imageMessage(localId: number, conversationId: string): chat.FormattedMessage { return { localId: String(localId), @@ -215,9 +213,14 @@ describe('§2-B 生产路径:清理必须逐会话重建 Knowledge', () => { await service.startPass() await vi.waitFor(() => expect(service.isRunning()).toBe(false)) - // 两个会话都真的产生了绑定。 + /** + * 这个 fixture **刻意没有解密服务** ⇒ 所有图片都落成 `image_missing`,没有一条 + * 可搜索的 OCR 文字。所以本遍**不应该**叫醒 Knowledge: + * 索引侧没有可搜索内容变化,重建纯属白读一遍 WCDB。 + * ("有文字 ⇒ 必须重建"由 image-text-index-store-cache 的门控用例覆盖。) + */ const databasePath = getImageTextIndexDatabasePath(databaseRoot, ACCOUNT) - expect(onConversationIndexed).toHaveBeenCalledTimes(2) + expect(onConversationIndexed).toHaveBeenCalledTimes(0) onConversationIndexed.mockClear() const result = await service.clear() diff --git a/tests/integration/image-text-index-progress-notify.test.ts b/tests/integration/image-text-index-progress-notify.test.ts new file mode 100644 index 0000000..008be9a --- /dev/null +++ b/tests/integration/image-text-index-progress-notify.test.ts @@ -0,0 +1,240 @@ +/** + * 图片文字索引的**进度通知节流**契约。 + * + * 背景:后台按 `BATCH_SIZE = 12` 推进,但 UI 不该感知 batch 大小 —— + * 每批都推会让计数以「+12」的粒度跳动。 + * + * 硬要求: + * 1. 正常运行态最多每 `progressNotifyIntervalMs` 推一次**最新权威快照**; + * 2. 状态变化(开始/暂停/继续/取消/完成/失败)**立即**推,不等窗口; + * 3. 完成必须立即给出最终值; + * 4. 同时最多一个 timer,pass 结束后不留残留定时器; + * 5. 推的是快照,不是把窗口内几十个 delta 重放给 Renderer。 + * + * 为了避免时序脆弱的测试,最关键的一条用**把窗口设得极大**来表达: + * 如果节流正确,整遍 pass 里只应有「开始」和「完成」两次状态变化通知。 + */ +import { mkdtempSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type * as chat from '../../src/main/services/chat-service' +import { ImageTextIndexService } from '../../src/main/services/image-text-index-service' +import type { ImageTextIndexStatus } from '../../src/shared/image-text-index' + +const ACCOUNT = 'wxid_progress_notify_fixture' +const CONVERSATION = 'md5-progress-notify' +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +function makeRoot(): string { + const root = mkdtempSync(join(tmpdir(), 'tm-progress-notify-')) + roots.push(root) + return root +} + +function pngBytes(seed: number): Buffer { + return Buffer.from([ + 0x89, + 0x50, + 0x4e, + 0x47, + 0x0d, + 0x0a, + 0x1a, + 0x0a, + seed & 0xff, + (seed >> 8) & 0xff + ]) +} + +function imageMessage(localId: number): chat.FormattedMessage { + return { + id: String(localId), + localId: String(localId), + from: 'user', + type: '图片', + content: '', + isSender: false, + name: '对方', + contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` }, + createTime: 1_700_000_000 + localId + } as unknown as chat.FormattedMessage +} + +const sleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)) + +/** 每张图 sleep 一点,让 pass 有可观测的持续时间。 */ +function createService(options: { count: number; notifyIntervalMs: number; perImageMs?: number }): { + service: ImageTextIndexService + notifications: Array<{ at: number; processed: number; state: string }> +} { + const databaseRoot = makeRoot() + const messages = Array.from({ length: options.count }, (_, index) => imageMessage(index + 1)) + const perImageMs = options.perImageMs ?? 0 + + const service = new ImageTextIndexService() + service.bind({ + databaseRoot, + progressNotifyIntervalMs: options.notifyIntervalMs, + resolveAccountId: () => ACCOUNT, + ocrConcurrency: 2, + listContacts: async () => [{ md5: CONVERSATION, m_nsUsrName: 'notify', type: 'user' as const }], + listMessages: async () => messages, + countConversationImages: async () => ({ count: messages.length, typeColumn: 'local_type' }), + imageWatermark: async () => ({ count: messages.length, maxLocalId: messages.length }), + capability: async () => ({ + available: true, + engine: 'windows-system-ocr', + platform: 'win32', + runtimeVersion: '1.2.0', + language: 'zh-Hans-CN' + }), + decryptService: () => + ({ + findImageFile: (md5: string) => `C:/fake/${md5}.dat`, + decryptImage: (path: string) => pngBytes(Number(/(\d+)\.dat$/.exec(String(path))?.[1] ?? 1)) + }) as never, + recognize: async () => { + if (perImageMs > 0) await sleep(perImageMs) + return { success: true, text: 'TEXT', language: null } + } + }) + + const notifications: Array<{ at: number; processed: number; state: string }> = [] + service.onStatusChange((status: ImageTextIndexStatus) => { + notifications.push({ + at: Date.now(), + processed: status.progress.processed, + state: status.progress.state + }) + }) + + return { service, notifications } +} + +const finish = async (service: ImageTextIndexService): Promise => { + await vi.waitFor(() => expect(service.isRunning()).toBe(false)) +} + +describe('进度通知节流', () => { + it('窗口极大时,整遍 pass 只推「开始」和「完成」两次状态变化', async () => { + // 240 张 = 20 个 batch。若每批都推会有 20+ 次通知;节流正确则只有状态变化。 + const { service, notifications } = createService({ + count: 240, + notifyIntervalMs: 100_000, + perImageMs: 2 + }) + + await service.startPass() + await finish(service) + + expect(notifications.map((n) => n.state)).toEqual(['running', 'completed']) + // 完成必须带**最终值**,不能等下一次 5 秒 timer。 + expect(notifications[notifications.length - 1].processed).toBe(240) + service.resetAccount() + }) + + it('窗口为 0 时不做节流(每次都推最新快照)', async () => { + const { service, notifications } = createService({ + count: 240, + notifyIntervalMs: 0, + perImageMs: 2 + }) + + await service.startPass() + await finish(service) + + // 不节流 ⇒ 通知数应远多于「仅两次状态变化」,且处理量单调不减。 + expect(notifications.length).toBeGreaterThan(4) + const processed = notifications.map((n) => n.processed) + expect([...processed].sort((a, b) => a - b)).toEqual(processed) + expect(processed[processed.length - 1]).toBe(240) + service.resetAccount() + }) + + it('暂停立即推送,不等窗口', async () => { + const { service, notifications } = createService({ + count: 600, + notifyIntervalMs: 100_000, + perImageMs: 4 + }) + + void service.startPass() + // 注意:窗口是 100 秒,**进度通知按设计不会来**,所以不能用通知当等待条件。 + await vi.waitFor(async () => { + const status = await service.getStatus() + expect(status.progress.processed).toBeGreaterThan(0) + }) + service.pause() + await finish(service) + + const pausedAt = notifications.findIndex((n) => n.state === 'paused') + expect(pausedAt).toBeGreaterThanOrEqual(0) + // 暂停之后不应再有「运行中」的进度推送(窗口是 100 秒,等不到)。 + expect(notifications.slice(pausedAt + 1).some((n) => n.state === 'running')).toBe(false) + service.resetAccount() + }) + + it('pass 结束后不留残留 timer:不再产生额外通知', async () => { + const { service, notifications } = createService({ + count: 120, + notifyIntervalMs: 30, + perImageMs: 1 + }) + + await service.startPass() + await finish(service) + const afterFinish = notifications.length + // 窗口只有 30ms,如果尾随 timer 没被清掉,这段时间里一定会再冒出通知。 + await sleep(200) + expect(notifications.length).toBe(afterFinish) + expect(notifications[notifications.length - 1].state).toBe('completed') + service.resetAccount() + }) + + it('continue(重新 startPass)不会叠加出第二个 timer', async () => { + const { service, notifications } = createService({ + count: 240, + notifyIntervalMs: 30, + perImageMs: 1 + }) + + void service.startPass() + await vi.waitFor(async () => { + const status = await service.getStatus() + expect(status.progress.processed).toBeGreaterThan(0) + }) + service.pause() + await finish(service) + + const pausedCount = notifications.length + await service.resume() + await finish(service) + await sleep(200) + + // 恢复后应重新开始推送,但不应因"两个 interval 并存"而翻倍: + // 第一遍以 paused 收尾、第二遍以 completed 收尾 ⇒ completed 恰好 1 次。 + expect(notifications.length).toBeGreaterThan(pausedCount) + const completed = notifications.filter((n) => n.state === 'completed') + expect(completed).toHaveLength(1) + expect(notifications.filter((n) => n.state === 'paused')).toHaveLength(1) + service.resetAccount() + }) + + it('速度与 ETA 只在窗口样本足够时给出,否则为 null', async () => { + const { service } = createService({ count: 120, notifyIntervalMs: 0, perImageMs: 1 }) + await service.startPass() + await finish(service) + + const status = await service.getStatus() + // 这遍跑得太快,窗口跨度不足 ⇒ 必须如实为 null(UI 显示"计算中"),不许编数。 + expect(status.progress.speedPerSec).toBeNull() + expect(status.progress.etaMs).toBeNull() + service.resetAccount() + }) +}) diff --git a/tests/integration/image-text-index-store-cache.test.ts b/tests/integration/image-text-index-store-cache.test.ts new file mode 100644 index 0000000..ef643eb --- /dev/null +++ b/tests/integration/image-text-index-store-cache.test.ts @@ -0,0 +1,572 @@ +/** + * 派生库句柄的**账号身份缓存契约**。 + * + * `resolveAccountId()` 不是廉价 getter:main 把它绑定成同步 WCDB 查询。 + * 而 `ensureStore()` 在图片处理热路径上会被每张图片调用多次,所以身份解析 + * **必须**只发生常数次;否则它就成了每张图片的固定成本,且完全不随 OCR 并发改善。 + * + * 这一组用例把契约钉住: + * 1. 身份解析只发生常数次(不是每张图片一次); + * 2. 解析本身再慢,也只能让整遍多付一次; + * 3. 切账号仍然换库 —— `resetAccount()` 是权威的失效信号; + * 4. 解析失败(空结果)不被永久缓存; + * 5. 进入流水线之前的一次性成本不算进单张净耗时。 + */ +import { existsSync, mkdtempSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { DatabaseSync } from 'node:sqlite' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type * as chat from '../../src/main/services/chat-service' +import { ImageTextIndexService } from '../../src/main/services/image-text-index-service' +import { getImageTextIndexDatabasePath } from '../../src/main/services/image-text-index-store' + +const ACCOUNT = 'wxid_store_cache_fixture' +const IMAGES = 24 +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +const sleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)) + +/** 同步阻塞:模拟同步 WCDB 查询(不是 await,是实打实占住主线程)。 */ +function blockFor(ms: number): void { + const end = Date.now() + ms + while (Date.now() < end) { + /* busy */ + } +} + +/** 合法 PNG 头 + 唯一尾部:不同 seed → 不同内容身份(sha256)。 */ +function pngBytes(seed: number): Buffer { + return Buffer.from([ + 0x89, + 0x50, + 0x4e, + 0x47, + 0x0d, + 0x0a, + 0x1a, + 0x0a, + seed & 0xff, + (seed >> 8) & 0xff + ]) +} + +function imageMessage(localId: number): chat.FormattedMessage { + return { + id: String(localId), + localId: String(localId), + from: 'user', + type: '图片', + content: '', + isSender: false, + name: '对方', + contentData: { type: 'image', md5: `md5-${localId}`, datName: `dat-${localId}` }, + createTime: 1_700_000_000 + localId + } as unknown as chat.FormattedMessage +} + +const seedFromPath = (path: string): number => Number(/(\d+)\.dat$/.exec(String(path))?.[1] ?? 1) + +interface Harness { + service: ImageTextIndexService + databaseRoot: string + /** 切换当前会话:让下一遍不被 checkpoint 跳过,用来验证"换库后重新处理"。 */ + setConversation: (conversationId: string) => void + bindingCount: (accountId: string) => number +} + +function createHarness(options: { + resolveAccountId: () => string + /** 注入到"每个会话一次"的前置路径上(countConversationImages / listMessages)。 */ + preLoopDelayMs?: number + /** 注入到会话完成后的 Knowledge 重建回调(属于别的模块的成本)。 */ + onConversationIndexedDelayMs?: number + /** OCR 返回的文字;空串 = 识别成"没有文字"(非可搜索结果)。 */ + recognizeText?: string + ocrMs?: number +}): Harness { + const databaseRoot = mkdtempSync(join(tmpdir(), 'tm-store-cache-')) + roots.push(databaseRoot) + const messages = Array.from({ length: IMAGES }, (_, index) => imageMessage(index + 1)) + let conversation = 'conv-initial' + + const service = new ImageTextIndexService() + service.bind({ + databaseRoot, + progressNotifyIntervalMs: 0, + resolveAccountId: options.resolveAccountId, + ocrConcurrency: 1, + listContacts: async () => [ + { md5: conversation, m_nsUsrName: conversation, type: 'user' as const } + ], + countConversationImages: async () => { + if (options.preLoopDelayMs) await sleep(options.preLoopDelayMs) + return { count: messages.length, typeColumn: 'local_type' } + }, + imageWatermark: async () => ({ count: messages.length, maxLocalId: messages.length }), + listMessages: async () => { + if (options.preLoopDelayMs) await sleep(options.preLoopDelayMs) + return messages + }, + capability: async () => ({ + available: true, + engine: 'macos-system-ocr', + platform: 'darwin', + runtimeVersion: '1.2.0', + language: null + }), + decryptService: () => + ({ + findImageFile: (md5: string) => `C:/fake/${md5}.dat`, + decryptImage: (path: string) => pngBytes(seedFromPath(path)) + }) as never, + recognize: async () => { + if (options.ocrMs) await sleep(options.ocrMs) + return { success: true, text: options.recognizeText ?? 'TEXT', language: null } + }, + onConversationIndexed: async () => { + if (options.onConversationIndexedDelayMs) await sleep(options.onConversationIndexedDelayMs) + } + }) + + return { + service, + databaseRoot, + setConversation: (conversationId) => { + conversation = conversationId + }, + bindingCount: (accountId) => { + const path = getImageTextIndexDatabasePath(databaseRoot, accountId) + if (!existsSync(path)) return 0 + const db = new DatabaseSync(path) + try { + const row = db.prepare('SELECT COUNT(*) AS n FROM image_ocr_bindings').get() as { + n: number + } + return Number(row?.n ?? 0) + } finally { + db.close() + } + } + } +} + +const runPass = async ( + service: ImageTextIndexService, + options: { sinceMs?: number } = {} +): Promise => { + await service.startPass(options) + await vi.waitFor(() => expect(service.isRunning()).toBe(false), { timeout: 60_000 }) +} + +describe('派生库句柄的账号身份缓存', () => { + it('caches account identity until reset', async () => { + let resolveCalls = 0 + const h = createHarness({ + resolveAccountId: () => { + resolveCalls += 1 + blockFor(30) + return ACCOUNT + }, + ocrMs: 5 + }) + + await runPass(h.service) + + expect((await h.service.getStatus()).progress.processed).toBe(IMAGES) + console.log(`[store-cache] 张数=${IMAGES} resolveAccountId 调用=${resolveCalls}`) + // 每张图片都要重新解析的话,这里会是 2 × IMAGES 的量级。 + expect(resolveCalls).toBeLessThanOrEqual(3) + + h.service.resetAccount() + }) + + it('identity resolution cost is paid once per pass, not per image', async () => { + const RESOLVE_MS = 30 + const OCR_MS = 5 + const slow = createHarness({ + resolveAccountId: () => { + blockFor(RESOLVE_MS) + return ACCOUNT + }, + ocrMs: OCR_MS + }) + const fast = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: OCR_MS }) + + await runPass(fast.service) + await runPass(slow.service) + + const fastPerImage = (await fast.service.getStatus()).stageTimings?.perImageMs ?? 0 + const slowPerImage = (await slow.service.getStatus()).stageTimings?.perImageMs ?? 0 + + /** + * 用**差值**断言而不是绝对阈值,避免锁住某台机器的速度: + * 身份解析变慢 30ms,若只付一次,单张净耗时最多只涨这一点点; + * 若每张都要付(甚至两次),差值会是 30ms 的倍数。 + */ + expect(slowPerImage - fastPerImage).toBeLessThan(RESOLVE_MS + 70) + + fast.service.resetAccount() + slow.service.resetAccount() + }) + + it('invalidates store cache on reset', async () => { + let account = 'account-A' + const h = createHarness({ resolveAccountId: () => account }) + + await runPass(h.service) + expect(h.bindingCount('account-A')).toBe(IMAGES) + expect(h.bindingCount('account-B')).toBe(0) + + // 切账号:换身份 + 显式失效(main 在全部切换路径上都会这样做)。 + account = 'account-B' + h.setConversation('conv-B') + h.service.resetAccount() + await runPass(h.service) + + expect(h.bindingCount('account-B')).toBe(IMAGES) + // A 的库不得被继续写入 —— 这就是"切账号必须换句柄"要防的串账号。 + expect(h.bindingCount('account-A')).toBe(IMAGES) + + h.service.resetAccount() + }) + + it('does not cache unresolved account identity', async () => { + let account = '' + let resolveCalls = 0 + const h = createHarness({ + resolveAccountId: () => { + resolveCalls += 1 + return account + } + }) + + // 微信还没就绪:解析结果为空 → 不建库、也不该把空结果记成"已解析"。 + await runPass(h.service) + expect((await h.service.getStatus()).progress.state).toBe('error') + const callsWhileUnresolved = resolveCalls + expect(callsWhileUnresolved).toBeGreaterThan(0) + + // 数据就绪之后再跑:必须能重新解析出来(空结果没有被永久缓存)。 + account = ACCOUNT + h.service.resetAccount() + await runPass(h.service) + expect((await h.service.getStatus()).progress.processed).toBe(IMAGES) + expect(resolveCalls).toBeGreaterThan(callsWhileUnresolved) + + h.service.resetAccount() + }) + + it('pre-loop cost is excluded from per-image cost', async () => { + const PRE_LOOP_MS = 400 + const withPreLoop = createHarness({ + resolveAccountId: () => ACCOUNT, + preLoopDelayMs: PRE_LOOP_MS, + ocrMs: 5 + }) + const control = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 5 }) + + const wallStartedAt = Date.now() + await runPass(withPreLoop.service) + const wallMs = Date.now() - wallStartedAt + await runPass(control.service) + + const status = await withPreLoop.service.getStatus() + const timings = status.stageTimings + expect(timings).toBeDefined() + if (!timings) return + const processed = status.progress.processed + expect(processed).toBe(IMAGES) + + const controlPerImage = (await control.service.getStatus()).stageTimings?.perImageMs ?? 0 + console.log( + `[store-cache] 整遍 wall=${wallMs}ms 整遍/张=${(wallMs / processed).toFixed(1)}ms ` + + `单张净=${timings.perImageMs}ms(对照 ${controlPerImage}ms)startup=${timings.preLoop.startupMs}ms` + ) + + // 前置成本必须被单独量出来(本用例在 countConversationImages 与 listMessages 各注入一次)。 + expect(timings.preLoop.startupMs).toBeGreaterThanOrEqual(PRE_LOOP_MS * 2 - 100) + expect(timings.preLoop.countImageMessagesMs).toBeGreaterThanOrEqual(PRE_LOOP_MS - 100) + expect(timings.preLoop.listMessagesMs).toBeGreaterThanOrEqual(PRE_LOOP_MS - 100) + // 会话级准备必须覆盖 listMessages,否则"每会话一次"的成本会漏出去。 + expect(timings.preLoop.conversationSetupMs).toBeGreaterThanOrEqual( + timings.preLoop.listMessagesMs + ) + + /** + * 单张净耗时**不能**因为前置成本变贵而变贵 —— 用对照实例做差值断言, + * 这样既锁住了语义,又不会锁住某台机器的速度。 + */ + expect(timings.perImageMs).toBeLessThan(controlPerImage + 60) + // 反过来,"整遍 ÷ 张数"必然被前置成本抬高 —— 这正是它不能当成单张成本的原因。 + expect(wallMs / processed).toBeGreaterThan(timings.perImageMs * 5) + + withPreLoop.service.resetAccount() + control.service.resetAccount() + }) +}) + +/** + * 会话完成后等待 Knowledge 重建的成本(`onConversationIndexed`)。 + * + * 这一段是**别的模块**的成本:Knowledge 会对同一个会话再全量读一遍消息并整篇写索引, + * 而且如果此时有索引在跑还会先等它。它不在 batch 循环里。 + * + * 这一组锁两件容易同时搞砸的事: + * 1. 它必须被**单独量出来**(否则"单张很快、整遍很慢"无从归因); + * 2. 它**不能**被算进单张净耗时(否则单张数字会被别的模块污染)。 + */ +describe('会话级 Knowledge 重建成本的归因', () => { + it('attributes the Knowledge callback to preLoop, not to per-image cost', async () => { + const INDEXED_MS = 300 + const slow = createHarness({ + resolveAccountId: () => ACCOUNT, + onConversationIndexedDelayMs: INDEXED_MS, + ocrMs: 5 + }) + const control = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 5 }) + + await runPass(control.service) + await runPass(slow.service) + + const timings = (await slow.service.getStatus()).stageTimings + expect(timings).toBeDefined() + if (!timings) return + const controlPerImage = (await control.service.getStatus()).stageTimings?.perImageMs ?? 0 + + // 必须被单独量出来:本用例只跑了一个会话,回调延迟应完整落在该桶里。 + expect(timings.preLoop.onConversationIndexedMs).toBeGreaterThanOrEqual(INDEXED_MS - 60) + console.log( + `[store-cache] onConversationIndexedMs=${timings.preLoop.onConversationIndexedMs}ms ` + + `单张净=${timings.perImageMs}ms(对照 ${controlPerImage}ms)` + ) + + /** + * 用差值断言:回调再慢,单张净耗时也不该跟着涨。 + * 若有人把这段并进 `perImageMs`,这里会立刻红 —— 那正是"图片索引变慢"被误判成 + * "OCR 变慢"的起因。 + */ + expect(timings.perImageMs).toBeLessThan(controlPerImage + 60) + + slow.service.resetAccount() + control.service.resetAccount() + }) +}) + +/** + * 可观测性契约:**慢步骤必须自己说出"卡在哪一步"**。 + * + * 背景:实测出现过"一次运行 81 分钟零落库"。没有这条日志时只能看到"没有进度", + * 无法区分"自己慢"和"被别的模块按住" —— 而这两者的修法完全不同。 + */ +describe('慢步骤告警', () => { + it('reports which step is blocking when it exceeds the threshold', async () => { + const h = createHarness({ + resolveAccountId: () => ACCOUNT, + onConversationIndexedDelayMs: 400 + }) + // 自己接管这一步:先证明它真的被调用了,再断言告警。 + const indexed = vi.fn(async () => { + await sleep(400) + }) + h.service.bind({ onConversationIndexed: indexed, slowStepWarnMs: 100 }) + + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + await runPass(h.service) + // 前提:这一步确实被调用了 —— 否则下面找不到告警的原因是错的。 + expect(indexed).toHaveBeenCalled() + + const lines = warn.mock.calls + .map((call) => String(call[0] ?? '')) + .filter((message) => message.includes('[ImageTextIndex] slow step=')) + + // 必须报出被按住的那一步,并且带耗时。 + expect(lines.find((line) => line.includes('knowledge-index'))).toBeDefined() + expect(lines.some((line) => /elapsedMs=\d+/.test(line))).toBe(true) + // 只报步骤名与耗时,不得出现会话标识 / 内容。 + expect(lines.join(' ')).not.toContain('conv-initial') + } finally { + warn.mockRestore() + } + + h.service.resetAccount() + }) + + it('stays quiet when every step is fast', async () => { + const h = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 1 }) + h.service.bind({ slowStepWarnMs: 100_000 }) + + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + await runPass(h.service) + expect( + warn.mock.calls + .map((call) => String(call[0] ?? '')) + .filter((message) => message.includes('slow step=')) + ).toEqual([]) + } finally { + warn.mockRestore() + } + + h.service.resetAccount() + }) +}) + +/** + * 画像触发的时间兜底。 + * + * 只按处理量触发时,吞吐掉到个位数会让画像十几分钟才出一条 —— + * 而那正是最需要画像的时刻。 + */ +describe('画像的时间兜底触发', () => { + it('emits a profile after the idle window even when the image count is far below the interval', async () => { + const h = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 30 }) + const profiles: unknown[] = [] + h.service.bind({ + logStageProfile: (profile) => profiles.push(profile), + // 张数阈值仍是 500(默认),本用例只有 IMAGES 张 ⇒ 只能靠时间触发。 + stageProfileMaxIdleMs: 1 + }) + + await runPass(h.service) + + expect(IMAGES).toBeLessThan(500) + expect(profiles.length).toBeGreaterThan(0) + + h.service.resetAccount() + }) +}) + +/** + * Knowledge 重建的门控:**没有可搜索内容变化就不要叫醒它。** + * + * 为什么这是硬要求:Knowledge 侧是"读整个会话 → 重分片整会话",代价与消息数成正比, + * 而且这一步在 `await` 路径上。图片索引走过的大多数会话里,被识别的图片要么没有文字、 + * 要么图片文件已被清理 —— 那些结果不改变可搜索内容,重建纯属白做,却会把索引按住十几秒。 + * + * 这一组锁三件事: + * 1. 全部结果都不可搜索(无文字 / 图片缺失)⇒ **不调用** Knowledge; + * 2. 只要有**一条**识别出文字 ⇒ 必须调用(可搜索内容变了); + * 3. **已知终态直接跳过**的图片不得让会话被判成 dirty(用户明确要求的那条)。 + */ +describe('Knowledge 重建门控', () => { + const runOnceWith = async (options: { + recognizeText?: string + }): Promise<{ indexed: ReturnType; skipped: number }> => { + const h = createHarness({ + resolveAccountId: () => ACCOUNT, + ocrMs: 1, + ...(options.recognizeText === undefined ? {} : { recognizeText: options.recognizeText }) + }) + const indexed = vi.fn(async () => undefined) + h.service.bind({ onConversationIndexed: indexed }) + + await runPass(h.service) + const timings = (await h.service.getStatus()).stageTimings + h.service.resetAccount() + return { indexed, skipped: timings?.knowledgeIndexSkipped ?? -1 } + } + + it('all non-searchable results → Knowledge is not woken up', async () => { + const { indexed, skipped } = await runOnceWith({ recognizeText: '' }) + + expect(indexed).not.toHaveBeenCalled() + expect(skipped).toBe(1) + }) + + it('a single searchable result → Knowledge must be rebuilt', async () => { + const { indexed, skipped } = await runOnceWith({ recognizeText: '识别出来的文字' }) + + expect(indexed).toHaveBeenCalledTimes(1) + expect(skipped).toBe(0) + }) + + it('images skipped as already-terminal do not mark the conversation dirty', async () => { + const h = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 1 }) + const indexed = vi.fn(async () => undefined) + h.service.bind({ onConversationIndexed: indexed }) + + // 第一遍:正常识别出文字 ⇒ 会调用一次。 + await runPass(h.service) + expect(indexed).toHaveBeenCalledTimes(1) + + /** + * 第二遍带时间窗 ⇒ **刻意绕过 checkpoint 跳过**(窗口模式下不做增量跳过), + * 这样才会真的进到"逐张检查"这一步:所有图片都已是终态 ⇒ `fill()` 直接短路、 + * 不调 `settle()` ⇒ 会话不得被判成 dirty ⇒ 不得再叫 Knowledge。 + * + * 若不带窗口,整个会话会被 checkpoint 整体跳过 —— 那也安全,但测不到这条规则。 + */ + await runPass(h.service, { sinceMs: 1 }) + expect(indexed).toHaveBeenCalledTimes(1) + + const timings = (await h.service.getStatus()).stageTimings + expect(timings?.knowledgeIndexSkipped).toBeGreaterThanOrEqual(1) + + h.service.resetAccount() + }) +}) + +/** + * 「可搜索 → 不可搜索」是否可能**经由 settle()** 发生。 + * + * 为什么必须证明它:dirty 判定是 `state === 'indexed' && text.trim()`。 + * 如果存在一条路径能让一条**原本有 OCR 文字**的消息重新进 settle() 并落成 + * empty / image_missing / 空文字,那么旧的可搜索文字就会留在 Knowledge 里 —— 搜索能搜到、 + * 但索引已经"没有"那段文字,属于静默的 stale 结果。 + * + * 结论:**不可达**。依据是全量穷举写入/删除面(见下两条用例)。 + */ +describe('dirty 判定的安全边界', () => { + it('an already-indexed message never re-enters settle() on a later pass', async () => { + const h = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 1 }) + const recognize = vi.fn(async () => ({ success: true, text: '识别出来的文字', language: null })) + h.service.bind({ recognize }) + + await runPass(h.service) + const firstCalls = recognize.mock.calls.length + expect(firstCalls).toBeGreaterThan(0) + + // 带时间窗 ⇒ 绕过 checkpoint 跳过,逼它逐张检查(否则整个会话被 skip,测不到这条规则)。 + await runPass(h.service, { sinceMs: 1 }) + + /** + * 关键断言:第二遍**一次 OCR 都不该发生**。 + * 只要 binding 是 `indexed`(终态)就会被 `fill()` 短路 —— 短路即不 settle, + * 也就不可能把"有文字"改写成"没有文字"。 + */ + expect(recognize.mock.calls.length).toBe(firstCalls) + + h.service.resetAccount() + }) + + it('resetRetriableFailures leaves searchable results untouched', async () => { + const h = createHarness({ resolveAccountId: () => ACCOUNT, ocrMs: 1 }) + await runPass(h.service) + + const before = (await h.service.getStatus()).coverage + // 本用例识别出的全是"有文字",所以 indexed === 全部绑定。 + expect(before.indexed).toBe(IMAGES) + expect(before.failed).toBe(0) + + /** + * 走生产路径复位失败记录(「修复图片搜索索引」用的就是它)。 + * 它按**可重试失败状态**删除;`indexed` 不在那个集合里 ⇒ 一条都不会被删。 + * 只要 binding 还在且是终态,那条消息就永远不会重新进 `settle()`。 + */ + const reset = await h.service.resetRetriableFailures() + expect(reset.reset).toBe(0) + + const after = (await h.service.getStatus()).coverage + expect(after.indexed).toBe(IMAGES) + expect(after.processed).toBe(before.processed) + + h.service.resetAccount() + }) +}) diff --git a/tests/integration/image-text-index-synthetic-e2e.test.ts b/tests/integration/image-text-index-synthetic-e2e.test.ts index bb08645..cec0936 100644 --- a/tests/integration/image-text-index-synthetic-e2e.test.ts +++ b/tests/integration/image-text-index-synthetic-e2e.test.ts @@ -1,5 +1,5 @@ /** - * §2:图片 OCR 来源语义的 **deterministic synthetic E2E**。 + * 图片 OCR 来源语义的 **deterministic synthetic E2E**。 * * 硬要求是"不依赖真实线上 AI 模型也能 PASS",所以这里把两个外部边界**确定性**地固定住: * - WCDB(chat-service)→ 用合成联系人 / 合成消息; @@ -124,7 +124,7 @@ function makeKnowledge() { const NOW = new Date('2026-09-16T09:00:00+08:00') -describe('§2 图片文字索引 synthetic E2E(确定性,不依赖真模型)', () => { +describe('图片文字索引 synthetic E2E(确定性,不依赖真模型)', () => { let knowledge: ReturnType let service: LocalQueryApiService @@ -244,7 +244,7 @@ describe('§2 图片文字索引 synthetic E2E(确定性,不依赖真模型 }) }) -describe('§3 partial coverage honesty(确定性,不依赖真模型)', () => { +describe('partial coverage honesty(确定性,不依赖真模型)', () => { const NOT_INDEXED_KEYWORD = 'TRACE_NOT_YET_INDEXED_IMAGE' function partialImageCoverage() { diff --git a/tests/integration/system-ocr-macos.test.ts b/tests/integration/system-ocr-macos.test.ts new file mode 100644 index 0000000..1efdd70 --- /dev/null +++ b/tests/integration/system-ocr-macos.test.ts @@ -0,0 +1,131 @@ +// 【macOS】System OCR native fidelity。 +// +// capability-gated 的原生冒烟测试:只有在「macOS + native 运行时可用」时才真正跑。 +// macOS 的 Apple Vision 后端没有"语言包缺失"这一失败模式,所以门槛只有运行时本身; +// mock 单元测试仍然是 mandatory(tests/unit/system-ocr-service.test.ts)。 +// +// 这里断言的是 **macOS 专有**的性质,与 system-ocr-windows.test.ts 刻意不同: +// - 引擎标识是 macos-system-ocr; +// - 不做任何图片归一化:Vision 原生接受 PNG / JPEG,不得转码、不得起 ffmpeg; +// - capability.language 恒为 null(识别语言由 Vision 决定); +// - line.confidence 是 Vision 的真实置信度,不像 Windows 恒为 1.0。 +// +// fixture 全部是 synthetic 图片(tests/fixtures/ocr/*),不含任何真实聊天数据。 + +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { SYSTEM_OCR_ENGINE_MACOS } from '../../src/shared/system-ocr' + +vi.mock('../../src/main/image-decrypt-service', () => ({ + resolveFfmpegExecutable: (): string => 'ffmpeg' +})) + +import { SystemOcrService, systemOcrService } from '../../src/main/services/system-ocr-service' + +const fixtureDirectory = join(__dirname, '..', 'fixtures', 'ocr') + +const toDataUrl = (fileName: string, mimeType: string): string => + `data:${mimeType};base64,${readFileSync(join(fixtureDirectory, fileName)).toString('base64')}` + +/** 只比较"主要 token",避免识别微差造成脆弱测试。 */ +const expectContainsTokens = (text: string, tokens: string[]): void => { + const normalized = text.replace(/[\s\u3000]+/g, '').toLowerCase() + for (const token of tokens) { + expect(normalized).toContain(token.replace(/[\s\u3000]+/g, '').toLowerCase()) + } +} + +const capability = await systemOcrService.getCapability() +const onMac = process.platform === 'darwin' +const platformGate = onMac ? it : it.skip +const nativeGate = onMac && capability.available ? it : it.skip + +/** + * 走「真实 process.platform/arch + 真实 native binding + 默认归一化路径」的实例, + * 只把 ffmpeg 解析器换成 spy —— 用来证明 macOS 路径根本没有碰归一化。 + */ +const ffmpegResolver = vi.fn(() => 'ffmpeg') +const nativeService = new SystemOcrService({ resolveFfmpegExecutable: ffmpegResolver }) + +describe('macOS System OCR native fidelity', () => { + platformGate('reports a usable capability backed by Apple Vision', () => { + expect(capability.engine).toBe(SYSTEM_OCR_ENGINE_MACOS) + expect(capability.platform).toBe('darwin') + expect(capability.runtimeVersion).not.toBeNull() + // Vision 自行决定识别语言,不声称任何语言包。 + expect(capability.language).toBeNull() + if (!capability.available) { + console.warn(`[integration] macOS System OCR smoke skipped: ${capability.message}`) + } + }) + + nativeGate('recognizes simplified Chinese text', async () => { + const result = await nativeService.recognize({ + imageDataUrl: toDataUrl('system-ocr-zh.png', 'image/png') + }) + expect(result.success).toBe(true) + expectContainsTokens(result.text, ['TraceMemo', '本地', '文字', '识别']) + expect(result.language).toBeNull() + expect(result.durationMs).toBeGreaterThan(0) + }) + + nativeGate('recognizes English text', async () => { + const result = await nativeService.recognize({ + imageDataUrl: toDataUrl('system-ocr-en.png', 'image/png') + }) + expect(result.success).toBe(true) + expectContainsTokens(result.text, ['TraceMemo', 'System', 'OCR']) + }) + + nativeGate('recognizes mixed Chinese/English text', async () => { + const result = await nativeService.recognize({ + imageDataUrl: toDataUrl('system-ocr-mixed.png', 'image/png') + }) + expect(result.success).toBe(true) + expectContainsTokens(result.text, ['TraceMemo', '本地', 'OCR', '2026']) + }) + + /** + * Vision 原生接受 JPEG —— 这条用例同时是「macOS 不做归一化」的回归保护: + * 一旦有人把 Windows 的 PNG-only 假设搬过来,ffmpeg 解析器就会被调用。 + */ + nativeGate('accepts JPEG directly without any image normalization', async () => { + ffmpegResolver.mockClear() + const result = await nativeService.recognize({ + imageDataUrl: toDataUrl('system-ocr-mixed.jpg', 'image/jpeg') + }) + expect(result.success).toBe(true) + expectContainsTokens(result.text, ['TraceMemo', 'OCR', '2026']) + expect(ffmpegResolver).not.toHaveBeenCalled() + }) + + nativeGate('reports real Vision confidence and top-left-origin boxes', async () => { + const result = await nativeService.recognize({ + imageDataUrl: toDataUrl('system-ocr-mixed.png', 'image/png') + }) + expect(result.success).toBe(true) + expect(result.lines.length).toBeGreaterThan(0) + for (const line of result.lines) { + // Windows 恒为 1.0;macOS 必须给出真实置信度。 + expect(line.confidence).toBeGreaterThanOrEqual(0) + expect(line.confidence).toBeLessThanOrEqual(1) + const { x, y, width, height } = line.boundingBox + for (const value of [x, y, width, height]) { + expect(Number.isFinite(value)).toBe(true) + } + expect(x).toBeGreaterThanOrEqual(0) + expect(y).toBeGreaterThanOrEqual(0) + expect(x + width).toBeLessThanOrEqual(1.0001) + expect(y + height).toBeLessThanOrEqual(1.0001) + } + }) + + nativeGate('caches an identical repeat request', async () => { + const request = { imageDataUrl: toDataUrl('system-ocr-en.png', 'image/png') } + const first = await systemOcrService.recognize(request) + const second = await systemOcrService.recognize(request) + expect(first.success).toBe(true) + expect(second.fromCache).toBe(true) + }) +}) diff --git a/tests/integration/system-ocr-windows.test.ts b/tests/integration/system-ocr-windows.test.ts index 43bcb54..74756a6 100644 --- a/tests/integration/system-ocr-windows.test.ts +++ b/tests/integration/system-ocr-windows.test.ts @@ -1,17 +1,21 @@ -// Windows System OCR native fidelity。 +// 【Windows】System OCR native fidelity。 // // 这是 capability-gated 的原生冒烟测试: -// - 只有在「当前平台支持 + native 运行时可用 + 有可用 OCR 语言包」时才真正跑; +// - 只有在「Windows + native 运行时可用 + 有可用 OCR 语言包」时才真正跑; // - CI 环境无法保证 Windows OCR 语言包,所以中文识别不作为所有 CI 的硬门槛 // (mock 单元测试才是 mandatory,见 tests/unit/system-ocr-service.test.ts); // - 在 Windows 真机上必须实际通过。 // +// 这个文件断言的是 **Windows 专有**的性质:Windows 引擎标识、以及「引擎只吃 PNG, +// JPEG 必须走本服务归一化」这条约束。macOS 的对应测试见 system-ocr-macos.test.ts +// (macOS 不做归一化,不要把这个文件里的约束套到 macOS 上)。 +// // fixture 全部是 synthetic 图片(tests/fixtures/ocr/*),不含任何真实聊天数据。 import { readFileSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it, vi } from 'vitest' -import { SYSTEM_OCR_ENGINE } from '../../src/shared/system-ocr' +import { SYSTEM_OCR_ENGINE_WINDOWS } from '../../src/shared/system-ocr' vi.mock('../../src/main/image-decrypt-service', () => ({ resolveFfmpegExecutable: (): string => 'ffmpeg' @@ -33,11 +37,13 @@ const expectContainsTokens = (text: string, tokens: string[]): void => { } const capability = await systemOcrService.getCapability() -const nativeGate = capability.available ? it : it.skip +const onWindows = process.platform === 'win32' +const platformGate = onWindows ? it : it.skip +const nativeGate = onWindows && capability.available ? it : it.skip describe('Windows System OCR native fidelity', () => { - it('reports a usable capability on this machine', () => { - expect(capability.engine).toBe(SYSTEM_OCR_ENGINE) + platformGate('reports a usable capability on this machine', () => { + expect(capability.engine).toBe(SYSTEM_OCR_ENGINE_WINDOWS) if (!capability.available) { console.warn(`[integration] System OCR native smoke skipped: ${capability.message}`) } diff --git a/tests/unit/chat-service-list-messages-perf.test.ts b/tests/unit/chat-service-list-messages-perf.test.ts new file mode 100644 index 0000000..89d014e --- /dev/null +++ b/tests/unit/chat-service-list-messages-perf.test.ts @@ -0,0 +1,102 @@ +/** + * 大会话读取的性能日志与隐私契约。 + * + * 这一组锁两件事: + * 1. 大会话必须留下**可归因**的一行(谁读的 / 各阶段耗时 / 行数), + * 否则"图片索引卡住 10 秒"永远只能靠猜; + * 2. 那一行里**不能**出现会话 md5 —— 稳定会话标识不进日志。 + * + * 单独一个文件:`chat-service.test.ts` 里会调用 `closeChatDbForQuit()`, + * 那会把进程级的关闭标志置上,后续任何 `setChatDb` 都会被拒。 + */ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { WechatDb } from '../../src/main/wechat-db' +import { listMessagesAsync, setChatDb } from '../../src/main/services/chat-service' + +const FIXTURE_MD5 = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa' + +const makeMessages = (count: number): Array> => + Array.from({ length: count }, (_, index) => ({ + messageType: '1', + msgCreateTime: String(1_700_000_000 + index), + mesDes: '0', + mesLocalID: String(index + 1), + msgContent: '普通文本', + sender: 'wxid_fixture', + serverId: String(index + 1) + })) + +const installDb = (raw: Array>): void => { + const fakeDb = { + close: vi.fn(), + md5: () => FIXTURE_MD5, + getWcdb4Client: () => ({ + getUsernameByMd5: () => 'fixture@chatroom', + resolveEmoticonCdnUrl: () => '' + }), + getUserMessagesAsync: vi.fn(async () => raw) + } as unknown as WechatDb + setChatDb(fakeDb) +} + +const perfLines = (log: ReturnType): string[] => + log.mock.calls + .map((call) => String(call[0] ?? '')) + .filter((message) => message.startsWith('[ChatServicePerf]')) + +describe('chat service listMessages perf log', () => { + afterEach(() => setChatDb(null)) + + it('leaves one attributable line for a large read, without the conversation md5', async () => { + installDb(makeMessages(20_000)) + const log = vi.spyOn(console, 'log').mockImplementation(() => undefined) + try { + await listMessagesAsync( + FIXTURE_MD5, + undefined, + undefined, + undefined, + undefined, + 'image-text-index' + ) + + const lines = perfLines(log) + expect(lines).toHaveLength(1) + + const line = lines[0] + expect(line).toContain('caller=image-text-index') + expect(line).toContain('rows=20000') + expect(line).toContain('rawRows=20000') + // 拆解字段必须齐全,否则"这几秒花在哪"还是答不出来。 + for (const field of [ + 'totalMs=', + 'rawReadMs=', + 'formatMs=', + 'dateFormatMs=', + 'contentParseMs=', + 'sortMs=', + 'otherMs=' + ]) { + expect(line).toContain(field) + } + // 隐私:稳定会话标识绝不出现。 + expect(line).not.toContain(FIXTURE_MD5) + expect(line).not.toContain('md5') + // 关联用进程内序号(`request-N`),不是稳定标识。 + expect(line).toMatch(/request=request-\d+/) + } finally { + log.mockRestore() + } + }) + + it('stays silent for a small read so normal usage does not spam the log', async () => { + installDb(makeMessages(10)) + const log = vi.spyOn(console, 'log').mockImplementation(() => undefined) + try { + await listMessagesAsync(FIXTURE_MD5, undefined, undefined, undefined, undefined, 'knowledge') + expect(perfLines(log)).toEqual([]) + } finally { + log.mockRestore() + } + }) +}) diff --git a/tests/unit/knowledge-search-service.test.ts b/tests/unit/knowledge-search-service.test.ts index 4955df6..1cefc34 100644 --- a/tests/unit/knowledge-search-service.test.ts +++ b/tests/unit/knowledge-search-service.test.ts @@ -123,7 +123,14 @@ describe('KnowledgeSearchService legacy fallback', () => { startTime: 1785800000, limit: 10 }) - expect(listMessagesAsync).toHaveBeenCalledWith('fixture-conversation', 1785800000, undefined) + expect(listMessagesAsync).toHaveBeenCalledWith( + 'fixture-conversation', + 1785800000, + undefined, + undefined, + undefined, + 'knowledge' + ) expect(result).toMatchObject({ source: 'fallback', fallbackReason: 'unavailable', @@ -151,7 +158,14 @@ describe('KnowledgeSearchService legacy fallback', () => { limit: 10 }) - expect(listMessagesAsync).toHaveBeenCalledWith('fixture-conversation', undefined, undefined) + expect(listMessagesAsync).toHaveBeenCalledWith( + 'fixture-conversation', + undefined, + undefined, + undefined, + undefined, + 'knowledge' + ) expect(result.evidence).toHaveLength(1) await service.dispose() }) diff --git a/tests/unit/query-agent-answer-rules.test.ts b/tests/unit/query-agent-answer-rules.test.ts new file mode 100644 index 0000000..4f30fd7 --- /dev/null +++ b/tests/unit/query-agent-answer-rules.test.ts @@ -0,0 +1,49 @@ +/** + * 回答规则的语义断言。 + * + * 这几条是**产品契约**,不是措辞偏好 —— 换行、改写都可以,但实质约束不能丢: + * + * 1. 当前是单次检索回答,没有自动连续的多轮工具执行; + * 2. 因此禁止任何"下一步还能帮你继续"的邀约(那是能力幻觉); + * 3. 多条命中结果不能用 Markdown 表格承载(结果栏放不下,会错位)。 + * + * 这里只断言关键语义片段,不做巨型 snapshot —— 措辞会调整,语义不会。 + */ +import { describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ app: { getPath: () => '/tmp' } })) + +import { ANSWER_RULES } from '../../src/main/services/query-agent-service' + +describe('Query Agent 回答规则', () => { + it('明确当前是单次检索回答,没有连续多轮执行', () => { + expect(ANSWER_RULES).toContain('单次检索回答') + expect(ANSWER_RULES).toContain('没有自动连续的多轮工具执行') + }) + + it('禁止续问邀约,并点名常见的错误句式', () => { + expect(ANSWER_RULES).toContain('禁止') + for (const phrase of ['如果你需要,我可以', '要不要我继续', '我还可以帮你进一步', '需要的话我再查']) { + expect(ANSWER_RULES).toContain(phrase) + } + }) + + it('禁止用 Markdown 表格承载多条命中结果', () => { + expect(ANSWER_RULES).toContain('不要用 Markdown 表格') + }) + + it('派生内容要自报来源,不伪装成原始聊天文本', () => { + expect(ANSWER_RULES).toContain('图片 OCR') + expect(ANSWER_RULES).toContain('语音转写') + expect(ANSWER_RULES).toContain('派生内容') + }) + + it('范围说明按需给,不机械复读', () => { + expect(ANSWER_RULES).toContain('范围说明只在确有必要时给') + expect(ANSWER_RULES).toContain('不要') + }) + + it('没有同一性证据时不把"疑似同图"写成确定事实', () => { + expect(ANSWER_RULES).toContain('内容高度相似') + }) +}) diff --git a/tests/unit/query-agent-image-ocr-coverage.test.ts b/tests/unit/query-agent-image-ocr-coverage.test.ts index 2cb94e8..76d5ac0 100644 --- a/tests/unit/query-agent-image-ocr-coverage.test.ts +++ b/tests/unit/query-agent-image-ocr-coverage.test.ts @@ -1,5 +1,5 @@ /** - * §6 / §7 / §18:图片文字索引覆盖度在 Query Agent 这一层的语义。 + * 图片文字索引覆盖度在 Query Agent 这一层的语义。 * * 这里能确定性验证的是"覆盖度**真的进到了模型上下文**",而不是只做了个 UI 数字: * - 系统提示词把 imageOcrCoverage 定义成**独立于文字索引**的维度,并禁止凭零结果下"没有"; diff --git a/tests/unit/system-ocr-service.test.ts b/tests/unit/system-ocr-service.test.ts index 52f9564..1c40336 100644 --- a/tests/unit/system-ocr-service.test.ts +++ b/tests/unit/system-ocr-service.test.ts @@ -2,14 +2,18 @@ import { readFileSync } from 'node:fs' import { join } from 'node:path' import { beforeEach, describe, expect, it, vi } from 'vitest' import { - SYSTEM_OCR_ENGINE, + SYSTEM_OCR_ENGINE_MACOS, + SYSTEM_OCR_ENGINE_WINDOWS, SYSTEM_OCR_PROBE_PNG_BASE64, buildSystemOcrCacheKey, detectSystemOcrImageFormat, mapSystemOcrNativeError, normalizeSystemOcrText, parseImageDataUrl, - resolveSystemOcrLanguageTag + resolveMacOcrLanguageTag, + resolveSystemOcrEngine, + resolveSystemOcrLanguageTag, + resolveWindowsOcrLanguageTag } from '../../src/shared/system-ocr' vi.mock('../../src/main/image-decrypt-service', () => ({ @@ -106,6 +110,27 @@ describe('system-ocr shared helpers', () => { expect(resolveSystemOcrLanguageTag('en')).toBe('en-US') expect(resolveSystemOcrLanguageTag('')).toBeNull() expect(resolveSystemOcrLanguageTag('xx-YY')).toBeNull() + expect(resolveWindowsOcrLanguageTag('zh-HK')).toBe('zh-Hant-HK') + }) + + it('maps system locale onto Apple Vision language tags without region suffixes', () => { + // Vision 只认脚本级中文字标签,`zh-Hans-CN` 这类组合不是合法输入。 + expect(resolveMacOcrLanguageTag('zh-CN')).toBe('zh-Hans') + expect(resolveMacOcrLanguageTag('zh-Hans-CN')).toBe('zh-Hans') + expect(resolveMacOcrLanguageTag('zh_TW')).toBe('zh-Hant') + expect(resolveMacOcrLanguageTag('zh-HK')).toBe('zh-Hant') + expect(resolveMacOcrLanguageTag('en-US')).toBe('en-US') + expect(resolveMacOcrLanguageTag('')).toBeNull() + expect(resolveMacOcrLanguageTag('xx-YY')).toBeNull() + // 同一个 locale 在两个平台上必须给出各自的标签,不能串用。 + expect(resolveSystemOcrLanguageTag('zh-TW', 'darwin')).toBe('zh-Hant') + expect(resolveSystemOcrLanguageTag('zh-TW', 'win32')).toBe('zh-Hant-TW') + }) + + it('resolves a distinct engine identity per platform', () => { + expect(resolveSystemOcrEngine('win32')).toBe(SYSTEM_OCR_ENGINE_WINDOWS) + expect(resolveSystemOcrEngine('darwin')).toBe(SYSTEM_OCR_ENGINE_MACOS) + expect(SYSTEM_OCR_ENGINE_WINDOWS).not.toBe(SYSTEM_OCR_ENGINE_MACOS) }) it('maps native Windows errors onto product error codes', () => { @@ -117,6 +142,23 @@ describe('system-ocr shared helpers', () => { expect(mapSystemOcrNativeError('')).toBe('OCR_FAILED') }) + it('maps native macOS Vision errors onto product error codes', () => { + // 实测自 1.2.0 / macOS 15:畸形图片与伪造魔数都走这条。 + expect( + mapSystemOcrNativeError('CRImage Reader Detector was given zero-dimensioned image (0 x 0)') + ).toBe('IMAGE_DECODE_FAILED') + expect( + mapSystemOcrNativeError( + 'The image is too small in at least one dimension 2 x 2 (each dimension has to be more than 2 pixels)' + ) + ).toBe('IMAGE_DECODE_FAILED') + expect(mapSystemOcrNativeError('Cannot find native binding.')).toBe('SYSTEM_OCR_UNAVAILABLE') + // macOS 没有语言包概念:不能把普通失败误判成语言不可用。 + expect(mapSystemOcrNativeError('Vision request failed')).toBe('OCR_FAILED') + // "图里没有文字"是正常终态,不是失败 —— 否则表情包会落成可重试失败。 + expect(mapSystemOcrNativeError('No text recognized')).toBe('OCR_EMPTY_RESULT') + }) + it('parses image data urls and rejects other payloads', () => { expect(parseImageDataUrl(PNG_DATA_URL)).toMatchObject({ mimeType: 'image/png' }) expect(parseImageDataUrl('data:text/plain;base64,aGk=')).toBeNull() @@ -139,7 +181,7 @@ describe('system-ocr shared helpers', () => { const base = { imageHash: 'a'.repeat(32), language: 'zh-Hans-CN', runtimeVersion: '1.2.0' } const key = buildSystemOcrCacheKey({ ...base, platform: 'win32' }) expect(key).not.toBe(base.imageHash) - expect(key).toContain(SYSTEM_OCR_ENGINE) + expect(key).toContain(SYSTEM_OCR_ENGINE_WINDOWS) expect(key).toContain('zh-Hans-CN') expect(key).toContain('1.2.0') // 语言或运行时版本变化必须换 key,避免复用过期 / 跨引擎结果。 @@ -148,6 +190,19 @@ describe('system-ocr shared helpers', () => { buildSystemOcrCacheKey({ ...base, runtimeVersion: '1.3.0', platform: 'win32' }) ).not.toBe(key) }) + + it('never shares a cache key between the Windows and macOS engines', () => { + // 同一张图、同一 runtime 版本:平台不同 → key 必须不同,否则 macOS 会直接 + // 复用 Windows 变体算出的 artifact,用户永远看不到新引擎的结果。 + const shared = { imageHash: 'a'.repeat(32), language: null, runtimeVersion: '1.2.0' } + const windows = buildSystemOcrCacheKey({ ...shared, platform: 'win32' }) + const macos = buildSystemOcrCacheKey({ ...shared, platform: 'darwin' }) + expect(windows).not.toBe(macos) + expect(windows).toContain(SYSTEM_OCR_ENGINE_WINDOWS) + expect(macos).toContain(SYSTEM_OCR_ENGINE_MACOS) + // 同一个平台重启后必须给出同一个 key —— artifact 要能正常复用。 + expect(buildSystemOcrCacheKey({ ...shared, platform: 'darwin' })).toBe(macos) + }) }) describe('SystemOcrService capability detection', () => { @@ -156,7 +211,7 @@ describe('SystemOcrService capability detection', () => { const capability = await service.getCapability() expect(capability).toMatchObject({ available: true, - engine: SYSTEM_OCR_ENGINE, + engine: SYSTEM_OCR_ENGINE_WINDOWS, platform: 'win32', arch: 'x64', runtimeVersion: '1.2.0', @@ -164,6 +219,23 @@ describe('SystemOcrService capability detection', () => { }) }) + it('reports available on macOS without a language hint', async () => { + const runtime = createRuntime() + const service = createService(runtime, { platform: 'darwin', arch: 'arm64' }) + const capability = await service.getCapability() + expect(capability).toMatchObject({ + available: true, + engine: SYSTEM_OCR_ENGINE_MACOS, + platform: 'darwin', + arch: 'arm64', + runtimeVersion: '1.2.0', + // Vision 自行决定识别语言,capability 不再声称某个语言包。 + language: null + }) + // 探测本身也要走 native 运行时,而不是凭平台就宣称可用。 + expect(runtime.recognize).toHaveBeenCalled() + }) + it('is unavailable on unsupported platforms without loading a runtime', async () => { const loadRuntime = vi.fn(() => null) const service = createService(null, { platform: 'linux', loadRuntime }) @@ -173,6 +245,29 @@ describe('SystemOcrService capability detection', () => { expect(loadRuntime).not.toHaveBeenCalled() }) + it('reports a failed macOS probe as an engine failure, never as a missing language pack', async () => { + const service = createService(createRuntime({ probeError: 'Vision request failed' }), { + platform: 'darwin', + arch: 'arm64' + }) + const capability = await service.getCapability() + expect(capability.available).toBe(false) + expect(capability.reason).toBe('NATIVE_MODULE_MISSING') + expect(capability.message).not.toContain('语言包') + }) + + it('treats a macOS "No text recognized" probe as proof the recognizer works', async () => { + // 探测图是纯白图,真机 Vision 对它就是抛 `No text recognized`。 + // 这是 macOS 上 capability 探测的**正常路径**,不是故障。 + const service = createService(createRuntime({ probeError: 'No text recognized' }), { + platform: 'darwin', + arch: 'arm64' + }) + const capability = await service.getCapability() + expect(capability.available).toBe(true) + expect(capability.engine).toBe(SYSTEM_OCR_ENGINE_MACOS) + }) + it('is unavailable when the native runtime cannot be loaded', async () => { const service = createService(null) const capability = await service.getCapability() @@ -210,7 +305,7 @@ describe('SystemOcrService recognition', () => { success: true, text: 'TraceMemo 本地 OCR 2026', language: 'zh-Hans-CN', - engine: SYSTEM_OCR_ENGINE + engine: SYSTEM_OCR_ENGINE_WINDOWS }) expect(result.lines[0].boundingBox).toEqual({ x: 0.1, y: 0.2, width: 0.3, height: 0.4 }) expect(result.durationMs).toBeGreaterThanOrEqual(0) @@ -224,12 +319,66 @@ describe('SystemOcrService recognition', () => { expect(result.text).toBe('') }) - it('returns UNSUPPORTED_PLATFORM on non-Windows platforms', async () => { - const service = createService(null, { platform: 'darwin', arch: 'arm64' }) + it('returns UNSUPPORTED_PLATFORM on platforms without a system OCR backend', async () => { + const service = createService(null, { platform: 'linux', arch: 'x64' }) const result = await service.recognize({ imageDataUrl: PNG_DATA_URL }) expect(result.success).toBe(false) expect(result.errorCode).toBe('UNSUPPORTED_PLATFORM') - expect(result.engine).toBe(SYSTEM_OCR_ENGINE) + expect(result.engine).toBe(SYSTEM_OCR_ENGINE_WINDOWS) + }) + + it('recognizes on macOS and skips image normalization entirely', async () => { + const runtime = createRuntime({ text: 'TraceMemo 图 片 OCR 2026' }) + const resolveFfmpegExecutable = vi.fn(() => 'ffmpeg') + // 刻意不注入 toPngBytes:要验证的就是**默认归一化路径**在 macOS 上被绕过。 + const service = new SystemOcrService({ + platform: 'darwin', + arch: 'arm64', + locale: () => 'zh-CN', + loadRuntime: () => ({ ...runtime }), + resolveFfmpegExecutable + }) + + const result = await service.recognize({ imageDataUrl: JPEG_DATA_URL }) + + expect(result).toMatchObject({ + success: true, + text: 'TraceMemo 图片 OCR 2026', + language: null, + engine: SYSTEM_OCR_ENGINE_MACOS + }) + // Vision 原生接受 JPEG:不转码、不起 ffmpeg 子进程。 + expect(resolveFfmpegExecutable).not.toHaveBeenCalled() + // 而且送给引擎的就是原始 JPEG 字节,没有被换成 PNG。 + const businessCall = runtime.recognize.mock.calls.find( + (call) => !Buffer.from(call[0] as Uint8Array).equals(PROBE_BYTES) + ) + expect(Buffer.from(businessCall?.[0] as Uint8Array)).toEqual( + Buffer.from(JPEG_DATA_URL.split(',')[1], 'base64') + ) + }) + + it('maps a macOS Vision decode failure onto IMAGE_DECODE_FAILED', async () => { + const runtime = createRuntime({ + error: 'CRImage Reader Detector was given zero-dimensioned image (0 x 0)' + }) + const service = createService(runtime, { platform: 'darwin', arch: 'arm64' }) + const result = await service.recognize({ imageDataUrl: PNG_DATA_URL }) + expect(result.success).toBe(false) + expect(result.errorCode).toBe('IMAGE_DECODE_FAILED') + // 不把 native 堆栈透给用户。 + expect(result.error).not.toContain('CRImage') + }) + + it('treats a macOS "No text recognized" throw as OCR_EMPTY_RESULT, not a failure', async () => { + // Windows 对无文字图片返回空文本;macOS 的 Vision 是抛错。 + // 两者必须是同一个终态,否则表情包 / 风景图会全部落成可重试失败。 + const runtime = createRuntime({ error: 'No text recognized' }) + const service = createService(runtime, { platform: 'darwin', arch: 'arm64' }) + const result = await service.recognize({ imageDataUrl: PNG_DATA_URL }) + expect(result.success).toBe(false) + expect(result.errorCode).toBe('OCR_EMPTY_RESULT') + expect(result.error).toContain('没有在这张图片里识别到文字') }) it('returns OCR_LANGUAGE_UNAVAILABLE when no OCR language pack is installed', async () => { @@ -337,13 +486,76 @@ describe('ImageInsightService local OCR orchestration', () => { }) const capability = await imageInsightService.getSystemOcrCapability() - expect(capability.engine).toBe(SYSTEM_OCR_ENGINE) + // 单例用的是真实平台,断言也按平台推导,避免变成"只能在这台机器上过"的测试。 + expect(capability.engine).toBe(resolveSystemOcrEngine(process.platform)) const result = await imageInsightService.extractLocalText({ imageDataUrl: PNG_DATA_URL }) - expect(result.engine).toBe(SYSTEM_OCR_ENGINE) + expect(result.engine).toBe(resolveSystemOcrEngine(process.platform)) // 关键约束:本地 OCR 路径绝不调用远端 Vision Provider。 expect(analyzeImage).not.toHaveBeenCalled() // 也不写 Vision 的 insight 缓存。 expect(upsert).not.toHaveBeenCalled() }) }) + +/** + * 日志契约:后台回填会连续识别几万张,默认输出**不能**逐张留痕。 + * + * 判据是"默认输出里一条成功日志都没有",而不是"日志看起来还行" —— + * 这条约束一旦破了,跑一次全量回填就会把日志刷爆。 + */ +describe('system-ocr 日志契约', () => { + const runOnce = async ( + service: SystemOcrService + ): Promise<{ log: ReturnType; warn: ReturnType }> => { + const log = vi.spyOn(console, 'log').mockImplementation(() => undefined) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + await service.getCapability() + await service.recognize({ imageDataUrl: PNG_DATA_URL }) + return { log, warn } + } finally { + log.mockRestore() + warn.mockRestore() + } + } + + it('识别成功时不写任何 console.log(逐张成功日志是纯噪声)', async () => { + const service = createService(createRuntime({ text: '本地图片文字识别' })) + const { log } = await runOnce(service) + + const messages = log.mock.calls.map((call) => String(call[0] ?? '')) + expect(messages.filter((message) => message.includes('[SystemOcrService]'))).toEqual([]) + }) + + it('单张的耗时与字数仍然通过返回值给出(设置页诊断不依赖日志)', async () => { + const service = createService(createRuntime({ text: '本地图片文字识别' })) + const result = await service.recognize({ imageDataUrl: PNG_DATA_URL }) + + expect(result.success).toBe(true) + expect(result.text).toBe('本地图片文字识别') + expect(typeof result.durationMs).toBe('number') + expect(result.durationMs).toBeGreaterThanOrEqual(0) + }) + + it('失败时保留一条 warn,且只含 error code / engine / platform / duration', async () => { + const service = createService(createRuntime({ error: 'Windows error 拒绝访问 (0x80070005)' })) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + await service.getCapability() + const result = await service.recognize({ imageDataUrl: PNG_DATA_URL }) + expect(result.success).toBe(false) + + const failedLines = warn.mock.calls + .map((call) => String(call[0] ?? '')) + .filter((message) => message.includes('[SystemOcrService] failed')) + expect(failedLines).toHaveLength(1) + // 绝不出现识别正文 / 图片内容 / 稳定标识。 + const joined = failedLines.join(' ') + expect(joined).not.toContain('base64') + expect(joined).not.toContain('data:image') + } finally { + warn.mockRestore() + } + }) +})