mirror of
https://wget.la/https://github.com/Wxw-Gu/WechatExplorer
synced 2026-10-06 13:54:10 +08:00
feat: 新增mac ocr转文字,增加到问一问微信 图片索引优化速度
- 图片文字索引性能与进度诚实化 - 问问微信:证据卡区分「消息类型」与「派生来源」,派生命中内容自报来源 - 问问微信:回答规则禁止未真实执行的多轮承诺 - 本地图片文字识别:支持 macOS 系统 OCR(Apple Vision)
This commit is contained in:
@@ -0,0 +1,102 @@
|
||||
/**
|
||||
* 大会话读取的性能日志与隐私契约。
|
||||
*
|
||||
* 这一组锁两件事:
|
||||
* 1. 大会话必须留下**可归因**的一行(谁读的 / 各阶段耗时 / 行数),
|
||||
* 否则"图片索引卡住 10 秒"永远只能靠猜;
|
||||
* 2. 那一行里**不能**出现会话 md5 —— 稳定会话标识不进日志。
|
||||
*
|
||||
* 单独一个文件:`chat-service.test.ts` 里会调用 `closeChatDbForQuit()`,
|
||||
* 那会把进程级的关闭标志置上,后续任何 `setChatDb` 都会被拒。
|
||||
*/
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import type { WechatDb } from '../../src/main/wechat-db'
|
||||
import { listMessagesAsync, setChatDb } from '../../src/main/services/chat-service'
|
||||
|
||||
const FIXTURE_MD5 = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa'
|
||||
|
||||
const makeMessages = (count: number): Array<Record<string, unknown>> =>
|
||||
Array.from({ length: count }, (_, index) => ({
|
||||
messageType: '1',
|
||||
msgCreateTime: String(1_700_000_000 + index),
|
||||
mesDes: '0',
|
||||
mesLocalID: String(index + 1),
|
||||
msgContent: '普通文本',
|
||||
sender: 'wxid_fixture',
|
||||
serverId: String(index + 1)
|
||||
}))
|
||||
|
||||
const installDb = (raw: Array<Record<string, unknown>>): void => {
|
||||
const fakeDb = {
|
||||
close: vi.fn(),
|
||||
md5: () => FIXTURE_MD5,
|
||||
getWcdb4Client: () => ({
|
||||
getUsernameByMd5: () => 'fixture@chatroom',
|
||||
resolveEmoticonCdnUrl: () => ''
|
||||
}),
|
||||
getUserMessagesAsync: vi.fn(async () => raw)
|
||||
} as unknown as WechatDb
|
||||
setChatDb(fakeDb)
|
||||
}
|
||||
|
||||
const perfLines = (log: ReturnType<typeof vi.spyOn>): string[] =>
|
||||
log.mock.calls
|
||||
.map((call) => String(call[0] ?? ''))
|
||||
.filter((message) => message.startsWith('[ChatServicePerf]'))
|
||||
|
||||
describe('chat service listMessages perf log', () => {
|
||||
afterEach(() => setChatDb(null))
|
||||
|
||||
it('leaves one attributable line for a large read, without the conversation md5', async () => {
|
||||
installDb(makeMessages(20_000))
|
||||
const log = vi.spyOn(console, 'log').mockImplementation(() => undefined)
|
||||
try {
|
||||
await listMessagesAsync(
|
||||
FIXTURE_MD5,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
'image-text-index'
|
||||
)
|
||||
|
||||
const lines = perfLines(log)
|
||||
expect(lines).toHaveLength(1)
|
||||
|
||||
const line = lines[0]
|
||||
expect(line).toContain('caller=image-text-index')
|
||||
expect(line).toContain('rows=20000')
|
||||
expect(line).toContain('rawRows=20000')
|
||||
// 拆解字段必须齐全,否则"这几秒花在哪"还是答不出来。
|
||||
for (const field of [
|
||||
'totalMs=',
|
||||
'rawReadMs=',
|
||||
'formatMs=',
|
||||
'dateFormatMs=',
|
||||
'contentParseMs=',
|
||||
'sortMs=',
|
||||
'otherMs='
|
||||
]) {
|
||||
expect(line).toContain(field)
|
||||
}
|
||||
// 隐私:稳定会话标识绝不出现。
|
||||
expect(line).not.toContain(FIXTURE_MD5)
|
||||
expect(line).not.toContain('md5')
|
||||
// 关联用进程内序号(`request-N`),不是稳定标识。
|
||||
expect(line).toMatch(/request=request-\d+/)
|
||||
} finally {
|
||||
log.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('stays silent for a small read so normal usage does not spam the log', async () => {
|
||||
installDb(makeMessages(10))
|
||||
const log = vi.spyOn(console, 'log').mockImplementation(() => undefined)
|
||||
try {
|
||||
await listMessagesAsync(FIXTURE_MD5, undefined, undefined, undefined, undefined, 'knowledge')
|
||||
expect(perfLines(log)).toEqual([])
|
||||
} finally {
|
||||
log.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -123,7 +123,14 @@ describe('KnowledgeSearchService legacy fallback', () => {
|
||||
startTime: 1785800000,
|
||||
limit: 10
|
||||
})
|
||||
expect(listMessagesAsync).toHaveBeenCalledWith('fixture-conversation', 1785800000, undefined)
|
||||
expect(listMessagesAsync).toHaveBeenCalledWith(
|
||||
'fixture-conversation',
|
||||
1785800000,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
'knowledge'
|
||||
)
|
||||
expect(result).toMatchObject({
|
||||
source: 'fallback',
|
||||
fallbackReason: 'unavailable',
|
||||
@@ -151,7 +158,14 @@ describe('KnowledgeSearchService legacy fallback', () => {
|
||||
limit: 10
|
||||
})
|
||||
|
||||
expect(listMessagesAsync).toHaveBeenCalledWith('fixture-conversation', undefined, undefined)
|
||||
expect(listMessagesAsync).toHaveBeenCalledWith(
|
||||
'fixture-conversation',
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
'knowledge'
|
||||
)
|
||||
expect(result.evidence).toHaveLength(1)
|
||||
await service.dispose()
|
||||
})
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
/**
|
||||
* 回答规则的语义断言。
|
||||
*
|
||||
* 这几条是**产品契约**,不是措辞偏好 —— 换行、改写都可以,但实质约束不能丢:
|
||||
*
|
||||
* 1. 当前是单次检索回答,没有自动连续的多轮工具执行;
|
||||
* 2. 因此禁止任何"下一步还能帮你继续"的邀约(那是能力幻觉);
|
||||
* 3. 多条命中结果不能用 Markdown 表格承载(结果栏放不下,会错位)。
|
||||
*
|
||||
* 这里只断言关键语义片段,不做巨型 snapshot —— 措辞会调整,语义不会。
|
||||
*/
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
|
||||
vi.mock('electron', () => ({ app: { getPath: () => '/tmp' } }))
|
||||
|
||||
import { ANSWER_RULES } from '../../src/main/services/query-agent-service'
|
||||
|
||||
describe('Query Agent 回答规则', () => {
|
||||
it('明确当前是单次检索回答,没有连续多轮执行', () => {
|
||||
expect(ANSWER_RULES).toContain('单次检索回答')
|
||||
expect(ANSWER_RULES).toContain('没有自动连续的多轮工具执行')
|
||||
})
|
||||
|
||||
it('禁止续问邀约,并点名常见的错误句式', () => {
|
||||
expect(ANSWER_RULES).toContain('禁止')
|
||||
for (const phrase of ['如果你需要,我可以', '要不要我继续', '我还可以帮你进一步', '需要的话我再查']) {
|
||||
expect(ANSWER_RULES).toContain(phrase)
|
||||
}
|
||||
})
|
||||
|
||||
it('禁止用 Markdown 表格承载多条命中结果', () => {
|
||||
expect(ANSWER_RULES).toContain('不要用 Markdown 表格')
|
||||
})
|
||||
|
||||
it('派生内容要自报来源,不伪装成原始聊天文本', () => {
|
||||
expect(ANSWER_RULES).toContain('图片 OCR')
|
||||
expect(ANSWER_RULES).toContain('语音转写')
|
||||
expect(ANSWER_RULES).toContain('派生内容')
|
||||
})
|
||||
|
||||
it('范围说明按需给,不机械复读', () => {
|
||||
expect(ANSWER_RULES).toContain('范围说明只在确有必要时给')
|
||||
expect(ANSWER_RULES).toContain('不要')
|
||||
})
|
||||
|
||||
it('没有同一性证据时不把"疑似同图"写成确定事实', () => {
|
||||
expect(ANSWER_RULES).toContain('内容高度相似')
|
||||
})
|
||||
})
|
||||
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* §6 / §7 / §18:图片文字索引覆盖度在 Query Agent 这一层的语义。
|
||||
* 图片文字索引覆盖度在 Query Agent 这一层的语义。
|
||||
*
|
||||
* 这里能确定性验证的是"覆盖度**真的进到了模型上下文**",而不是只做了个 UI 数字:
|
||||
* - 系统提示词把 imageOcrCoverage 定义成**独立于文字索引**的维度,并禁止凭零结果下"没有";
|
||||
|
||||
@@ -2,14 +2,18 @@ import { readFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
SYSTEM_OCR_ENGINE,
|
||||
SYSTEM_OCR_ENGINE_MACOS,
|
||||
SYSTEM_OCR_ENGINE_WINDOWS,
|
||||
SYSTEM_OCR_PROBE_PNG_BASE64,
|
||||
buildSystemOcrCacheKey,
|
||||
detectSystemOcrImageFormat,
|
||||
mapSystemOcrNativeError,
|
||||
normalizeSystemOcrText,
|
||||
parseImageDataUrl,
|
||||
resolveSystemOcrLanguageTag
|
||||
resolveMacOcrLanguageTag,
|
||||
resolveSystemOcrEngine,
|
||||
resolveSystemOcrLanguageTag,
|
||||
resolveWindowsOcrLanguageTag
|
||||
} from '../../src/shared/system-ocr'
|
||||
|
||||
vi.mock('../../src/main/image-decrypt-service', () => ({
|
||||
@@ -106,6 +110,27 @@ describe('system-ocr shared helpers', () => {
|
||||
expect(resolveSystemOcrLanguageTag('en')).toBe('en-US')
|
||||
expect(resolveSystemOcrLanguageTag('')).toBeNull()
|
||||
expect(resolveSystemOcrLanguageTag('xx-YY')).toBeNull()
|
||||
expect(resolveWindowsOcrLanguageTag('zh-HK')).toBe('zh-Hant-HK')
|
||||
})
|
||||
|
||||
it('maps system locale onto Apple Vision language tags without region suffixes', () => {
|
||||
// Vision 只认脚本级中文字标签,`zh-Hans-CN` 这类组合不是合法输入。
|
||||
expect(resolveMacOcrLanguageTag('zh-CN')).toBe('zh-Hans')
|
||||
expect(resolveMacOcrLanguageTag('zh-Hans-CN')).toBe('zh-Hans')
|
||||
expect(resolveMacOcrLanguageTag('zh_TW')).toBe('zh-Hant')
|
||||
expect(resolveMacOcrLanguageTag('zh-HK')).toBe('zh-Hant')
|
||||
expect(resolveMacOcrLanguageTag('en-US')).toBe('en-US')
|
||||
expect(resolveMacOcrLanguageTag('')).toBeNull()
|
||||
expect(resolveMacOcrLanguageTag('xx-YY')).toBeNull()
|
||||
// 同一个 locale 在两个平台上必须给出各自的标签,不能串用。
|
||||
expect(resolveSystemOcrLanguageTag('zh-TW', 'darwin')).toBe('zh-Hant')
|
||||
expect(resolveSystemOcrLanguageTag('zh-TW', 'win32')).toBe('zh-Hant-TW')
|
||||
})
|
||||
|
||||
it('resolves a distinct engine identity per platform', () => {
|
||||
expect(resolveSystemOcrEngine('win32')).toBe(SYSTEM_OCR_ENGINE_WINDOWS)
|
||||
expect(resolveSystemOcrEngine('darwin')).toBe(SYSTEM_OCR_ENGINE_MACOS)
|
||||
expect(SYSTEM_OCR_ENGINE_WINDOWS).not.toBe(SYSTEM_OCR_ENGINE_MACOS)
|
||||
})
|
||||
|
||||
it('maps native Windows errors onto product error codes', () => {
|
||||
@@ -117,6 +142,23 @@ describe('system-ocr shared helpers', () => {
|
||||
expect(mapSystemOcrNativeError('')).toBe('OCR_FAILED')
|
||||
})
|
||||
|
||||
it('maps native macOS Vision errors onto product error codes', () => {
|
||||
// 实测自 1.2.0 / macOS 15:畸形图片与伪造魔数都走这条。
|
||||
expect(
|
||||
mapSystemOcrNativeError('CRImage Reader Detector was given zero-dimensioned image (0 x 0)')
|
||||
).toBe('IMAGE_DECODE_FAILED')
|
||||
expect(
|
||||
mapSystemOcrNativeError(
|
||||
'The image is too small in at least one dimension 2 x 2 (each dimension has to be more than 2 pixels)'
|
||||
)
|
||||
).toBe('IMAGE_DECODE_FAILED')
|
||||
expect(mapSystemOcrNativeError('Cannot find native binding.')).toBe('SYSTEM_OCR_UNAVAILABLE')
|
||||
// macOS 没有语言包概念:不能把普通失败误判成语言不可用。
|
||||
expect(mapSystemOcrNativeError('Vision request failed')).toBe('OCR_FAILED')
|
||||
// "图里没有文字"是正常终态,不是失败 —— 否则表情包会落成可重试失败。
|
||||
expect(mapSystemOcrNativeError('No text recognized')).toBe('OCR_EMPTY_RESULT')
|
||||
})
|
||||
|
||||
it('parses image data urls and rejects other payloads', () => {
|
||||
expect(parseImageDataUrl(PNG_DATA_URL)).toMatchObject({ mimeType: 'image/png' })
|
||||
expect(parseImageDataUrl('data:text/plain;base64,aGk=')).toBeNull()
|
||||
@@ -139,7 +181,7 @@ describe('system-ocr shared helpers', () => {
|
||||
const base = { imageHash: 'a'.repeat(32), language: 'zh-Hans-CN', runtimeVersion: '1.2.0' }
|
||||
const key = buildSystemOcrCacheKey({ ...base, platform: 'win32' })
|
||||
expect(key).not.toBe(base.imageHash)
|
||||
expect(key).toContain(SYSTEM_OCR_ENGINE)
|
||||
expect(key).toContain(SYSTEM_OCR_ENGINE_WINDOWS)
|
||||
expect(key).toContain('zh-Hans-CN')
|
||||
expect(key).toContain('1.2.0')
|
||||
// 语言或运行时版本变化必须换 key,避免复用过期 / 跨引擎结果。
|
||||
@@ -148,6 +190,19 @@ describe('system-ocr shared helpers', () => {
|
||||
buildSystemOcrCacheKey({ ...base, runtimeVersion: '1.3.0', platform: 'win32' })
|
||||
).not.toBe(key)
|
||||
})
|
||||
|
||||
it('never shares a cache key between the Windows and macOS engines', () => {
|
||||
// 同一张图、同一 runtime 版本:平台不同 → key 必须不同,否则 macOS 会直接
|
||||
// 复用 Windows 变体算出的 artifact,用户永远看不到新引擎的结果。
|
||||
const shared = { imageHash: 'a'.repeat(32), language: null, runtimeVersion: '1.2.0' }
|
||||
const windows = buildSystemOcrCacheKey({ ...shared, platform: 'win32' })
|
||||
const macos = buildSystemOcrCacheKey({ ...shared, platform: 'darwin' })
|
||||
expect(windows).not.toBe(macos)
|
||||
expect(windows).toContain(SYSTEM_OCR_ENGINE_WINDOWS)
|
||||
expect(macos).toContain(SYSTEM_OCR_ENGINE_MACOS)
|
||||
// 同一个平台重启后必须给出同一个 key —— artifact 要能正常复用。
|
||||
expect(buildSystemOcrCacheKey({ ...shared, platform: 'darwin' })).toBe(macos)
|
||||
})
|
||||
})
|
||||
|
||||
describe('SystemOcrService capability detection', () => {
|
||||
@@ -156,7 +211,7 @@ describe('SystemOcrService capability detection', () => {
|
||||
const capability = await service.getCapability()
|
||||
expect(capability).toMatchObject({
|
||||
available: true,
|
||||
engine: SYSTEM_OCR_ENGINE,
|
||||
engine: SYSTEM_OCR_ENGINE_WINDOWS,
|
||||
platform: 'win32',
|
||||
arch: 'x64',
|
||||
runtimeVersion: '1.2.0',
|
||||
@@ -164,6 +219,23 @@ describe('SystemOcrService capability detection', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('reports available on macOS without a language hint', async () => {
|
||||
const runtime = createRuntime()
|
||||
const service = createService(runtime, { platform: 'darwin', arch: 'arm64' })
|
||||
const capability = await service.getCapability()
|
||||
expect(capability).toMatchObject({
|
||||
available: true,
|
||||
engine: SYSTEM_OCR_ENGINE_MACOS,
|
||||
platform: 'darwin',
|
||||
arch: 'arm64',
|
||||
runtimeVersion: '1.2.0',
|
||||
// Vision 自行决定识别语言,capability 不再声称某个语言包。
|
||||
language: null
|
||||
})
|
||||
// 探测本身也要走 native 运行时,而不是凭平台就宣称可用。
|
||||
expect(runtime.recognize).toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('is unavailable on unsupported platforms without loading a runtime', async () => {
|
||||
const loadRuntime = vi.fn(() => null)
|
||||
const service = createService(null, { platform: 'linux', loadRuntime })
|
||||
@@ -173,6 +245,29 @@ describe('SystemOcrService capability detection', () => {
|
||||
expect(loadRuntime).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('reports a failed macOS probe as an engine failure, never as a missing language pack', async () => {
|
||||
const service = createService(createRuntime({ probeError: 'Vision request failed' }), {
|
||||
platform: 'darwin',
|
||||
arch: 'arm64'
|
||||
})
|
||||
const capability = await service.getCapability()
|
||||
expect(capability.available).toBe(false)
|
||||
expect(capability.reason).toBe('NATIVE_MODULE_MISSING')
|
||||
expect(capability.message).not.toContain('语言包')
|
||||
})
|
||||
|
||||
it('treats a macOS "No text recognized" probe as proof the recognizer works', async () => {
|
||||
// 探测图是纯白图,真机 Vision 对它就是抛 `No text recognized`。
|
||||
// 这是 macOS 上 capability 探测的**正常路径**,不是故障。
|
||||
const service = createService(createRuntime({ probeError: 'No text recognized' }), {
|
||||
platform: 'darwin',
|
||||
arch: 'arm64'
|
||||
})
|
||||
const capability = await service.getCapability()
|
||||
expect(capability.available).toBe(true)
|
||||
expect(capability.engine).toBe(SYSTEM_OCR_ENGINE_MACOS)
|
||||
})
|
||||
|
||||
it('is unavailable when the native runtime cannot be loaded', async () => {
|
||||
const service = createService(null)
|
||||
const capability = await service.getCapability()
|
||||
@@ -210,7 +305,7 @@ describe('SystemOcrService recognition', () => {
|
||||
success: true,
|
||||
text: 'TraceMemo 本地 OCR 2026',
|
||||
language: 'zh-Hans-CN',
|
||||
engine: SYSTEM_OCR_ENGINE
|
||||
engine: SYSTEM_OCR_ENGINE_WINDOWS
|
||||
})
|
||||
expect(result.lines[0].boundingBox).toEqual({ x: 0.1, y: 0.2, width: 0.3, height: 0.4 })
|
||||
expect(result.durationMs).toBeGreaterThanOrEqual(0)
|
||||
@@ -224,12 +319,66 @@ describe('SystemOcrService recognition', () => {
|
||||
expect(result.text).toBe('')
|
||||
})
|
||||
|
||||
it('returns UNSUPPORTED_PLATFORM on non-Windows platforms', async () => {
|
||||
const service = createService(null, { platform: 'darwin', arch: 'arm64' })
|
||||
it('returns UNSUPPORTED_PLATFORM on platforms without a system OCR backend', async () => {
|
||||
const service = createService(null, { platform: 'linux', arch: 'x64' })
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('UNSUPPORTED_PLATFORM')
|
||||
expect(result.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
expect(result.engine).toBe(SYSTEM_OCR_ENGINE_WINDOWS)
|
||||
})
|
||||
|
||||
it('recognizes on macOS and skips image normalization entirely', async () => {
|
||||
const runtime = createRuntime({ text: 'TraceMemo 图 片 OCR 2026' })
|
||||
const resolveFfmpegExecutable = vi.fn(() => 'ffmpeg')
|
||||
// 刻意不注入 toPngBytes:要验证的就是**默认归一化路径**在 macOS 上被绕过。
|
||||
const service = new SystemOcrService({
|
||||
platform: 'darwin',
|
||||
arch: 'arm64',
|
||||
locale: () => 'zh-CN',
|
||||
loadRuntime: () => ({ ...runtime }),
|
||||
resolveFfmpegExecutable
|
||||
})
|
||||
|
||||
const result = await service.recognize({ imageDataUrl: JPEG_DATA_URL })
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: true,
|
||||
text: 'TraceMemo 图片 OCR 2026',
|
||||
language: null,
|
||||
engine: SYSTEM_OCR_ENGINE_MACOS
|
||||
})
|
||||
// Vision 原生接受 JPEG:不转码、不起 ffmpeg 子进程。
|
||||
expect(resolveFfmpegExecutable).not.toHaveBeenCalled()
|
||||
// 而且送给引擎的就是原始 JPEG 字节,没有被换成 PNG。
|
||||
const businessCall = runtime.recognize.mock.calls.find(
|
||||
(call) => !Buffer.from(call[0] as Uint8Array).equals(PROBE_BYTES)
|
||||
)
|
||||
expect(Buffer.from(businessCall?.[0] as Uint8Array)).toEqual(
|
||||
Buffer.from(JPEG_DATA_URL.split(',')[1], 'base64')
|
||||
)
|
||||
})
|
||||
|
||||
it('maps a macOS Vision decode failure onto IMAGE_DECODE_FAILED', async () => {
|
||||
const runtime = createRuntime({
|
||||
error: 'CRImage Reader Detector was given zero-dimensioned image (0 x 0)'
|
||||
})
|
||||
const service = createService(runtime, { platform: 'darwin', arch: 'arm64' })
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('IMAGE_DECODE_FAILED')
|
||||
// 不把 native 堆栈透给用户。
|
||||
expect(result.error).not.toContain('CRImage')
|
||||
})
|
||||
|
||||
it('treats a macOS "No text recognized" throw as OCR_EMPTY_RESULT, not a failure', async () => {
|
||||
// Windows 对无文字图片返回空文本;macOS 的 Vision 是抛错。
|
||||
// 两者必须是同一个终态,否则表情包 / 风景图会全部落成可重试失败。
|
||||
const runtime = createRuntime({ error: 'No text recognized' })
|
||||
const service = createService(runtime, { platform: 'darwin', arch: 'arm64' })
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('OCR_EMPTY_RESULT')
|
||||
expect(result.error).toContain('没有在这张图片里识别到文字')
|
||||
})
|
||||
|
||||
it('returns OCR_LANGUAGE_UNAVAILABLE when no OCR language pack is installed', async () => {
|
||||
@@ -337,13 +486,76 @@ describe('ImageInsightService local OCR orchestration', () => {
|
||||
})
|
||||
|
||||
const capability = await imageInsightService.getSystemOcrCapability()
|
||||
expect(capability.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
// 单例用的是真实平台,断言也按平台推导,避免变成"只能在这台机器上过"的测试。
|
||||
expect(capability.engine).toBe(resolveSystemOcrEngine(process.platform))
|
||||
|
||||
const result = await imageInsightService.extractLocalText({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
expect(result.engine).toBe(resolveSystemOcrEngine(process.platform))
|
||||
// 关键约束:本地 OCR 路径绝不调用远端 Vision Provider。
|
||||
expect(analyzeImage).not.toHaveBeenCalled()
|
||||
// 也不写 Vision 的 insight 缓存。
|
||||
expect(upsert).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
/**
|
||||
* 日志契约:后台回填会连续识别几万张,默认输出**不能**逐张留痕。
|
||||
*
|
||||
* 判据是"默认输出里一条成功日志都没有",而不是"日志看起来还行" ——
|
||||
* 这条约束一旦破了,跑一次全量回填就会把日志刷爆。
|
||||
*/
|
||||
describe('system-ocr 日志契约', () => {
|
||||
const runOnce = async (
|
||||
service: SystemOcrService
|
||||
): Promise<{ log: ReturnType<typeof vi.spyOn>; warn: ReturnType<typeof vi.spyOn> }> => {
|
||||
const log = vi.spyOn(console, 'log').mockImplementation(() => undefined)
|
||||
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
|
||||
try {
|
||||
await service.getCapability()
|
||||
await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
return { log, warn }
|
||||
} finally {
|
||||
log.mockRestore()
|
||||
warn.mockRestore()
|
||||
}
|
||||
}
|
||||
|
||||
it('识别成功时不写任何 console.log(逐张成功日志是纯噪声)', async () => {
|
||||
const service = createService(createRuntime({ text: '本地图片文字识别' }))
|
||||
const { log } = await runOnce(service)
|
||||
|
||||
const messages = log.mock.calls.map((call) => String(call[0] ?? ''))
|
||||
expect(messages.filter((message) => message.includes('[SystemOcrService]'))).toEqual([])
|
||||
})
|
||||
|
||||
it('单张的耗时与字数仍然通过返回值给出(设置页诊断不依赖日志)', async () => {
|
||||
const service = createService(createRuntime({ text: '本地图片文字识别' }))
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
|
||||
expect(result.success).toBe(true)
|
||||
expect(result.text).toBe('本地图片文字识别')
|
||||
expect(typeof result.durationMs).toBe('number')
|
||||
expect(result.durationMs).toBeGreaterThanOrEqual(0)
|
||||
})
|
||||
|
||||
it('失败时保留一条 warn,且只含 error code / engine / platform / duration', async () => {
|
||||
const service = createService(createRuntime({ error: 'Windows error 拒绝访问 (0x80070005)' }))
|
||||
const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined)
|
||||
try {
|
||||
await service.getCapability()
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
|
||||
const failedLines = warn.mock.calls
|
||||
.map((call) => String(call[0] ?? ''))
|
||||
.filter((message) => message.includes('[SystemOcrService] failed'))
|
||||
expect(failedLines).toHaveLength(1)
|
||||
// 绝不出现识别正文 / 图片内容 / 稳定标识。
|
||||
const joined = failedLines.join(' ')
|
||||
expect(joined).not.toContain('base64')
|
||||
expect(joined).not.toContain('data:image')
|
||||
} finally {
|
||||
warn.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user