mirror of
https://wget.la/https://github.com/Wxw-Gu/WechatExplorer
synced 2026-10-04 03:01:42 +08:00
feat: 新增 Windows 本地图片文字识别能力
This commit is contained in:
@@ -6,6 +6,7 @@ import { AIProviderCard } from '../../src/renderer/src/features/settings/ai-mode
|
||||
import { AIModelPage } from '../../src/renderer/src/features/settings/pages/AIModelPage'
|
||||
import type { AIProviderSummary, AIRuntimeModelConfig } from '../../src/shared/ai-provider'
|
||||
import type { AIVisionTestState } from '../../src/renderer/src/features/settings/ai-model/types'
|
||||
import type { SystemOcrCapability } from '../../src/shared/system-ocr'
|
||||
|
||||
const runtime: AIRuntimeModelConfig = {
|
||||
providerName: 'Not configured',
|
||||
@@ -15,6 +16,16 @@ const runtime: AIRuntimeModelConfig = {
|
||||
status: 'untested'
|
||||
}
|
||||
|
||||
const systemOcrCapability: SystemOcrCapability = {
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
arch: 'x64',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN',
|
||||
message: '本地图片文字识别可用(Windows 系统 OCR,zh-Hans-CN)。'
|
||||
}
|
||||
|
||||
const provider: AIProviderSummary = {
|
||||
id: 'fixture-provider',
|
||||
name: '本地假服务',
|
||||
@@ -56,7 +67,8 @@ describe('AI model settings', () => {
|
||||
})
|
||||
window.api = {
|
||||
listAIProviders: vi.fn().mockResolvedValue({ success: true, providers: [] }),
|
||||
getAIRuntimeConfig: vi.fn().mockResolvedValue(runtime)
|
||||
getAIRuntimeConfig: vi.fn().mockResolvedValue(runtime),
|
||||
getSystemOcrCapability: vi.fn().mockResolvedValue(systemOcrCapability)
|
||||
} as typeof window.api
|
||||
})
|
||||
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
import { fireEvent, render, screen, waitFor } from '@testing-library/react'
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { LocalImageTextRecognition } from '../../src/renderer/src/features/settings/ai-model/LocalImageTextRecognition'
|
||||
import type { SystemOcrCapability, SystemOcrResult } from '../../src/shared/system-ocr'
|
||||
|
||||
const availableCapability: SystemOcrCapability = {
|
||||
available: true,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
arch: 'x64',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN',
|
||||
message: '本地图片文字识别可用(Windows 系统 OCR,zh-Hans-CN)。'
|
||||
}
|
||||
|
||||
const unavailableCapability: SystemOcrCapability = {
|
||||
available: false,
|
||||
engine: 'windows-system-ocr',
|
||||
platform: 'win32',
|
||||
arch: 'x64',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: null,
|
||||
reason: 'LANGUAGE_UNAVAILABLE',
|
||||
message: '当前 Windows 未安装可用的 OCR 语言支持。'
|
||||
}
|
||||
|
||||
const successResult: SystemOcrResult = {
|
||||
success: true,
|
||||
text: 'TraceMemo 本地文字识别',
|
||||
lines: [],
|
||||
language: 'zh-Hans-CN',
|
||||
engine: 'windows-system-ocr',
|
||||
durationMs: 42
|
||||
}
|
||||
|
||||
const emptyResult: SystemOcrResult = {
|
||||
success: false,
|
||||
text: '',
|
||||
lines: [],
|
||||
language: 'zh-Hans-CN',
|
||||
engine: 'windows-system-ocr',
|
||||
durationMs: 12,
|
||||
errorCode: 'OCR_EMPTY_RESULT',
|
||||
error: '没有在这张图片里识别到文字。'
|
||||
}
|
||||
|
||||
const selectImage = (): void => {
|
||||
const input = document.querySelector<HTMLInputElement>('input[type="file"]')
|
||||
if (!input) throw new Error('file input missing')
|
||||
const file = new File([new Uint8Array([0x89, 0x50, 0x4e, 0x47])], 'fixture.png', {
|
||||
type: 'image/png'
|
||||
})
|
||||
fireEvent.change(input, { target: { files: [file] } })
|
||||
}
|
||||
|
||||
describe('LocalImageTextRecognition', () => {
|
||||
beforeEach(() => {
|
||||
window.api = {
|
||||
getSystemOcrCapability: vi.fn().mockResolvedValue(availableCapability),
|
||||
recognizeLocalImageText: vi.fn().mockResolvedValue(successResult)
|
||||
} as typeof window.api
|
||||
})
|
||||
|
||||
it('shows local availability without requiring any AI provider', async () => {
|
||||
render(<LocalImageTextRecognition />)
|
||||
expect(await screen.findByText('本机可用')).toBeInTheDocument()
|
||||
expect(screen.getByText(/组件 1\.2\.0/)).toBeInTheDocument()
|
||||
expect(screen.getByRole('button', { name: '本地文字识别' })).toBeDisabled()
|
||||
})
|
||||
|
||||
it('recognizes text locally and renders the extracted text', async () => {
|
||||
render(<LocalImageTextRecognition />)
|
||||
await screen.findByText('本机可用')
|
||||
|
||||
selectImage()
|
||||
const button = await screen.findByRole('button', { name: '本地文字识别' })
|
||||
await waitFor(() => expect(button).toBeEnabled())
|
||||
fireEvent.click(button)
|
||||
|
||||
expect(await screen.findByText('TraceMemo 本地文字识别')).toBeInTheDocument()
|
||||
expect(window.api.recognizeLocalImageText).toHaveBeenCalledWith({
|
||||
imageDataUrl: expect.stringContaining('data:image/png;base64,')
|
||||
})
|
||||
})
|
||||
|
||||
it('surfaces an empty OCR result as a product message, not a native error', async () => {
|
||||
vi.mocked(window.api.recognizeLocalImageText).mockResolvedValue(emptyResult)
|
||||
render(<LocalImageTextRecognition />)
|
||||
await screen.findByText('本机可用')
|
||||
|
||||
selectImage()
|
||||
const button = await screen.findByRole('button', { name: '本地文字识别' })
|
||||
await waitFor(() => expect(button).toBeEnabled())
|
||||
fireEvent.click(button)
|
||||
|
||||
expect(await screen.findByText('没有在这张图片里识别到文字。')).toBeInTheDocument()
|
||||
})
|
||||
|
||||
it('keeps the entry disabled and explains why when the language pack is missing', async () => {
|
||||
vi.mocked(window.api.getSystemOcrCapability).mockResolvedValue(unavailableCapability)
|
||||
render(<LocalImageTextRecognition />)
|
||||
|
||||
expect(await screen.findByText('本机不可用')).toBeInTheDocument()
|
||||
expect(screen.getByText('当前 Windows 未安装可用的 OCR 语言支持。')).toBeInTheDocument()
|
||||
expect(screen.getByRole('button', { name: '本地文字识别' })).toBeDisabled()
|
||||
})
|
||||
})
|
||||
Vendored
BIN
Binary file not shown.
|
After Width: | Height: | Size: 10 KiB |
BIN
Binary file not shown.
|
After Width: | Height: | Size: 20 KiB |
BIN
Binary file not shown.
|
After Width: | Height: | Size: 12 KiB |
Vendored
BIN
Binary file not shown.
|
After Width: | Height: | Size: 14 KiB |
@@ -0,0 +1,96 @@
|
||||
// Windows System OCR native fidelity。
|
||||
//
|
||||
// 这是 capability-gated 的原生冒烟测试:
|
||||
// - 只有在「当前平台支持 + native 运行时可用 + 有可用 OCR 语言包」时才真正跑;
|
||||
// - CI 环境无法保证 Windows OCR 语言包,所以中文识别不作为所有 CI 的硬门槛
|
||||
// (mock 单元测试才是 mandatory,见 tests/unit/system-ocr-service.test.ts);
|
||||
// - 在 Windows 真机上必须实际通过。
|
||||
//
|
||||
// fixture 全部是 synthetic 图片(tests/fixtures/ocr/*),不含任何真实聊天数据。
|
||||
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
import { SYSTEM_OCR_ENGINE } from '../../src/shared/system-ocr'
|
||||
|
||||
vi.mock('../../src/main/image-decrypt-service', () => ({
|
||||
resolveFfmpegExecutable: (): string => 'ffmpeg'
|
||||
}))
|
||||
|
||||
import { systemOcrService } from '../../src/main/services/system-ocr-service'
|
||||
|
||||
const fixtureDirectory = join(__dirname, '..', 'fixtures', 'ocr')
|
||||
|
||||
const toDataUrl = (fileName: string, mimeType: string): string =>
|
||||
`data:${mimeType};base64,${readFileSync(join(fixtureDirectory, fileName)).toString('base64')}`
|
||||
|
||||
/** 只比较"主要 token",避免系统字体 / 识别微差造成脆弱测试。 */
|
||||
const expectContainsTokens = (text: string, tokens: string[]): void => {
|
||||
const normalized = text.replace(/[\s\u3000]+/g, '').toLowerCase()
|
||||
for (const token of tokens) {
|
||||
expect(normalized).toContain(token.replace(/[\s\u3000]+/g, '').toLowerCase())
|
||||
}
|
||||
}
|
||||
|
||||
const capability = await systemOcrService.getCapability()
|
||||
const nativeGate = capability.available ? it : it.skip
|
||||
|
||||
describe('Windows System OCR native fidelity', () => {
|
||||
it('reports a usable capability on this machine', () => {
|
||||
expect(capability.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
if (!capability.available) {
|
||||
console.warn(`[integration] System OCR native smoke skipped: ${capability.message}`)
|
||||
}
|
||||
})
|
||||
|
||||
nativeGate('recognizes simplified Chinese text', async () => {
|
||||
const result = await systemOcrService.recognize({
|
||||
imageDataUrl: toDataUrl('system-ocr-zh.png', 'image/png')
|
||||
})
|
||||
expect(result.success).toBe(true)
|
||||
expectContainsTokens(result.text, ['TraceMemo', '本地', '文字', '识别'])
|
||||
// 语言要么是探测到的语言包,要么是"跟随系统用户语言"(null)。
|
||||
if (capability.language) {
|
||||
expect(result.language).toBe(capability.language)
|
||||
} else {
|
||||
expect(result.language).toBeNull()
|
||||
}
|
||||
expect(result.durationMs).toBeGreaterThan(0)
|
||||
})
|
||||
|
||||
nativeGate('recognizes English text', async () => {
|
||||
const result = await systemOcrService.recognize({
|
||||
imageDataUrl: toDataUrl('system-ocr-en.png', 'image/png')
|
||||
})
|
||||
expect(result.success).toBe(true)
|
||||
expectContainsTokens(result.text, ['TraceMemo', 'System', 'OCR'])
|
||||
})
|
||||
|
||||
nativeGate('recognizes mixed Chinese/English text', async () => {
|
||||
const result = await systemOcrService.recognize({
|
||||
imageDataUrl: toDataUrl('system-ocr-mixed.png', 'image/png')
|
||||
})
|
||||
expect(result.success).toBe(true)
|
||||
expectContainsTokens(result.text, ['TraceMemo', '本地', 'OCR', '2026'])
|
||||
})
|
||||
|
||||
/**
|
||||
* 引擎的 Buffer 输入只接受 PNG,所以 JPEG 必须走本服务的归一化路径。
|
||||
* 这条用例就是那个约束的回归保护。
|
||||
*/
|
||||
nativeGate('normalizes a JPEG source before OCR', async () => {
|
||||
const result = await systemOcrService.recognize({
|
||||
imageDataUrl: toDataUrl('system-ocr-mixed.jpg', 'image/jpeg')
|
||||
})
|
||||
expect(result.success).toBe(true)
|
||||
expectContainsTokens(result.text, ['TraceMemo', 'OCR', '2026'])
|
||||
})
|
||||
|
||||
nativeGate('caches an identical repeat request', async () => {
|
||||
const request = { imageDataUrl: toDataUrl('system-ocr-mixed.png', 'image/png') }
|
||||
const first = await systemOcrService.recognize(request)
|
||||
const second = await systemOcrService.recognize(request)
|
||||
expect(first.success).toBe(true)
|
||||
expect(second.fromCache).toBe(true)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,349 @@
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import {
|
||||
SYSTEM_OCR_ENGINE,
|
||||
SYSTEM_OCR_PROBE_PNG_BASE64,
|
||||
buildSystemOcrCacheKey,
|
||||
detectSystemOcrImageFormat,
|
||||
mapSystemOcrNativeError,
|
||||
normalizeSystemOcrText,
|
||||
parseImageDataUrl,
|
||||
resolveSystemOcrLanguageTag
|
||||
} from '../../src/shared/system-ocr'
|
||||
|
||||
vi.mock('../../src/main/image-decrypt-service', () => ({
|
||||
resolveFfmpegExecutable: (): string => 'ffmpeg'
|
||||
}))
|
||||
|
||||
const { getByHash, upsert } = vi.hoisted(() => ({
|
||||
getByHash: vi.fn(),
|
||||
upsert: vi.fn()
|
||||
}))
|
||||
|
||||
vi.mock('../../src/main/db/image-insights-store', () => ({
|
||||
imageInsightsStore: {
|
||||
getByHash,
|
||||
upsert,
|
||||
listBySession: vi.fn(() => [])
|
||||
}
|
||||
}))
|
||||
|
||||
import { SystemOcrService } from '../../src/main/services/system-ocr-service'
|
||||
import { imageInsightService } from '../../src/main/services/image-insight-service'
|
||||
|
||||
const PROBE_BYTES = Buffer.from(SYSTEM_OCR_PROBE_PNG_BASE64, 'base64')
|
||||
const FIXTURE_PNG = readFileSync(join(__dirname, '..', 'fixtures', 'ocr', 'system-ocr-zh.png'))
|
||||
const PNG_DATA_URL = `data:image/png;base64,${FIXTURE_PNG.toString('base64')}`
|
||||
const JPEG_DATA_URL = `data:image/jpeg;base64,${Buffer.from([0xff, 0xd8, 0xff, 0xe0, 0x00, 0x10]).toString('base64')}`
|
||||
const GARBAGE_DATA_URL = `data:image/png;base64,${Buffer.from('definitely-not-an-image').toString('base64')}`
|
||||
|
||||
const LANGUAGE_UNAVAILABLE_MESSAGE = 'Windows error 操作成功完成。 (0x00000000)'
|
||||
const DECODE_FAILED_MESSAGE = 'Windows error Could not recognize file (0x80070005)'
|
||||
|
||||
/**
|
||||
* mock 运行时按「探测图字节」区分 capability probe 和业务调用,
|
||||
* 这样 probeError 才能稳定复现「语言包缺失」的场景。
|
||||
*/
|
||||
const createRuntime = (
|
||||
options: {
|
||||
text?: string
|
||||
lines?: Array<{ text: string; confidence?: number }>
|
||||
probeError?: string
|
||||
error?: string
|
||||
version?: string | null
|
||||
} = {}
|
||||
): { version: string | null; recognize: ReturnType<typeof vi.fn> } => {
|
||||
const recognize = vi.fn(
|
||||
async (image: Uint8Array, _accuracy?: number, _languages?: string[]): Promise<unknown> => {
|
||||
if (Buffer.from(image).equals(PROBE_BYTES)) {
|
||||
if (options.probeError) throw new Error(options.probeError)
|
||||
return { text: '', confidence: 1, lines: [] }
|
||||
}
|
||||
if (options.error) throw new Error(options.error)
|
||||
const text = options.text ?? ''
|
||||
return {
|
||||
text,
|
||||
confidence: 1,
|
||||
lines: (options.lines ?? [{ text, confidence: 1 }]).map((line) => ({
|
||||
text: line.text,
|
||||
confidence: line.confidence ?? 1,
|
||||
boundingBox: { x: 0.1, y: 0.2, width: 0.3, height: 0.4 }
|
||||
}))
|
||||
}
|
||||
}
|
||||
)
|
||||
return { version: options.version === undefined ? '1.2.0' : options.version, recognize }
|
||||
}
|
||||
|
||||
const createService = (
|
||||
runtime: { version: string | null; recognize: ReturnType<typeof vi.fn> } | null,
|
||||
overrides: Partial<ConstructorParameters<typeof SystemOcrService>[0]> = {}
|
||||
): SystemOcrService =>
|
||||
new SystemOcrService({
|
||||
platform: 'win32',
|
||||
arch: 'x64',
|
||||
locale: () => 'zh-CN',
|
||||
loadRuntime: () => (runtime ? { ...runtime } : null),
|
||||
toPngBytes: async ({ buffer }) => buffer,
|
||||
...overrides
|
||||
})
|
||||
|
||||
describe('system-ocr shared helpers', () => {
|
||||
it('removes only the engine-inserted spaces between CJK glyphs', () => {
|
||||
expect(normalizeSystemOcrText('TraceMemo 本 地 图 片 文 字 识 别')).toBe(
|
||||
'TraceMemo 本地图片文字识别'
|
||||
)
|
||||
expect(normalizeSystemOcrText('TraceMemo System OCR')).toBe('TraceMemo System OCR')
|
||||
expect(normalizeSystemOcrText(' 本 地 ')).toBe('本地')
|
||||
})
|
||||
|
||||
it('maps system locale onto Windows OCR language tags', () => {
|
||||
expect(resolveSystemOcrLanguageTag('zh-CN')).toBe('zh-Hans-CN')
|
||||
expect(resolveSystemOcrLanguageTag('zh-Hans-CN')).toBe('zh-Hans-CN')
|
||||
expect(resolveSystemOcrLanguageTag('zh_TW')).toBe('zh-Hant-TW')
|
||||
expect(resolveSystemOcrLanguageTag('en-US')).toBe('en-US')
|
||||
expect(resolveSystemOcrLanguageTag('en')).toBe('en-US')
|
||||
expect(resolveSystemOcrLanguageTag('')).toBeNull()
|
||||
expect(resolveSystemOcrLanguageTag('xx-YY')).toBeNull()
|
||||
})
|
||||
|
||||
it('maps native Windows errors onto product error codes', () => {
|
||||
expect(mapSystemOcrNativeError(LANGUAGE_UNAVAILABLE_MESSAGE)).toBe('OCR_LANGUAGE_UNAVAILABLE')
|
||||
expect(mapSystemOcrNativeError(DECODE_FAILED_MESSAGE)).toBe('IMAGE_DECODE_FAILED')
|
||||
expect(mapSystemOcrNativeError('Cannot find native binding.')).toBe('SYSTEM_OCR_UNAVAILABLE')
|
||||
expect(mapSystemOcrNativeError('Failed to load native binding')).toBe('SYSTEM_OCR_UNAVAILABLE')
|
||||
expect(mapSystemOcrNativeError('Windows error something broke (0x80070057)')).toBe('OCR_FAILED')
|
||||
expect(mapSystemOcrNativeError('')).toBe('OCR_FAILED')
|
||||
})
|
||||
|
||||
it('parses image data urls and rejects other payloads', () => {
|
||||
expect(parseImageDataUrl(PNG_DATA_URL)).toMatchObject({ mimeType: 'image/png' })
|
||||
expect(parseImageDataUrl('data:text/plain;base64,aGk=')).toBeNull()
|
||||
expect(parseImageDataUrl('not-a-data-url')).toBeNull()
|
||||
})
|
||||
|
||||
it('detects supported container formats by magic bytes', () => {
|
||||
expect(detectSystemOcrImageFormat(Buffer.from([0x89, 0x50, 0x4e, 0x47]))).toBe('png')
|
||||
expect(detectSystemOcrImageFormat(Buffer.from([0xff, 0xd8, 0xff, 0xe0]))).toBe('jpeg')
|
||||
expect(detectSystemOcrImageFormat(Buffer.from('GIF89a'))).toBe('gif')
|
||||
expect(detectSystemOcrImageFormat(Buffer.from('BM1234'))).toBe('bmp')
|
||||
expect(
|
||||
detectSystemOcrImageFormat(Buffer.concat([Buffer.from('RIFF'), Buffer.alloc(4), Buffer.from('WEBP')]))
|
||||
).toBe('webp')
|
||||
expect(detectSystemOcrImageFormat(Buffer.from([0x49, 0x49, 0x2a, 0x00]))).toBe('tiff')
|
||||
expect(detectSystemOcrImageFormat(Buffer.from('nope'))).toBeNull()
|
||||
})
|
||||
|
||||
it('keeps the local OCR cache keyspace separate from the vision imageHash', () => {
|
||||
const base = { imageHash: 'a'.repeat(32), language: 'zh-Hans-CN', runtimeVersion: '1.2.0' }
|
||||
const key = buildSystemOcrCacheKey({ ...base, platform: 'win32' })
|
||||
expect(key).not.toBe(base.imageHash)
|
||||
expect(key).toContain(SYSTEM_OCR_ENGINE)
|
||||
expect(key).toContain('zh-Hans-CN')
|
||||
expect(key).toContain('1.2.0')
|
||||
// 语言或运行时版本变化必须换 key,避免复用过期 / 跨引擎结果。
|
||||
expect(buildSystemOcrCacheKey({ ...base, language: 'en-US', platform: 'win32' })).not.toBe(key)
|
||||
expect(
|
||||
buildSystemOcrCacheKey({ ...base, runtimeVersion: '1.3.0', platform: 'win32' })
|
||||
).not.toBe(key)
|
||||
})
|
||||
})
|
||||
|
||||
describe('SystemOcrService capability detection', () => {
|
||||
it('reports available with the probed language on Windows', async () => {
|
||||
const service = createService(createRuntime())
|
||||
const capability = await service.getCapability()
|
||||
expect(capability).toMatchObject({
|
||||
available: true,
|
||||
engine: SYSTEM_OCR_ENGINE,
|
||||
platform: 'win32',
|
||||
arch: 'x64',
|
||||
runtimeVersion: '1.2.0',
|
||||
language: 'zh-Hans-CN'
|
||||
})
|
||||
})
|
||||
|
||||
it('is unavailable on unsupported platforms without loading a runtime', async () => {
|
||||
const loadRuntime = vi.fn(() => null)
|
||||
const service = createService(null, { platform: 'linux', loadRuntime })
|
||||
const capability = await service.getCapability()
|
||||
expect(capability.available).toBe(false)
|
||||
expect(capability.reason).toBe('UNSUPPORTED_PLATFORM')
|
||||
expect(loadRuntime).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('is unavailable when the native runtime cannot be loaded', async () => {
|
||||
const service = createService(null)
|
||||
const capability = await service.getCapability()
|
||||
expect(capability.available).toBe(false)
|
||||
expect(capability.reason).toBe('NATIVE_MODULE_MISSING')
|
||||
})
|
||||
|
||||
it('reports a missing Windows OCR language pack as LANGUAGE_UNAVAILABLE', async () => {
|
||||
const runtime = createRuntime({ probeError: LANGUAGE_UNAVAILABLE_MESSAGE })
|
||||
const service = createService(runtime)
|
||||
const capability = await service.getCapability()
|
||||
expect(capability.available).toBe(false)
|
||||
expect(capability.reason).toBe('LANGUAGE_UNAVAILABLE')
|
||||
expect(capability.message).toContain('OCR 语言')
|
||||
})
|
||||
})
|
||||
|
||||
describe('SystemOcrService recognition', () => {
|
||||
beforeEach(() => {
|
||||
getByHash.mockReset()
|
||||
getByHash.mockReturnValue(null)
|
||||
upsert.mockReset()
|
||||
})
|
||||
|
||||
it('normalizes a successful result and reports runtime metadata', async () => {
|
||||
const runtime = createRuntime({
|
||||
text: 'TraceMemo 本 地 OCR 2026',
|
||||
lines: [{ text: 'TraceMemo 本 地 OCR 2026' }]
|
||||
})
|
||||
const service = createService(runtime)
|
||||
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: true,
|
||||
text: 'TraceMemo 本地 OCR 2026',
|
||||
language: 'zh-Hans-CN',
|
||||
engine: SYSTEM_OCR_ENGINE
|
||||
})
|
||||
expect(result.lines[0].boundingBox).toEqual({ x: 0.1, y: 0.2, width: 0.3, height: 0.4 })
|
||||
expect(result.durationMs).toBeGreaterThanOrEqual(0)
|
||||
})
|
||||
|
||||
it('returns OCR_EMPTY_RESULT when the engine finds no text', async () => {
|
||||
const service = createService(createRuntime({ text: '' }))
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('OCR_EMPTY_RESULT')
|
||||
expect(result.text).toBe('')
|
||||
})
|
||||
|
||||
it('returns UNSUPPORTED_PLATFORM on non-Windows platforms', async () => {
|
||||
const service = createService(null, { platform: 'darwin', arch: 'arm64' })
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('UNSUPPORTED_PLATFORM')
|
||||
expect(result.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
})
|
||||
|
||||
it('returns OCR_LANGUAGE_UNAVAILABLE when no OCR language pack is installed', async () => {
|
||||
const service = createService(createRuntime({ probeError: LANGUAGE_UNAVAILABLE_MESSAGE }))
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('OCR_LANGUAGE_UNAVAILABLE')
|
||||
})
|
||||
|
||||
it('rejects payloads that are not a supported image container', async () => {
|
||||
const runtime = createRuntime()
|
||||
const service = createService(runtime)
|
||||
|
||||
const badUrl = await service.recognize({ imageDataUrl: 'nope' })
|
||||
expect(badUrl.errorCode).toBe('UNSUPPORTED_IMAGE')
|
||||
|
||||
const garbage = await service.recognize({ imageDataUrl: GARBAGE_DATA_URL })
|
||||
expect(garbage.errorCode).toBe('UNSUPPORTED_IMAGE')
|
||||
expect(runtime.recognize).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('maps an image preparation failure onto IMAGE_DECODE_FAILED', async () => {
|
||||
const runtime = createRuntime()
|
||||
const service = createService(runtime, { toPngBytes: async () => null })
|
||||
const result = await service.recognize({ imageDataUrl: JPEG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('IMAGE_DECODE_FAILED')
|
||||
})
|
||||
|
||||
it('maps a native OCR failure onto a product error code', async () => {
|
||||
const runtime = createRuntime({ error: DECODE_FAILED_MESSAGE })
|
||||
const service = createService(runtime)
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.success).toBe(false)
|
||||
expect(result.errorCode).toBe('IMAGE_DECODE_FAILED')
|
||||
expect(result.error).not.toContain('0x80070005')
|
||||
})
|
||||
|
||||
it('caches by image identity + engine + language + runtime version only', async () => {
|
||||
const runtime = createRuntime({ text: 'TraceMemo 本 地' })
|
||||
const service = createService(runtime)
|
||||
|
||||
const first = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
const callsAfterFirst = runtime.recognize.mock.calls.length
|
||||
const second = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
|
||||
expect(first.success).toBe(true)
|
||||
expect(second.fromCache).toBe(true)
|
||||
expect(runtime.recognize.mock.calls.length).toBe(callsAfterFirst)
|
||||
|
||||
// 运行时版本升级 → 缓存 key 变化 → 必须重新识别,不能永远吃旧结果。
|
||||
const upgradedRuntime = createRuntime({ text: 'TraceMemo 本 地' })
|
||||
service.bind({
|
||||
loadRuntime: () => ({ version: '1.3.0', recognize: upgradedRuntime.recognize })
|
||||
})
|
||||
const afterUpgrade = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
|
||||
expect(afterUpgrade.success).toBe(true)
|
||||
expect(afterUpgrade.fromCache).toBeUndefined()
|
||||
expect(upgradedRuntime.recognize).toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('never reuses a remote Vision insight as a local OCR result', async () => {
|
||||
getByHash.mockReturnValue({
|
||||
imageHash: 'a'.repeat(32),
|
||||
description: '远端 Vision 旧结果',
|
||||
ocrText: '远端 OCR 文本',
|
||||
updatedAt: Date.now()
|
||||
})
|
||||
const runtime = createRuntime({ text: '本地 文 字' })
|
||||
const service = createService(runtime)
|
||||
|
||||
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
|
||||
|
||||
expect(result.text).toBe('本地文字')
|
||||
expect(getByHash).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
describe('ImageInsightService local OCR orchestration', () => {
|
||||
beforeEach(() => {
|
||||
getByHash.mockReset()
|
||||
getByHash.mockReturnValue(null)
|
||||
upsert.mockReset()
|
||||
})
|
||||
|
||||
it('exposes capability and never falls back to the remote vision provider', async () => {
|
||||
const analyzeImage = vi.fn(async () => ({ success: true, data: '{}' }))
|
||||
imageInsightService.bind({
|
||||
providerService: {
|
||||
list: () => ({ providers: [], defaultProviderId: 'vision-provider' }),
|
||||
getVisionRuntimeConfig: () => ({
|
||||
providerId: 'vision-provider',
|
||||
providerName: 'OpenAI',
|
||||
model: 'gpt-vision',
|
||||
modelName: 'gpt-vision',
|
||||
configured: true
|
||||
}),
|
||||
analyzeImage
|
||||
},
|
||||
decryptService: {
|
||||
findImageFile: () => null,
|
||||
decryptImageToBase64: () => null
|
||||
}
|
||||
})
|
||||
|
||||
const capability = await imageInsightService.getSystemOcrCapability()
|
||||
expect(capability.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
|
||||
const result = await imageInsightService.extractLocalText({ imageDataUrl: PNG_DATA_URL })
|
||||
expect(result.engine).toBe(SYSTEM_OCR_ENGINE)
|
||||
// 关键约束:本地 OCR 路径绝不调用远端 Vision Provider。
|
||||
expect(analyzeImage).not.toHaveBeenCalled()
|
||||
// 也不写 Vision 的 insight 缓存。
|
||||
expect(upsert).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
Reference in New Issue
Block a user