feat: 新增 Windows 本地图片文字识别能力

This commit is contained in:
电摇小子
2026-09-16 00:23:23 +08:00
parent e6db4de711
commit 24399f1d70
23 changed files with 1984 additions and 7 deletions
+2
View File
@@ -19,6 +19,8 @@ asarUnpack:
- node_modules/silk-wasm/**
- node_modules/sherpa-onnx-node/**
- node_modules/sherpa-onnx-*/**
- node_modules/@napi-rs/system-ocr/**
- node_modules/@napi-rs/system-ocr-*/**
extraResources:
# Keep the updater provider in every packaged Windows app. electron-builder also
# regenerates this file during publish, using the same release configuration.
+1 -1
View File
@@ -16,7 +16,7 @@ export default defineConfig({
output: {
entryFileNames: '[name].js'
},
external: ['koffi', 'sherpa-onnx-node']
external: ['koffi', 'sherpa-onnx-node', '@napi-rs/system-ocr']
}
}
},
+2
View File
@@ -79,6 +79,8 @@
"@electron-toolkit/preload": "^3.0.2",
"@electron-toolkit/utils": "^4.0.0",
"@koromix/koffi-win32-x64": "3.1.0",
"@napi-rs/system-ocr": "1.2.0",
"@napi-rs/system-ocr-win32-x64-msvc": "1.2.0",
"@radix-ui/react-alert-dialog": "^1.1.23",
"@radix-ui/react-checkbox": "^1.3.11",
"@radix-ui/react-dialog": "^1.1.23",
+42
View File
@@ -12,6 +12,8 @@ specifiers:
'@electron-toolkit/tsconfig': ^2.0.0
'@electron-toolkit/utils': ^4.0.0
'@koromix/koffi-win32-x64': 3.1.0
'@napi-rs/system-ocr': 1.2.0
'@napi-rs/system-ocr-win32-x64-msvc': 1.2.0
'@playwright/test': ^1.62.1
'@radix-ui/react-alert-dialog': ^1.1.23
'@radix-ui/react-checkbox': ^1.3.11
@@ -86,6 +88,8 @@ dependencies:
'@electron-toolkit/preload': 3.0.2_electron@43.1.0
'@electron-toolkit/utils': 4.0.0_electron@43.1.0
'@koromix/koffi-win32-x64': 3.1.0
'@napi-rs/system-ocr': 1.2.0
'@napi-rs/system-ocr-win32-x64-msvc': 1.2.0
'@radix-ui/react-alert-dialog': 1.1.23_eijghdl4n2x4hz6j4cg7ctgbuu
'@radix-ui/react-checkbox': 1.3.11_eijghdl4n2x4hz6j4cg7ctgbuu
'@radix-ui/react-dialog': 1.1.23_eijghdl4n2x4hz6j4cg7ctgbuu
@@ -1685,6 +1689,44 @@ packages:
- supports-color
dev: true
/@napi-rs/system-ocr/1.2.0:
resolution: {integrity: sha512-r0f2xNH6U+sth44qF+lUP+2WuHSGUBAry5KSCNuaLDGRbgslFqeROr/qJJ/fb6AjBp3Ov+CJP5MdrOWoaoM3cw==}
engines: {node: '>= 10'}
optionalDependencies:
'@napi-rs/system-ocr-darwin-arm64': 1.2.0
'@napi-rs/system-ocr-darwin-x64': 1.2.0
'@napi-rs/system-ocr-win32-arm64-msvc': 1.2.0
'@napi-rs/system-ocr-win32-x64-msvc': 1.2.0
dev: false
/@napi-rs/system-ocr-darwin-arm64/1.2.0:
resolution: {integrity: sha512-cK8dcDBEl3P4A04xmFJSHEJQxfDytaAIFyDCLqavTp92FVU5plESttWzZsqtTkS81/kzKiBfHyPQffSIndfWbQ==}
cpu: [arm64]
os: [darwin]
engines: {node: '>= 10'}
dev: false
/@napi-rs/system-ocr-darwin-x64/1.2.0:
resolution: {integrity: sha512-u3TBvBGrhmT5Os6AfaxbUEg6VHe8lvrFJNPgThJgshJHyRXUx/wCfTyOroJ22KdVCP5AE4GpwS5tFHMb6p6iaQ==}
cpu: [x64]
os: [darwin]
engines: {node: '>= 10'}
dev: false
/@napi-rs/system-ocr-win32-arm64-msvc/1.2.0:
resolution: {integrity: sha512-7ej8uMvmXomw3NXo5gZ5p2Nl6UKsHI+VRU3ELv0mhcxR0sJ6wFifYTu5bJrM1TGcz1/RsaX+TjWMmsDq8vriKQ==}
cpu: [arm64]
os: [win32]
engines: {node: '>= 10'}
dev: false
/@napi-rs/system-ocr-win32-x64-msvc/1.2.0:
resolution: {integrity: sha512-oOoCj3FPWDVctTxx98vMBiMI6m51U+w7SMmMefvmtpcpLelzZ/zYTqdwtWZFAjShaHO+RdaKkcpeVcQuBQiVbA==}
cpu: [x64]
os: [win32]
engines: {node: '>= 10'}
dev: false
/@nodelib/fs.scandir/2.1.5:
resolution: {integrity: sha512-vq24Bq3ym5HEQm2NKCr3yXDwjc7vTsEThRDnkp2DK9p1uqLR+DHurm/NOTo0KG7HYHU7eppKZj3MyqYuMBf62g==}
engines: {node: '>= 8'}
+54 -4
View File
@@ -78,6 +78,44 @@ function validateSherpaRuntime(runtimeResources, platform, arch) {
}
}
/**
* System OCR 用 native package(@napi-rs/system-ocr)。它是 external + asarUnpack,
* 打包后必须以 unpacked 形式存在,否则运行时会 MODULE_NOT_FOUND / native binding missing。
* 本轮只有 Windows 是 supported target,所以只在 Windows 上做硬校验。
*/
function systemOcrTarget(platform, arch) {
return platform === 'win32' ? `${platform}-${arch}-msvc` : `${platform}-${arch}`
}
function validateSystemOcrRuntime(runtimeResources, platform, arch) {
if (platform !== 'win32') return
const target = systemOcrTarget(platform, arch)
const basePath = path.join(
runtimeResources,
'app.asar.unpacked',
'node_modules',
'@napi-rs',
'system-ocr'
)
const nativePath = path.join(
runtimeResources,
'app.asar.unpacked',
'node_modules',
'@napi-rs',
`system-ocr-${target}`
)
const requiredFiles = [
path.join(basePath, 'package.json'),
path.join(basePath, 'index.js'),
path.join(nativePath, 'package.json'),
path.join(nativePath, `system-ocr.${target}.node`)
]
const missingFiles = requiredFiles.filter((filePath) => !existsSync(filePath))
if (missingFiles.length > 0) {
throw new Error(`Missing unpacked System OCR runtime: ${missingFiles.join(', ')}`)
}
}
function normalizeBuilderArch(arch) {
if (typeof arch === 'string') return arch
return { 0: 'ia32', 1: 'x64', 2: 'armv7l', 3: 'arm64', 4: 'universal' }[arch] || String(arch)
@@ -152,16 +190,24 @@ function validateReaderSkillRuntime(runtimeResources) {
* The loaders pick their package from process.platform/arch, so the siblings
* are dead weight — drop them.
*/
// 每个条目返回 platform package 的**完整后缀**(不含 package 前缀与连字符)。
const NATIVE_RUNTIME_PACKAGES = [
{
modules: [],
prefix: 'sherpa-onnx',
platformName: (platform) => (platform === 'win32' ? 'win' : platform)
platformName: (platform, arch) => `${platform === 'win32' ? 'win' : platform}-${arch}`
},
{
modules: ['@koromix'],
prefix: 'koffi',
platformName: (platform) => platform
platformName: (platform, arch) => `${platform}-${arch}`
},
{
// @napi-rs 的 platform package 目录名带 -msvc 后缀(win32-x64-msvc)。
modules: ['@napi-rs'],
prefix: 'system-ocr',
platformName: (platform, arch) => systemOcrTarget(platform, arch),
foreignPattern: /^system-ocr-[a-z0-9]+-(arm64|x64|ia32|loong64|riscv64)(-msvc)?$/
}
]
@@ -173,8 +219,10 @@ function pruneForeignArchNativeRuntimes(runtimeResources, platform, arch) {
for (const runtime of NATIVE_RUNTIME_PACKAGES) {
const modulesRoot = path.join(unpackedRoot, ...runtime.modules)
if (!existsSync(modulesRoot)) continue
const expected = `${runtime.prefix}-${runtime.platformName(platform)}-${arch}`
const foreign = new RegExp(`^${runtime.prefix}-[a-z0-9]+-(arm64|x64|ia32|loong64|riscv64)$`)
const expected = `${runtime.prefix}-${runtime.platformName(platform, arch)}`
const foreign =
runtime.foreignPattern ||
new RegExp(`^${runtime.prefix}-[a-z0-9]+-(arm64|x64|ia32|loong64|riscv64)$`)
for (const entry of readdirSync(modulesRoot, { withFileTypes: true })) {
if (!entry.isDirectory() || entry.name === expected || !foreign.test(entry.name)) continue
rmSync(path.join(modulesRoot, entry.name), { recursive: true, force: true })
@@ -223,6 +271,7 @@ exports.default = async function afterPack(context) {
'Bundled ffmpeg'
)
validateSherpaRuntime(runtimeResources, context.electronPlatformName, arch)
validateSystemOcrRuntime(runtimeResources, context.electronPlatformName, arch)
pruneIntelMacKeyTool(runtimeResources, context.electronPlatformName, arch)
pruneForeignArchConnectors(runtimeResources, context.electronPlatformName, arch)
pruneForeignArchNativeRuntimes(runtimeResources, context.electronPlatformName, arch)
@@ -258,6 +307,7 @@ exports.validateReaderSkillRuntime = validateReaderSkillRuntime
exports.validateFfmpegRuntime = validateFfmpegRuntime
exports.validateSilkWasmRuntime = validateSilkWasmRuntime
exports.validateSherpaRuntime = validateSherpaRuntime
exports.validateSystemOcrRuntime = validateSystemOcrRuntime
exports.pruneIntelMacKeyTool = pruneIntelMacKeyTool
exports.pruneForeignArchConnectors = pruneForeignArchConnectors
exports.pruneForeignArchNativeRuntimes = pruneForeignArchNativeRuntimes
+5 -1
View File
@@ -114,7 +114,11 @@ function getFfmpegCandidates(selectedPath = loadSettings().ffmpegPath): FfmpegCa
)
}
function resolveFfmpegExecutable(): string {
/**
* 解析可用的 ffmpeg 可执行文件。除图片解密自身使用外,也供 System OCR 的
* 图片归一化(GIF/BMP/WebP/TIFF → PNG)复用,避免重复一套路径探测逻辑。
*/
export function resolveFfmpegExecutable(): string {
for (const candidate of getFfmpegCandidates()) {
const pathLike = candidate.executable.includes('/') || candidate.executable.includes('\\')
if (pathLike) {
+30
View File
@@ -29,6 +29,7 @@ import {
ImageDecryptService,
inspectImageDecoderExecutable,
inspectImageDecoderStatus,
resolveFfmpegExecutable,
type DecodedImage
} from './image-decrypt-service'
import {
@@ -64,6 +65,7 @@ import { apiTokenStore } from './api-token-store'
import { ImageKeyConfigService } from './services/image-key-config-service'
import { AIProviderService } from './services/ai-provider-service'
import { imageInsightService } from './services/image-insight-service'
import { systemOcrService } from './services/system-ocr-service'
import type {
ImageAnalysisRequest,
ImageAnalysisResponse,
@@ -71,6 +73,7 @@ import type {
ImageCandidateQuery,
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import { KeyServiceMac } from './key-service-mac'
import { KeyService as KeyServiceWin } from './key-service-win'
import * as chat from './services/chat-service'
@@ -1814,6 +1817,13 @@ app.whenReady().then(async () => {
}
})
// System OCR 是独立的本地 Runtime(不是 AI Provider):只注入项目统一的 ffmpeg
// 解析逻辑(GIF/BMP/WebP/TIFF → PNG 归一化)和系统 locale(OCR 语言包探测)。
systemOcrService.bind({
resolveFfmpegExecutable,
locale: () => app.getLocale()
})
/** 日报入口:取会话 Top N 热点图片 + 已缓存的 Insight */
ipcMain.handle(
'image:listCandidates',
@@ -1885,6 +1895,26 @@ app.whenReady().then(async () => {
}
)
// ============================================================
// 本地图片文字识别(System OCR / Windows System OCR Runtime)
// ============================================================
// 这是本地 Runtime,不是 AI Vision Provider:
// - 不联网、不上传原图;
// - 不读写 AI Provider / Vision 模型配置;
// - 结果不落库(派生内容,本轮只做内存级闭环)。
ipcMain.handle('system-ocr:getCapability', async (): Promise<SystemOcrCapability> => {
return imageInsightService.getSystemOcrCapability()
})
ipcMain.handle(
'system-ocr:recognize',
async (_, request: SystemOcrRequest): Promise<SystemOcrResult> => {
// 日志只记录结构性信息,不记录 base64、不记录识别正文。
console.log('[IPC] system-ocr:recognize hash=%s', request?.imageHash || 'auto')
return imageInsightService.extractLocalText(request)
}
)
ipcMain.handle('db:getSticker', async (_, cdnUrl?: string, md5?: string) => {
if (!stickerService) {
stickerService = new StickerService(chat.getChatDb()?.getWcdb4Client())
@@ -27,6 +27,8 @@ import {
isFreshImageInsight,
isHotImageCandidate
} from '../../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../../shared/system-ocr'
import { systemOcrService } from './system-ocr-service'
/**
* 单张图片的最小信息(由 renderer 从已加载的 messages 中提取并传入 main)。
@@ -312,6 +314,35 @@ class ImageInsightService {
listBySession(sessionId: string, limit?: number): ImageInsight[] {
return imageInsightsStore.listBySession(sessionId, limit)
}
// ============================================================
// 本地图片文字识别(System OCR)
// ============================================================
//
// 与 Vision 路径的关系:
// ImageInsightService 是统一编排入口,下面挂两条互不干扰的运行时——
// - Vision Model Runtime(AIProviderService,走 AI Provider,可能联网)
// - Windows System OCR Runtime(SystemOcrService,纯本地,不联网)
//
// 边界与约束:
// 1. 本地 OCR 结果属于 **派生内容**,原始消息始终是权威来源;
// 本轮不落库、不写 Knowledge、不做历史图片 backfill。
// 2. 本地 OCR 结果 **不会** 写入 image-insights.json——那是 Vision 结果的缓存,
// 两者的缓存键空间也不同(见 buildSystemOcrCacheKey)。
// 3. 这里不读取也绝不修改 AI Vision Provider / 模型配置。
/** 本机是否支持本地图片文字识别(Windows System OCR)。 */
getSystemOcrCapability(): Promise<SystemOcrCapability> {
return systemOcrService.getCapability()
}
/**
* 只做「把图片里的文字读出来」。不发网络请求,不动 AI Provider 配置。
* 失败不抛,返回带 errorCode 的结果。
*/
extractLocalText(request: SystemOcrRequest): Promise<SystemOcrResult> {
return systemOcrService.recognize(request)
}
}
export const imageInsightService = new ImageInsightService()
+558
View File
@@ -0,0 +1,558 @@
// src/main/services/system-ocr-service.ts
//
// System OCR Runtime(本地图片文字识别)。
//
// 职责边界(只做这些事):
// 1. capability detection
// 2. image normalization / preparation
// 3. OCR execution
// 4. result normalization
// 5. runtime metadata
// 6. error mapping
//
// 明确不做:
// - 不伪装成 AI Provider / Vision Model;不读写 AIVisionRuntimeConfig;
// - 不发任何网络请求;不上传原图;
// - 不遍历历史图片、不做 backfill、不写 Knowledge;
// - 不把 OCR 文本写进 image-insights.json(那是 Vision 结果的缓存)。
//
// Windows 后端:Windows.Media.Ocr.OcrEngine(经 @napi-rs/system-ocr)。
// 已实测的引擎行为(@napi-rs/system-ocr 1.2.0 / Electron 43 / Windows x64):
// - Buffer 输入只接受 PNG;JPEG / WEBP / BMP 会被判为不可识别,
// 所以本服务在边界上统一归一化成 PNG 字节再调用(不落盘)。
// - preferredLangs 只使用第一个语言;语言包缺失时引擎创建失败,
// 抛出的错误是 `Windows error 操作成功完成。 (0x00000000)`(HRESULT 为 S_OK)。
// - 空白图不会报错,返回空文本 → 映射成 OCR_EMPTY_RESULT。
// - CJK 字符之间会被引擎插入空格,结果里做归一化。
import crypto from 'node:crypto'
import { spawn } from 'node:child_process'
import {
SYSTEM_OCR_CACHE_TTL_MS,
SYSTEM_OCR_ENGINE,
SYSTEM_OCR_PROBE_PNG_BASE64,
buildSystemOcrCacheKey,
detectSystemOcrImageFormat,
mapSystemOcrNativeError,
normalizeSystemOcrText,
parseImageDataUrl,
resolveSystemOcrLanguageTag
} from '../../shared/system-ocr'
import type {
SystemOcrCapability,
SystemOcrErrorCode,
SystemOcrImageFormat,
SystemOcrLine,
SystemOcrRequest,
SystemOcrResult
} from '../../shared/system-ocr'
const NATIVE_PACKAGE = '@napi-rs/system-ocr'
const MAX_CACHE_ENTRIES = 32
const FFMPEG_TIMEOUT_MS = 10_000
interface NativeLine {
text: string
confidence: number
boundingBox: { x: number; y: number; width: number; height: number }
}
interface NativeResult {
text: string
confidence: number
lines: NativeLine[]
}
interface NativeRuntime {
version: string | null
recognize: (image: Uint8Array, accuracy?: number, languages?: string[]) => Promise<NativeResult>
}
export interface SystemOcrServiceDeps {
/** 加载 native 运行时;不可用时返回 null(不允许抛) */
loadRuntime?: () => NativeRuntime | null
/** 把输入图片转成 PNG 字节;失败返回 null */
toPngBytes?: (input: {
buffer: Buffer
format: SystemOcrImageFormat
}) => Promise<Buffer | null>
/**
* ffmpeg 可执行文件解析器。只用于 GIF/BMP/WebP/TIFF → PNG 的兜底归一化。
* main/index.ts 会注入项目统一的解析逻辑(与图片解密共用一套候选路径)。
*/
resolveFfmpegExecutable?: () => string
platform?: NodeJS.Platform
arch?: string
/** 系统 locale(如 zh-CN),用于推导 OCR 语言标签 */
locale?: () => string
}
/** 未被显式注入时的兜底:环境变量 → 打包内 ffmpeg-static → PATH。 */
const defaultResolveFfmpegExecutable = (): string => {
const fromEnvironment = String(process.env['FFMPEG_BIN'] || '').trim()
if (fromEnvironment) return fromEnvironment
try {
const bundled = require('ffmpeg-static') as string | null
if (bundled) return bundled
} catch {
// 忽略:退回到 PATH 上的 ffmpeg
}
return process.platform === 'win32' ? 'ffmpeg.exe' : 'ffmpeg'
}
const toLines = (lines: NativeLine[] | undefined): SystemOcrLine[] =>
Array.isArray(lines)
? lines.map((line) => ({
text: normalizeSystemOcrText(line.text),
confidence: typeof line.confidence === 'number' ? line.confidence : 1,
boundingBox: {
x: Number(line.boundingBox?.x ?? 0),
y: Number(line.boundingBox?.y ?? 0),
width: Number(line.boundingBox?.width ?? 0),
height: Number(line.boundingBox?.height ?? 0)
}
}))
: []
const failure = (
errorCode: SystemOcrErrorCode,
error: string,
startedAt: number
): SystemOcrResult => ({
success: false,
text: '',
lines: [],
language: null,
engine: SYSTEM_OCR_ENGINE,
durationMs: Date.now() - startedAt,
errorCode,
error
})
/** 把任意容器(gif/bmp/webp/tiff)用 ffmpeg 走内存管道转成 PNG。不落盘。 */
const convertWithFfmpeg = (buffer: Buffer, executable: string): Promise<Buffer | null> =>
new Promise((resolve) => {
let settled = false
const finish = (value: Buffer | null): void => {
if (settled) return
settled = true
resolve(value)
}
let child: ReturnType<typeof spawn>
try {
child = spawn(
executable,
[
'-hide_banner',
'-loglevel',
'error',
'-i',
'pipe:0',
'-frames:v',
'1',
'-f',
'image2pipe',
'-vcodec',
'png',
'pipe:1'
],
{ windowsHide: true }
)
} catch {
finish(null)
return
}
const chunks: Buffer[] = []
const timeout = setTimeout(() => {
try {
child.kill()
} catch {
// best-effort
}
finish(null)
}, FFMPEG_TIMEOUT_MS)
child.stdout?.on('data', (chunk: Buffer) => chunks.push(chunk))
child.on('error', () => {
clearTimeout(timeout)
finish(null)
})
child.on('close', (code) => {
clearTimeout(timeout)
finish(code === 0 && chunks.length > 0 ? Buffer.concat(chunks) : null)
})
child.stdin?.on('error', () => undefined)
child.stdin?.end(buffer)
})
class SystemOcrService {
private runtime: NativeRuntime | null = null
private runtimeLoaded = false
private capability: SystemOcrCapability | null = null
private capabilityPromise: Promise<SystemOcrCapability> | null = null
private readonly cache = new Map<string, { value: SystemOcrResult; expireAt: number }>()
private deps: SystemOcrServiceDeps = {}
constructor(deps: SystemOcrServiceDeps = {}) {
this.deps = deps
}
/** 由 main/index.ts 在 app ready 后调用(可选,用于注入 app.getLocale 等)。 */
bind(deps: SystemOcrServiceDeps): void {
this.deps = { ...this.deps, ...deps }
this.runtime = null
this.runtimeLoaded = false
this.capability = null
this.capabilityPromise = null
}
/** 仅测试用:清空探测与缓存状态。 */
reset(): void {
this.runtime = null
this.runtimeLoaded = false
this.capability = null
this.capabilityPromise = null
this.cache.clear()
}
private get platform(): NodeJS.Platform {
return this.deps.platform ?? process.platform
}
private get arch(): string {
return this.deps.arch ?? process.arch
}
private get locale(): string {
if (this.deps.locale) {
try {
return this.deps.locale()
} catch {
return ''
}
}
try {
const { app } = require('electron') as typeof import('electron')
return app?.getLocale?.() ?? ''
} catch {
return ''
}
}
private loadRuntime(): NativeRuntime | null {
if (this.runtimeLoaded) return this.runtime
this.runtimeLoaded = true
if (this.deps.loadRuntime) {
this.runtime = this.deps.loadRuntime()
return this.runtime
}
if (this.platform !== 'win32') {
this.runtime = null
return this.runtime
}
try {
// 原生模块必须在打包时 external + asarUnpack,否则这里会 MODULE_NOT_FOUND。
const nativeModule = require(NATIVE_PACKAGE) as {
recognize: NativeRuntime['recognize']
}
let version: string | null = null
try {
version = (require(`${NATIVE_PACKAGE}/package.json`) as { version?: string }).version ?? null
} catch {
version = null
}
this.runtime =
nativeModule && typeof nativeModule.recognize === 'function'
? { version, recognize: nativeModule.recognize.bind(nativeModule) }
: null
} catch (error) {
console.warn(
'[SystemOcrService] native runtime unavailable engine=%s platform=%s reason=%s',
SYSTEM_OCR_ENGINE,
this.platform,
error instanceof Error ? error.message.split('\n')[0] : String(error)
)
this.runtime = null
}
return this.runtime
}
private async toPngBytes(
buffer: Buffer,
format: SystemOcrImageFormat
): Promise<Buffer | null> {
if (this.deps.toPngBytes) return this.deps.toPngBytes({ buffer, format })
if (format === 'png') return buffer
if (format === 'jpeg') {
// 项目内已有的进程内解码能力,优先于 ffmpeg(更快、无子进程)。
try {
const { nativeImage } = require('electron') as typeof import('electron')
const image = nativeImage.createFromBuffer(buffer)
if (!image.isEmpty()) {
const png = image.toPNG()
if (png && png.length > 0) return png
}
} catch {
// 继续走 ffmpeg 兜底
}
}
try {
const resolveFfmpeg = this.deps.resolveFfmpegExecutable ?? defaultResolveFfmpegExecutable
const png = await convertWithFfmpeg(buffer, resolveFfmpeg())
return png
} catch {
return null
}
}
/** capability 探测:平台 → native 运行时 → 至少一个可用 OCR 语言。 */
async getCapability(force = false): Promise<SystemOcrCapability> {
if (!force && this.capability) return this.capability
if (!force && this.capabilityPromise) return this.capabilityPromise
this.capabilityPromise = this.detectCapability()
try {
this.capability = await this.capabilityPromise
} finally {
this.capabilityPromise = null
}
return this.capability
}
private async detectCapability(): Promise<SystemOcrCapability> {
const base: Pick<
SystemOcrCapability,
'engine' | 'platform' | 'arch' | 'runtimeVersion' | 'language'
> = {
engine: SYSTEM_OCR_ENGINE,
platform: this.platform,
arch: this.arch,
runtimeVersion: null,
language: null
}
if (this.platform !== 'win32') {
return {
...base,
available: false,
reason: 'UNSUPPORTED_PLATFORM',
message: '本地图片文字识别目前仅支持 Windows。'
}
}
const runtime = this.loadRuntime()
if (!runtime) {
return {
...base,
available: false,
reason: 'NATIVE_MODULE_MISSING',
message: '本地文字识别组件不可用,请重新安装 TraceMemo。'
}
}
const probed = await this.probeLanguage(runtime)
if (probed.reason) {
return {
...base,
runtimeVersion: runtime.version,
available: false,
reason: probed.reason,
message: probed.message
}
}
return {
...base,
runtimeVersion: runtime.version,
available: true,
language: probed.language,
message: probed.language
? `本地图片文字识别可用(Windows 系统 OCR,${probed.language})。`
: '本地图片文字识别可用(Windows 系统 OCR,跟随系统语言)。'
}
}
/**
* 用一个 64x32 纯白 PNG 探测语言可用性:引擎能创建即说明语言包可用。
* 首选「系统 locale 推导出的标签」,失败再退回「系统用户语言配置」。
*/
private async probeLanguage(
runtime: NativeRuntime
): Promise<
| { language: string | null; reason?: undefined; message?: undefined }
| { language: null; reason: 'LANGUAGE_UNAVAILABLE' | 'NATIVE_MODULE_MISSING'; message: string }
> {
const probeBuffer = Buffer.from(SYSTEM_OCR_PROBE_PNG_BASE64, 'base64')
const preferred = resolveSystemOcrLanguageTag(this.locale)
const candidates: Array<string | null> = preferred ? [preferred, null] : [null]
let lastCode: SystemOcrErrorCode = 'OCR_FAILED'
for (const candidate of candidates) {
try {
await runtime.recognize(
probeBuffer,
undefined,
candidate ? [candidate] : undefined
)
return { language: candidate }
} catch (error) {
lastCode = mapSystemOcrNativeError(
error instanceof Error ? error.message : String(error)
)
}
}
if (lastCode === 'OCR_LANGUAGE_UNAVAILABLE') {
return {
language: null,
reason: 'LANGUAGE_UNAVAILABLE',
message:
'当前 Windows 未安装可用的 OCR 语言支持,请在系统「语言和区域」里安装简体中文或英文的 OCR 语言包后重试。'
}
}
return {
language: null,
reason: 'NATIVE_MODULE_MISSING',
message: '本地文字识别引擎初始化失败,请重启 TraceMemo 或重新安装。'
}
}
private readCache(key: string, startedAt: number): SystemOcrResult | null {
const hit = this.cache.get(key)
if (!hit) return null
if (hit.expireAt <= Date.now()) {
this.cache.delete(key)
return null
}
return { ...hit.value, durationMs: Date.now() - startedAt, fromCache: true }
}
private writeCache(key: string, value: SystemOcrResult): void {
if (this.cache.size >= MAX_CACHE_ENTRIES) {
const oldest = this.cache.keys().next()
if (!oldest.done) this.cache.delete(oldest.value)
}
this.cache.set(key, { value, expireAt: Date.now() + SYSTEM_OCR_CACHE_TTL_MS })
}
/**
* 识别一张图片里的文字。
* 任意失败都不抛,统一返回 success=false + 产品级 errorCode。
*/
async recognize(request: SystemOcrRequest): Promise<SystemOcrResult> {
const startedAt = Date.now()
const parsed = parseImageDataUrl(request.imageDataUrl)
if (!parsed) {
return failure('UNSUPPORTED_IMAGE', '仅支持 PNG、JPG、JPEG、WebP、GIF、BMP 图片。', startedAt)
}
let sourceBuffer: Buffer
try {
sourceBuffer = Buffer.from(parsed.base64, 'base64')
} catch {
return failure('IMAGE_DECODE_FAILED', '图片数据无法解码。', startedAt)
}
if (sourceBuffer.length === 0) {
return failure('IMAGE_DECODE_FAILED', '图片数据为空。', startedAt)
}
const format = detectSystemOcrImageFormat(sourceBuffer)
if (!format) {
return failure('UNSUPPORTED_IMAGE', '无法识别的图片格式。', startedAt)
}
const imageHash =
request.imageHash?.trim() ||
crypto.createHash('sha256').update(sourceBuffer).digest('hex').slice(0, 32)
const requestedLanguage = request.language?.trim() || null
const capability = await this.getCapability()
if (!capability.available) {
const errorCode: SystemOcrErrorCode =
capability.reason === 'UNSUPPORTED_PLATFORM'
? 'UNSUPPORTED_PLATFORM'
: capability.reason === 'LANGUAGE_UNAVAILABLE'
? 'OCR_LANGUAGE_UNAVAILABLE'
: 'SYSTEM_OCR_UNAVAILABLE'
return failure(errorCode, capability.message, startedAt)
}
const languageForCache = requestedLanguage ?? capability.language
const cacheKey = buildSystemOcrCacheKey({
imageHash,
language: languageForCache,
runtimeVersion: capability.runtimeVersion,
platform: capability.platform
})
if (requestedLanguage === null) {
const cached = this.readCache(cacheKey, startedAt)
if (cached) return cached
}
const runtime = this.loadRuntime()
if (!runtime) {
return failure('SYSTEM_OCR_UNAVAILABLE', '本地文字识别组件不可用。', startedAt)
}
const png = await this.toPngBytes(sourceBuffer, format)
if (!png || png.length === 0 || !detectSystemOcrImageFormat(png)) {
return failure('IMAGE_DECODE_FAILED', '图片解码失败,无法读取这张图片。', startedAt)
}
const candidates: Array<string | null> = requestedLanguage
? [requestedLanguage]
: capability.language
? [capability.language, null]
: [null]
let lastErrorCode: SystemOcrErrorCode = 'OCR_FAILED'
let lastErrorMessage = ''
let usedLanguage: string | null = null
for (const candidate of candidates) {
try {
const result = await runtime.recognize(
png,
undefined,
candidate ? [candidate] : undefined
)
const text = normalizeSystemOcrText(result?.text ?? '')
const lines = toLines(result?.lines)
usedLanguage = candidate
if (!text) {
return failure('OCR_EMPTY_RESULT', '没有在这张图片里识别到文字。', startedAt)
}
const succeeded: SystemOcrResult = {
success: true,
text,
lines,
language: usedLanguage,
engine: SYSTEM_OCR_ENGINE,
durationMs: Date.now() - startedAt
}
// 生产日志只记录 error code / engine / platform / duration,绝不记录识别正文。
console.log(
'[SystemOcrService] ok engine=%s platform=%s language=%s chars=%d durationMs=%d',
SYSTEM_OCR_ENGINE,
capability.platform,
usedLanguage ?? 'system-default',
text.length,
succeeded.durationMs
)
this.writeCache(cacheKey, succeeded)
return succeeded
} catch (error) {
lastErrorMessage = error instanceof Error ? error.message : String(error)
lastErrorCode = mapSystemOcrNativeError(lastErrorMessage)
// 语言不可用才值得换下一个候选;其它错误直接结束,避免无意义重试。
if (lastErrorCode !== 'OCR_LANGUAGE_UNAVAILABLE') break
}
}
console.warn(
'[SystemOcrService] failed engine=%s platform=%s errorCode=%s durationMs=%d',
SYSTEM_OCR_ENGINE,
capability.platform,
lastErrorCode,
Date.now() - startedAt
)
return failure(
lastErrorCode,
lastErrorCode === 'OCR_LANGUAGE_UNAVAILABLE'
? '当前 Windows 未安装可用的 OCR 语言支持。'
: '本地文字识别失败,请稍后重试。',
startedAt
)
}
}
export { SystemOcrService }
export const systemOcrService = new SystemOcrService()
+4
View File
@@ -57,6 +57,7 @@ import type {
ImageCandidateQuery,
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import type { AgentHubActionResult, AgentHubLogEntry, AgentHubStatus } from '../shared/agent-hub'
import type {
PersonalWechatGeneratedTtsVoiceRequest,
@@ -660,6 +661,9 @@ declare global {
sessionId: string,
limit?: number
) => Promise<{ success: boolean; insights: ImageInsight[] }>
// 本地图片文字识别(System OCR,本地 Runtime,非 AI Provider)
getSystemOcrCapability: () => Promise<SystemOcrCapability>
recognizeLocalImageText: (request: SystemOcrRequest) => Promise<SystemOcrResult>
getPersonalWechatSenderStatus: () => Promise<PersonalWechatSenderStatus>
getPersonalWechatSendCapability: () => Promise<PersonalWechatSendCapability>
getPersonalWechatKeepOneBotProcess: () => Promise<boolean>
+6
View File
@@ -30,6 +30,7 @@ import type {
ImageCandidateQuery,
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import type { AgentHubLogEntry, AgentHubStatus } from '../shared/agent-hub'
import type {
PersonalWechatGeneratedTtsVoiceRequest,
@@ -462,6 +463,11 @@ const api = {
limit?: number
): Promise<{ success: boolean; insights: ImageInsight[] }> =>
ipcRenderer.invoke('image:listInsights', sessionId, limit),
// 本地图片文字识别(System OCR,本地 Runtime,非 AI Provider)
getSystemOcrCapability: (): Promise<SystemOcrCapability> =>
ipcRenderer.invoke('system-ocr:getCapability'),
recognizeLocalImageText: (request: SystemOcrRequest): Promise<SystemOcrResult> =>
ipcRenderer.invoke('system-ocr:recognize', request),
getPersonalWechatSenderStatus: (): Promise<PersonalWechatSenderStatus> =>
ipcRenderer.invoke('wechat-personal:getStatus'),
getPersonalWechatSendCapability: (): Promise<PersonalWechatSendCapability> =>
@@ -0,0 +1,255 @@
import { useCallback, useEffect, useState } from 'react'
import type { SystemOcrCapability, SystemOcrResult } from '../../../../../shared/system-ocr'
import { Button } from '../../../components/ui'
const MAX_FILE_BYTES = 10 * 1024 * 1024
type LocalOcrStatus = 'idle' | 'reading' | 'ready' | 'running' | 'done' | 'error'
interface LocalOcrState {
status: LocalOcrStatus
image?: {
dataUrl: string
fileName: string
size: number
}
result?: SystemOcrResult
error?: string
}
/**
* 本地图片文字识别(Windows 系统 OCR)。
*
* 这是**本地 Runtime**,不是 AI 图片理解:
* - 只把图片里的文字读出来;不描述画面、人物、场景,也不做视觉推理;
* - 原始图片不会因为这一步发给任何 AI Provider;
* - 结果只是派生内容,不会写进本地知识库。
*/
export function LocalImageTextRecognition(): React.ReactElement {
const [capability, setCapability] = useState<SystemOcrCapability | null>(null)
const [state, setState] = useState<LocalOcrState>({ status: 'idle' })
useEffect(() => {
let alive = true
void window.api
.getSystemOcrCapability()
.then((value) => {
if (alive) setCapability(value)
})
.catch(() => {
if (alive) setCapability(null)
})
return () => {
alive = false
}
}, [])
const selectImage = useCallback(async (file: File): Promise<void> => {
const extension = file.name.split('.').pop()?.toLowerCase()
const inferredType =
extension === 'png'
? 'image/png'
: extension === 'jpg' || extension === 'jpeg'
? 'image/jpeg'
: extension === 'webp'
? 'image/webp'
: extension === 'bmp'
? 'image/bmp'
: extension === 'gif'
? 'image/gif'
: ''
const mimeType = file.type === 'image/jpg' ? 'image/jpeg' : file.type || inferredType
const supportedTypes = new Set([
'image/png',
'image/jpeg',
'image/webp',
'image/bmp',
'image/gif'
])
if (!supportedTypes.has(mimeType)) {
setState({ status: 'error', error: '请选择 PNG、JPG、JPEG、WebP、BMP 或 GIF 图片' })
return
}
if (!file.size || file.size > MAX_FILE_BYTES) {
setState({ status: 'error', error: '图片大小必须在 10 MB 以内' })
return
}
setState({ status: 'reading' })
try {
const rawDataUrl = await readFileAsDataUrl(file)
const dataUrl = rawDataUrl.replace(/^data:[^;]*;/, `data:${mimeType};`)
setState({
status: 'ready',
image: { dataUrl, fileName: file.name, size: file.size }
})
} catch {
setState({ status: 'error', error: '图片无法读取,请重新选择' })
}
}, [])
const run = useCallback(async (): Promise<void> => {
const image = state.image
if (!image) return
if (!capability?.available) {
setState((current) => ({
...current,
status: 'error',
error: capability?.message || '本机当前不支持本地图片文字识别'
}))
return
}
setState((current) => ({ ...current, status: 'running', result: undefined, error: undefined }))
try {
const result = await window.api.recognizeLocalImageText({ imageDataUrl: image.dataUrl })
setState((current) =>
result.success
? { ...current, status: 'done', result, error: undefined }
: { ...current, status: 'error', result: undefined, error: localOcrErrorMessage(result) }
)
} catch {
setState((current) => ({
...current,
status: 'error',
result: undefined,
error: '本地文字识别调用失败,请重试'
}))
}
}, [capability, state.image])
const clear = useCallback(() => setState({ status: 'idle' }), [])
const running = state.status === 'running'
const result = state.result
return (
<section className="settings-card local-ocr-test">
<header>
<div>
<h2>本地图片文字识别</h2>
<p>使用 Windows 系统 OCR 在本机读取图片中的文字,原始图片无需发送给 AI Provider。</p>
</div>
<span className={`local-ocr-capability ${capability?.available ? 'supported' : ''}`}>
{capability ? (capability.available ? '本机可用' : '本机不可用') : '检测中…'}
</span>
</header>
{capability && !capability.available ? (
<p className="local-ocr-notice">{capability.message}</p>
) : null}
{capability?.available ? (
<p className="local-ocr-runtime">
引擎:Windows 系统 OCR
{capability.runtimeVersion ? ` · 组件 ${capability.runtimeVersion}` : ''}
{capability.language ? ` · 语言 ${capability.language}` : ' · 语言跟随系统'}
</p>
) : null}
<label className={`local-ocr-upload ${state.image ? 'has-image' : ''}`}>
<input
type="file"
accept=".png,.jpg,.jpeg,.webp,.bmp,.gif,image/png,image/jpeg,image/webp,image/bmp,image/gif"
onChange={(event) => {
const file = event.currentTarget.files?.[0]
if (file) void selectImage(file)
event.currentTarget.value = ''
}}
/>
{state.image ? (
<>
<img src={state.image.dataUrl} alt="本地文字识别测试预览" />
<div>
<strong>{state.image.fileName}</strong>
<small>{formatFileSize(state.image.size)} · 仅保存在内存中</small>
</div>
</>
) : (
<div>
<strong>{state.status === 'reading' ? '正在读取图片…' : '选择本地图片'}</strong>
<small>支持 PNG、JPG、JPEG、WebP、BMP、GIF,最大 10 MB</small>
</div>
)}
</label>
<p className="local-ocr-privacy">
使用本地 OCR 时,原始图片无需发送给 AI Provider,也不会写入本地缓存或知识库。如果后续继续使用云端
AI 分析,提取出的文字可能按当前 Provider 配置发送。
</p>
{state.error ? <p className="local-ocr-error">{state.error}</p> : null}
{result?.success ? (
<div className="local-ocr-result">
<h3>识别结果</h3>
<dl>
<div>
<dt>引擎</dt>
<dd>Windows 系统 OCR</dd>
</div>
<div>
<dt>语言</dt>
<dd>{result.language || '跟随系统'}</dd>
</div>
<div>
<dt>耗时</dt>
<dd>{Math.round(result.durationMs)} ms</dd>
</div>
</dl>
<pre className="local-ocr-text">{result.text}</pre>
<p className="local-ocr-hint">
本地文字识别只读取图片中的文字内容,不会描述画面、人物或场景。
</p>
</div>
) : null}
<footer>
{state.image ? (
<Button variant="outline" onClick={clear}>
移除图片
</Button>
) : null}
<Button
disabled={!state.image || running || !capability?.available}
onClick={() => void run()}
>
{running ? '识别中…' : '本地文字识别'}
</Button>
</footer>
</section>
)
}
function localOcrErrorMessage(result: SystemOcrResult): string {
switch (result.errorCode) {
case 'UNSUPPORTED_PLATFORM':
return '本地图片文字识别目前仅支持 Windows。'
case 'SYSTEM_OCR_UNAVAILABLE':
return '本地文字识别组件不可用,请重新安装 TraceMemo。'
case 'OCR_LANGUAGE_UNAVAILABLE':
return '当前 Windows 未安装可用的 OCR 语言支持,请在系统「语言和区域」中安装中文或英文语言包。'
case 'UNSUPPORTED_IMAGE':
return '这张图片的格式暂不支持本地文字识别。'
case 'IMAGE_DECODE_FAILED':
return '图片解码失败,无法读取这张图片。'
case 'OCR_EMPTY_RESULT':
return '没有在这张图片里识别到文字。'
default:
return result.error || '本地文字识别失败,请重试。'
}
}
function formatFileSize(bytes: number): string {
return bytes < 1024 * 1024
? `${Math.max(1, Math.round(bytes / 1024))} KB`
: `${(bytes / 1024 / 1024).toFixed(1)} MB`
}
function readFileAsDataUrl(file: File): Promise<string> {
return new Promise((resolve, reject) => {
const reader = new FileReader()
reader.addEventListener('load', () =>
typeof reader.result === 'string' ? resolve(reader.result) : reject(new Error('invalid image'))
)
reader.addEventListener('error', () => reject(reader.error || new Error('read failed')))
reader.readAsDataURL(file)
})
}
@@ -4,6 +4,7 @@ import { Button } from '../../../components/ui'
import { AIProviderCard } from '../ai-model/AIProviderCard'
import { AIProviderEditor } from '../ai-model/AIProviderEditor'
import { AIImageUnderstandingTest } from '../ai-model/AIImageUnderstandingTest'
import { LocalImageTextRecognition } from '../ai-model/LocalImageTextRecognition'
import { useAIModelSettingsController } from '../ai-model/useAIModelSettingsController'
export function AIModelPage({
@@ -68,6 +69,7 @@ export function AIModelPage({
onTest={() => void controller.runVisionTest()}
onClear={controller.clearVisionImage}
/>
<LocalImageTextRecognition />
{controller.state.error ? (
<p className="ai-model-page-error">{controller.state.error}</p>
) : null}
+161
View File
@@ -1605,6 +1605,167 @@
justify-content: flex-end;
}
/* 本地图片文字识别(System OCR,本地 Runtime,与 AI 图片理解刻意区分) */
.local-ocr-test {
display: grid;
gap: 16px;
}
.local-ocr-test > header,
.local-ocr-test > footer {
display: flex;
align-items: center;
justify-content: space-between;
gap: 16px;
}
.local-ocr-test h2,
.local-ocr-test h3,
.local-ocr-test p {
margin: 0;
}
.local-ocr-test h2 {
color: var(--wxex-text-primary);
font-size: 15px;
}
.local-ocr-test header p,
.local-ocr-runtime,
.local-ocr-upload small,
.local-ocr-hint {
margin-top: 4px;
color: var(--wxex-text-secondary);
font-size: 12px;
}
.local-ocr-capability {
border-radius: 999px;
padding: 5px 9px;
background: var(--wxex-bg-sidebar);
color: var(--wxex-text-secondary);
white-space: nowrap;
font-size: 12px;
}
.local-ocr-capability.supported {
background: var(--wxex-brand-soft);
color: var(--wxex-brand);
}
.local-ocr-notice {
border-left: 3px solid var(--wxex-danger);
padding: 9px 11px;
background: color-mix(in srgb, var(--wxex-danger) 12%, transparent);
color: var(--wxex-danger);
font-size: 12px;
line-height: 1.6;
}
.local-ocr-upload {
display: flex;
min-height: 112px;
align-items: center;
justify-content: center;
gap: 14px;
border: 1px dashed var(--wxex-border);
border-radius: var(--wxex-radius-md);
padding: 14px;
background: var(--wxex-bg-sidebar);
color: var(--wxex-text-primary);
text-align: center;
cursor: pointer;
}
.local-ocr-upload:hover {
border-color: var(--wxex-brand);
background: var(--wxex-brand-soft);
}
.local-ocr-upload input {
display: none;
}
.local-ocr-upload.has-image {
justify-content: flex-start;
text-align: left;
}
.local-ocr-upload img {
width: 112px;
height: 82px;
flex: 0 0 auto;
border-radius: 8px;
object-fit: cover;
}
.local-ocr-upload strong,
.local-ocr-upload small {
display: block;
}
.local-ocr-privacy {
color: var(--wxex-text-secondary);
font-size: 12px;
line-height: 1.6;
}
.local-ocr-error {
border-left: 3px solid var(--wxex-danger);
padding: 9px 11px;
background: color-mix(in srgb, var(--wxex-danger) 12%, transparent);
color: var(--wxex-danger);
font-size: 12px;
}
.local-ocr-result {
display: grid;
gap: 12px;
border: 1px solid var(--wxex-border);
border-radius: var(--wxex-radius-md);
padding: 14px;
background: var(--wxex-bg-sidebar);
}
.local-ocr-result dl {
display: grid;
grid-template-columns: repeat(3, minmax(0, 1fr));
gap: 12px;
margin: 0;
}
.local-ocr-result dt {
color: var(--wxex-text-muted);
font-size: 11px;
}
.local-ocr-result dd {
margin: 4px 0 0;
color: var(--wxex-text-primary);
font-size: 12px;
font-weight: 600;
}
.local-ocr-text {
max-height: 240px;
overflow: auto;
margin: 0;
border: 1px solid var(--wxex-border);
border-radius: 8px;
padding: 12px;
background: var(--wxex-bg-app);
color: var(--wxex-text-primary);
font-family: inherit;
font-size: 13px;
line-height: 1.7;
white-space: pre-wrap;
word-break: break-word;
}
.local-ocr-test > footer {
justify-content: flex-end;
}
.report-history-sidebar {
display: flex;
min-width: 0;
+266
View File
@@ -0,0 +1,266 @@
// src/shared/system-ocr.ts
//
// 本地系统 OCR(System OCR)共享契约。
//
// 架构边界(不要混淆):
// - System OCR 是**本地 Runtime**,不是 AI Provider,也不是 Vision Model。
// 它不占用 AIVisionRuntimeConfig.source,也不产生任何网络请求。
// - 能力边界:只把图片里的文字读出来。它不等于「理解人物 / 理解场景 /
// 描述照片 / 理解表情包语义 / 视觉推理」——那些仍然属于 Vision Model。
// - Windows 后端为 Windows.Media.Ocr.OcrEngine(经 @napi-rs/system-ocr 调用)。
// macOS 本轮只保留架构位置,未实现;Linux 不支持。
//
// 数据边界(本轮不做):
// - 不做历史图片全量 OCR、不做 Knowledge 回填、不把 OCR 文字伪装成原始聊天文字。
// 原始消息始终是权威来源,OCR 文字只是派生内容(本轮仅存在于内存)。
/** System OCR 引擎标识。这是本地 Runtime,不是 provider id。 */
export const SYSTEM_OCR_ENGINE = 'windows-system-ocr'
/** 本地 OCR 结果在内存中的缓存时长。 */
export const SYSTEM_OCR_CACHE_TTL_MS = 10 * 60 * 1000
/**
* 产品级错误码。用户可见文案由 error 字段承载,任何 native 堆栈 / HRESULT
* 都不会直接透出到 Renderer。
*/
export type SystemOcrErrorCode =
/** 运行时不可用(native binding 缺失 / 加载失败) */
| 'SYSTEM_OCR_UNAVAILABLE'
/** 当前平台不支持(Linux,或非 Windows 平台) */
| 'UNSUPPORTED_PLATFORM'
/** 图片格式不在支持范围内 */
| 'UNSUPPORTED_IMAGE'
/** 图片解码失败(格式可识别但内容损坏或无法转成 PNG) */
| 'IMAGE_DECODE_FAILED'
/** 当前 Windows 未安装对应的 OCR 语言支持 */
| 'OCR_LANGUAGE_UNAVAILABLE'
/** 引擎执行失败 */
| 'OCR_FAILED'
/** 识别成功执行,但图里没有文字 */
| 'OCR_EMPTY_RESULT'
export type SystemOcrUnavailableReason =
| 'UNSUPPORTED_PLATFORM'
| 'NATIVE_MODULE_MISSING'
| 'LANGUAGE_UNAVAILABLE'
/** 本机 System OCR 能力。UI 只用它决定是否展示「本地文字识别」入口。 */
export interface SystemOcrCapability {
/** 本机当前是否真的可以识别图片文字 */
available: boolean
engine: typeof SYSTEM_OCR_ENGINE
platform: NodeJS.Platform
arch: string
/** @napi-rs/system-ocr 运行时版本;无法读取时为 null */
runtimeVersion: string | null
/** 实际可用的 OCR 语言标签(对应 Windows 语言包);null 表示走系统用户语言 */
language: string | null
reason?: SystemOcrUnavailableReason
/** 面向用户的中文说明,可直接展示 */
message: string
}
export interface SystemOcrBoundingBox {
/** 归一化到 0..1,原点在左上角 */
x: number
y: number
width: number
height: number
}
export interface SystemOcrLine {
text: string
/** Windows 恒为 1.0 */
confidence: number
boundingBox: SystemOcrBoundingBox
}
/** 本地 OCR 结果。不包含任何 Windows handle / native 内部对象。 */
export interface SystemOcrResult {
success: boolean
/** 归一化后的文本(去掉 CJK 字符之间的引擎伪空格) */
text: string
lines: SystemOcrLine[]
/** 实际使用的 OCR 语言标签;null 表示由系统用户语言决定 */
language: string | null
engine: typeof SYSTEM_OCR_ENGINE
durationMs: number
/** 命中内存缓存时为 true */
fromCache?: boolean
errorCode?: SystemOcrErrorCode
error?: string
}
export interface SystemOcrRequest {
/** data URL(data:image/png;base64,...)。base64 只在 main 内部流转,不回传 Renderer。 */
imageDataUrl: string
/** 调用方已经算好的图片内容哈希;未传时由 main 内部计算 */
imageHash?: string
/** 指定 OCR 语言标签;默认按系统语言解析 */
language?: string
}
/**
* 缓存 key 组合。刻意与 ImageInsight 的 `imageHash` 保持不同的键空间,
* 保证远端 Vision 的旧结果永远不会被当成"本地 OCR 结果"复用,
* 也保证 System OCR 运行时升级后不会永远命中旧结果。
*/
export const buildSystemOcrCacheKey = (input: {
imageHash: string
language: string | null
runtimeVersion: string | null
platform?: string
}): string =>
[
input.imageHash,
SYSTEM_OCR_ENGINE,
input.platform ?? 'unknown',
input.language ?? 'auto',
input.runtimeVersion ?? 'unknown'
].join('|')
const CJK_CHAR =
/[\u3000-\u303f\u3040-\u30ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff00-\uffef\uac00-\ud7af]/
/**
* Windows OCR 会在每个 CJK 字符之间插入空格("本 地 图 片")。
* 这里只删除 **两侧都是 CJK** 的空格,保留 "TraceMemo 本地图片文字识别" 里的真实分隔。
*/
export const normalizeSystemOcrText = (value: string): string => {
const source = String(value ?? '')
if (!source) return ''
let result = ''
for (let index = 0; index < source.length; index += 1) {
const char = source[index]
if (char === ' ' || char === '\u3000') {
const previous = result[result.length - 1]
let next = ''
for (let lookahead = index + 1; lookahead < source.length; lookahead += 1) {
if (source[lookahead] !== ' ' && source[lookahead] !== '\u3000') {
next = source[lookahead]
break
}
}
if (previous && next && CJK_CHAR.test(previous) && CJK_CHAR.test(next)) continue
}
result += char
}
return result.trim()
}
/** 把系统 locale(如 zh-CN / en-US)映射成 Windows OCR 语言标签。 */
const LANGUAGE_TAG_BY_LOCALE: Record<string, string> = {
zh: 'zh-Hans-CN',
'zh-cn': 'zh-Hans-CN',
'zh-hans': 'zh-Hans-CN',
'zh-hans-cn': 'zh-Hans-CN',
'zh-sg': 'zh-Hans-CN',
'zh-tw': 'zh-Hant-TW',
'zh-hant': 'zh-Hant-TW',
'zh-hant-tw': 'zh-Hant-TW',
'zh-hk': 'zh-Hant-HK',
'zh-hant-hk': 'zh-Hant-HK',
'zh-mo': 'zh-Hant-MO',
'zh-hant-mo': 'zh-Hant-MO',
en: 'en-US',
'en-us': 'en-US',
'en-gb': 'en-GB',
'en-au': 'en-AU',
'en-ca': 'en-CA',
ja: 'ja-JP',
'ja-jp': 'ja-JP',
ko: 'ko-KR',
'ko-kr': 'ko-KR',
fr: 'fr-FR',
'fr-fr': 'fr-FR',
de: 'de-DE',
'de-de': 'de-DE',
es: 'es-ES',
'es-es': 'es-ES',
it: 'it-IT',
'it-it': 'it-IT',
pt: 'pt-BR',
'pt-br': 'pt-BR',
ru: 'ru-RU',
'ru-ru': 'ru-RU'
}
export const resolveSystemOcrLanguageTag = (
locale: string | null | undefined
): string | null => {
const normalized = String(locale ?? '')
.trim()
.toLowerCase()
.replace(/_/g, '-')
if (!normalized) return null
if (LANGUAGE_TAG_BY_LOCALE[normalized]) return LANGUAGE_TAG_BY_LOCALE[normalized]
const primary = normalized.split('-')[0]
return LANGUAGE_TAG_BY_LOCALE[primary] ?? null
}
/**
* 把 native 错误映射成产品级错误码。
*
* 已确认的 Windows 行为(1.2.0):
* - 语言包缺失 / 引擎无法创建:`Windows error 操作成功完成。 (0x00000000)`
* —— TryCreateFromLanguage 返回 null 引擎但 HRESULT 是 S_OK,非常容易误判。
* - 送给解码器的字节不是可识别的图片:`Windows error Could not recognize file (0x80070005)`
*/
export const mapSystemOcrNativeError = (message: string): SystemOcrErrorCode => {
const detail = String(message ?? '')
if (!detail) return 'OCR_FAILED'
if (/Cannot find native binding|Failed to load native binding|MODULE_NOT_FOUND/i.test(detail)) {
return 'SYSTEM_OCR_UNAVAILABLE'
}
if (/\(0x00000000\)/.test(detail)) return 'OCR_LANGUAGE_UNAVAILABLE'
if (/Could not recognize file/i.test(detail)) return 'IMAGE_DECODE_FAILED'
if (/Could not open file/i.test(detail)) return 'IMAGE_DECODE_FAILED'
return 'OCR_FAILED'
}
/** 解析 data URL;只接受图片 MIME。 */
export const parseImageDataUrl = (
dataUrl: string
): { mimeType: string; base64: string } | null => {
const matched = /^data:(image\/[a-z0-9.+-]+);base64,(.+)$/i.exec(String(dataUrl ?? '').trim())
if (!matched) return null
return { mimeType: matched[1].toLowerCase(), base64: matched[2] }
}
export type SystemOcrImageFormat = 'png' | 'jpeg' | 'gif' | 'bmp' | 'webp' | 'tiff'
/** 按魔数识别格式。返回 null 表示不在支持范围内。 */
export const detectSystemOcrImageFormat = (buffer: Uint8Array): SystemOcrImageFormat | null => {
if (!buffer || buffer.length < 4) return null
const byte = (index: number): number => buffer[index]
if (byte(0) === 0x89 && byte(1) === 0x50 && byte(2) === 0x4e && byte(3) === 0x47) return 'png'
if (byte(0) === 0xff && byte(1) === 0xd8 && byte(2) === 0xff) return 'jpeg'
if (byte(0) === 0x47 && byte(1) === 0x49 && byte(2) === 0x46) return 'gif'
if (byte(0) === 0x42 && byte(1) === 0x4d) return 'bmp'
if (
byte(0) === 0x52 &&
byte(1) === 0x49 &&
byte(2) === 0x46 &&
byte(3) === 0x46 &&
buffer.length > 11 &&
byte(8) === 0x57 &&
byte(9) === 0x45 &&
byte(10) === 0x42 &&
byte(11) === 0x50
) {
return 'webp'
}
if ((byte(0) === 0x49 && byte(1) === 0x49) || (byte(0) === 0x4d && byte(1) === 0x4d)) {
return 'tiff'
}
return null
}
/**
* 64x32 纯白 PNG。仅用于 language capability 探测:
* 引擎能创建 → 该语言包可用;引擎创建失败 → 语言不可用。
* 探测耗时量级为个位数毫秒。
*/
export const SYSTEM_OCR_PROBE_PNG_BASE64 =
'iVBORw0KGgoAAAANSUhEUgAAAEAAAAAgCAIAAAAt/+nTAAAANUlEQVR42u3PAQkAAAgDMLV/59tCELYG6yT12dRzAgICAgICAgICAgICAgICAgICAgICAvcWMisDPdIJjMIAAAAASUVORK5CYII='
+13 -1
View File
@@ -6,6 +6,7 @@ import { AIProviderCard } from '../../src/renderer/src/features/settings/ai-mode
import { AIModelPage } from '../../src/renderer/src/features/settings/pages/AIModelPage'
import type { AIProviderSummary, AIRuntimeModelConfig } from '../../src/shared/ai-provider'
import type { AIVisionTestState } from '../../src/renderer/src/features/settings/ai-model/types'
import type { SystemOcrCapability } from '../../src/shared/system-ocr'
const runtime: AIRuntimeModelConfig = {
providerName: 'Not configured',
@@ -15,6 +16,16 @@ const runtime: AIRuntimeModelConfig = {
status: 'untested'
}
const systemOcrCapability: SystemOcrCapability = {
available: true,
engine: 'windows-system-ocr',
platform: 'win32',
arch: 'x64',
runtimeVersion: '1.2.0',
language: 'zh-Hans-CN',
message: '本地图片文字识别可用(Windows 系统 OCR,zh-Hans-CN)。'
}
const provider: AIProviderSummary = {
id: 'fixture-provider',
name: '本地假服务',
@@ -56,7 +67,8 @@ describe('AI model settings', () => {
})
window.api = {
listAIProviders: vi.fn().mockResolvedValue({ success: true, providers: [] }),
getAIRuntimeConfig: vi.fn().mockResolvedValue(runtime)
getAIRuntimeConfig: vi.fn().mockResolvedValue(runtime),
getSystemOcrCapability: vi.fn().mockResolvedValue(systemOcrCapability)
} as typeof window.api
})
@@ -0,0 +1,107 @@
import { fireEvent, render, screen, waitFor } from '@testing-library/react'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { LocalImageTextRecognition } from '../../src/renderer/src/features/settings/ai-model/LocalImageTextRecognition'
import type { SystemOcrCapability, SystemOcrResult } from '../../src/shared/system-ocr'
const availableCapability: SystemOcrCapability = {
available: true,
engine: 'windows-system-ocr',
platform: 'win32',
arch: 'x64',
runtimeVersion: '1.2.0',
language: 'zh-Hans-CN',
message: '本地图片文字识别可用(Windows 系统 OCR,zh-Hans-CN)。'
}
const unavailableCapability: SystemOcrCapability = {
available: false,
engine: 'windows-system-ocr',
platform: 'win32',
arch: 'x64',
runtimeVersion: '1.2.0',
language: null,
reason: 'LANGUAGE_UNAVAILABLE',
message: '当前 Windows 未安装可用的 OCR 语言支持。'
}
const successResult: SystemOcrResult = {
success: true,
text: 'TraceMemo 本地文字识别',
lines: [],
language: 'zh-Hans-CN',
engine: 'windows-system-ocr',
durationMs: 42
}
const emptyResult: SystemOcrResult = {
success: false,
text: '',
lines: [],
language: 'zh-Hans-CN',
engine: 'windows-system-ocr',
durationMs: 12,
errorCode: 'OCR_EMPTY_RESULT',
error: '没有在这张图片里识别到文字。'
}
const selectImage = (): void => {
const input = document.querySelector<HTMLInputElement>('input[type="file"]')
if (!input) throw new Error('file input missing')
const file = new File([new Uint8Array([0x89, 0x50, 0x4e, 0x47])], 'fixture.png', {
type: 'image/png'
})
fireEvent.change(input, { target: { files: [file] } })
}
describe('LocalImageTextRecognition', () => {
beforeEach(() => {
window.api = {
getSystemOcrCapability: vi.fn().mockResolvedValue(availableCapability),
recognizeLocalImageText: vi.fn().mockResolvedValue(successResult)
} as typeof window.api
})
it('shows local availability without requiring any AI provider', async () => {
render(<LocalImageTextRecognition />)
expect(await screen.findByText('本机可用')).toBeInTheDocument()
expect(screen.getByText(/组件 1\.2\.0/)).toBeInTheDocument()
expect(screen.getByRole('button', { name: '本地文字识别' })).toBeDisabled()
})
it('recognizes text locally and renders the extracted text', async () => {
render(<LocalImageTextRecognition />)
await screen.findByText('本机可用')
selectImage()
const button = await screen.findByRole('button', { name: '本地文字识别' })
await waitFor(() => expect(button).toBeEnabled())
fireEvent.click(button)
expect(await screen.findByText('TraceMemo 本地文字识别')).toBeInTheDocument()
expect(window.api.recognizeLocalImageText).toHaveBeenCalledWith({
imageDataUrl: expect.stringContaining('data:image/png;base64,')
})
})
it('surfaces an empty OCR result as a product message, not a native error', async () => {
vi.mocked(window.api.recognizeLocalImageText).mockResolvedValue(emptyResult)
render(<LocalImageTextRecognition />)
await screen.findByText('本机可用')
selectImage()
const button = await screen.findByRole('button', { name: '本地文字识别' })
await waitFor(() => expect(button).toBeEnabled())
fireEvent.click(button)
expect(await screen.findByText('没有在这张图片里识别到文字。')).toBeInTheDocument()
})
it('keeps the entry disabled and explains why when the language pack is missing', async () => {
vi.mocked(window.api.getSystemOcrCapability).mockResolvedValue(unavailableCapability)
render(<LocalImageTextRecognition />)
expect(await screen.findByText('本机不可用')).toBeInTheDocument()
expect(screen.getByText('当前 Windows 未安装可用的 OCR 语言支持。')).toBeInTheDocument()
expect(screen.getByRole('button', { name: '本地文字识别' })).toBeDisabled()
})
})
Binary file not shown.

After

Width:  |  Height:  |  Size: 10 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 20 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 12 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 14 KiB

@@ -0,0 +1,96 @@
// Windows System OCR native fidelity。
//
// 这是 capability-gated 的原生冒烟测试:
// - 只有在「当前平台支持 + native 运行时可用 + 有可用 OCR 语言包」时才真正跑;
// - CI 环境无法保证 Windows OCR 语言包,所以中文识别不作为所有 CI 的硬门槛
// (mock 单元测试才是 mandatory,见 tests/unit/system-ocr-service.test.ts);
// - 在 Windows 真机上必须实际通过。
//
// fixture 全部是 synthetic 图片(tests/fixtures/ocr/*),不含任何真实聊天数据。
import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { describe, expect, it, vi } from 'vitest'
import { SYSTEM_OCR_ENGINE } from '../../src/shared/system-ocr'
vi.mock('../../src/main/image-decrypt-service', () => ({
resolveFfmpegExecutable: (): string => 'ffmpeg'
}))
import { systemOcrService } from '../../src/main/services/system-ocr-service'
const fixtureDirectory = join(__dirname, '..', 'fixtures', 'ocr')
const toDataUrl = (fileName: string, mimeType: string): string =>
`data:${mimeType};base64,${readFileSync(join(fixtureDirectory, fileName)).toString('base64')}`
/** 只比较"主要 token",避免系统字体 / 识别微差造成脆弱测试。 */
const expectContainsTokens = (text: string, tokens: string[]): void => {
const normalized = text.replace(/[\s\u3000]+/g, '').toLowerCase()
for (const token of tokens) {
expect(normalized).toContain(token.replace(/[\s\u3000]+/g, '').toLowerCase())
}
}
const capability = await systemOcrService.getCapability()
const nativeGate = capability.available ? it : it.skip
describe('Windows System OCR native fidelity', () => {
it('reports a usable capability on this machine', () => {
expect(capability.engine).toBe(SYSTEM_OCR_ENGINE)
if (!capability.available) {
console.warn(`[integration] System OCR native smoke skipped: ${capability.message}`)
}
})
nativeGate('recognizes simplified Chinese text', async () => {
const result = await systemOcrService.recognize({
imageDataUrl: toDataUrl('system-ocr-zh.png', 'image/png')
})
expect(result.success).toBe(true)
expectContainsTokens(result.text, ['TraceMemo', '本地', '文字', '识别'])
// 语言要么是探测到的语言包,要么是"跟随系统用户语言"(null)。
if (capability.language) {
expect(result.language).toBe(capability.language)
} else {
expect(result.language).toBeNull()
}
expect(result.durationMs).toBeGreaterThan(0)
})
nativeGate('recognizes English text', async () => {
const result = await systemOcrService.recognize({
imageDataUrl: toDataUrl('system-ocr-en.png', 'image/png')
})
expect(result.success).toBe(true)
expectContainsTokens(result.text, ['TraceMemo', 'System', 'OCR'])
})
nativeGate('recognizes mixed Chinese/English text', async () => {
const result = await systemOcrService.recognize({
imageDataUrl: toDataUrl('system-ocr-mixed.png', 'image/png')
})
expect(result.success).toBe(true)
expectContainsTokens(result.text, ['TraceMemo', '本地', 'OCR', '2026'])
})
/**
* 引擎的 Buffer 输入只接受 PNG,所以 JPEG 必须走本服务的归一化路径。
* 这条用例就是那个约束的回归保护。
*/
nativeGate('normalizes a JPEG source before OCR', async () => {
const result = await systemOcrService.recognize({
imageDataUrl: toDataUrl('system-ocr-mixed.jpg', 'image/jpeg')
})
expect(result.success).toBe(true)
expectContainsTokens(result.text, ['TraceMemo', 'OCR', '2026'])
})
nativeGate('caches an identical repeat request', async () => {
const request = { imageDataUrl: toDataUrl('system-ocr-mixed.png', 'image/png') }
const first = await systemOcrService.recognize(request)
const second = await systemOcrService.recognize(request)
expect(first.success).toBe(true)
expect(second.fromCache).toBe(true)
})
})
+349
View File
@@ -0,0 +1,349 @@
import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import {
SYSTEM_OCR_ENGINE,
SYSTEM_OCR_PROBE_PNG_BASE64,
buildSystemOcrCacheKey,
detectSystemOcrImageFormat,
mapSystemOcrNativeError,
normalizeSystemOcrText,
parseImageDataUrl,
resolveSystemOcrLanguageTag
} from '../../src/shared/system-ocr'
vi.mock('../../src/main/image-decrypt-service', () => ({
resolveFfmpegExecutable: (): string => 'ffmpeg'
}))
const { getByHash, upsert } = vi.hoisted(() => ({
getByHash: vi.fn(),
upsert: vi.fn()
}))
vi.mock('../../src/main/db/image-insights-store', () => ({
imageInsightsStore: {
getByHash,
upsert,
listBySession: vi.fn(() => [])
}
}))
import { SystemOcrService } from '../../src/main/services/system-ocr-service'
import { imageInsightService } from '../../src/main/services/image-insight-service'
const PROBE_BYTES = Buffer.from(SYSTEM_OCR_PROBE_PNG_BASE64, 'base64')
const FIXTURE_PNG = readFileSync(join(__dirname, '..', 'fixtures', 'ocr', 'system-ocr-zh.png'))
const PNG_DATA_URL = `data:image/png;base64,${FIXTURE_PNG.toString('base64')}`
const JPEG_DATA_URL = `data:image/jpeg;base64,${Buffer.from([0xff, 0xd8, 0xff, 0xe0, 0x00, 0x10]).toString('base64')}`
const GARBAGE_DATA_URL = `data:image/png;base64,${Buffer.from('definitely-not-an-image').toString('base64')}`
const LANGUAGE_UNAVAILABLE_MESSAGE = 'Windows error 操作成功完成。 (0x00000000)'
const DECODE_FAILED_MESSAGE = 'Windows error Could not recognize file (0x80070005)'
/**
* mock 运行时按「探测图字节」区分 capability probe 和业务调用,
* 这样 probeError 才能稳定复现「语言包缺失」的场景。
*/
const createRuntime = (
options: {
text?: string
lines?: Array<{ text: string; confidence?: number }>
probeError?: string
error?: string
version?: string | null
} = {}
): { version: string | null; recognize: ReturnType<typeof vi.fn> } => {
const recognize = vi.fn(
async (image: Uint8Array, _accuracy?: number, _languages?: string[]): Promise<unknown> => {
if (Buffer.from(image).equals(PROBE_BYTES)) {
if (options.probeError) throw new Error(options.probeError)
return { text: '', confidence: 1, lines: [] }
}
if (options.error) throw new Error(options.error)
const text = options.text ?? ''
return {
text,
confidence: 1,
lines: (options.lines ?? [{ text, confidence: 1 }]).map((line) => ({
text: line.text,
confidence: line.confidence ?? 1,
boundingBox: { x: 0.1, y: 0.2, width: 0.3, height: 0.4 }
}))
}
}
)
return { version: options.version === undefined ? '1.2.0' : options.version, recognize }
}
const createService = (
runtime: { version: string | null; recognize: ReturnType<typeof vi.fn> } | null,
overrides: Partial<ConstructorParameters<typeof SystemOcrService>[0]> = {}
): SystemOcrService =>
new SystemOcrService({
platform: 'win32',
arch: 'x64',
locale: () => 'zh-CN',
loadRuntime: () => (runtime ? { ...runtime } : null),
toPngBytes: async ({ buffer }) => buffer,
...overrides
})
describe('system-ocr shared helpers', () => {
it('removes only the engine-inserted spaces between CJK glyphs', () => {
expect(normalizeSystemOcrText('TraceMemo 本 地 图 片 文 字 识 别')).toBe(
'TraceMemo 本地图片文字识别'
)
expect(normalizeSystemOcrText('TraceMemo System OCR')).toBe('TraceMemo System OCR')
expect(normalizeSystemOcrText(' 本 地 ')).toBe('本地')
})
it('maps system locale onto Windows OCR language tags', () => {
expect(resolveSystemOcrLanguageTag('zh-CN')).toBe('zh-Hans-CN')
expect(resolveSystemOcrLanguageTag('zh-Hans-CN')).toBe('zh-Hans-CN')
expect(resolveSystemOcrLanguageTag('zh_TW')).toBe('zh-Hant-TW')
expect(resolveSystemOcrLanguageTag('en-US')).toBe('en-US')
expect(resolveSystemOcrLanguageTag('en')).toBe('en-US')
expect(resolveSystemOcrLanguageTag('')).toBeNull()
expect(resolveSystemOcrLanguageTag('xx-YY')).toBeNull()
})
it('maps native Windows errors onto product error codes', () => {
expect(mapSystemOcrNativeError(LANGUAGE_UNAVAILABLE_MESSAGE)).toBe('OCR_LANGUAGE_UNAVAILABLE')
expect(mapSystemOcrNativeError(DECODE_FAILED_MESSAGE)).toBe('IMAGE_DECODE_FAILED')
expect(mapSystemOcrNativeError('Cannot find native binding.')).toBe('SYSTEM_OCR_UNAVAILABLE')
expect(mapSystemOcrNativeError('Failed to load native binding')).toBe('SYSTEM_OCR_UNAVAILABLE')
expect(mapSystemOcrNativeError('Windows error something broke (0x80070057)')).toBe('OCR_FAILED')
expect(mapSystemOcrNativeError('')).toBe('OCR_FAILED')
})
it('parses image data urls and rejects other payloads', () => {
expect(parseImageDataUrl(PNG_DATA_URL)).toMatchObject({ mimeType: 'image/png' })
expect(parseImageDataUrl('data:text/plain;base64,aGk=')).toBeNull()
expect(parseImageDataUrl('not-a-data-url')).toBeNull()
})
it('detects supported container formats by magic bytes', () => {
expect(detectSystemOcrImageFormat(Buffer.from([0x89, 0x50, 0x4e, 0x47]))).toBe('png')
expect(detectSystemOcrImageFormat(Buffer.from([0xff, 0xd8, 0xff, 0xe0]))).toBe('jpeg')
expect(detectSystemOcrImageFormat(Buffer.from('GIF89a'))).toBe('gif')
expect(detectSystemOcrImageFormat(Buffer.from('BM1234'))).toBe('bmp')
expect(
detectSystemOcrImageFormat(Buffer.concat([Buffer.from('RIFF'), Buffer.alloc(4), Buffer.from('WEBP')]))
).toBe('webp')
expect(detectSystemOcrImageFormat(Buffer.from([0x49, 0x49, 0x2a, 0x00]))).toBe('tiff')
expect(detectSystemOcrImageFormat(Buffer.from('nope'))).toBeNull()
})
it('keeps the local OCR cache keyspace separate from the vision imageHash', () => {
const base = { imageHash: 'a'.repeat(32), language: 'zh-Hans-CN', runtimeVersion: '1.2.0' }
const key = buildSystemOcrCacheKey({ ...base, platform: 'win32' })
expect(key).not.toBe(base.imageHash)
expect(key).toContain(SYSTEM_OCR_ENGINE)
expect(key).toContain('zh-Hans-CN')
expect(key).toContain('1.2.0')
// 语言或运行时版本变化必须换 key,避免复用过期 / 跨引擎结果。
expect(buildSystemOcrCacheKey({ ...base, language: 'en-US', platform: 'win32' })).not.toBe(key)
expect(
buildSystemOcrCacheKey({ ...base, runtimeVersion: '1.3.0', platform: 'win32' })
).not.toBe(key)
})
})
describe('SystemOcrService capability detection', () => {
it('reports available with the probed language on Windows', async () => {
const service = createService(createRuntime())
const capability = await service.getCapability()
expect(capability).toMatchObject({
available: true,
engine: SYSTEM_OCR_ENGINE,
platform: 'win32',
arch: 'x64',
runtimeVersion: '1.2.0',
language: 'zh-Hans-CN'
})
})
it('is unavailable on unsupported platforms without loading a runtime', async () => {
const loadRuntime = vi.fn(() => null)
const service = createService(null, { platform: 'linux', loadRuntime })
const capability = await service.getCapability()
expect(capability.available).toBe(false)
expect(capability.reason).toBe('UNSUPPORTED_PLATFORM')
expect(loadRuntime).not.toHaveBeenCalled()
})
it('is unavailable when the native runtime cannot be loaded', async () => {
const service = createService(null)
const capability = await service.getCapability()
expect(capability.available).toBe(false)
expect(capability.reason).toBe('NATIVE_MODULE_MISSING')
})
it('reports a missing Windows OCR language pack as LANGUAGE_UNAVAILABLE', async () => {
const runtime = createRuntime({ probeError: LANGUAGE_UNAVAILABLE_MESSAGE })
const service = createService(runtime)
const capability = await service.getCapability()
expect(capability.available).toBe(false)
expect(capability.reason).toBe('LANGUAGE_UNAVAILABLE')
expect(capability.message).toContain('OCR 语言')
})
})
describe('SystemOcrService recognition', () => {
beforeEach(() => {
getByHash.mockReset()
getByHash.mockReturnValue(null)
upsert.mockReset()
})
it('normalizes a successful result and reports runtime metadata', async () => {
const runtime = createRuntime({
text: 'TraceMemo 本 地 OCR 2026',
lines: [{ text: 'TraceMemo 本 地 OCR 2026' }]
})
const service = createService(runtime)
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(result).toMatchObject({
success: true,
text: 'TraceMemo 本地 OCR 2026',
language: 'zh-Hans-CN',
engine: SYSTEM_OCR_ENGINE
})
expect(result.lines[0].boundingBox).toEqual({ x: 0.1, y: 0.2, width: 0.3, height: 0.4 })
expect(result.durationMs).toBeGreaterThanOrEqual(0)
})
it('returns OCR_EMPTY_RESULT when the engine finds no text', async () => {
const service = createService(createRuntime({ text: '' }))
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(result.success).toBe(false)
expect(result.errorCode).toBe('OCR_EMPTY_RESULT')
expect(result.text).toBe('')
})
it('returns UNSUPPORTED_PLATFORM on non-Windows platforms', async () => {
const service = createService(null, { platform: 'darwin', arch: 'arm64' })
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(result.success).toBe(false)
expect(result.errorCode).toBe('UNSUPPORTED_PLATFORM')
expect(result.engine).toBe(SYSTEM_OCR_ENGINE)
})
it('returns OCR_LANGUAGE_UNAVAILABLE when no OCR language pack is installed', async () => {
const service = createService(createRuntime({ probeError: LANGUAGE_UNAVAILABLE_MESSAGE }))
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(result.success).toBe(false)
expect(result.errorCode).toBe('OCR_LANGUAGE_UNAVAILABLE')
})
it('rejects payloads that are not a supported image container', async () => {
const runtime = createRuntime()
const service = createService(runtime)
const badUrl = await service.recognize({ imageDataUrl: 'nope' })
expect(badUrl.errorCode).toBe('UNSUPPORTED_IMAGE')
const garbage = await service.recognize({ imageDataUrl: GARBAGE_DATA_URL })
expect(garbage.errorCode).toBe('UNSUPPORTED_IMAGE')
expect(runtime.recognize).not.toHaveBeenCalled()
})
it('maps an image preparation failure onto IMAGE_DECODE_FAILED', async () => {
const runtime = createRuntime()
const service = createService(runtime, { toPngBytes: async () => null })
const result = await service.recognize({ imageDataUrl: JPEG_DATA_URL })
expect(result.success).toBe(false)
expect(result.errorCode).toBe('IMAGE_DECODE_FAILED')
})
it('maps a native OCR failure onto a product error code', async () => {
const runtime = createRuntime({ error: DECODE_FAILED_MESSAGE })
const service = createService(runtime)
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(result.success).toBe(false)
expect(result.errorCode).toBe('IMAGE_DECODE_FAILED')
expect(result.error).not.toContain('0x80070005')
})
it('caches by image identity + engine + language + runtime version only', async () => {
const runtime = createRuntime({ text: 'TraceMemo 本 地' })
const service = createService(runtime)
const first = await service.recognize({ imageDataUrl: PNG_DATA_URL })
const callsAfterFirst = runtime.recognize.mock.calls.length
const second = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(first.success).toBe(true)
expect(second.fromCache).toBe(true)
expect(runtime.recognize.mock.calls.length).toBe(callsAfterFirst)
// 运行时版本升级 → 缓存 key 变化 → 必须重新识别,不能永远吃旧结果。
const upgradedRuntime = createRuntime({ text: 'TraceMemo 本 地' })
service.bind({
loadRuntime: () => ({ version: '1.3.0', recognize: upgradedRuntime.recognize })
})
const afterUpgrade = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(afterUpgrade.success).toBe(true)
expect(afterUpgrade.fromCache).toBeUndefined()
expect(upgradedRuntime.recognize).toHaveBeenCalled()
})
it('never reuses a remote Vision insight as a local OCR result', async () => {
getByHash.mockReturnValue({
imageHash: 'a'.repeat(32),
description: '远端 Vision 旧结果',
ocrText: '远端 OCR 文本',
updatedAt: Date.now()
})
const runtime = createRuntime({ text: '本地 文 字' })
const service = createService(runtime)
const result = await service.recognize({ imageDataUrl: PNG_DATA_URL })
expect(result.text).toBe('本地文字')
expect(getByHash).not.toHaveBeenCalled()
})
})
describe('ImageInsightService local OCR orchestration', () => {
beforeEach(() => {
getByHash.mockReset()
getByHash.mockReturnValue(null)
upsert.mockReset()
})
it('exposes capability and never falls back to the remote vision provider', async () => {
const analyzeImage = vi.fn(async () => ({ success: true, data: '{}' }))
imageInsightService.bind({
providerService: {
list: () => ({ providers: [], defaultProviderId: 'vision-provider' }),
getVisionRuntimeConfig: () => ({
providerId: 'vision-provider',
providerName: 'OpenAI',
model: 'gpt-vision',
modelName: 'gpt-vision',
configured: true
}),
analyzeImage
},
decryptService: {
findImageFile: () => null,
decryptImageToBase64: () => null
}
})
const capability = await imageInsightService.getSystemOcrCapability()
expect(capability.engine).toBe(SYSTEM_OCR_ENGINE)
const result = await imageInsightService.extractLocalText({ imageDataUrl: PNG_DATA_URL })
expect(result.engine).toBe(SYSTEM_OCR_ENGINE)
// 关键约束:本地 OCR 路径绝不调用远端 Vision Provider。
expect(analyzeImage).not.toHaveBeenCalled()
// 也不写 Vision 的 insight 缓存。
expect(upsert).not.toHaveBeenCalled()
})
})