feat: 新增 Windows 本地图片文字识别能力

This commit is contained in:
电摇小子
2026-09-16 00:23:23 +08:00
parent e6db4de711
commit 24399f1d70
23 changed files with 1984 additions and 7 deletions
+5 -1
View File
@@ -114,7 +114,11 @@ function getFfmpegCandidates(selectedPath = loadSettings().ffmpegPath): FfmpegCa
)
}
function resolveFfmpegExecutable(): string {
/**
* 解析可用的 ffmpeg 可执行文件。除图片解密自身使用外,也供 System OCR 的
* 图片归一化(GIF/BMP/WebP/TIFF → PNG)复用,避免重复一套路径探测逻辑。
*/
export function resolveFfmpegExecutable(): string {
for (const candidate of getFfmpegCandidates()) {
const pathLike = candidate.executable.includes('/') || candidate.executable.includes('\\')
if (pathLike) {
+30
View File
@@ -29,6 +29,7 @@ import {
ImageDecryptService,
inspectImageDecoderExecutable,
inspectImageDecoderStatus,
resolveFfmpegExecutable,
type DecodedImage
} from './image-decrypt-service'
import {
@@ -64,6 +65,7 @@ import { apiTokenStore } from './api-token-store'
import { ImageKeyConfigService } from './services/image-key-config-service'
import { AIProviderService } from './services/ai-provider-service'
import { imageInsightService } from './services/image-insight-service'
import { systemOcrService } from './services/system-ocr-service'
import type {
ImageAnalysisRequest,
ImageAnalysisResponse,
@@ -71,6 +73,7 @@ import type {
ImageCandidateQuery,
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import { KeyServiceMac } from './key-service-mac'
import { KeyService as KeyServiceWin } from './key-service-win'
import * as chat from './services/chat-service'
@@ -1814,6 +1817,13 @@ app.whenReady().then(async () => {
}
})
// System OCR 是独立的本地 Runtime(不是 AI Provider):只注入项目统一的 ffmpeg
// 解析逻辑(GIF/BMP/WebP/TIFF → PNG 归一化)和系统 locale(OCR 语言包探测)。
systemOcrService.bind({
resolveFfmpegExecutable,
locale: () => app.getLocale()
})
/** 日报入口:取会话 Top N 热点图片 + 已缓存的 Insight */
ipcMain.handle(
'image:listCandidates',
@@ -1885,6 +1895,26 @@ app.whenReady().then(async () => {
}
)
// ============================================================
// 本地图片文字识别(System OCR / Windows System OCR Runtime)
// ============================================================
// 这是本地 Runtime,不是 AI Vision Provider:
// - 不联网、不上传原图;
// - 不读写 AI Provider / Vision 模型配置;
// - 结果不落库(派生内容,本轮只做内存级闭环)。
ipcMain.handle('system-ocr:getCapability', async (): Promise<SystemOcrCapability> => {
return imageInsightService.getSystemOcrCapability()
})
ipcMain.handle(
'system-ocr:recognize',
async (_, request: SystemOcrRequest): Promise<SystemOcrResult> => {
// 日志只记录结构性信息,不记录 base64、不记录识别正文。
console.log('[IPC] system-ocr:recognize hash=%s', request?.imageHash || 'auto')
return imageInsightService.extractLocalText(request)
}
)
ipcMain.handle('db:getSticker', async (_, cdnUrl?: string, md5?: string) => {
if (!stickerService) {
stickerService = new StickerService(chat.getChatDb()?.getWcdb4Client())
@@ -27,6 +27,8 @@ import {
isFreshImageInsight,
isHotImageCandidate
} from '../../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../../shared/system-ocr'
import { systemOcrService } from './system-ocr-service'
/**
* 单张图片的最小信息(由 renderer 从已加载的 messages 中提取并传入 main)。
@@ -312,6 +314,35 @@ class ImageInsightService {
listBySession(sessionId: string, limit?: number): ImageInsight[] {
return imageInsightsStore.listBySession(sessionId, limit)
}
// ============================================================
// 本地图片文字识别(System OCR)
// ============================================================
//
// 与 Vision 路径的关系:
// ImageInsightService 是统一编排入口,下面挂两条互不干扰的运行时——
// - Vision Model Runtime(AIProviderService,走 AI Provider,可能联网)
// - Windows System OCR Runtime(SystemOcrService,纯本地,不联网)
//
// 边界与约束:
// 1. 本地 OCR 结果属于 **派生内容**,原始消息始终是权威来源;
// 本轮不落库、不写 Knowledge、不做历史图片 backfill。
// 2. 本地 OCR 结果 **不会** 写入 image-insights.json——那是 Vision 结果的缓存,
// 两者的缓存键空间也不同(见 buildSystemOcrCacheKey)。
// 3. 这里不读取也绝不修改 AI Vision Provider / 模型配置。
/** 本机是否支持本地图片文字识别(Windows System OCR)。 */
getSystemOcrCapability(): Promise<SystemOcrCapability> {
return systemOcrService.getCapability()
}
/**
* 只做「把图片里的文字读出来」。不发网络请求,不动 AI Provider 配置。
* 失败不抛,返回带 errorCode 的结果。
*/
extractLocalText(request: SystemOcrRequest): Promise<SystemOcrResult> {
return systemOcrService.recognize(request)
}
}
export const imageInsightService = new ImageInsightService()
+558
View File
@@ -0,0 +1,558 @@
// src/main/services/system-ocr-service.ts
//
// System OCR Runtime(本地图片文字识别)。
//
// 职责边界(只做这些事):
// 1. capability detection
// 2. image normalization / preparation
// 3. OCR execution
// 4. result normalization
// 5. runtime metadata
// 6. error mapping
//
// 明确不做:
// - 不伪装成 AI Provider / Vision Model;不读写 AIVisionRuntimeConfig;
// - 不发任何网络请求;不上传原图;
// - 不遍历历史图片、不做 backfill、不写 Knowledge;
// - 不把 OCR 文本写进 image-insights.json(那是 Vision 结果的缓存)。
//
// Windows 后端:Windows.Media.Ocr.OcrEngine(经 @napi-rs/system-ocr)。
// 已实测的引擎行为(@napi-rs/system-ocr 1.2.0 / Electron 43 / Windows x64):
// - Buffer 输入只接受 PNG;JPEG / WEBP / BMP 会被判为不可识别,
// 所以本服务在边界上统一归一化成 PNG 字节再调用(不落盘)。
// - preferredLangs 只使用第一个语言;语言包缺失时引擎创建失败,
// 抛出的错误是 `Windows error 操作成功完成。 (0x00000000)`(HRESULT 为 S_OK)。
// - 空白图不会报错,返回空文本 → 映射成 OCR_EMPTY_RESULT。
// - CJK 字符之间会被引擎插入空格,结果里做归一化。
import crypto from 'node:crypto'
import { spawn } from 'node:child_process'
import {
SYSTEM_OCR_CACHE_TTL_MS,
SYSTEM_OCR_ENGINE,
SYSTEM_OCR_PROBE_PNG_BASE64,
buildSystemOcrCacheKey,
detectSystemOcrImageFormat,
mapSystemOcrNativeError,
normalizeSystemOcrText,
parseImageDataUrl,
resolveSystemOcrLanguageTag
} from '../../shared/system-ocr'
import type {
SystemOcrCapability,
SystemOcrErrorCode,
SystemOcrImageFormat,
SystemOcrLine,
SystemOcrRequest,
SystemOcrResult
} from '../../shared/system-ocr'
const NATIVE_PACKAGE = '@napi-rs/system-ocr'
const MAX_CACHE_ENTRIES = 32
const FFMPEG_TIMEOUT_MS = 10_000
interface NativeLine {
text: string
confidence: number
boundingBox: { x: number; y: number; width: number; height: number }
}
interface NativeResult {
text: string
confidence: number
lines: NativeLine[]
}
interface NativeRuntime {
version: string | null
recognize: (image: Uint8Array, accuracy?: number, languages?: string[]) => Promise<NativeResult>
}
export interface SystemOcrServiceDeps {
/** 加载 native 运行时;不可用时返回 null(不允许抛) */
loadRuntime?: () => NativeRuntime | null
/** 把输入图片转成 PNG 字节;失败返回 null */
toPngBytes?: (input: {
buffer: Buffer
format: SystemOcrImageFormat
}) => Promise<Buffer | null>
/**
* ffmpeg 可执行文件解析器。只用于 GIF/BMP/WebP/TIFF → PNG 的兜底归一化。
* main/index.ts 会注入项目统一的解析逻辑(与图片解密共用一套候选路径)。
*/
resolveFfmpegExecutable?: () => string
platform?: NodeJS.Platform
arch?: string
/** 系统 locale(如 zh-CN),用于推导 OCR 语言标签 */
locale?: () => string
}
/** 未被显式注入时的兜底:环境变量 → 打包内 ffmpeg-static → PATH。 */
const defaultResolveFfmpegExecutable = (): string => {
const fromEnvironment = String(process.env['FFMPEG_BIN'] || '').trim()
if (fromEnvironment) return fromEnvironment
try {
const bundled = require('ffmpeg-static') as string | null
if (bundled) return bundled
} catch {
// 忽略:退回到 PATH 上的 ffmpeg
}
return process.platform === 'win32' ? 'ffmpeg.exe' : 'ffmpeg'
}
const toLines = (lines: NativeLine[] | undefined): SystemOcrLine[] =>
Array.isArray(lines)
? lines.map((line) => ({
text: normalizeSystemOcrText(line.text),
confidence: typeof line.confidence === 'number' ? line.confidence : 1,
boundingBox: {
x: Number(line.boundingBox?.x ?? 0),
y: Number(line.boundingBox?.y ?? 0),
width: Number(line.boundingBox?.width ?? 0),
height: Number(line.boundingBox?.height ?? 0)
}
}))
: []
const failure = (
errorCode: SystemOcrErrorCode,
error: string,
startedAt: number
): SystemOcrResult => ({
success: false,
text: '',
lines: [],
language: null,
engine: SYSTEM_OCR_ENGINE,
durationMs: Date.now() - startedAt,
errorCode,
error
})
/** 把任意容器(gif/bmp/webp/tiff)用 ffmpeg 走内存管道转成 PNG。不落盘。 */
const convertWithFfmpeg = (buffer: Buffer, executable: string): Promise<Buffer | null> =>
new Promise((resolve) => {
let settled = false
const finish = (value: Buffer | null): void => {
if (settled) return
settled = true
resolve(value)
}
let child: ReturnType<typeof spawn>
try {
child = spawn(
executable,
[
'-hide_banner',
'-loglevel',
'error',
'-i',
'pipe:0',
'-frames:v',
'1',
'-f',
'image2pipe',
'-vcodec',
'png',
'pipe:1'
],
{ windowsHide: true }
)
} catch {
finish(null)
return
}
const chunks: Buffer[] = []
const timeout = setTimeout(() => {
try {
child.kill()
} catch {
// best-effort
}
finish(null)
}, FFMPEG_TIMEOUT_MS)
child.stdout?.on('data', (chunk: Buffer) => chunks.push(chunk))
child.on('error', () => {
clearTimeout(timeout)
finish(null)
})
child.on('close', (code) => {
clearTimeout(timeout)
finish(code === 0 && chunks.length > 0 ? Buffer.concat(chunks) : null)
})
child.stdin?.on('error', () => undefined)
child.stdin?.end(buffer)
})
class SystemOcrService {
private runtime: NativeRuntime | null = null
private runtimeLoaded = false
private capability: SystemOcrCapability | null = null
private capabilityPromise: Promise<SystemOcrCapability> | null = null
private readonly cache = new Map<string, { value: SystemOcrResult; expireAt: number }>()
private deps: SystemOcrServiceDeps = {}
constructor(deps: SystemOcrServiceDeps = {}) {
this.deps = deps
}
/** 由 main/index.ts 在 app ready 后调用(可选,用于注入 app.getLocale 等)。 */
bind(deps: SystemOcrServiceDeps): void {
this.deps = { ...this.deps, ...deps }
this.runtime = null
this.runtimeLoaded = false
this.capability = null
this.capabilityPromise = null
}
/** 仅测试用:清空探测与缓存状态。 */
reset(): void {
this.runtime = null
this.runtimeLoaded = false
this.capability = null
this.capabilityPromise = null
this.cache.clear()
}
private get platform(): NodeJS.Platform {
return this.deps.platform ?? process.platform
}
private get arch(): string {
return this.deps.arch ?? process.arch
}
private get locale(): string {
if (this.deps.locale) {
try {
return this.deps.locale()
} catch {
return ''
}
}
try {
const { app } = require('electron') as typeof import('electron')
return app?.getLocale?.() ?? ''
} catch {
return ''
}
}
private loadRuntime(): NativeRuntime | null {
if (this.runtimeLoaded) return this.runtime
this.runtimeLoaded = true
if (this.deps.loadRuntime) {
this.runtime = this.deps.loadRuntime()
return this.runtime
}
if (this.platform !== 'win32') {
this.runtime = null
return this.runtime
}
try {
// 原生模块必须在打包时 external + asarUnpack,否则这里会 MODULE_NOT_FOUND。
const nativeModule = require(NATIVE_PACKAGE) as {
recognize: NativeRuntime['recognize']
}
let version: string | null = null
try {
version = (require(`${NATIVE_PACKAGE}/package.json`) as { version?: string }).version ?? null
} catch {
version = null
}
this.runtime =
nativeModule && typeof nativeModule.recognize === 'function'
? { version, recognize: nativeModule.recognize.bind(nativeModule) }
: null
} catch (error) {
console.warn(
'[SystemOcrService] native runtime unavailable engine=%s platform=%s reason=%s',
SYSTEM_OCR_ENGINE,
this.platform,
error instanceof Error ? error.message.split('\n')[0] : String(error)
)
this.runtime = null
}
return this.runtime
}
private async toPngBytes(
buffer: Buffer,
format: SystemOcrImageFormat
): Promise<Buffer | null> {
if (this.deps.toPngBytes) return this.deps.toPngBytes({ buffer, format })
if (format === 'png') return buffer
if (format === 'jpeg') {
// 项目内已有的进程内解码能力,优先于 ffmpeg(更快、无子进程)。
try {
const { nativeImage } = require('electron') as typeof import('electron')
const image = nativeImage.createFromBuffer(buffer)
if (!image.isEmpty()) {
const png = image.toPNG()
if (png && png.length > 0) return png
}
} catch {
// 继续走 ffmpeg 兜底
}
}
try {
const resolveFfmpeg = this.deps.resolveFfmpegExecutable ?? defaultResolveFfmpegExecutable
const png = await convertWithFfmpeg(buffer, resolveFfmpeg())
return png
} catch {
return null
}
}
/** capability 探测:平台 → native 运行时 → 至少一个可用 OCR 语言。 */
async getCapability(force = false): Promise<SystemOcrCapability> {
if (!force && this.capability) return this.capability
if (!force && this.capabilityPromise) return this.capabilityPromise
this.capabilityPromise = this.detectCapability()
try {
this.capability = await this.capabilityPromise
} finally {
this.capabilityPromise = null
}
return this.capability
}
private async detectCapability(): Promise<SystemOcrCapability> {
const base: Pick<
SystemOcrCapability,
'engine' | 'platform' | 'arch' | 'runtimeVersion' | 'language'
> = {
engine: SYSTEM_OCR_ENGINE,
platform: this.platform,
arch: this.arch,
runtimeVersion: null,
language: null
}
if (this.platform !== 'win32') {
return {
...base,
available: false,
reason: 'UNSUPPORTED_PLATFORM',
message: '本地图片文字识别目前仅支持 Windows。'
}
}
const runtime = this.loadRuntime()
if (!runtime) {
return {
...base,
available: false,
reason: 'NATIVE_MODULE_MISSING',
message: '本地文字识别组件不可用,请重新安装 TraceMemo。'
}
}
const probed = await this.probeLanguage(runtime)
if (probed.reason) {
return {
...base,
runtimeVersion: runtime.version,
available: false,
reason: probed.reason,
message: probed.message
}
}
return {
...base,
runtimeVersion: runtime.version,
available: true,
language: probed.language,
message: probed.language
? `本地图片文字识别可用(Windows 系统 OCR,${probed.language})。`
: '本地图片文字识别可用(Windows 系统 OCR,跟随系统语言)。'
}
}
/**
* 用一个 64x32 纯白 PNG 探测语言可用性:引擎能创建即说明语言包可用。
* 首选「系统 locale 推导出的标签」,失败再退回「系统用户语言配置」。
*/
private async probeLanguage(
runtime: NativeRuntime
): Promise<
| { language: string | null; reason?: undefined; message?: undefined }
| { language: null; reason: 'LANGUAGE_UNAVAILABLE' | 'NATIVE_MODULE_MISSING'; message: string }
> {
const probeBuffer = Buffer.from(SYSTEM_OCR_PROBE_PNG_BASE64, 'base64')
const preferred = resolveSystemOcrLanguageTag(this.locale)
const candidates: Array<string | null> = preferred ? [preferred, null] : [null]
let lastCode: SystemOcrErrorCode = 'OCR_FAILED'
for (const candidate of candidates) {
try {
await runtime.recognize(
probeBuffer,
undefined,
candidate ? [candidate] : undefined
)
return { language: candidate }
} catch (error) {
lastCode = mapSystemOcrNativeError(
error instanceof Error ? error.message : String(error)
)
}
}
if (lastCode === 'OCR_LANGUAGE_UNAVAILABLE') {
return {
language: null,
reason: 'LANGUAGE_UNAVAILABLE',
message:
'当前 Windows 未安装可用的 OCR 语言支持,请在系统「语言和区域」里安装简体中文或英文的 OCR 语言包后重试。'
}
}
return {
language: null,
reason: 'NATIVE_MODULE_MISSING',
message: '本地文字识别引擎初始化失败,请重启 TraceMemo 或重新安装。'
}
}
private readCache(key: string, startedAt: number): SystemOcrResult | null {
const hit = this.cache.get(key)
if (!hit) return null
if (hit.expireAt <= Date.now()) {
this.cache.delete(key)
return null
}
return { ...hit.value, durationMs: Date.now() - startedAt, fromCache: true }
}
private writeCache(key: string, value: SystemOcrResult): void {
if (this.cache.size >= MAX_CACHE_ENTRIES) {
const oldest = this.cache.keys().next()
if (!oldest.done) this.cache.delete(oldest.value)
}
this.cache.set(key, { value, expireAt: Date.now() + SYSTEM_OCR_CACHE_TTL_MS })
}
/**
* 识别一张图片里的文字。
* 任意失败都不抛,统一返回 success=false + 产品级 errorCode。
*/
async recognize(request: SystemOcrRequest): Promise<SystemOcrResult> {
const startedAt = Date.now()
const parsed = parseImageDataUrl(request.imageDataUrl)
if (!parsed) {
return failure('UNSUPPORTED_IMAGE', '仅支持 PNG、JPG、JPEG、WebP、GIF、BMP 图片。', startedAt)
}
let sourceBuffer: Buffer
try {
sourceBuffer = Buffer.from(parsed.base64, 'base64')
} catch {
return failure('IMAGE_DECODE_FAILED', '图片数据无法解码。', startedAt)
}
if (sourceBuffer.length === 0) {
return failure('IMAGE_DECODE_FAILED', '图片数据为空。', startedAt)
}
const format = detectSystemOcrImageFormat(sourceBuffer)
if (!format) {
return failure('UNSUPPORTED_IMAGE', '无法识别的图片格式。', startedAt)
}
const imageHash =
request.imageHash?.trim() ||
crypto.createHash('sha256').update(sourceBuffer).digest('hex').slice(0, 32)
const requestedLanguage = request.language?.trim() || null
const capability = await this.getCapability()
if (!capability.available) {
const errorCode: SystemOcrErrorCode =
capability.reason === 'UNSUPPORTED_PLATFORM'
? 'UNSUPPORTED_PLATFORM'
: capability.reason === 'LANGUAGE_UNAVAILABLE'
? 'OCR_LANGUAGE_UNAVAILABLE'
: 'SYSTEM_OCR_UNAVAILABLE'
return failure(errorCode, capability.message, startedAt)
}
const languageForCache = requestedLanguage ?? capability.language
const cacheKey = buildSystemOcrCacheKey({
imageHash,
language: languageForCache,
runtimeVersion: capability.runtimeVersion,
platform: capability.platform
})
if (requestedLanguage === null) {
const cached = this.readCache(cacheKey, startedAt)
if (cached) return cached
}
const runtime = this.loadRuntime()
if (!runtime) {
return failure('SYSTEM_OCR_UNAVAILABLE', '本地文字识别组件不可用。', startedAt)
}
const png = await this.toPngBytes(sourceBuffer, format)
if (!png || png.length === 0 || !detectSystemOcrImageFormat(png)) {
return failure('IMAGE_DECODE_FAILED', '图片解码失败,无法读取这张图片。', startedAt)
}
const candidates: Array<string | null> = requestedLanguage
? [requestedLanguage]
: capability.language
? [capability.language, null]
: [null]
let lastErrorCode: SystemOcrErrorCode = 'OCR_FAILED'
let lastErrorMessage = ''
let usedLanguage: string | null = null
for (const candidate of candidates) {
try {
const result = await runtime.recognize(
png,
undefined,
candidate ? [candidate] : undefined
)
const text = normalizeSystemOcrText(result?.text ?? '')
const lines = toLines(result?.lines)
usedLanguage = candidate
if (!text) {
return failure('OCR_EMPTY_RESULT', '没有在这张图片里识别到文字。', startedAt)
}
const succeeded: SystemOcrResult = {
success: true,
text,
lines,
language: usedLanguage,
engine: SYSTEM_OCR_ENGINE,
durationMs: Date.now() - startedAt
}
// 生产日志只记录 error code / engine / platform / duration,绝不记录识别正文。
console.log(
'[SystemOcrService] ok engine=%s platform=%s language=%s chars=%d durationMs=%d',
SYSTEM_OCR_ENGINE,
capability.platform,
usedLanguage ?? 'system-default',
text.length,
succeeded.durationMs
)
this.writeCache(cacheKey, succeeded)
return succeeded
} catch (error) {
lastErrorMessage = error instanceof Error ? error.message : String(error)
lastErrorCode = mapSystemOcrNativeError(lastErrorMessage)
// 语言不可用才值得换下一个候选;其它错误直接结束,避免无意义重试。
if (lastErrorCode !== 'OCR_LANGUAGE_UNAVAILABLE') break
}
}
console.warn(
'[SystemOcrService] failed engine=%s platform=%s errorCode=%s durationMs=%d',
SYSTEM_OCR_ENGINE,
capability.platform,
lastErrorCode,
Date.now() - startedAt
)
return failure(
lastErrorCode,
lastErrorCode === 'OCR_LANGUAGE_UNAVAILABLE'
? '当前 Windows 未安装可用的 OCR 语言支持。'
: '本地文字识别失败,请稍后重试。',
startedAt
)
}
}
export { SystemOcrService }
export const systemOcrService = new SystemOcrService()
+4
View File
@@ -57,6 +57,7 @@ import type {
ImageCandidateQuery,
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import type { AgentHubActionResult, AgentHubLogEntry, AgentHubStatus } from '../shared/agent-hub'
import type {
PersonalWechatGeneratedTtsVoiceRequest,
@@ -660,6 +661,9 @@ declare global {
sessionId: string,
limit?: number
) => Promise<{ success: boolean; insights: ImageInsight[] }>
// 本地图片文字识别(System OCR,本地 Runtime,非 AI Provider)
getSystemOcrCapability: () => Promise<SystemOcrCapability>
recognizeLocalImageText: (request: SystemOcrRequest) => Promise<SystemOcrResult>
getPersonalWechatSenderStatus: () => Promise<PersonalWechatSenderStatus>
getPersonalWechatSendCapability: () => Promise<PersonalWechatSendCapability>
getPersonalWechatKeepOneBotProcess: () => Promise<boolean>
+6
View File
@@ -30,6 +30,7 @@ import type {
ImageCandidateQuery,
ImageInsight
} from '../shared/image-insight'
import type { SystemOcrCapability, SystemOcrRequest, SystemOcrResult } from '../shared/system-ocr'
import type { AgentHubLogEntry, AgentHubStatus } from '../shared/agent-hub'
import type {
PersonalWechatGeneratedTtsVoiceRequest,
@@ -462,6 +463,11 @@ const api = {
limit?: number
): Promise<{ success: boolean; insights: ImageInsight[] }> =>
ipcRenderer.invoke('image:listInsights', sessionId, limit),
// 本地图片文字识别(System OCR,本地 Runtime,非 AI Provider)
getSystemOcrCapability: (): Promise<SystemOcrCapability> =>
ipcRenderer.invoke('system-ocr:getCapability'),
recognizeLocalImageText: (request: SystemOcrRequest): Promise<SystemOcrResult> =>
ipcRenderer.invoke('system-ocr:recognize', request),
getPersonalWechatSenderStatus: (): Promise<PersonalWechatSenderStatus> =>
ipcRenderer.invoke('wechat-personal:getStatus'),
getPersonalWechatSendCapability: (): Promise<PersonalWechatSendCapability> =>
@@ -0,0 +1,255 @@
import { useCallback, useEffect, useState } from 'react'
import type { SystemOcrCapability, SystemOcrResult } from '../../../../../shared/system-ocr'
import { Button } from '../../../components/ui'
const MAX_FILE_BYTES = 10 * 1024 * 1024
type LocalOcrStatus = 'idle' | 'reading' | 'ready' | 'running' | 'done' | 'error'
interface LocalOcrState {
status: LocalOcrStatus
image?: {
dataUrl: string
fileName: string
size: number
}
result?: SystemOcrResult
error?: string
}
/**
* 本地图片文字识别(Windows 系统 OCR)。
*
* 这是**本地 Runtime**,不是 AI 图片理解:
* - 只把图片里的文字读出来;不描述画面、人物、场景,也不做视觉推理;
* - 原始图片不会因为这一步发给任何 AI Provider;
* - 结果只是派生内容,不会写进本地知识库。
*/
export function LocalImageTextRecognition(): React.ReactElement {
const [capability, setCapability] = useState<SystemOcrCapability | null>(null)
const [state, setState] = useState<LocalOcrState>({ status: 'idle' })
useEffect(() => {
let alive = true
void window.api
.getSystemOcrCapability()
.then((value) => {
if (alive) setCapability(value)
})
.catch(() => {
if (alive) setCapability(null)
})
return () => {
alive = false
}
}, [])
const selectImage = useCallback(async (file: File): Promise<void> => {
const extension = file.name.split('.').pop()?.toLowerCase()
const inferredType =
extension === 'png'
? 'image/png'
: extension === 'jpg' || extension === 'jpeg'
? 'image/jpeg'
: extension === 'webp'
? 'image/webp'
: extension === 'bmp'
? 'image/bmp'
: extension === 'gif'
? 'image/gif'
: ''
const mimeType = file.type === 'image/jpg' ? 'image/jpeg' : file.type || inferredType
const supportedTypes = new Set([
'image/png',
'image/jpeg',
'image/webp',
'image/bmp',
'image/gif'
])
if (!supportedTypes.has(mimeType)) {
setState({ status: 'error', error: '请选择 PNG、JPG、JPEG、WebP、BMP 或 GIF 图片' })
return
}
if (!file.size || file.size > MAX_FILE_BYTES) {
setState({ status: 'error', error: '图片大小必须在 10 MB 以内' })
return
}
setState({ status: 'reading' })
try {
const rawDataUrl = await readFileAsDataUrl(file)
const dataUrl = rawDataUrl.replace(/^data:[^;]*;/, `data:${mimeType};`)
setState({
status: 'ready',
image: { dataUrl, fileName: file.name, size: file.size }
})
} catch {
setState({ status: 'error', error: '图片无法读取,请重新选择' })
}
}, [])
const run = useCallback(async (): Promise<void> => {
const image = state.image
if (!image) return
if (!capability?.available) {
setState((current) => ({
...current,
status: 'error',
error: capability?.message || '本机当前不支持本地图片文字识别'
}))
return
}
setState((current) => ({ ...current, status: 'running', result: undefined, error: undefined }))
try {
const result = await window.api.recognizeLocalImageText({ imageDataUrl: image.dataUrl })
setState((current) =>
result.success
? { ...current, status: 'done', result, error: undefined }
: { ...current, status: 'error', result: undefined, error: localOcrErrorMessage(result) }
)
} catch {
setState((current) => ({
...current,
status: 'error',
result: undefined,
error: '本地文字识别调用失败,请重试'
}))
}
}, [capability, state.image])
const clear = useCallback(() => setState({ status: 'idle' }), [])
const running = state.status === 'running'
const result = state.result
return (
<section className="settings-card local-ocr-test">
<header>
<div>
<h2>本地图片文字识别</h2>
<p>使用 Windows 系统 OCR 在本机读取图片中的文字,原始图片无需发送给 AI Provider。</p>
</div>
<span className={`local-ocr-capability ${capability?.available ? 'supported' : ''}`}>
{capability ? (capability.available ? '本机可用' : '本机不可用') : '检测中…'}
</span>
</header>
{capability && !capability.available ? (
<p className="local-ocr-notice">{capability.message}</p>
) : null}
{capability?.available ? (
<p className="local-ocr-runtime">
引擎:Windows 系统 OCR
{capability.runtimeVersion ? ` · 组件 ${capability.runtimeVersion}` : ''}
{capability.language ? ` · 语言 ${capability.language}` : ' · 语言跟随系统'}
</p>
) : null}
<label className={`local-ocr-upload ${state.image ? 'has-image' : ''}`}>
<input
type="file"
accept=".png,.jpg,.jpeg,.webp,.bmp,.gif,image/png,image/jpeg,image/webp,image/bmp,image/gif"
onChange={(event) => {
const file = event.currentTarget.files?.[0]
if (file) void selectImage(file)
event.currentTarget.value = ''
}}
/>
{state.image ? (
<>
<img src={state.image.dataUrl} alt="本地文字识别测试预览" />
<div>
<strong>{state.image.fileName}</strong>
<small>{formatFileSize(state.image.size)} · 仅保存在内存中</small>
</div>
</>
) : (
<div>
<strong>{state.status === 'reading' ? '正在读取图片…' : '选择本地图片'}</strong>
<small>支持 PNG、JPG、JPEG、WebP、BMP、GIF,最大 10 MB</small>
</div>
)}
</label>
<p className="local-ocr-privacy">
使用本地 OCR 时,原始图片无需发送给 AI Provider,也不会写入本地缓存或知识库。如果后续继续使用云端
AI 分析,提取出的文字可能按当前 Provider 配置发送。
</p>
{state.error ? <p className="local-ocr-error">{state.error}</p> : null}
{result?.success ? (
<div className="local-ocr-result">
<h3>识别结果</h3>
<dl>
<div>
<dt>引擎</dt>
<dd>Windows 系统 OCR</dd>
</div>
<div>
<dt>语言</dt>
<dd>{result.language || '跟随系统'}</dd>
</div>
<div>
<dt>耗时</dt>
<dd>{Math.round(result.durationMs)} ms</dd>
</div>
</dl>
<pre className="local-ocr-text">{result.text}</pre>
<p className="local-ocr-hint">
本地文字识别只读取图片中的文字内容,不会描述画面、人物或场景。
</p>
</div>
) : null}
<footer>
{state.image ? (
<Button variant="outline" onClick={clear}>
移除图片
</Button>
) : null}
<Button
disabled={!state.image || running || !capability?.available}
onClick={() => void run()}
>
{running ? '识别中…' : '本地文字识别'}
</Button>
</footer>
</section>
)
}
function localOcrErrorMessage(result: SystemOcrResult): string {
switch (result.errorCode) {
case 'UNSUPPORTED_PLATFORM':
return '本地图片文字识别目前仅支持 Windows。'
case 'SYSTEM_OCR_UNAVAILABLE':
return '本地文字识别组件不可用,请重新安装 TraceMemo。'
case 'OCR_LANGUAGE_UNAVAILABLE':
return '当前 Windows 未安装可用的 OCR 语言支持,请在系统「语言和区域」中安装中文或英文语言包。'
case 'UNSUPPORTED_IMAGE':
return '这张图片的格式暂不支持本地文字识别。'
case 'IMAGE_DECODE_FAILED':
return '图片解码失败,无法读取这张图片。'
case 'OCR_EMPTY_RESULT':
return '没有在这张图片里识别到文字。'
default:
return result.error || '本地文字识别失败,请重试。'
}
}
function formatFileSize(bytes: number): string {
return bytes < 1024 * 1024
? `${Math.max(1, Math.round(bytes / 1024))} KB`
: `${(bytes / 1024 / 1024).toFixed(1)} MB`
}
function readFileAsDataUrl(file: File): Promise<string> {
return new Promise((resolve, reject) => {
const reader = new FileReader()
reader.addEventListener('load', () =>
typeof reader.result === 'string' ? resolve(reader.result) : reject(new Error('invalid image'))
)
reader.addEventListener('error', () => reject(reader.error || new Error('read failed')))
reader.readAsDataURL(file)
})
}
@@ -4,6 +4,7 @@ import { Button } from '../../../components/ui'
import { AIProviderCard } from '../ai-model/AIProviderCard'
import { AIProviderEditor } from '../ai-model/AIProviderEditor'
import { AIImageUnderstandingTest } from '../ai-model/AIImageUnderstandingTest'
import { LocalImageTextRecognition } from '../ai-model/LocalImageTextRecognition'
import { useAIModelSettingsController } from '../ai-model/useAIModelSettingsController'
export function AIModelPage({
@@ -68,6 +69,7 @@ export function AIModelPage({
onTest={() => void controller.runVisionTest()}
onClear={controller.clearVisionImage}
/>
<LocalImageTextRecognition />
{controller.state.error ? (
<p className="ai-model-page-error">{controller.state.error}</p>
) : null}
+161
View File
@@ -1605,6 +1605,167 @@
justify-content: flex-end;
}
/* 本地图片文字识别(System OCR,本地 Runtime,与 AI 图片理解刻意区分) */
.local-ocr-test {
display: grid;
gap: 16px;
}
.local-ocr-test > header,
.local-ocr-test > footer {
display: flex;
align-items: center;
justify-content: space-between;
gap: 16px;
}
.local-ocr-test h2,
.local-ocr-test h3,
.local-ocr-test p {
margin: 0;
}
.local-ocr-test h2 {
color: var(--wxex-text-primary);
font-size: 15px;
}
.local-ocr-test header p,
.local-ocr-runtime,
.local-ocr-upload small,
.local-ocr-hint {
margin-top: 4px;
color: var(--wxex-text-secondary);
font-size: 12px;
}
.local-ocr-capability {
border-radius: 999px;
padding: 5px 9px;
background: var(--wxex-bg-sidebar);
color: var(--wxex-text-secondary);
white-space: nowrap;
font-size: 12px;
}
.local-ocr-capability.supported {
background: var(--wxex-brand-soft);
color: var(--wxex-brand);
}
.local-ocr-notice {
border-left: 3px solid var(--wxex-danger);
padding: 9px 11px;
background: color-mix(in srgb, var(--wxex-danger) 12%, transparent);
color: var(--wxex-danger);
font-size: 12px;
line-height: 1.6;
}
.local-ocr-upload {
display: flex;
min-height: 112px;
align-items: center;
justify-content: center;
gap: 14px;
border: 1px dashed var(--wxex-border);
border-radius: var(--wxex-radius-md);
padding: 14px;
background: var(--wxex-bg-sidebar);
color: var(--wxex-text-primary);
text-align: center;
cursor: pointer;
}
.local-ocr-upload:hover {
border-color: var(--wxex-brand);
background: var(--wxex-brand-soft);
}
.local-ocr-upload input {
display: none;
}
.local-ocr-upload.has-image {
justify-content: flex-start;
text-align: left;
}
.local-ocr-upload img {
width: 112px;
height: 82px;
flex: 0 0 auto;
border-radius: 8px;
object-fit: cover;
}
.local-ocr-upload strong,
.local-ocr-upload small {
display: block;
}
.local-ocr-privacy {
color: var(--wxex-text-secondary);
font-size: 12px;
line-height: 1.6;
}
.local-ocr-error {
border-left: 3px solid var(--wxex-danger);
padding: 9px 11px;
background: color-mix(in srgb, var(--wxex-danger) 12%, transparent);
color: var(--wxex-danger);
font-size: 12px;
}
.local-ocr-result {
display: grid;
gap: 12px;
border: 1px solid var(--wxex-border);
border-radius: var(--wxex-radius-md);
padding: 14px;
background: var(--wxex-bg-sidebar);
}
.local-ocr-result dl {
display: grid;
grid-template-columns: repeat(3, minmax(0, 1fr));
gap: 12px;
margin: 0;
}
.local-ocr-result dt {
color: var(--wxex-text-muted);
font-size: 11px;
}
.local-ocr-result dd {
margin: 4px 0 0;
color: var(--wxex-text-primary);
font-size: 12px;
font-weight: 600;
}
.local-ocr-text {
max-height: 240px;
overflow: auto;
margin: 0;
border: 1px solid var(--wxex-border);
border-radius: 8px;
padding: 12px;
background: var(--wxex-bg-app);
color: var(--wxex-text-primary);
font-family: inherit;
font-size: 13px;
line-height: 1.7;
white-space: pre-wrap;
word-break: break-word;
}
.local-ocr-test > footer {
justify-content: flex-end;
}
.report-history-sidebar {
display: flex;
min-width: 0;
+266
View File
@@ -0,0 +1,266 @@
// src/shared/system-ocr.ts
//
// 本地系统 OCR(System OCR)共享契约。
//
// 架构边界(不要混淆):
// - System OCR 是**本地 Runtime**,不是 AI Provider,也不是 Vision Model。
// 它不占用 AIVisionRuntimeConfig.source,也不产生任何网络请求。
// - 能力边界:只把图片里的文字读出来。它不等于「理解人物 / 理解场景 /
// 描述照片 / 理解表情包语义 / 视觉推理」——那些仍然属于 Vision Model。
// - Windows 后端为 Windows.Media.Ocr.OcrEngine(经 @napi-rs/system-ocr 调用)。
// macOS 本轮只保留架构位置,未实现;Linux 不支持。
//
// 数据边界(本轮不做):
// - 不做历史图片全量 OCR、不做 Knowledge 回填、不把 OCR 文字伪装成原始聊天文字。
// 原始消息始终是权威来源,OCR 文字只是派生内容(本轮仅存在于内存)。
/** System OCR 引擎标识。这是本地 Runtime,不是 provider id。 */
export const SYSTEM_OCR_ENGINE = 'windows-system-ocr'
/** 本地 OCR 结果在内存中的缓存时长。 */
export const SYSTEM_OCR_CACHE_TTL_MS = 10 * 60 * 1000
/**
* 产品级错误码。用户可见文案由 error 字段承载,任何 native 堆栈 / HRESULT
* 都不会直接透出到 Renderer。
*/
export type SystemOcrErrorCode =
/** 运行时不可用(native binding 缺失 / 加载失败) */
| 'SYSTEM_OCR_UNAVAILABLE'
/** 当前平台不支持(Linux,或非 Windows 平台) */
| 'UNSUPPORTED_PLATFORM'
/** 图片格式不在支持范围内 */
| 'UNSUPPORTED_IMAGE'
/** 图片解码失败(格式可识别但内容损坏或无法转成 PNG) */
| 'IMAGE_DECODE_FAILED'
/** 当前 Windows 未安装对应的 OCR 语言支持 */
| 'OCR_LANGUAGE_UNAVAILABLE'
/** 引擎执行失败 */
| 'OCR_FAILED'
/** 识别成功执行,但图里没有文字 */
| 'OCR_EMPTY_RESULT'
export type SystemOcrUnavailableReason =
| 'UNSUPPORTED_PLATFORM'
| 'NATIVE_MODULE_MISSING'
| 'LANGUAGE_UNAVAILABLE'
/** 本机 System OCR 能力。UI 只用它决定是否展示「本地文字识别」入口。 */
export interface SystemOcrCapability {
/** 本机当前是否真的可以识别图片文字 */
available: boolean
engine: typeof SYSTEM_OCR_ENGINE
platform: NodeJS.Platform
arch: string
/** @napi-rs/system-ocr 运行时版本;无法读取时为 null */
runtimeVersion: string | null
/** 实际可用的 OCR 语言标签(对应 Windows 语言包);null 表示走系统用户语言 */
language: string | null
reason?: SystemOcrUnavailableReason
/** 面向用户的中文说明,可直接展示 */
message: string
}
export interface SystemOcrBoundingBox {
/** 归一化到 0..1,原点在左上角 */
x: number
y: number
width: number
height: number
}
export interface SystemOcrLine {
text: string
/** Windows 恒为 1.0 */
confidence: number
boundingBox: SystemOcrBoundingBox
}
/** 本地 OCR 结果。不包含任何 Windows handle / native 内部对象。 */
export interface SystemOcrResult {
success: boolean
/** 归一化后的文本(去掉 CJK 字符之间的引擎伪空格) */
text: string
lines: SystemOcrLine[]
/** 实际使用的 OCR 语言标签;null 表示由系统用户语言决定 */
language: string | null
engine: typeof SYSTEM_OCR_ENGINE
durationMs: number
/** 命中内存缓存时为 true */
fromCache?: boolean
errorCode?: SystemOcrErrorCode
error?: string
}
export interface SystemOcrRequest {
/** data URL(data:image/png;base64,...)。base64 只在 main 内部流转,不回传 Renderer。 */
imageDataUrl: string
/** 调用方已经算好的图片内容哈希;未传时由 main 内部计算 */
imageHash?: string
/** 指定 OCR 语言标签;默认按系统语言解析 */
language?: string
}
/**
* 缓存 key 组合。刻意与 ImageInsight 的 `imageHash` 保持不同的键空间,
* 保证远端 Vision 的旧结果永远不会被当成"本地 OCR 结果"复用,
* 也保证 System OCR 运行时升级后不会永远命中旧结果。
*/
export const buildSystemOcrCacheKey = (input: {
imageHash: string
language: string | null
runtimeVersion: string | null
platform?: string
}): string =>
[
input.imageHash,
SYSTEM_OCR_ENGINE,
input.platform ?? 'unknown',
input.language ?? 'auto',
input.runtimeVersion ?? 'unknown'
].join('|')
const CJK_CHAR =
/[\u3000-\u303f\u3040-\u30ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uff00-\uffef\uac00-\ud7af]/
/**
* Windows OCR 会在每个 CJK 字符之间插入空格("本 地 图 片")。
* 这里只删除 **两侧都是 CJK** 的空格,保留 "TraceMemo 本地图片文字识别" 里的真实分隔。
*/
export const normalizeSystemOcrText = (value: string): string => {
const source = String(value ?? '')
if (!source) return ''
let result = ''
for (let index = 0; index < source.length; index += 1) {
const char = source[index]
if (char === ' ' || char === '\u3000') {
const previous = result[result.length - 1]
let next = ''
for (let lookahead = index + 1; lookahead < source.length; lookahead += 1) {
if (source[lookahead] !== ' ' && source[lookahead] !== '\u3000') {
next = source[lookahead]
break
}
}
if (previous && next && CJK_CHAR.test(previous) && CJK_CHAR.test(next)) continue
}
result += char
}
return result.trim()
}
/** 把系统 locale(如 zh-CN / en-US)映射成 Windows OCR 语言标签。 */
const LANGUAGE_TAG_BY_LOCALE: Record<string, string> = {
zh: 'zh-Hans-CN',
'zh-cn': 'zh-Hans-CN',
'zh-hans': 'zh-Hans-CN',
'zh-hans-cn': 'zh-Hans-CN',
'zh-sg': 'zh-Hans-CN',
'zh-tw': 'zh-Hant-TW',
'zh-hant': 'zh-Hant-TW',
'zh-hant-tw': 'zh-Hant-TW',
'zh-hk': 'zh-Hant-HK',
'zh-hant-hk': 'zh-Hant-HK',
'zh-mo': 'zh-Hant-MO',
'zh-hant-mo': 'zh-Hant-MO',
en: 'en-US',
'en-us': 'en-US',
'en-gb': 'en-GB',
'en-au': 'en-AU',
'en-ca': 'en-CA',
ja: 'ja-JP',
'ja-jp': 'ja-JP',
ko: 'ko-KR',
'ko-kr': 'ko-KR',
fr: 'fr-FR',
'fr-fr': 'fr-FR',
de: 'de-DE',
'de-de': 'de-DE',
es: 'es-ES',
'es-es': 'es-ES',
it: 'it-IT',
'it-it': 'it-IT',
pt: 'pt-BR',
'pt-br': 'pt-BR',
ru: 'ru-RU',
'ru-ru': 'ru-RU'
}
export const resolveSystemOcrLanguageTag = (
locale: string | null | undefined
): string | null => {
const normalized = String(locale ?? '')
.trim()
.toLowerCase()
.replace(/_/g, '-')
if (!normalized) return null
if (LANGUAGE_TAG_BY_LOCALE[normalized]) return LANGUAGE_TAG_BY_LOCALE[normalized]
const primary = normalized.split('-')[0]
return LANGUAGE_TAG_BY_LOCALE[primary] ?? null
}
/**
* 把 native 错误映射成产品级错误码。
*
* 已确认的 Windows 行为(1.2.0):
* - 语言包缺失 / 引擎无法创建:`Windows error 操作成功完成。 (0x00000000)`
* —— TryCreateFromLanguage 返回 null 引擎但 HRESULT 是 S_OK,非常容易误判。
* - 送给解码器的字节不是可识别的图片:`Windows error Could not recognize file (0x80070005)`
*/
export const mapSystemOcrNativeError = (message: string): SystemOcrErrorCode => {
const detail = String(message ?? '')
if (!detail) return 'OCR_FAILED'
if (/Cannot find native binding|Failed to load native binding|MODULE_NOT_FOUND/i.test(detail)) {
return 'SYSTEM_OCR_UNAVAILABLE'
}
if (/\(0x00000000\)/.test(detail)) return 'OCR_LANGUAGE_UNAVAILABLE'
if (/Could not recognize file/i.test(detail)) return 'IMAGE_DECODE_FAILED'
if (/Could not open file/i.test(detail)) return 'IMAGE_DECODE_FAILED'
return 'OCR_FAILED'
}
/** 解析 data URL;只接受图片 MIME。 */
export const parseImageDataUrl = (
dataUrl: string
): { mimeType: string; base64: string } | null => {
const matched = /^data:(image\/[a-z0-9.+-]+);base64,(.+)$/i.exec(String(dataUrl ?? '').trim())
if (!matched) return null
return { mimeType: matched[1].toLowerCase(), base64: matched[2] }
}
export type SystemOcrImageFormat = 'png' | 'jpeg' | 'gif' | 'bmp' | 'webp' | 'tiff'
/** 按魔数识别格式。返回 null 表示不在支持范围内。 */
export const detectSystemOcrImageFormat = (buffer: Uint8Array): SystemOcrImageFormat | null => {
if (!buffer || buffer.length < 4) return null
const byte = (index: number): number => buffer[index]
if (byte(0) === 0x89 && byte(1) === 0x50 && byte(2) === 0x4e && byte(3) === 0x47) return 'png'
if (byte(0) === 0xff && byte(1) === 0xd8 && byte(2) === 0xff) return 'jpeg'
if (byte(0) === 0x47 && byte(1) === 0x49 && byte(2) === 0x46) return 'gif'
if (byte(0) === 0x42 && byte(1) === 0x4d) return 'bmp'
if (
byte(0) === 0x52 &&
byte(1) === 0x49 &&
byte(2) === 0x46 &&
byte(3) === 0x46 &&
buffer.length > 11 &&
byte(8) === 0x57 &&
byte(9) === 0x45 &&
byte(10) === 0x42 &&
byte(11) === 0x50
) {
return 'webp'
}
if ((byte(0) === 0x49 && byte(1) === 0x49) || (byte(0) === 0x4d && byte(1) === 0x4d)) {
return 'tiff'
}
return null
}
/**
* 64x32 纯白 PNG。仅用于 language capability 探测:
* 引擎能创建 → 该语言包可用;引擎创建失败 → 语言不可用。
* 探测耗时量级为个位数毫秒。
*/
export const SYSTEM_OCR_PROBE_PNG_BASE64 =
'iVBORw0KGgoAAAANSUhEUgAAAEAAAAAgCAIAAAAt/+nTAAAANUlEQVR42u3PAQkAAAgDMLV/59tCELYG6yT12dRzAgICAgICAgICAgICAgICAgICAgICAvcWMisDPdIJjMIAAAAASUVORK5CYII='