feat: 语音

This commit is contained in:
电摇小子
2026-08-05 09:45:59 +08:00
committed by Wxw-Gu
parent 69bc6f57e7
commit 7529a67f09
49 changed files with 3013 additions and 154 deletions
+2
View File
@@ -17,6 +17,8 @@ asarUnpack:
- resources/** - resources/**
- node_modules/ffmpeg-static/** - node_modules/ffmpeg-static/**
- node_modules/silk-wasm/** - node_modules/silk-wasm/**
- node_modules/sherpa-onnx-node/**
- node_modules/sherpa-onnx-*/**
extraResources: extraResources:
# Includes the optional WeChat connector binary for the target platform. # Includes the optional WeChat connector binary for the target platform.
- from: resources - from: resources
+3 -2
View File
@@ -7,12 +7,13 @@ export default defineConfig({
build: { build: {
rollupOptions: { rollupOptions: {
input: { input: {
index: resolve('src/main/index.ts') index: resolve('src/main/index.ts'),
voiceRecognitionWorker: resolve('src/main/voice-pipeline/voice-recognition-worker.ts')
}, },
output: { output: {
entryFileNames: '[name].js' entryFileNames: '[name].js'
}, },
external: ['koffi'] external: ['koffi', 'sherpa-onnx-node']
} }
} }
}, },
+1
View File
@@ -67,6 +67,7 @@
"jsonrepair": "^3.15.0", "jsonrepair": "^3.15.0",
"koffi": "^3.1.0", "koffi": "^3.1.0",
"openai": "^6.10.0", "openai": "^6.10.0",
"sherpa-onnx-node": "1.13.3",
"silk-wasm": "^3.7.1", "silk-wasm": "^3.7.1",
"wechat-emojis": "^1.0.2" "wechat-emojis": "^1.0.2"
}, },
+55
View File
@@ -45,6 +45,7 @@ specifiers:
react: ^19.2.1 react: ^19.2.1
react-dom: ^19.2.1 react-dom: ^19.2.1
sass: ^1.102.0 sass: ^1.102.0
sherpa-onnx-node: 1.13.3
silk-wasm: ^3.7.1 silk-wasm: ^3.7.1
typescript: ^5.9.3 typescript: ^5.9.3
vite: ^7.2.6 vite: ^7.2.6
@@ -64,6 +65,7 @@ dependencies:
jsonrepair: 3.15.0 jsonrepair: 3.15.0
koffi: 3.1.0 koffi: 3.1.0
openai: 6.10.0 openai: 6.10.0
sherpa-onnx-node: 1.13.3
silk-wasm: 3.7.1 silk-wasm: 3.7.1
wechat-emojis: 1.0.2 wechat-emojis: 1.0.2
@@ -5177,6 +5179,59 @@ packages:
resolution: {integrity: sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==} resolution: {integrity: sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==}
engines: {node: '>=8'} engines: {node: '>=8'}
/sherpa-onnx-darwin-arm64/1.13.4:
resolution: {integrity: sha512-QcYKzyrTzGSx6aKCD6hUODgRS1LetqfG57Z/+i5LCyfMlrgCvDc1lRcl9cdB+TozBsLha9QwLTlI0vmDcf5JKg==}
cpu: [arm64]
os: [darwin]
dev: false
optional: true
/sherpa-onnx-darwin-x64/1.13.3:
resolution: {integrity: sha512-TVQ35g7JIpDPB1lUDdcog+JtI0cI45ZzOnvHXm0DtWs/dgxnJXtWMY3uLRtBbLnysV9j5ljffwZ1IX9VDHsCzQ==}
cpu: [x64]
os: [darwin]
dev: false
optional: true
/sherpa-onnx-linux-arm64/1.13.4:
resolution: {integrity: sha512-RMjMRqT82BgTXypNNGmLe6ZFYhc3WEvnAGl3DdkK7qB/kuXwkL3iHhV31wAecbnWPsnEpUoD+8cFovWSBzsCuw==}
cpu: [arm64]
os: [linux]
dev: false
optional: true
/sherpa-onnx-linux-x64/1.13.4:
resolution: {integrity: sha512-WZh5NCkGPFHHpYSd78iN4OnmxQeSTGyt9uZskH+im/NFHQ7elQ7B0sLzCMeRpvJxiIKvd9C6WxIJ4hYaxClfsQ==}
cpu: [x64]
os: [linux]
dev: false
optional: true
/sherpa-onnx-node/1.13.3:
resolution: {integrity: sha512-3XEiRvfZ73QoKDZqweAkAOb+OA2LPMKcLcRY6zngjhwZ1ymV6xagdahepYzeDRIzVq0nRjhUe8IaRdlRMYoamw==}
optionalDependencies:
sherpa-onnx-darwin-arm64: 1.13.4
sherpa-onnx-darwin-x64: 1.13.3
sherpa-onnx-linux-arm64: 1.13.4
sherpa-onnx-linux-x64: 1.13.4
sherpa-onnx-win-ia32: 1.13.4
sherpa-onnx-win-x64: 1.13.4
dev: false
/sherpa-onnx-win-ia32/1.13.4:
resolution: {integrity: sha512-/JbPjldrfNv+t+uIS3MlkuhfIf5l3FHUGkRC2oRXgjRqOaVmEyP3vLlQ7dTa4J7raG5oB8c3GoPjuSWSqT9GOQ==}
cpu: [ia32]
os: [win32]
dev: false
optional: true
/sherpa-onnx-win-x64/1.13.4:
resolution: {integrity: sha512-R0PWby1VxC14TDZPq7GcfSyXSY6SAFO8Y4JwdCdqouFmeXkZ1L7Is9m98C9KxQ0dN7ZtDzhAmE/43FUs/elXRQ==}
cpu: [x64]
os: [win32]
dev: false
optional: true
/side-channel-list/1.0.0: /side-channel-list/1.0.0:
resolution: {integrity: sha512-FCLHtRD/gnpCiCHEiJLOwdmFP+wzCmDEkc9y7NsYxeF4u7Btsn1ZuwgwJGxImImHicJArLP4R0yX4c2KCrMrTA==} resolution: {integrity: sha512-FCLHtRD/gnpCiCHEiJLOwdmFP+wzCmDEkc9y7NsYxeF4u7Btsn1ZuwgwJGxImImHicJArLP4R0yX4c2KCrMrTA==}
engines: {node: '>= 0.4'} engines: {node: '>= 0.4'}
+37
View File
@@ -41,6 +41,37 @@ function validateFfmpegRuntime(runtimeResources, platform = process.platform) {
return ffmpegPath return ffmpegPath
} }
function validateSherpaRuntime(runtimeResources, platform, arch) {
const platformName = platform === 'win32' ? 'win' : platform
const basePath = path.join(
runtimeResources,
'app.asar.unpacked',
'node_modules',
'sherpa-onnx-node'
)
const nativePath = path.join(
runtimeResources,
'app.asar.unpacked',
'node_modules',
`sherpa-onnx-${platformName}-${arch}`
)
const requiredFiles = [
path.join(basePath, 'package.json'),
path.join(basePath, 'sherpa-onnx.js'),
path.join(nativePath, 'package.json'),
path.join(nativePath, 'sherpa-onnx.node')
]
const missingFiles = requiredFiles.filter((filePath) => !existsSync(filePath))
if (missingFiles.length > 0) {
throw new Error(`Missing unpacked sherpa-onnx runtime: ${missingFiles.join(', ')}`)
}
}
function normalizeBuilderArch(arch) {
if (typeof arch === 'string') return arch
return { 0: 'ia32', 1: 'x64', 2: 'armv7l', 3: 'arm64', 4: 'universal' }[arch] || String(arch)
}
function setPlistValue(plistPath, key, value) { function setPlistValue(plistPath, key, value) {
execFileSync('/usr/libexec/PlistBuddy', ['-c', `Set :${key} ${value}`, plistPath]) execFileSync('/usr/libexec/PlistBuddy', ['-c', `Set :${key} ${value}`, plistPath])
} }
@@ -49,6 +80,11 @@ exports.default = async function afterPack(context) {
const runtimeResources = getRuntimeResources(context) const runtimeResources = getRuntimeResources(context)
validateSilkWasmRuntime(runtimeResources) validateSilkWasmRuntime(runtimeResources)
const ffmpegPath = validateFfmpegRuntime(runtimeResources, context.electronPlatformName) const ffmpegPath = validateFfmpegRuntime(runtimeResources, context.electronPlatformName)
validateSherpaRuntime(
runtimeResources,
context.electronPlatformName,
normalizeBuilderArch(context.arch)
)
if (context.electronPlatformName === 'darwin') { if (context.electronPlatformName === 'darwin') {
execFileSync('/usr/bin/codesign', ['--force', '--sign', '-', ffmpegPath], { execFileSync('/usr/bin/codesign', ['--force', '--sign', '-', ffmpegPath], {
@@ -119,3 +155,4 @@ exports.default = async function afterPack(context) {
exports.getRuntimeResources = getRuntimeResources exports.getRuntimeResources = getRuntimeResources
exports.validateFfmpegRuntime = validateFfmpegRuntime exports.validateFfmpegRuntime = validateFfmpegRuntime
exports.validateSilkWasmRuntime = validateSilkWasmRuntime exports.validateSilkWasmRuntime = validateSilkWasmRuntime
exports.validateSherpaRuntime = validateSherpaRuntime
+38 -7
View File
@@ -120,8 +120,10 @@ body {
.message.system { align-items: center; } .message.system { align-items: center; }
.message.system .row { justify-content: center; } .message.system .row { justify-content: center; }
.message.system .avatar { display: none; } .message.system .avatar { display: none; }
.message.system .bubble { .message.system .message-stack {
max-width: 92%; max-width: 92%;
}
.message.system .bubble {
padding: 5px 10px; padding: 5px 10px;
border: 0; border: 0;
border-radius: 5px; border-radius: 5px;
@@ -146,9 +148,15 @@ body {
place-items: center; place-items: center;
} }
.avatar img { width: 100%; height: 100%; object-fit: cover; } .avatar img { width: 100%; height: 100%; object-fit: cover; }
.bubble { .message-stack {
min-width: 0; min-width: 0;
max-width: min(78%, 760px); max-width: min(78%, 760px);
display: grid;
gap: 6px;
}
.bubble {
min-width: 0;
max-width: 100%;
padding: 13px 15px; padding: 13px 15px;
border: 1px solid var(--border); border: 1px solid var(--border);
border-radius: 10px 18px 18px 18px; border-radius: 10px 18px 18px 18px;
@@ -158,8 +166,21 @@ body {
.sent .bubble { background: var(--mine); border-color: #c7e6d4; border-radius: 18px 10px 18px 18px; } .sent .bubble { background: var(--mine); border-color: #c7e6d4; border-radius: 18px 10px 18px 18px; }
.sender { color: var(--muted); font-size: 12px; margin-bottom: 5px; } .sender { color: var(--muted); font-size: 12px; margin-bottom: 5px; }
.content { line-height: 1.7; word-break: break-word; white-space: pre-wrap; } .content { line-height: 1.7; word-break: break-word; white-space: pre-wrap; }
.audio-wrap { width: 260px; max-width: 100%; min-width: 0; } .audio-wrap { width: 380px; max-width: 100%; min-width: 0; }
.audio { display: block; width: 100%; max-width: 100%; height: 38px; } .audio { display: block; width: 100%; max-width: 100%; height: 38px; }
.voice-transcript {
width: 100%;
max-width: 100%;
margin-top: 8px;
padding-top: 8px;
border-top: 1px solid #cad8d1;
color: #3c4742;
font-size: 13px;
line-height: 1.55;
white-space: pre-wrap;
word-break: break-word;
}
.voice-transcript.error { color: #8a5a16; border-top-color: #e4c88f; }
.media-status { .media-status {
margin-top: 8px; margin-top: 8px;
padding: 6px 8px; padding: 6px 8px;
@@ -352,6 +373,7 @@ const renderExportScript = (name: string): string => `
message.contentData && message.contentData.title, message.contentData && message.contentData.title,
message.contentData && message.contentData.quotedSender, message.contentData && message.contentData.quotedSender,
message.contentData && message.contentData.quotedContent, message.contentData && message.contentData.quotedContent,
message.voiceTranscript,
message.exportMediaName message.exportMediaName
].filter(Boolean).join(' ').toLowerCase() ].filter(Boolean).join(' ').toLowerCase()
@@ -373,6 +395,11 @@ const renderExportScript = (name: string): string => `
const audio = message.voiceDataUrl const audio = message.voiceDataUrl
? '<div class="audio-wrap"><audio class="audio" controls preload="metadata" src="' + esc(message.voiceDataUrl) + '"></audio></div>' ? '<div class="audio-wrap"><audio class="audio" controls preload="metadata" src="' + esc(message.voiceDataUrl) + '"></audio></div>'
: '' : ''
const voiceTranscript = message.voiceTranscript
? '<div class="voice-transcript">' + esc(message.voiceTranscript) + '</div>'
: message.voiceTranscriptError
? '<div class="voice-transcript error">' + esc(message.voiceTranscriptError) + '</div>'
: ''
const mediaStatus = message.exportMediaError const mediaStatus = message.exportMediaError
? '<div class="media-status">' + esc(message.exportMediaError) + '</div>' ? '<div class="media-status">' + esc(message.exportMediaError) + '</div>'
: '' : ''
@@ -387,14 +414,18 @@ const renderExportScript = (name: string): string => `
: '<div class="avatar">' + (message.exportAvatarUrl : '<div class="avatar">' + (message.exportAvatarUrl
? '<img src="' + esc(message.exportAvatarUrl) + '" alt="">' ? '<img src="' + esc(message.exportAvatarUrl) + '" alt="">'
: avatarFallback) + '</div>' : avatarFallback) + '</div>'
const text = message.content || (data.type === 'quote' ? data.title : '') const rawText = message.content || (data.type === 'quote' ? data.title : '')
const text = kindOf(message) === 'voice' && /^\\[语音(?:消息)?\\]$/.test(String(rawText).trim())
? ''
: rawText
const content = esc(text || (!media && !audio && !quote ? '[' + (message.type || '消息') + ']' : '')) const content = esc(text || (!media && !audio && !quote ? '[' + (message.type || '消息') + ']' : ''))
const contentBlock = content ? '<div class="content">' + content + '</div>' : ''
return '<article class="message' + (message.isSender ? ' sent' : '') + (isSystem ? ' system' : '') + return '<article class="message' + (message.isSender ? ' sent' : '') + (isSystem ? ' system' : '') +
'" data-index="' + archiveIndex + '" data-month="' + esc(monthKey(message)) + '">' + '" data-index="' + archiveIndex + '" data-month="' + esc(monthKey(message)) + '">' +
'<div class="time">' + esc(fullTime(message)) + '</div><div class="row">' + '<div class="time">' + esc(fullTime(message)) + '</div><div class="row">' +
(isSystem ? '' : avatar) + '<div class="bubble"><div class="sender">' + (isSystem ? '' : avatar) + '<div class="message-stack"><div class="bubble"><div class="sender">' +
(isSystem ? '' : esc(sender)) + '</div>' + media + audio + quote + (isSystem ? '' : esc(sender)) + '</div>' + media + audio + voiceTranscript + quote +
'<div class="content">' + content + '</div>' + mediaStatus + '</div></div></article>' contentBlock + mediaStatus + '</div></div></div></article>'
} }
const renderTimeline = () => { const renderTimeline = () => {
+76 -8
View File
@@ -20,6 +20,7 @@ import { StickerService } from './sticker-service'
import { getImageExportAttempts } from '../shared/export-media' import { getImageExportAttempts } from '../shared/export-media'
import { FileAssetService } from './file-asset-service' import { FileAssetService } from './file-asset-service'
import { mergeCachedSelfInfo } from './services/bootstrap-cache' import { mergeCachedSelfInfo } from './services/bootstrap-cache'
import type { VoiceRecognitionUseCase } from './voice-pipeline/voice-recognition-use-case'
const jobs = new Set<string>() const jobs = new Set<string>()
const safeFilePart = (value: string): string => const safeFilePart = (value: string): string =>
@@ -31,6 +32,37 @@ const exportStamp = (): string => {
} }
const imageKeys = new ImageKeyConfigService() const imageKeys = new ImageKeyConfigService()
const copyExportAsset = async (
source: string,
destination: string
): Promise<{ success: true } | { success: false; error: string }> => {
try {
await fs.copyFile(source, destination)
return { success: true }
} catch (error) {
try {
const [sourceStat, destinationStat] = await Promise.all([
fs.stat(source),
fs.stat(destination)
])
if (
sourceStat.isFile() &&
destinationStat.isFile() &&
sourceStat.size > 0 &&
sourceStat.size === destinationStat.size
) {
return { success: true }
}
} catch {
// The original copy error below is more useful than a secondary stat error.
}
return {
success: false,
error: error instanceof Error ? error.message : String(error)
}
}
}
export interface HtmlExportArchive { export interface HtmlExportArchive {
version: 1 version: 1
sourceId: string sourceId: string
@@ -70,6 +102,8 @@ const mergeArchiveMessage = (previous: Message, current: Message): Message => {
const preserveWhenMissing: (keyof Message)[] = [ const preserveWhenMissing: (keyof Message)[] = [
'voiceDataUrl', 'voiceDataUrl',
'voiceDuration', 'voiceDuration',
'voiceTranscript',
'voiceTranscriptError',
'exportMediaUrl', 'exportMediaUrl',
'exportMediaType', 'exportMediaType',
'exportMediaName', 'exportMediaName',
@@ -292,7 +326,11 @@ function render(format: ExportRequest['format'], messages: Message[], name: stri
].join('\n') ].join('\n')
} }
export async function runExport(request: ExportRequest, win: BrowserWindow): Promise<ExportResult> { export async function runExport(
request: ExportRequest,
win: BrowserWindow,
voiceRecognition?: Pick<VoiceRecognitionUseCase, 'recognize'>
): Promise<ExportResult> {
jobs.add(request.jobId) jobs.add(request.jobId)
const send = (p: ExportJobProgress): void => { const send = (p: ExportJobProgress): void => {
if (!win.isDestroyed()) win.webContents.send('export:progress', p) if (!win.isDestroyed()) win.webContents.send('export:progress', p)
@@ -318,6 +356,8 @@ export async function runExport(request: ExportRequest, win: BrowserWindow): Pro
message.exportMediaName = undefined message.exportMediaName = undefined
message.exportMediaError = undefined message.exportMediaError = undefined
message.voiceDataUrl = undefined message.voiceDataUrl = undefined
message.voiceTranscript = undefined
message.voiceTranscriptError = undefined
message.exportShowAvatar = request.includeAvatars !== false message.exportShowAvatar = request.includeAvatars !== false
const mappedName = message.senderId ? request.nameMap?.[message.senderId] : undefined const mappedName = message.senderId ? request.nameMap?.[message.senderId] : undefined
if (mappedName && (!message.isSender || isUsableSelfName(mappedName))) { if (mappedName && (!message.isSender || isUsableSelfName(mappedName))) {
@@ -415,6 +455,23 @@ export async function runExport(request: ExportRequest, win: BrowserWindow): Pro
await fs.writeFile(join(outputDir, 'voices', voiceName), audioBuffer) await fs.writeFile(join(outputDir, 'voices', voiceName), audioBuffer)
message.voiceDataUrl = `voices/${voiceName}` message.voiceDataUrl = `voices/${voiceName}`
message.voiceDuration = Math.max(1, Math.round(audioBuffer.length / (24000 * 2))) message.voiceDuration = Math.max(1, Math.round(audioBuffer.length / (24000 * 2)))
if (request.includeVoiceTranscripts) {
if (!voiceRecognition) {
message.voiceTranscriptError = '语音转文字服务不可用'
} else {
const recognition = await voiceRecognition.recognize({
sessionId: message.sessionId,
localId: message.localId,
createTime: message.createTime,
svrId: message.serverId
})
if (recognition.success) {
message.voiceTranscript = recognition.transcript?.trim() || '未识别出文字'
} else {
message.voiceTranscriptError = recognition.error || '语音识别失败'
}
}
}
} catch (error) { } catch (error) {
keepMediaError( keepMediaError(
request, request,
@@ -535,9 +592,13 @@ export async function runExport(request: ExportRequest, win: BrowserWindow): Pro
keepMediaError(request, message, '视频格式不支持,仅支持本地 MP4 文件') keepMediaError(request, message, '视频格式不支持,仅支持本地 MP4 文件')
} else { } else {
const name = `video_${hashPart(exportMessageKey(message, request.userMd5))}.mp4` const name = `video_${hashPart(exportMessageKey(message, request.userMd5))}.mp4`
await fs.copyFile(source, join(outputDir, 'media', name)) const copied = await copyExportAsset(source, join(outputDir, 'media', name))
message.exportMediaUrl = `media/${name}` if (copied.success) {
message.exportMediaType = 'video' message.exportMediaUrl = `media/${name}`
message.exportMediaType = 'video'
} else {
keepMediaError(request, message, `视频复制失败:${copied.error}`)
}
} }
} }
} else if (message.contentData.type === 'sticker' && stickerService) { } else if (message.contentData.type === 'sticker' && stickerService) {
@@ -565,10 +626,17 @@ export async function runExport(request: ExportRequest, win: BrowserWindow): Pro
keepMediaError(request, message, resolved.error || '本地文件附件缺失') keepMediaError(request, message, resolved.error || '本地文件附件缺失')
} else { } else {
const name = `file_${hashPart(exportMessageKey(message, request.userMd5))}_${safeFilePart(resolved.fileName)}` const name = `file_${hashPart(exportMessageKey(message, request.userMd5))}_${safeFilePart(resolved.fileName)}`
await fs.copyFile(resolved.filePath, join(outputDir, 'media', name)) const copied = await copyExportAsset(
message.exportMediaUrl = `media/${name}` resolved.filePath,
message.exportMediaType = 'file' join(outputDir, 'media', name)
message.exportMediaName = message.contentData.title || resolved.fileName )
if (copied.success) {
message.exportMediaUrl = `media/${name}`
message.exportMediaType = 'file'
message.exportMediaName = message.contentData.title || resolved.fileName
} else {
keepMediaError(request, message, `附件复制失败:${copied.error}`)
}
} }
} }
} }
+65 -2
View File
@@ -102,6 +102,8 @@ import { VideoAssetService } from './video-asset-service'
import { cancelExport, revealExport, runExport } from './export-service' import { cancelExport, revealExport, runExport } from './export-service'
import type { ExportRequest } from '../shared/export' import type { ExportRequest } from '../shared/export'
import { discoverAccounts } from './services/account-discovery' import { discoverAccounts } from './services/account-discovery'
import { VoiceRecognitionUseCase } from './voice-pipeline/voice-recognition-use-case'
import type { VoiceMessageReference } from '../shared/voice-recognition'
// electron-vite can close the child's stdout/stderr after spawning Electron. // electron-vite can close the child's stdout/stderr after spawning Electron.
// Plain console.error then throws EPIPE on a closed pipe and crashes the IPC // Plain console.error then throws EPIPE on a closed pipe and crashes the IPC
@@ -109,6 +111,7 @@ import { discoverAccounts } from './services/account-discovery'
installSafeConsole() installSafeConsole()
let voiceService: VoiceService | null = null let voiceService: VoiceService | null = null
let voiceRecognition: VoiceRecognitionUseCase | null = null
let imageDecryptService: ImageDecryptService | null = null let imageDecryptService: ImageDecryptService | null = null
let stickerService: StickerService | null = null let stickerService: StickerService | null = null
let videoAssetService: VideoAssetService | null = null let videoAssetService: VideoAssetService | null = null
@@ -422,6 +425,16 @@ function createWindow(): void {
// Electron 初始化完成并准备创建浏览器窗口后,将调用此方法 // Electron 初始化完成并准备创建浏览器窗口后,将调用此方法
// 某些 API 只能在此事件发生后使用 // 某些 API 只能在此事件发生后使用
app.whenReady().then(async () => { app.whenReady().then(async () => {
voiceRecognition = new VoiceRecognitionUseCase({
modelRoot: join(app.getPath('userData'), 'models', 'sensevoice-small-int8'),
databasePath: join(app.getPath('userData'), 'cache', 'voice-transcripts.sqlite'),
workerPath: join(__dirname, 'voiceRecognitionWorker.js')
})
voiceRecognition.modelManager.setProgressListener((status) => {
for (const window of BrowserWindow.getAllWindows()) {
if (!window.isDestroyed()) window.webContents.send('voice:modelProgress', status)
}
})
protocol.handle('wxe-media', async (request) => { protocol.handle('wxe-media', async (request) => {
const filePath = videoAssetService?.pathForUrl(request.url) const filePath = videoAssetService?.pathForUrl(request.url)
if (!filePath) return new Response('Not found', { status: 404 }) if (!filePath) return new Response('Not found', { status: 404 })
@@ -557,6 +570,7 @@ app.whenReady().then(async () => {
const sessions = await wcdb4Client.getSessionsAsync({ hydrateDisplayNames: false }) const sessions = await wcdb4Client.getSessionsAsync({ hydrateDisplayNames: false })
configureRecallProtection(wcdb4Client, resolvedRoot, settings.recallProtectionEnabled) configureRecallProtection(wcdb4Client, resolvedRoot, settings.recallProtectionEnabled)
voiceService = new VoiceService(wcdb4Client) voiceService = new VoiceService(wcdb4Client)
voiceRecognition?.connect(voiceService, resolvedRoot)
stickerService = new StickerService(wcdb4Client) stickerService = new StickerService(wcdb4Client)
videoAssetService = new VideoAssetService(wcdb4Client) videoAssetService = new VideoAssetService(wcdb4Client)
const monitoring = await wcdb4Client.startMonitor((type, json) => { const monitoring = await wcdb4Client.startMonitor((type, json) => {
@@ -926,7 +940,7 @@ app.whenReady().then(async () => {
ipcMain.handle('export:start', async (event, request: ExportRequest) => { ipcMain.handle('export:start', async (event, request: ExportRequest) => {
const window = BrowserWindow.fromWebContents(event.sender) const window = BrowserWindow.fromWebContents(event.sender)
if (!window) return { success: false, error: '窗口不可用' } if (!window) return { success: false, error: '窗口不可用' }
return runExport(request, window) return runExport(request, window, voiceRecognition || undefined)
}) })
ipcMain.handle('export:cancel', (_, jobId: string) => { ipcMain.handle('export:cancel', (_, jobId: string) => {
cancelExport(jobId) cancelExport(jobId)
@@ -971,6 +985,47 @@ app.whenReady().then(async () => {
} }
) )
ipcMain.handle('voice:getModelStatus', async () => {
if (!voiceRecognition) throw new Error('Voice recognition is not initialized')
return voiceRecognition.getModelStatus()
})
ipcMain.handle('voice:downloadModel', async () => {
if (!voiceRecognition) throw new Error('Voice recognition is not initialized')
return voiceRecognition.downloadModel()
})
ipcMain.handle(
'voice:cancelModelDownload',
() => voiceRecognition?.cancelModelDownload() || { success: false }
)
ipcMain.handle('voice:removeModel', async () => {
if (!voiceRecognition) throw new Error('Voice recognition is not initialized')
return voiceRecognition.removeModel()
})
ipcMain.handle('voice:openModelDirectory', async () => {
if (!voiceRecognition) return { success: false, error: '语音识别服务尚未初始化' }
const directory = voiceRecognition.modelManager.directory
await fsPromises.mkdir(directory, { recursive: true })
const error = await shell.openPath(directory)
return error ? { success: false, error } : { success: true }
})
ipcMain.handle('voice:recognize', (_, reference: VoiceMessageReference) => {
if (!voiceRecognition) {
return { success: false, code: 'NOT_CONNECTED', error: '语音识别服务尚未初始化' }
}
return voiceRecognition.recognize(reference)
})
ipcMain.handle(
'voice:cancelRecognition',
(_, reference: VoiceMessageReference) =>
voiceRecognition?.cancelRecognition(reference) || { success: false }
)
ipcMain.handle('db:parseMessage', async (_, content: string, messageType: number) => { ipcMain.handle('db:parseMessage', async (_, content: string, messageType: number) => {
return parseMessageContent(content, messageType) return parseMessageContent(content, messageType)
}) })
@@ -1237,6 +1292,11 @@ app.whenReady().then(async () => {
ipcMain.handle('db:reopenWithRoot', async (_, accountRoot: string) => { ipcMain.handle('db:reopenWithRoot', async (_, accountRoot: string) => {
const ok = chat.reopenWithRoot(accountRoot) const ok = chat.reopenWithRoot(accountRoot)
if (!ok) return { success: false, error: '数据库未初始化或重新打开失败' } if (!ok) return { success: false, error: '数据库未初始化或重新打开失败' }
const client = chat.getChatDb()?.getWcdb4Client()
if (client) {
voiceService = new VoiceService(client)
voiceRecognition?.connect(voiceService, client.getAccountRoot())
}
// 同步 imageKeyRoot,避免自动获取扫描到旧目录 // 同步 imageKeyRoot,避免自动获取扫描到旧目录
const settings = loadSettings() const settings = loadSettings()
if (accountRoot && accountRoot !== settings.imageKeyRoot) { if (accountRoot && accountRoot !== settings.imageKeyRoot) {
@@ -1266,6 +1326,8 @@ app.whenReady().then(async () => {
ipcMain.handle('db:disconnect', (_, options?: { closeNative?: boolean }) => { ipcMain.handle('db:disconnect', (_, options?: { closeNative?: boolean }) => {
// 断开操作保持幂等:渲染进程可能已标记断开,或主进程连接已先行失效。 // 断开操作保持幂等:渲染进程可能已标记断开,或主进程连接已先行失效。
// 即使当前未就绪,也应让用户正常返回登录页。 // 即使当前未就绪,也应让用户正常返回登录页。
voiceRecognition?.disconnect()
voiceService = null
if (options?.closeNative !== false && chat.isReady()) chat.setChatDb(null) if (options?.closeNative !== false && chat.isReady()) chat.setChatDb(null)
return { success: true } return { success: true }
}) })
@@ -1374,7 +1436,8 @@ app.on('before-quit', (event) => {
flushBootstrapCacheWritesSync() flushBootstrapCacheWritesSync()
const [, nativeCallsDrained] = await Promise.all([ const [, nativeCallsDrained] = await Promise.all([
apiServer.stop().catch(() => undefined), apiServer.stop().catch(() => undefined),
chat.closeChatDbForQuit().catch(() => false) chat.closeChatDbForQuit().catch(() => false),
voiceRecognition?.dispose().catch(() => undefined)
]) ])
if (!nativeCallsDrained) { if (!nativeCallsDrained) {
console.warn('[Shutdown] WCDB async calls did not fully drain before quit') console.warn('[Shutdown] WCDB async calls did not fully drain before quit')
+2
View File
@@ -54,6 +54,8 @@ export interface FormattedMessage {
contentData?: ReturnType<typeof parseMessageContent> contentData?: ReturnType<typeof parseMessageContent>
voiceDataUrl?: string voiceDataUrl?: string
voiceDuration?: number voiceDuration?: number
voiceTranscript?: string
voiceTranscriptError?: string
exportMediaUrl?: string exportMediaUrl?: string
exportMediaType?: 'image' | 'video' | 'sticker' | 'file' exportMediaType?: 'image' | 'video' | 'sticker' | 'file'
exportMediaName?: string exportMediaName?: string
+108
View File
@@ -0,0 +1,108 @@
import { app } from 'electron'
import { existsSync } from 'fs'
import { createRequire } from 'module'
import { join } from 'path'
import { isPackagedRuntime } from '../runtime-mode'
const nodeRequire = createRequire(import.meta.url)
export interface EncodedVoiceSource {
data: Buffer
codec: string
sourceHash: string
}
export interface DecodedVoiceAudio {
pcm: Buffer
sampleRate: number
channels: number
sourceHash: string
}
export interface VoiceAudioDecoder {
readonly codec: string
decode(source: EncodedVoiceSource): Promise<DecodedVoiceAudio>
}
export type SilkWasmRuntimeLocation = {
packagePath: string
wasmPath: string
source: 'unpacked' | 'resources' | 'asar' | 'development'
}
export function getSilkWasmRuntimeLocations(options?: {
packaged?: boolean
resourcesPath?: string
appPath?: string
}): SilkWasmRuntimeLocation[] {
const packaged = options?.packaged ?? isPackagedRuntime()
const resourcesPath = options?.resourcesPath ?? process.resourcesPath
const appPath = options?.appPath ?? app.getAppPath()
const location = (
packagePath: string,
source: SilkWasmRuntimeLocation['source']
): SilkWasmRuntimeLocation => ({
packagePath,
wasmPath: join(packagePath, 'lib', 'silk.wasm'),
source
})
if (!packaged) {
return [location(join(appPath, 'node_modules', 'silk-wasm'), 'development')]
}
return [
location(join(resourcesPath, 'app.asar.unpacked', 'node_modules', 'silk-wasm'), 'unpacked'),
location(join(resourcesPath, 'node_modules', 'silk-wasm'), 'resources'),
location(join(appPath, 'node_modules', 'silk-wasm'), 'asar')
]
}
export function findSilkWasmRuntimeLocation(
locations: SilkWasmRuntimeLocation[]
): SilkWasmRuntimeLocation | null {
return locations.find((location) => existsSync(location.wasmPath)) || null
}
export class SilkAudioDecoder implements VoiceAudioDecoder {
readonly codec = 'silk'
async decode(source: EncodedVoiceSource): Promise<DecodedVoiceAudio> {
const locations = getSilkWasmRuntimeLocations()
const runtime = findSilkWasmRuntimeLocation(locations)
if (!runtime) throw new Error('silk.wasm 未找到')
const silkWasm = nodeRequire(runtime.packagePath) as {
decode?: (data: Buffer, sampleRate: number) => Promise<{ data: Uint8Array }>
}
if (!silkWasm.decode) throw new Error('silk-wasm 运行时无效')
const result = await silkWasm.decode(source.data, 24000)
const pcm = Buffer.from(result.data)
if (!pcm.length) throw new Error('Silk 解码结果为空')
return {
pcm,
sampleRate: 24000,
channels: 1,
sourceHash: source.sourceHash
}
}
}
export class AudioDecoderRegistry {
private readonly decoders = new Map<string, VoiceAudioDecoder>()
register(decoder: VoiceAudioDecoder): this {
if (this.decoders.has(decoder.codec))
throw new Error(`Decoder already registered: ${decoder.codec}`)
this.decoders.set(decoder.codec, decoder)
return this
}
decode(source: EncodedVoiceSource): Promise<DecodedVoiceAudio> {
const decoder = this.decoders.get(source.codec)
if (!decoder) throw new Error(`Unsupported voice codec: ${source.codec}`)
return decoder.decode(source)
}
}
export function createDefaultAudioDecoderRegistry(): AudioDecoderRegistry {
return new AudioDecoderRegistry().register(new SilkAudioDecoder())
}
@@ -0,0 +1,90 @@
import type { AudioProcessor, PipelineAudio } from './types'
export const VOICE_PROCESSOR_VERSION = 'pcm16-mono-16k-v1'
export interface PcmProcessorOptions {
targetSampleRate?: number
silenceThreshold?: number
silencePaddingMs?: number
normalizePeak?: number
}
export class PcmAudioProcessor implements AudioProcessor {
private readonly targetSampleRate: number
private readonly silenceThreshold: number
private readonly silencePaddingMs: number
private readonly normalizePeak: number
constructor(options: PcmProcessorOptions = {}) {
this.targetSampleRate = options.targetSampleRate ?? 16000
this.silenceThreshold = options.silenceThreshold ?? 0.008
this.silencePaddingMs = options.silencePaddingMs ?? 80
this.normalizePeak = options.normalizePeak ?? 0.92
}
process(input: {
pcm: Buffer
sampleRate: number
channels: number
sourceHash: string
}): PipelineAudio {
if (input.channels !== 1) throw new Error('Only mono PCM is supported')
if (input.pcm.length < 2) throw new Error('PCM audio is empty')
const decoded = this.decodePcm16(input.pcm)
const trimmed = this.trimSilence(decoded, input.sampleRate)
const resampled = this.resample(trimmed, input.sampleRate, this.targetSampleRate)
const normalized = this.normalize(resampled)
return {
samples: normalized,
sampleRate: this.targetSampleRate,
channels: 1,
sourceHash: input.sourceHash,
processorVersion: VOICE_PROCESSOR_VERSION,
durationMs: Math.round((normalized.length / this.targetSampleRate) * 1000)
}
}
private decodePcm16(buffer: Buffer): Float32Array {
const output = new Float32Array(Math.floor(buffer.length / 2))
for (let index = 0; index < output.length; index += 1) {
output[index] = buffer.readInt16LE(index * 2) / 32768
}
return output
}
private trimSilence(samples: Float32Array, sampleRate: number): Float32Array {
let first = 0
while (first < samples.length && Math.abs(samples[first]) < this.silenceThreshold) first += 1
if (first === samples.length) return new Float32Array(0)
let last = samples.length - 1
while (last > first && Math.abs(samples[last]) < this.silenceThreshold) last -= 1
const padding = Math.round((sampleRate * this.silencePaddingMs) / 1000)
return samples.slice(Math.max(0, first - padding), Math.min(samples.length, last + padding + 1))
}
private resample(samples: Float32Array, sourceRate: number, targetRate: number): Float32Array {
if (sourceRate === targetRate || samples.length === 0) return samples.slice()
const outputLength = Math.max(1, Math.round((samples.length * targetRate) / sourceRate))
const output = new Float32Array(outputLength)
const ratio = sourceRate / targetRate
for (let index = 0; index < outputLength; index += 1) {
const position = index * ratio
const left = Math.min(samples.length - 1, Math.floor(position))
const right = Math.min(samples.length - 1, left + 1)
const fraction = position - left
output[index] = samples[left] + (samples[right] - samples[left]) * fraction
}
return output
}
private normalize(samples: Float32Array): Float32Array {
let peak = 0
for (const sample of samples) peak = Math.max(peak, Math.abs(sample))
if (peak < 0.001 || peak <= this.normalizePeak) return samples
const scale = this.normalizePeak / peak
return samples.map((sample) => sample * scale)
}
}
+305
View File
@@ -0,0 +1,305 @@
import { createHash } from 'crypto'
import { createReadStream } from 'fs'
import { mkdir, open, readFile, rename, rm, stat, writeFile } from 'fs/promises'
import { join } from 'path'
import type { VoiceModelDownloadResult, VoiceModelStatus } from '../../shared/voice-recognition'
import { DEFAULT_VOICE_MODEL_ID } from '../../shared/voice-recognition'
const MODEL_VERSION = '2024-07-17'
// SHA-256 values come from the repository's Git LFS object IDs. Hugging Face's
// xetHash is a storage-level hash and does not match the downloaded file bytes.
export const SENSEVOICE_MODEL_FILES = [
{
name: 'model.int8.onnx',
size: 239_233_841,
sha256: 'c71f0ce00bec95b07744e116345e33d8cbbe08cef896382cf907bf4b51a2cd51',
url: 'https://huggingface.co/csukuangfj/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17/resolve/main/model.int8.onnx'
},
{
name: 'tokens.txt',
size: 315_894,
sha256: 'f449eb28dc567533d7fa59be34e2abca8784f771850c78a47fb731a31429a1dc',
url: 'https://huggingface.co/csukuangfj/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17/resolve/main/tokens.txt'
}
] as const
const TOTAL_BYTES = SENSEVOICE_MODEL_FILES.reduce((total, file) => total + file.size, 0)
const MODEL_FINGERPRINT = createHash('sha256')
.update(SENSEVOICE_MODEL_FILES.map((file) => `${file.name}:${file.sha256}`).join('|'))
.digest('hex')
interface VerifiedManifest {
modelId: string
version: string
fingerprint: string
files: Record<string, { size: number; sha256: string }>
}
export interface VoiceModelPaths {
model: string
tokens: string
}
export class VoiceModelManager {
readonly modelId = DEFAULT_VOICE_MODEL_ID
readonly version = MODEL_VERSION
readonly fingerprint = MODEL_FINGERPRINT
private readonly modelRoot: string
private downloadController: AbortController | null = null
private downloadPromise: Promise<VoiceModelDownloadResult> | null = null
private progressBytes = 0
private lastProgressAt = 0
private progressListener: ((status: VoiceModelStatus) => void) | null = null
constructor(modelRoot: string) {
this.modelRoot = modelRoot
}
get directory(): string {
return this.modelRoot
}
setProgressListener(listener: ((status: VoiceModelStatus) => void) | null): void {
this.progressListener = listener
}
async getStatus(): Promise<VoiceModelStatus> {
if (!this.isRuntimeSupported()) {
return this.buildStatus(
'unsupported',
0,
`当前系统暂不支持离线语音识别:${process.platform} ${process.arch}`
)
}
if (this.downloadPromise) return this.buildStatus('downloading', this.progressBytes)
const verified = await this.isVerified()
if (verified) return this.buildStatus('ready', TOTAL_BYTES)
const hasFiles = await this.hasAnyModelFile()
return this.buildStatus(
hasFiles ? 'invalid' : 'missing',
0,
hasFiles ? '模型文件不完整或校验失败,请重新下载' : undefined
)
}
async getPaths(): Promise<VoiceModelPaths | null> {
if (!(await this.isVerified())) return null
return {
model: join(this.modelRoot, SENSEVOICE_MODEL_FILES[0].name),
tokens: join(this.modelRoot, SENSEVOICE_MODEL_FILES[1].name)
}
}
download(): Promise<VoiceModelDownloadResult> {
if (!this.isRuntimeSupported()) {
const status = this.buildStatus(
'unsupported',
0,
`当前系统暂不支持离线语音识别:${process.platform} ${process.arch}`
)
return Promise.resolve({ success: false, status, error: status.error })
}
if (this.downloadPromise) return this.downloadPromise
this.downloadController = new AbortController()
this.progressBytes = 0
this.downloadPromise = this.runDownload(this.downloadController.signal).finally(() => {
this.downloadPromise = null
this.downloadController = null
})
return this.downloadPromise
}
cancelDownload(): boolean {
if (!this.downloadController) return false
this.downloadController.abort()
return true
}
async remove(): Promise<VoiceModelStatus> {
if (this.downloadPromise) return this.buildStatus('downloading', this.progressBytes)
await Promise.all([
...SENSEVOICE_MODEL_FILES.flatMap((file) => [
rm(join(this.modelRoot, file.name), { force: true }),
rm(join(this.modelRoot, `${file.name}.partial`), { force: true })
]),
rm(join(this.modelRoot, 'verified.json'), { force: true }),
rm(join(this.modelRoot, 'verified.json.partial'), { force: true })
])
return this.getStatus()
}
private async runDownload(signal: AbortSignal): Promise<VoiceModelDownloadResult> {
try {
await mkdir(this.modelRoot, { recursive: true })
for (const file of SENSEVOICE_MODEL_FILES) {
await this.downloadFile(file, signal)
}
await this.writeVerifiedManifest()
const status = this.buildStatus('ready', TOTAL_BYTES)
this.reportProgress(status, true)
return { success: true, status }
} catch (error) {
await Promise.all(
SENSEVOICE_MODEL_FILES.map((file) =>
rm(join(this.modelRoot, `${file.name}.partial`), { force: true })
)
)
const cancelled = signal.aborted
const message = cancelled
? '模型下载已取消'
: error instanceof Error
? error.message
: String(error)
const status = this.buildStatus(cancelled ? 'missing' : 'error', this.progressBytes, message)
this.reportProgress(status, true)
return { success: false, status, error: message }
}
}
private async downloadFile(
file: (typeof SENSEVOICE_MODEL_FILES)[number],
signal: AbortSignal
): Promise<void> {
const target = join(this.modelRoot, file.name)
const partial = `${target}.partial`
await rm(partial, { force: true })
const response = await fetch(file.url, { signal })
if (!response.ok || !response.body) throw new Error(`模型下载失败:HTTP ${response.status}`)
const handle = await open(partial, 'w')
const hash = createHash('sha256')
let fileBytes = 0
try {
const reader = response.body.getReader()
while (true) {
const { done, value } = await reader.read()
if (done) break
if (signal.aborted) throw new DOMException('Download cancelled', 'AbortError')
const chunk = Buffer.from(value)
await handle.write(chunk)
hash.update(chunk)
fileBytes += chunk.length
this.progressBytes += chunk.length
this.reportProgress(this.buildStatus('downloading', this.progressBytes))
}
} finally {
await handle.close()
}
const digest = hash.digest('hex')
if (fileBytes !== file.size || digest !== file.sha256) {
await rm(partial, { force: true })
throw new Error(`模型文件校验失败:${file.name}`)
}
await rm(target, { force: true })
await rename(partial, target)
}
private async isVerified(): Promise<boolean> {
try {
const manifest = JSON.parse(
await readFile(join(this.modelRoot, 'verified.json'), 'utf8')
) as VerifiedManifest
if (
manifest.modelId !== this.modelId ||
manifest.version !== this.version ||
manifest.fingerprint !== this.fingerprint
) {
return false
}
for (const file of SENSEVOICE_MODEL_FILES) {
const info = await stat(join(this.modelRoot, file.name))
if (info.size !== file.size || manifest.files[file.name]?.sha256 !== file.sha256)
return false
}
return true
} catch {
return this.verifyExistingFiles()
}
}
private async verifyExistingFiles(): Promise<boolean> {
try {
for (const file of SENSEVOICE_MODEL_FILES) {
const path = join(this.modelRoot, file.name)
const info = await stat(path)
if (info.size !== file.size || (await this.hashFile(path)) !== file.sha256) return false
}
await this.writeVerifiedManifest()
return true
} catch {
return false
}
}
private async hasAnyModelFile(): Promise<boolean> {
for (const file of SENSEVOICE_MODEL_FILES) {
try {
await stat(join(this.modelRoot, file.name))
return true
} catch {
// Continue checking the remaining model files.
}
}
return false
}
private hashFile(path: string): Promise<string> {
return new Promise((resolve, reject) => {
const hash = createHash('sha256')
const stream = createReadStream(path)
stream.on('data', (chunk) => hash.update(chunk))
stream.on('error', reject)
stream.on('end', () => resolve(hash.digest('hex')))
})
}
private async writeVerifiedManifest(): Promise<void> {
const manifest: VerifiedManifest = {
modelId: this.modelId,
version: this.version,
fingerprint: this.fingerprint,
files: Object.fromEntries(
SENSEVOICE_MODEL_FILES.map((file) => [file.name, { size: file.size, sha256: file.sha256 }])
)
}
const temporary = join(this.modelRoot, 'verified.json.partial')
const target = join(this.modelRoot, 'verified.json')
await writeFile(temporary, JSON.stringify(manifest, null, 2), 'utf8')
await rm(target, { force: true })
await rename(temporary, target)
}
private buildStatus(
state: VoiceModelStatus['state'],
downloadedBytes: number,
error?: string
): VoiceModelStatus {
return {
modelId: this.modelId,
version: this.version,
state,
downloadedBytes,
totalBytes: TOTAL_BYTES,
progress: TOTAL_BYTES ? Math.min(1, downloadedBytes / TOTAL_BYTES) : 0,
platform: process.platform,
architecture: process.arch,
supported: this.isRuntimeSupported(),
error
}
}
private isRuntimeSupported(): boolean {
return (
(process.platform === 'win32' && process.arch === 'x64') ||
(process.platform === 'darwin' && (process.arch === 'x64' || process.arch === 'arm64'))
)
}
private reportProgress(status: VoiceModelStatus, force = false): void {
const now = Date.now()
if (!force && now - this.lastProgressAt < 100) return
this.lastProgressAt = now
this.progressListener?.(status)
}
}
+179
View File
@@ -0,0 +1,179 @@
import { fork, type ChildProcess } from 'child_process'
import { randomUUID } from 'crypto'
import type {
PipelineAudio,
RecognitionMetadata,
RecognitionOutput,
SpeechRecognizer
} from './types'
import type { VoiceModelManager } from './model-manager'
import {
VOICE_WORKER_PROTOCOL_VERSION,
type WorkerRecognitionRequest,
type WorkerRecognitionResponse
} from './worker-protocol'
type PendingRequest = {
resolve: (result: RecognitionOutput) => void
reject: (error: Error) => void
timer: NodeJS.Timeout
removeAbortListener: () => void
}
export class RecognitionHost {
private child: ChildProcess | null = null
private readonly pending = new Map<string, PendingRequest>()
private idleTimer: NodeJS.Timeout | null = null
constructor(
private readonly workerPath: string,
private readonly timeoutMs = 120_000,
private readonly idleTimeoutMs = 60_000
) {}
async recognize(
audio: PipelineAudio,
model: { modelPath: string; tokensPath: string; fingerprint: string },
signal?: AbortSignal
): Promise<RecognitionOutput> {
if (signal?.aborted) throw new DOMException('Recognition cancelled', 'AbortError')
const child = this.ensureChild()
const requestId = randomUUID()
const request: WorkerRecognitionRequest = {
version: VOICE_WORKER_PROTOCOL_VERSION,
type: 'recognize',
requestId,
payload: {
recognizerId: 'sensevoice',
samples: audio.samples,
sampleRate: audio.sampleRate,
modelPath: model.modelPath,
tokensPath: model.tokensPath,
modelFingerprint: model.fingerprint
}
}
return new Promise<RecognitionOutput>((resolve, reject) => {
const abort = (): void => {
this.terminate(new DOMException('Recognition cancelled', 'AbortError'))
}
signal?.addEventListener('abort', abort, { once: true })
const timer = setTimeout(() => {
this.terminate(new Error('Voice recognition timed out'))
}, this.timeoutMs)
this.pending.set(requestId, {
resolve,
reject,
timer,
removeAbortListener: () => signal?.removeEventListener('abort', abort)
})
child.send(request, (error) => {
if (error) this.finish(requestId, null, error)
})
})
}
async dispose(): Promise<void> {
this.terminate(new Error('Voice recognition host disposed'))
}
private ensureChild(): ChildProcess {
if (this.idleTimer) {
clearTimeout(this.idleTimer)
this.idleTimer = null
}
if (this.child?.connected) return this.child
const child = fork(this.workerPath, [], {
stdio: ['ignore', 'ignore', 'ignore', 'ipc'],
serialization: 'advanced',
env: { ...process.env, ELECTRON_RUN_AS_NODE: '1' }
})
child.on('message', (message: WorkerRecognitionResponse) => {
if (message?.version !== VOICE_WORKER_PROTOCOL_VERSION) return
if (message.type === 'result') {
this.finish(message.requestId, {
text: message.transcript,
language: message.language
})
} else {
this.finish(message.requestId, null, new Error(message.error))
}
})
child.once('error', (error) => this.terminate(error))
child.once('exit', (code) => {
if (this.child === child) {
this.child = null
this.rejectAll(new Error(`Voice recognition worker exited (${code ?? 'unknown'})`))
}
})
this.child = child
return child
}
private finish(requestId: string, result: RecognitionOutput | null, error?: Error): void {
const pending = this.pending.get(requestId)
if (!pending) return
this.pending.delete(requestId)
clearTimeout(pending.timer)
pending.removeAbortListener()
if (error) pending.reject(error)
else pending.resolve(result || { text: '' })
if (this.pending.size === 0) this.scheduleIdleExit()
}
private terminate(error: Error): void {
if (this.idleTimer) {
clearTimeout(this.idleTimer)
this.idleTimer = null
}
const child = this.child
this.child = null
if (child && !child.killed) child.kill()
this.rejectAll(error)
}
private rejectAll(error: Error): void {
for (const [requestId] of this.pending) this.finish(requestId, null, error)
}
private scheduleIdleExit(): void {
if (!this.child || this.idleTimer) return
this.idleTimer = setTimeout(() => {
this.idleTimer = null
this.terminate(new Error('Voice recognition worker idle timeout'))
}, this.idleTimeoutMs)
}
}
export class WorkerSpeechRecognizer implements SpeechRecognizer {
readonly metadata: RecognitionMetadata
constructor(
private readonly host: RecognitionHost,
private readonly modelManager: VoiceModelManager
) {
this.metadata = {
recognizerId: 'sensevoice',
modelVersion: modelManager.version,
modelFingerprint: modelManager.fingerprint
}
}
async recognize(audio: PipelineAudio, signal?: AbortSignal): Promise<RecognitionOutput> {
const paths = await this.modelManager.getPaths()
if (!paths) throw new Error('Voice recognition model is not ready')
return this.host.recognize(
audio,
{
modelPath: paths.model,
tokensPath: paths.tokens,
fingerprint: this.modelManager.fingerprint
},
signal
)
}
dispose(): Promise<void> {
return this.host.dispose()
}
}
@@ -0,0 +1,58 @@
import { createRequire } from 'module'
import type { WorkerRecognizerEngine, WorkerRecognizerInput } from './worker-recognizer-registry'
const nodeRequire = createRequire(import.meta.url)
interface OfflineRecognitionResult {
text?: string
lang?: string
}
interface OfflineStream {
acceptWaveform(input: { samples: Float32Array; sampleRate: number }): void
}
interface OfflineRecognizerInstance {
createStream(): OfflineStream
decodeAsync(stream: OfflineStream): Promise<OfflineRecognitionResult>
}
interface OfflineRecognizerConstructor {
createAsync(config: Record<string, unknown>): Promise<OfflineRecognizerInstance>
}
export class SenseVoiceRecognizer implements WorkerRecognizerEngine {
readonly id = 'sensevoice'
private recognizer: OfflineRecognizerInstance | null = null
private fingerprint = ''
async recognize(
input: WorkerRecognizerInput
): Promise<{ transcript: string; language?: string }> {
if (!this.recognizer || this.fingerprint !== input.modelFingerprint) {
const sherpa = nodeRequire('sherpa-onnx-node') as {
OfflineRecognizer: OfflineRecognizerConstructor
}
this.recognizer = await sherpa.OfflineRecognizer.createAsync({
featConfig: { sampleRate: input.sampleRate, featureDim: 80 },
modelConfig: {
senseVoice: {
model: input.modelPath,
language: 'auto',
useInverseTextNormalization: 1
},
tokens: input.tokensPath,
numThreads: Math.max(1, Math.min(4, Number(process.env.WXE_VOICE_THREADS) || 2)),
provider: 'cpu',
debug: 0
}
})
this.fingerprint = input.modelFingerprint
}
const stream = this.recognizer.createStream()
stream.acceptWaveform({ samples: input.samples, sampleRate: input.sampleRate })
const result = await this.recognizer.decodeAsync(stream)
return { transcript: String(result.text || '').trim(), language: result.lang || undefined }
}
}
+61
View File
@@ -0,0 +1,61 @@
type ScheduledTask<T> = {
key: string
run: (signal: AbortSignal) => Promise<T>
controller: AbortController
resolve: (value: T) => void
reject: (reason: unknown) => void
}
export class VoiceTaskScheduler {
private readonly queue: ScheduledTask<unknown>[] = []
private active: ScheduledTask<unknown> | null = null
schedule<T>(key: string, run: (signal: AbortSignal) => Promise<T>): Promise<T> {
return new Promise<T>((resolve, reject) => {
this.queue.push({
key,
run,
controller: new AbortController(),
resolve: resolve as (value: unknown) => void,
reject
})
this.pump()
})
}
cancel(key: string): boolean {
if (this.active?.key === key) {
this.active.controller.abort()
return true
}
const index = this.queue.findIndex((task) => task.key === key)
if (index < 0) return false
const [task] = this.queue.splice(index, 1)
task.controller.abort()
task.reject(new DOMException('Recognition cancelled', 'AbortError'))
return true
}
cancelAll(): void {
this.active?.controller.abort()
while (this.queue.length) {
const task = this.queue.shift()
task?.controller.abort()
task?.reject(new DOMException('Recognition cancelled', 'AbortError'))
}
}
private pump(): void {
if (this.active || this.queue.length === 0) return
const task = this.queue.shift()
if (!task) return
this.active = task
void task
.run(task.controller.signal)
.then(task.resolve, task.reject)
.finally(() => {
this.active = null
this.pump()
})
}
}
@@ -0,0 +1,113 @@
import { dirname } from 'path'
import { mkdirSync } from 'fs'
import { DatabaseSync } from 'node:sqlite'
import type { TranscriptRecord, TranscriptRepository } from './types'
type TranscriptKey = Omit<
TranscriptRecord,
'transcript' | 'language' | 'durationMs' | 'createdAt' | 'updatedAt'
>
export class SqliteTranscriptRepository implements TranscriptRepository {
private readonly database: DatabaseSync
constructor(databasePath: string) {
mkdirSync(dirname(databasePath), { recursive: true })
this.database = new DatabaseSync(databasePath)
this.database.exec(`
PRAGMA journal_mode = WAL;
CREATE TABLE IF NOT EXISTS voice_transcripts (
account_id TEXT NOT NULL,
message_identity TEXT NOT NULL,
audio_hash TEXT NOT NULL,
processor_version TEXT NOT NULL,
recognizer_id TEXT NOT NULL,
model_version TEXT NOT NULL,
model_fingerprint TEXT NOT NULL,
transcript TEXT NOT NULL,
language TEXT,
duration_ms INTEGER NOT NULL,
created_at INTEGER NOT NULL,
updated_at INTEGER NOT NULL,
PRIMARY KEY (
account_id, message_identity, audio_hash, processor_version,
recognizer_id, model_version, model_fingerprint
)
) STRICT;
`)
}
find(key: TranscriptKey): TranscriptRecord | null {
const row = this.database
.prepare(
`SELECT account_id, message_identity, audio_hash, processor_version,
recognizer_id, model_version, model_fingerprint, transcript,
language, duration_ms, created_at, updated_at
FROM voice_transcripts
WHERE account_id = ? AND message_identity = ? AND audio_hash = ?
AND processor_version = ? AND recognizer_id = ? AND model_version = ?
AND model_fingerprint = ?`
)
.get(
key.accountId,
key.messageIdentity,
key.audioHash,
key.processorVersion,
key.recognizerId,
key.modelVersion,
key.modelFingerprint
) as Record<string, unknown> | undefined
if (!row) return null
return {
accountId: String(row.account_id),
messageIdentity: String(row.message_identity),
audioHash: String(row.audio_hash),
processorVersion: String(row.processor_version),
recognizerId: String(row.recognizer_id),
modelVersion: String(row.model_version),
modelFingerprint: String(row.model_fingerprint),
transcript: String(row.transcript),
language: row.language ? String(row.language) : undefined,
durationMs: Number(row.duration_ms),
createdAt: Number(row.created_at),
updatedAt: Number(row.updated_at)
}
}
save(record: TranscriptRecord): void {
this.database
.prepare(
`INSERT INTO voice_transcripts (
account_id, message_identity, audio_hash, processor_version,
recognizer_id, model_version, model_fingerprint, transcript,
language, duration_ms, created_at, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT (
account_id, message_identity, audio_hash, processor_version,
recognizer_id, model_version, model_fingerprint
) DO UPDATE SET
transcript = excluded.transcript,
language = excluded.language,
duration_ms = excluded.duration_ms,
updated_at = excluded.updated_at`
)
.run(
record.accountId,
record.messageIdentity,
record.audioHash,
record.processorVersion,
record.recognizerId,
record.modelVersion,
record.modelFingerprint,
record.transcript,
record.language ?? null,
record.durationMs,
record.createdAt,
record.updatedAt
)
}
close(): void {
this.database.close()
}
}
+81
View File
@@ -0,0 +1,81 @@
import type { VoiceMessageReference } from '../../shared/voice-recognition'
import type { EncodedVoiceSource } from './audio-decoder'
export interface PipelineAudio {
samples: Float32Array
sampleRate: number
channels: 1
sourceHash: string
processorVersion: string
durationMs: number
}
export interface RecognitionMetadata {
recognizerId: string
modelVersion: string
modelFingerprint: string
}
export interface RecognitionOutput {
text: string
language?: string
}
export interface SourceResolver {
resolve(reference: VoiceMessageReference): Promise<EncodedVoiceSource>
}
export class SpeechRecognizerRegistry {
private readonly recognizers = new Map<string, SpeechRecognizer>()
register(recognizer: SpeechRecognizer): this {
const id = recognizer.metadata.recognizerId
if (this.recognizers.has(id)) throw new Error(`Recognizer already registered: ${id}`)
this.recognizers.set(id, recognizer)
return this
}
get(recognizerId: string): SpeechRecognizer {
const recognizer = this.recognizers.get(recognizerId)
if (!recognizer) throw new Error(`Recognizer is not registered: ${recognizerId}`)
return recognizer
}
}
export interface AudioProcessor {
process(input: {
pcm: Buffer
sampleRate: number
channels: number
sourceHash: string
}): PipelineAudio
}
export interface SpeechRecognizer {
readonly metadata: RecognitionMetadata
recognize(audio: PipelineAudio, signal?: AbortSignal): Promise<RecognitionOutput>
dispose(): Promise<void>
}
export interface TranscriptRecord extends RecognitionMetadata {
accountId: string
messageIdentity: string
audioHash: string
processorVersion: string
transcript: string
language?: string
durationMs: number
createdAt: number
updatedAt: number
}
export interface TranscriptRepository {
find(
key: Omit<
TranscriptRecord,
'transcript' | 'language' | 'durationMs' | 'createdAt' | 'updatedAt'
>
): TranscriptRecord | null
save(record: TranscriptRecord): void
close(): void
}
+88
View File
@@ -0,0 +1,88 @@
import { createHash } from 'crypto'
import type { VoiceMessageReference } from '../../shared/voice-recognition'
import type { VoiceService } from '../voice-service'
import type { AudioDecoderRegistry, EncodedVoiceSource } from './audio-decoder'
import type {
AudioProcessor,
SourceResolver,
SpeechRecognizer,
TranscriptRecord,
TranscriptRepository
} from './types'
export class VoiceSourceResolver implements SourceResolver {
constructor(private readonly voiceService: VoiceService) {}
async resolve(reference: VoiceMessageReference): Promise<EncodedVoiceSource> {
const result = await this.voiceService.resolveSource(
reference.sessionId,
reference.localId,
reference.createTime,
reference.svrId
)
if (!result.success) throw new Error(result.error)
return result.source
}
}
export class VoicePipeline {
constructor(
private readonly sourceResolver: SourceResolver,
private readonly decoderRegistry: AudioDecoderRegistry,
private readonly audioProcessor: AudioProcessor,
private readonly recognizer: SpeechRecognizer,
private readonly transcripts: TranscriptRepository
) {}
async run(
accountId: string,
reference: VoiceMessageReference,
signal?: AbortSignal
): Promise<{ transcript: string; language?: string; durationMs: number; cached: boolean }> {
const source = await this.sourceResolver.resolve(reference)
if (signal?.aborted) throw new DOMException('Recognition cancelled', 'AbortError')
const decoded = await this.decoderRegistry.decode(source)
if (signal?.aborted) throw new DOMException('Recognition cancelled', 'AbortError')
const audio = this.audioProcessor.process(decoded)
if (audio.samples.length === 0) throw new Error('Voice audio is empty after processing')
const messageIdentity = createHash('sha256')
.update(
`${reference.sessionId}|${reference.localId}|${reference.createTime}|${reference.svrId ?? ''}`
)
.digest('hex')
const key = {
accountId,
messageIdentity,
audioHash: audio.sourceHash,
processorVersion: audio.processorVersion,
...this.recognizer.metadata
}
const cached = this.transcripts.find(key)
if (cached) {
return {
transcript: cached.transcript,
language: cached.language,
durationMs: cached.durationMs,
cached: true
}
}
const output = await this.recognizer.recognize(audio, signal)
const now = Date.now()
const record: TranscriptRecord = {
...key,
transcript: output.text,
language: output.language,
durationMs: audio.durationMs,
createdAt: now,
updatedAt: now
}
this.transcripts.save(record)
return {
transcript: output.text,
language: output.language,
durationMs: audio.durationMs,
cached: false
}
}
}
@@ -0,0 +1,119 @@
import { createHash } from 'crypto'
import type {
VoiceMessageReference,
VoiceModelDownloadResult,
VoiceModelStatus,
VoiceRecognitionResult
} from '../../shared/voice-recognition'
import type { VoiceService } from '../voice-service'
import { PcmAudioProcessor } from './audio-processor'
import { createDefaultAudioDecoderRegistry } from './audio-decoder'
import { VoiceModelManager } from './model-manager'
import { RecognitionHost, WorkerSpeechRecognizer } from './recognition-host'
import { VoiceTaskScheduler } from './task-scheduler'
import { SqliteTranscriptRepository } from './transcript-repository'
import { VoicePipeline, VoiceSourceResolver } from './voice-pipeline'
import { SpeechRecognizerRegistry } from './types'
export class VoiceRecognitionUseCase {
readonly modelManager: VoiceModelManager
private readonly scheduler = new VoiceTaskScheduler()
private readonly transcripts: SqliteTranscriptRepository
private readonly recognizer: WorkerSpeechRecognizer
private readonly recognizers = new SpeechRecognizerRegistry()
private pipeline: VoicePipeline | null = null
private accountId = ''
constructor(options: { modelRoot: string; databasePath: string; workerPath: string }) {
this.modelManager = new VoiceModelManager(options.modelRoot)
this.transcripts = new SqliteTranscriptRepository(options.databasePath)
this.recognizer = new WorkerSpeechRecognizer(
new RecognitionHost(options.workerPath),
this.modelManager
)
this.recognizers.register(this.recognizer)
}
connect(voiceService: VoiceService, accountRoot: string): void {
this.scheduler.cancelAll()
this.accountId = createHash('sha256')
.update(
accountRoot
.trim()
.replace(/[\\/]+$/, '')
.toLowerCase()
)
.digest('hex')
this.pipeline = new VoicePipeline(
new VoiceSourceResolver(voiceService),
createDefaultAudioDecoderRegistry(),
new PcmAudioProcessor(),
this.recognizers.get('sensevoice'),
this.transcripts
)
}
disconnect(): void {
this.scheduler.cancelAll()
this.pipeline = null
this.accountId = ''
}
getModelStatus(): Promise<VoiceModelStatus> {
return this.modelManager.getStatus()
}
downloadModel(): Promise<VoiceModelDownloadResult> {
return this.modelManager.download()
}
cancelModelDownload(): { success: boolean } {
return { success: this.modelManager.cancelDownload() }
}
async removeModel(): Promise<VoiceModelStatus> {
this.scheduler.cancelAll()
await this.recognizer.dispose()
return this.modelManager.remove()
}
recognize(reference: VoiceMessageReference): Promise<VoiceRecognitionResult> {
const pipeline = this.pipeline
const accountId = this.accountId
if (!pipeline || !accountId) {
return Promise.resolve({ success: false, code: 'NOT_CONNECTED', error: '请先连接微信数据库' })
}
const key = this.taskKey(reference)
return this.scheduler
.schedule(key, async (signal) => {
const status = await this.modelManager.getStatus()
if (status.state !== 'ready') {
return { success: false, code: 'MODEL_NOT_READY', error: '请先下载语音识别模型' } as const
}
const result = await pipeline.run(accountId, reference, signal)
return { success: true, ...result } as const
})
.catch((error): VoiceRecognitionResult => {
if (error instanceof DOMException && error.name === 'AbortError') {
return { success: false, code: 'CANCELLED', error: '语音识别已取消' }
}
const message = error instanceof Error ? error.message : String(error)
const code = message.toLowerCase().includes('timed out') ? 'TIMEOUT' : 'RECOGNITION_FAILED'
return { success: false, code, error: message }
})
}
cancelRecognition(reference: VoiceMessageReference): { success: boolean } {
return { success: this.scheduler.cancel(this.taskKey(reference)) }
}
async dispose(): Promise<void> {
this.scheduler.cancelAll()
await this.recognizer.dispose()
this.transcripts.close()
}
private taskKey(reference: VoiceMessageReference): string {
return `${this.accountId}:${reference.sessionId}:${reference.localId}:${reference.createTime}`
}
}
@@ -0,0 +1,43 @@
import { SenseVoiceRecognizer } from './sensevoice-recognizer'
import { WorkerRecognizerRegistry } from './worker-recognizer-registry'
import {
VOICE_WORKER_PROTOCOL_VERSION,
type WorkerRecognitionRequest,
type WorkerRecognitionResponse
} from './worker-protocol'
const recognizers = new WorkerRecognizerRegistry().register(new SenseVoiceRecognizer())
function send(response: WorkerRecognitionResponse): void {
if (process.send) process.send(response)
}
process.on('message', async (message: WorkerRecognitionRequest) => {
if (
message?.version !== VOICE_WORKER_PROTOCOL_VERSION ||
message.type !== 'recognize' ||
!message.requestId
) {
return
}
try {
const fakeTranscript = process.env.WXE_VOICE_RECOGNITION_FAKE_TEXT
const result = fakeTranscript
? { transcript: fakeTranscript, language: 'zh' }
: await recognizers.get(message.payload.recognizerId).recognize(message.payload)
send({
version: VOICE_WORKER_PROTOCOL_VERSION,
type: 'result',
requestId: message.requestId,
...result
})
} catch (error) {
send({
version: VOICE_WORKER_PROTOCOL_VERSION,
type: 'error',
requestId: message.requestId,
error: error instanceof Error ? error.message : String(error)
})
}
})
@@ -0,0 +1,30 @@
export const VOICE_WORKER_PROTOCOL_VERSION = 1
export interface WorkerRecognitionRequest {
version: typeof VOICE_WORKER_PROTOCOL_VERSION
type: 'recognize'
requestId: string
payload: {
recognizerId: string
samples: Float32Array
sampleRate: number
modelPath: string
tokensPath: string
modelFingerprint: string
}
}
export type WorkerRecognitionResponse =
| {
version: typeof VOICE_WORKER_PROTOCOL_VERSION
type: 'result'
requestId: string
transcript: string
language?: string
}
| {
version: typeof VOICE_WORKER_PROTOCOL_VERSION
type: 'error'
requestId: string
error: string
}
@@ -0,0 +1,29 @@
export interface WorkerRecognizerInput {
samples: Float32Array
sampleRate: number
modelPath: string
tokensPath: string
modelFingerprint: string
}
export interface WorkerRecognizerEngine {
readonly id: string
recognize(input: WorkerRecognizerInput): Promise<{ transcript: string; language?: string }>
}
export class WorkerRecognizerRegistry {
private readonly engines = new Map<string, WorkerRecognizerEngine>()
register(engine: WorkerRecognizerEngine): this {
if (this.engines.has(engine.id))
throw new Error(`Worker recognizer already registered: ${engine.id}`)
this.engines.set(engine.id, engine)
return this
}
get(id: string): WorkerRecognizerEngine {
const engine = this.engines.get(id)
if (!engine) throw new Error(`Worker recognizer is not registered: ${id}`)
return engine
}
}
+82 -102
View File
@@ -1,55 +1,37 @@
import { app } from 'electron' import { createHash } from 'crypto'
import { join } from 'path'
import { existsSync } from 'fs'
import { createRequire } from 'module'
import { Wcdb4Client } from './wcdb4-client' import { Wcdb4Client } from './wcdb4-client'
import { isPackagedRuntime } from './runtime-mode' import {
createDefaultAudioDecoderRegistry,
type EncodedVoiceSource
} from './voice-pipeline/audio-decoder'
const nodeRequire = createRequire(import.meta.url) export {
findSilkWasmRuntimeLocation,
getSilkWasmRuntimeLocations,
type SilkWasmRuntimeLocation
} from './voice-pipeline/audio-decoder'
export type SilkWasmRuntimeLocation = { export interface ResolvedPcmAudio {
packagePath: string pcm: Buffer
wasmPath: string sampleRate: number
source: 'unpacked' | 'resources' | 'asar' | 'development' channels: number
codec: 'silk'
sourceHash: string
} }
export function getSilkWasmRuntimeLocations(options?: { export type ResolvePcmResult =
packaged?: boolean | { success: true; audio: ResolvedPcmAudio }
resourcesPath?: string | { success: false; error: string }
appPath?: string
}): SilkWasmRuntimeLocation[] {
const packaged = options?.packaged ?? isPackagedRuntime()
const resourcesPath = options?.resourcesPath ?? process.resourcesPath
const appPath = options?.appPath ?? app.getAppPath()
const location = (
packagePath: string,
source: SilkWasmRuntimeLocation['source']
): SilkWasmRuntimeLocation => ({
packagePath,
wasmPath: join(packagePath, 'lib', 'silk.wasm'),
source
})
if (!packaged) { export type ResolveSourceResult =
return [location(join(appPath, 'node_modules', 'silk-wasm'), 'development')] | { success: true; source: EncodedVoiceSource }
} | { success: false; error: string }
return [
location(join(resourcesPath, 'app.asar.unpacked', 'node_modules', 'silk-wasm'), 'unpacked'),
location(join(resourcesPath, 'node_modules', 'silk-wasm'), 'resources'),
location(join(appPath, 'node_modules', 'silk-wasm'), 'asar')
]
}
export function findSilkWasmRuntimeLocation(
locations: SilkWasmRuntimeLocation[]
): SilkWasmRuntimeLocation | null {
return locations.find((location) => existsSync(location.wasmPath)) || null
}
export class VoiceService { export class VoiceService {
private wcdb4Client: Wcdb4Client private wcdb4Client: Wcdb4Client
private voiceCache = new Map<string, string>() private voiceCache = new Map<string, string>()
private pcmCache = new Map<string, ResolvedPcmAudio>()
private readonly decoderRegistry = createDefaultAudioDecoderRegistry()
constructor(wcdb4Client: Wcdb4Client) { constructor(wcdb4Client: Wcdb4Client) {
this.wcdb4Client = wcdb4Client this.wcdb4Client = wcdb4Client
@@ -69,39 +51,9 @@ export class VoiceService {
return { success: true, data: cached } return { success: true, data: cached }
} }
const candidates = this.buildCandidates(sessionId) const pcmResult = await this.resolvePcm(sessionId, localId, createTime, svrId)
console.log('[VoiceService] resolving voice:', { sessionId, localId, createTime, candidates }) if (!pcmResult.success) return pcmResult
const pcmData = pcmResult.audio.pcm
const voiceResult = await this.wcdb4Client.getVoiceData(
sessionId,
createTime,
candidates,
localId,
svrId || 0
)
if (!voiceResult.success || !voiceResult.hex) {
console.log('[VoiceService] getVoiceData failed:', voiceResult.error)
return { success: false, error: voiceResult.error || '获取语音数据失败' }
}
console.log('[VoiceService] got hex data, length:', voiceResult.hex.length)
const silkData = this.decodeVoiceBlob(voiceResult.hex)
if (!silkData || silkData.length === 0) {
console.log('[VoiceService] decodeVoiceBlob failed, hex:', voiceResult.hex.substring(0, 100))
return { success: false, error: '语音数据为空' }
}
console.log('[VoiceService] silkData length:', silkData.length)
const pcmData = await this.decodeSilkToPcm(silkData, 24000)
if (!pcmData || pcmData.length === 0) {
console.log('[VoiceService] decodeSilkToPcm failed')
return { success: false, error: 'Silk 解码失败' }
}
console.log('[VoiceService] pcmData length:', pcmData.length)
const wavData = this.createWavBuffer(pcmData, 24000) const wavData = this.createWavBuffer(pcmData, 24000)
console.log( console.log(
@@ -118,6 +70,61 @@ export class VoiceService {
return { success: true, data: base64Data } return { success: true, data: base64Data }
} }
async resolvePcm(
sessionId: string,
localId: number,
createTime: number,
svrId?: string | number
): Promise<ResolvePcmResult> {
const cacheKey = this.buildCacheKey(sessionId, localId, createTime)
const cached = this.pcmCache.get(cacheKey)
if (cached) return { success: true, audio: cached }
const sourceResult = await this.resolveSource(sessionId, localId, createTime, svrId)
if (!sourceResult.success) return sourceResult
try {
const decoded = await this.decoderRegistry.decode(sourceResult.source)
const audio: ResolvedPcmAudio = { ...decoded, codec: 'silk' }
this.pcmCache.set(cacheKey, audio)
return { success: true, audio }
} catch (error) {
console.error('[VoiceService] audio decode failed:', error)
return { success: false, error: error instanceof Error ? error.message : 'Silk 解码失败' }
}
}
async resolveSource(
sessionId: string,
localId: number,
createTime: number,
svrId?: string | number
): Promise<ResolveSourceResult> {
const candidates = this.buildCandidates(sessionId)
const voiceResult = await this.wcdb4Client.getVoiceData(
sessionId,
createTime,
candidates,
localId,
svrId || 0
)
if (!voiceResult.success || !voiceResult.hex) {
return { success: false, error: voiceResult.error || '获取语音数据失败' }
}
const silkData = this.decodeVoiceBlob(voiceResult.hex)
if (!silkData?.length) return { success: false, error: '语音数据为空' }
return {
success: true,
source: {
data: silkData,
codec: 'silk',
sourceHash: createHash('sha256').update(silkData).digest('hex')
}
}
}
private buildCacheKey(sessionId: string, localId: number, createTime: number): string { private buildCacheKey(sessionId: string, localId: number, createTime: number): string {
return `${sessionId}-${localId}-${createTime}` return `${sessionId}-${localId}-${createTime}`
} }
@@ -142,33 +149,6 @@ export class VoiceService {
} }
} }
private async decodeSilkToPcm(silkData: Buffer, sampleRate: number): Promise<Buffer | null> {
try {
const locations = getSilkWasmRuntimeLocations()
const runtime = findSilkWasmRuntimeLocation(locations)
if (!runtime) {
console.error(
'[VoiceService] silk.wasm not found. checked:',
locations.map((location) => location.wasmPath)
)
return null
}
const silkWasm = nodeRequire(runtime.packagePath)
if (!silkWasm || !silkWasm.decode) {
console.error('[VoiceService] silk-wasm module invalid:', runtime.packagePath)
return null
}
console.log('[VoiceService] using silk-wasm runtime:', runtime.source)
const result = await silkWasm.decode(silkData, sampleRate)
return Buffer.from(result.data)
} catch (e) {
console.error('[VoiceService] decodeSilkToPcm error:', e)
return null
}
}
private createWavBuffer( private createWavBuffer(
pcmData: Buffer, pcmData: Buffer,
sampleRate: number = 24000, sampleRate: number = 24000,
+15
View File
@@ -45,6 +45,13 @@ import type { AppLogEntry } from '../shared/app-log'
import type { AppUpdateCheckResult, AppUpdateState } from '../shared/app-update' import type { AppUpdateCheckResult, AppUpdateState } from '../shared/app-update'
import type { CacheSummary } from '../shared/cache' import type { CacheSummary } from '../shared/cache'
import type { ExportRequest, ExportJobProgress, ExportResult } from '../shared/export' import type { ExportRequest, ExportJobProgress, ExportResult } from '../shared/export'
import type {
VoiceMessageReference,
VoiceModelDownloadResult,
VoiceModelProgressEvent,
VoiceModelStatus,
VoiceRecognitionResult
} from '../shared/voice-recognition'
export type ParsedContent = export type ParsedContent =
| { type: 'text'; content: string } | { type: 'text'; content: string }
@@ -205,6 +212,14 @@ declare global {
createTime: number, createTime: number,
svrId?: string | number svrId?: string | number
) => Promise<{ success: boolean; data?: string; error?: string }> ) => Promise<{ success: boolean; data?: string; error?: string }>
getVoiceModelStatus: () => Promise<VoiceModelStatus>
downloadVoiceModel: () => Promise<VoiceModelDownloadResult>
cancelVoiceModelDownload: () => Promise<{ success: boolean }>
removeVoiceModel: () => Promise<VoiceModelStatus>
openVoiceModelDirectory: () => Promise<{ success: boolean; error?: string }>
recognizeVoice: (reference: VoiceMessageReference) => Promise<VoiceRecognitionResult>
cancelVoiceRecognition: (reference: VoiceMessageReference) => Promise<{ success: boolean }>
onVoiceModelProgress: (callback: (status: VoiceModelProgressEvent) => void) => () => void
parseMessage: (content: string, messageType: number) => Promise<ParsedContent> parseMessage: (content: string, messageType: number) => Promise<ParsedContent>
getImage: ( getImage: (
imageMd5?: string, imageMd5?: string,
+25
View File
@@ -22,6 +22,13 @@ import type { CacheSummary } from '../shared/cache'
import type { ExportRequest, ExportJobProgress } from '../shared/export' import type { ExportRequest, ExportJobProgress } from '../shared/export'
import type { ImageDecoderSelectionResult, ImageDecoderStatus } from '../shared/image-decryption' import type { ImageDecoderSelectionResult, ImageDecoderStatus } from '../shared/image-decryption'
import type { AccountDiscoveryResult } from '../shared/database-key' import type { AccountDiscoveryResult } from '../shared/database-key'
import type {
VoiceMessageReference,
VoiceModelDownloadResult,
VoiceModelProgressEvent,
VoiceModelStatus,
VoiceRecognitionResult
} from '../shared/voice-recognition'
// 渲染器的自定义 API // 渲染器的自定义 API
const api = { const api = {
@@ -74,6 +81,24 @@ const api = {
copyImage: (base64String) => ipcRenderer.invoke('copy-image', base64String), copyImage: (base64String) => ipcRenderer.invoke('copy-image', base64String),
getVoiceData: (sessionId: string, localId: number, createTime: number, svrId?: string | number) => getVoiceData: (sessionId: string, localId: number, createTime: number, svrId?: string | number) =>
ipcRenderer.invoke('db:getVoiceData', sessionId, localId, createTime, svrId), ipcRenderer.invoke('db:getVoiceData', sessionId, localId, createTime, svrId),
getVoiceModelStatus: (): Promise<VoiceModelStatus> => ipcRenderer.invoke('voice:getModelStatus'),
downloadVoiceModel: (): Promise<VoiceModelDownloadResult> =>
ipcRenderer.invoke('voice:downloadModel'),
cancelVoiceModelDownload: (): Promise<{ success: boolean }> =>
ipcRenderer.invoke('voice:cancelModelDownload'),
removeVoiceModel: (): Promise<VoiceModelStatus> => ipcRenderer.invoke('voice:removeModel'),
openVoiceModelDirectory: (): Promise<{ success: boolean; error?: string }> =>
ipcRenderer.invoke('voice:openModelDirectory'),
recognizeVoice: (reference: VoiceMessageReference): Promise<VoiceRecognitionResult> =>
ipcRenderer.invoke('voice:recognize', reference),
cancelVoiceRecognition: (reference: VoiceMessageReference): Promise<{ success: boolean }> =>
ipcRenderer.invoke('voice:cancelRecognition', reference),
onVoiceModelProgress: (callback: (status: VoiceModelProgressEvent) => void) => {
const listener = (_event: Electron.IpcRendererEvent, status: VoiceModelProgressEvent): void =>
callback(status)
ipcRenderer.on('voice:modelProgress', listener)
return () => ipcRenderer.removeListener('voice:modelProgress', listener)
},
parseMessage: (content: string, messageType: number) => parseMessage: (content: string, messageType: number) =>
ipcRenderer.invoke('db:parseMessage', content, messageType), ipcRenderer.invoke('db:parseMessage', content, messageType),
getImage: ( getImage: (
+12
View File
@@ -244,6 +244,18 @@ function App(): React.ReactElement {
const timer = window.setTimeout(() => setReportNotice(''), 3200) const timer = window.setTimeout(() => setReportNotice(''), 3200)
return () => window.clearTimeout(timer) return () => window.clearTimeout(timer)
}, [reportNotice]) }, [reportNotice])
React.useEffect(() => {
const openVoiceRecognitionSettings = (): void => {
setSettingsCategory('voice-recognition')
setActivePage('settings')
}
window.addEventListener('wxe:open-voice-recognition-settings', openVoiceRecognitionSettings)
return () =>
window.removeEventListener(
'wxe:open-voice-recognition-settings',
openVoiceRecognitionSettings
)
}, [])
React.useEffect(() => { React.useEffect(() => {
void window.api.getSettings().then((result) => { void window.api.getSettings().then((result) => {
setAppearanceSettings({ setAppearanceSettings({
+97 -27
View File
@@ -1,5 +1,6 @@
import { useState, useRef, useEffect, useCallback } from 'react' import { useState, useRef, useEffect, useCallback, useMemo } from 'react'
import type { JSX } from 'react' import type { JSX, MouseEvent as ReactMouseEvent } from 'react'
import type { VoiceMessageReference, VoiceModelStatus } from '../../../shared/voice-recognition'
interface VoicePlayerProps { interface VoicePlayerProps {
sessionId: string sessionId: string
@@ -24,8 +25,16 @@ export function VoicePlayer({
const [loading, setLoading] = useState(false) const [loading, setLoading] = useState(false)
const [error, setError] = useState<string | null>(null) const [error, setError] = useState<string | null>(null)
const [audioDuration, setAudioDuration] = useState<number | undefined>(duration) const [audioDuration, setAudioDuration] = useState<number | undefined>(duration)
const [modelStatus, setModelStatus] = useState<VoiceModelStatus | null>(null)
const [transcribing, setTranscribing] = useState(false)
const [transcript, setTranscript] = useState<string | null>(null)
const [transcriptError, setTranscriptError] = useState<string | null>(null)
const audioRef = useRef<HTMLAudioElement | null>(null) const audioRef = useRef<HTMLAudioElement | null>(null)
const objectUrlRef = useRef<string | null>(null) const objectUrlRef = useRef<string | null>(null)
const voiceReference = useMemo<VoiceMessageReference>(
() => ({ sessionId, localId, createTime, svrId }),
[createTime, localId, sessionId, svrId]
)
const stopCurrentAndPlay = useCallback((audio: HTMLAudioElement) => { const stopCurrentAndPlay = useCallback((audio: HTMLAudioElement) => {
if (globalCurrentAudio && globalCurrentAudio !== audio) { if (globalCurrentAudio && globalCurrentAudio !== audio) {
@@ -144,6 +153,46 @@ export function VoicePlayer({
} }
}, []) }, [])
const handleTranscribe = useCallback(
async (event: ReactMouseEvent<HTMLButtonElement>) => {
event.stopPropagation()
setTranscriptError(null)
const status = await window.api.getVoiceModelStatus()
setModelStatus(status)
if (status.state !== 'ready') return
setTranscribing(true)
try {
const result = await window.api.recognizeVoice(voiceReference)
if (result.success) {
setTranscript(result.transcript?.trim() || '未识别出文字')
setModelStatus(null)
} else if (result.code !== 'CANCELLED') {
setTranscriptError(result.error || '语音识别失败')
}
} catch (recognitionError) {
console.warn('[VoicePlayer] recognition failed:', recognitionError)
setTranscriptError('语音识别失败,请重试')
} finally {
setTranscribing(false)
}
},
[voiceReference]
)
const handleCancelRecognition = useCallback(
async (event: ReactMouseEvent<HTMLButtonElement>) => {
event.stopPropagation()
await window.api.cancelVoiceRecognition(voiceReference)
},
[voiceReference]
)
const handleOpenVoiceSettings = useCallback((event: ReactMouseEvent<HTMLButtonElement>) => {
event.stopPropagation()
window.dispatchEvent(new Event('wxe:open-voice-recognition-settings'))
}, [])
const formatDuration = (seconds: number | undefined): string => { const formatDuration = (seconds: number | undefined): string => {
if (!seconds || !isFinite(seconds)) return '0:00' if (!seconds || !isFinite(seconds)) return '0:00'
const mins = Math.floor(seconds / 60) const mins = Math.floor(seconds / 60)
@@ -151,33 +200,54 @@ export function VoicePlayer({
return `${mins}:${secs.toString().padStart(2, '0')}` return `${mins}:${secs.toString().padStart(2, '0')}`
} }
if (loading) {
return (
<div className="voice-message voice-loading">
<span className="voice-icon"></span>
<span className="voice-loading-text">...</span>
</div>
)
}
if (error && !audioUrl) {
return (
<div className="voice-message voice-error" onClick={handlePlayPause}>
<span className="voice-icon"></span>
<span className="voice-error-text"></span>
</div>
)
}
return ( return (
<div className="voice-message" onClick={handlePlayPause}> <div className="voice-player">
<span className={`voice-icon ${isPlaying ? 'playing' : ''}`}>{isPlaying ? '⏸' : '▶'}</span> <div
<div className="voice-bars" aria-hidden="true"> className={`voice-message ${loading ? 'voice-loading' : ''} ${error && !audioUrl ? 'voice-error' : ''}`}
<i></i> onClick={handlePlayPause}
<i></i> >
<i></i> <span className={`voice-icon ${isPlaying ? 'playing' : ''}`}>{isPlaying ? '⏸' : '▶'}</span>
{loading ? (
<span className="voice-loading-text">...</span>
) : error && !audioUrl ? (
<span className="voice-error-text"></span>
) : (
<>
<div className="voice-bars" aria-hidden="true">
<i></i>
<i></i>
<i></i>
</div>
<span className="voice-duration">{formatDuration(audioDuration)}</span>
</>
)}
{transcribing ? (
<button className="voice-text-action" type="button" onClick={handleCancelRecognition}>
</button>
) : (
<button className="voice-text-action" type="button" onClick={handleTranscribe}>
{transcript ? '重新识别' : '转文字'}
</button>
)}
</div> </div>
<span className="voice-duration">{formatDuration(audioDuration)}</span> {modelStatus && modelStatus.state !== 'ready' && (
<div className="voice-model-panel" onClick={(event) => event.stopPropagation()}>
<span>
{modelStatus.state === 'downloading'
? `离线模型正在下载 ${Math.round(modelStatus.progress * 100)}%`
: modelStatus.state === 'unsupported'
? '当前系统暂不支持语音转文字'
: '请先在设置中准备离线语音模型'}
</span>
<button type="button" onClick={handleOpenVoiceSettings}>
</button>
</div>
)}
{transcribing && <div className="voice-transcript-status">...</div>}
{transcript && <div className="voice-transcript">{transcript}</div>}
{transcriptError && <div className="voice-transcript-error">{transcriptError}</div>}
</div> </div>
) )
} }
@@ -16,6 +16,22 @@ export function ExportTaskCenter({
onToggle, onToggle,
onCancel onCancel
}: ExportTaskCenterProps): React.ReactElement { }: ExportTaskCenterProps): React.ReactElement {
const [copiedJobId, setCopiedJobId] = React.useState('')
const copyTaskLog = async (task: ExportTaskRecord): Promise<void> => {
const log = [
'WechatExplorer 导出任务日志',
`时间:${new Date(task.createdAt).toLocaleString('zh-CN')}`,
`会话:${task.contactName}`,
`格式:${task.format.toUpperCase()}`,
`状态:${task.progress.phase}`,
`进度:${task.progress.percent ?? 0}%`,
`错误:${task.progress.error || '未记录具体错误'}`
].join('\n')
await navigator.clipboard.writeText(log)
setCopiedJobId(task.jobId)
}
return ( return (
<> <>
<button type="button" className="export-task-center-button" onClick={onToggle}> <button type="button" className="export-task-center-button" onClick={onToggle}>
@@ -37,6 +53,11 @@ export function ExportTaskCenter({
<small> <small>
{task.format.toUpperCase()} · {task.progress.phase} {task.format.toUpperCase()} · {task.progress.phase}
</small> </small>
{task.progress.error && (
<small className="export-task-error" title={task.progress.error}>
{task.progress.error}
</small>
)}
</span> </span>
<span className="export-task-progress"> <span className="export-task-progress">
<i style={{ width: `${task.progress.percent ?? 0}%` }} /> <i style={{ width: `${task.progress.percent ?? 0}%` }} />
@@ -47,6 +68,11 @@ export function ExportTaskCenter({
</button> </button>
)} )}
{task.status === 'failed' && (
<button type="button" onClick={() => void copyTaskLog(task)}>
{copiedJobId === task.jobId ? '已复制' : '复制日志'}
</button>
)}
</div> </div>
)) ))
)} )}
@@ -16,6 +16,7 @@ import type {
GroupMemberName GroupMemberName
} from './exportTypes' } from './exportTypes'
import { displayName, formatLabels, formatOrder, messageKinds } from './exportUtils' import { displayName, formatLabels, formatOrder, messageKinds } from './exportUtils'
import type { VoiceModelStatus } from '../../../../shared/voice-recognition'
export function ExportWorkspace({ export function ExportWorkspace({
contacts, contacts,
@@ -38,6 +39,8 @@ export function ExportWorkspace({
const [nameMode, setNameMode] = useState<ExportNameMode>('remark') const [nameMode, setNameMode] = useState<ExportNameMode>('remark')
const [groupMembers, setGroupMembers] = useState<GroupMemberName[]>([]) const [groupMembers, setGroupMembers] = useState<GroupMemberName[]>([])
const [includeMedia, setIncludeMedia] = useState(true) const [includeMedia, setIncludeMedia] = useState(true)
const [includeVoiceTranscripts, setIncludeVoiceTranscripts] = useState(true)
const [voiceModelStatus, setVoiceModelStatus] = useState<VoiceModelStatus | null>(null)
const [includeAvatars, setIncludeAvatars] = useState(true) const [includeAvatars, setIncludeAvatars] = useState(true)
const [preferOriginal, setPreferOriginal] = useState(true) const [preferOriginal, setPreferOriginal] = useState(true)
const [fallbackThumbnail, setFallbackThumbnail] = useState(true) const [fallbackThumbnail, setFallbackThumbnail] = useState(true)
@@ -149,6 +152,17 @@ export function ExportWorkspace({
return () => window.clearTimeout(timer) return () => window.clearTimeout(timer)
}, [activeContact]) }, [activeContact])
React.useEffect(() => {
let active = true
void window.api
.getVoiceModelStatus()
.then((next) => active && setVoiceModelStatus(next))
.catch(() => undefined)
return () => {
active = false
}
}, [])
const toggleKind = (value: string): void => { const toggleKind = (value: string): void => {
setSelectedKinds((current) => { setSelectedKinds((current) => {
const next = new Set(current) const next = new Set(current)
@@ -208,6 +222,12 @@ export function ExportWorkspace({
: undefined, : undefined,
kinds: Array.from(selectedKinds) as ExportMessageKind[], kinds: Array.from(selectedKinds) as ExportMessageKind[],
includeMedia, includeMedia,
includeVoiceTranscripts:
includeVoiceTranscripts &&
includeMedia &&
format === 'html' &&
selectedKinds.has('voice') &&
voiceModelStatus?.state === 'ready',
preferOriginal, preferOriginal,
fallbackThumbnail, fallbackThumbnail,
keepMissing, keepMissing,
@@ -484,6 +504,20 @@ export function ExportWorkspace({
/> />
<span></span> <span></span>
</label> </label>
<label className="export-check-row">
<input
type="checkbox"
checked={includeVoiceTranscripts && voiceModelStatus?.state === 'ready'}
disabled={
!includeMedia ||
format !== 'html' ||
!selectedKinds.has('voice') ||
voiceModelStatus?.state !== 'ready'
}
onChange={(event) => setIncludeVoiceTranscripts(event.target.checked)}
/>
<span></span>
</label>
</div> </div>
<p className="export-helper-text"> <p className="export-helper-text">
HTML CSVJSON Markdown HTML CSVJSON Markdown
@@ -492,6 +526,10 @@ export function ExportWorkspace({
<span></span> <span></span>
<span></span> <span></span>
<span></span> <span></span>
<span>
{voiceModelStatus?.state === 'ready' ? '已就绪' : '请先在设置中准备模型'}
</span>
<span></span> <span></span>
<span></span> <span></span>
</div> </div>
@@ -11,6 +11,7 @@ import { AdvancedPage } from './pages/AdvancedPage'
import { CacheCleanupPage } from './pages/CacheCleanupPage' import { CacheCleanupPage } from './pages/CacheCleanupPage'
import { AppearancePage } from './pages/AppearancePage' import { AppearancePage } from './pages/AppearancePage'
import { AboutPage } from './pages/AboutPage' import { AboutPage } from './pages/AboutPage'
import { VoiceRecognitionPage } from './pages/VoiceRecognitionPage'
import type { Contact } from '../../../../shared/types' import type { Contact } from '../../../../shared/types'
import type { AIRuntimeModelConfig } from '../../../../shared/ai-provider' import type { AIRuntimeModelConfig } from '../../../../shared/ai-provider'
@@ -88,6 +89,8 @@ export function SettingsWorkspace({
return <ImageDecryptionPage selfInfo={selfInfo} onNotice={onNotice} /> return <ImageDecryptionPage selfInfo={selfInfo} onNotice={onNotice} />
case 'ai-model': case 'ai-model':
return <AIModelPage onRuntimeChange={onAIRuntimeChange} onNotice={onNotice} /> return <AIModelPage onRuntimeChange={onAIRuntimeChange} onNotice={onNotice} />
case 'voice-recognition':
return <VoiceRecognitionPage onNotice={onNotice} />
case 'recall-protection': case 'recall-protection':
return <RecallProtectionPage onNotice={onNotice} /> return <RecallProtectionPage onNotice={onNotice} />
case 'advanced': case 'advanced':
@@ -17,7 +17,8 @@ export const SETTINGS_NAVIGATION: SettingsNavigationGroup[] = [
{ {
label: '智能能力', label: '智能能力',
items: [ items: [
{ id: 'ai-model', label: 'AI 模型' }, { id: 'voice-recognition', label: '语音转文字' },
{ id: 'ai-model', label: 'AI 模型' }
] ]
}, },
{ {
@@ -2,6 +2,7 @@ export type SettingsCategoryId =
| 'account-database' | 'account-database'
| 'database-key' | 'database-key'
| 'image-key' | 'image-key'
| 'voice-recognition'
| 'ai-model' | 'ai-model'
| 'recall-protection' | 'recall-protection'
| 'local-api' | 'local-api'
@@ -0,0 +1,246 @@
import { useCallback, useEffect, useMemo, useState } from 'react'
import type { VoiceModelStatus } from '../../../../../shared/voice-recognition'
const SENSEVOICE_URL = 'https://github.com/FunAudioLLM/SenseVoice'
const SHERPA_URL = 'https://github.com/k2-fsa/sherpa-onnx'
const STATUS_LABELS: Record<VoiceModelStatus['state'], string> = {
missing: '未下载',
downloading: '下载中',
ready: '已就绪',
invalid: '需要修复',
error: '下载失败',
unsupported: '暂不支持'
}
function formatBytes(value: number): string {
if (value < 1024 * 1024) return `${Math.round(value / 1024)} KB`
return `${(value / 1024 / 1024).toFixed(1)} MB`
}
function formatPlatform(status: VoiceModelStatus): string {
if (status.platform === 'win32')
return `Windows ${status.architecture === 'x64' ? '64 位' : status.architecture}`
if (status.platform === 'darwin') {
return status.architecture === 'arm64' ? 'macOS Apple 芯片' : 'macOS Intel'
}
return `${status.platform} ${status.architecture}`
}
export function VoiceRecognitionPage({
onNotice
}: {
onNotice: (message: string) => void
}): React.ReactElement {
const [status, setStatus] = useState<VoiceModelStatus | null>(null)
const [busy, setBusy] = useState(false)
const refresh = useCallback(async (): Promise<void> => {
setStatus(await window.api.getVoiceModelStatus())
}, [])
useEffect(() => {
let active = true
void window.api.getVoiceModelStatus().then((next) => active && setStatus(next))
const unsubscribe = window.api.onVoiceModelProgress((next) => {
if (active) setStatus(next)
})
return () => {
active = false
unsubscribe()
}
}, [])
const badgeClass = useMemo(() => {
if (status?.state === 'ready') return 'ready'
if (status?.state === 'downloading') return 'checking'
if (status?.state === 'invalid' || status?.state === 'error') return 'error'
if (status?.state === 'unsupported') return 'unavailable'
return 'warning'
}, [status?.state])
const download = async (): Promise<void> => {
setBusy(true)
setStatus((current) =>
current ? { ...current, state: 'downloading', downloadedBytes: 0, progress: 0 } : current
)
try {
const result = await window.api.downloadVoiceModel()
setStatus(result.status)
onNotice(result.success ? '离线语音模型已准备好' : result.error || '模型下载失败')
} finally {
setBusy(false)
}
}
const cancelDownload = async (): Promise<void> => {
await window.api.cancelVoiceModelDownload()
onNotice('正在取消模型下载')
}
const removeModel = async (): Promise<void> => {
if (!window.confirm('删除离线语音模型?以后使用语音转文字时需要重新下载。')) return
setBusy(true)
try {
setStatus(await window.api.removeVoiceModel())
onNotice('离线语音模型已删除')
} catch (error) {
onNotice(error instanceof Error ? `模型删除失败:${error.message}` : '模型删除失败')
} finally {
setBusy(false)
}
}
const openDirectory = async (): Promise<void> => {
const result = await window.api.openVoiceModelDirectory()
if (!result.success) onNotice(result.error || '无法打开模型目录')
}
return (
<div className="settings-page voice-recognition-page">
<header className="settings-page-header">
<div>
<h1></h1>
<p>线</p>
</div>
<div className="voice-header-status">
<span className={`settings-status-badge ${badgeClass}`}>
{status?.state === 'downloading'
? `下载中 ${Math.round(status.progress * 100)}%`
: status
? STATUS_LABELS[status.state]
: '检测中'}
</span>
{status?.state === 'downloading' && (
<progress value={status.progress} max={1} aria-label="顶部语音模型下载进度" />
)}
</div>
</header>
<div className="settings-page-scroll">
<div className="settings-page-content voice-recognition-content">
<section className="settings-privacy-notice">
<svg viewBox="0 0 24 24" aria-hidden>
<path d="M12 3 5.5 5.7v5.2c0 4.3 2.7 8.2 6.5 10.1 3.8-1.9 6.5-5.8 6.5-10.1V5.7L12 3Z" />
</svg>
<div>
<strong></strong>
<p>线 AI </p>
</div>
</section>
<h2 className="settings-section-heading"></h2>
<section className="settings-card voice-runtime-card">
<dl>
<div>
<dt></dt>
<dd>{status ? formatPlatform(status) : '检测中...'}</dd>
</div>
<div>
<dt>线</dt>
<dd className={status?.supported ? 'voice-status-success' : 'voice-status-error'}>
{status?.supported ? '支持' : '暂不支持'}
</dd>
</div>
<div>
<dt></dt>
<dd>sherpa-onnx · SenseVoice</dd>
</div>
</dl>
</section>
<h2 className="settings-section-heading">线</h2>
<section className="settings-card voice-model-card">
<div className="voice-model-summary">
<span className="settings-card-kicker">SenseVoice Small INT8</span>
<strong>
{status?.state === 'downloading'
? `正在下载 ${Math.round(status.progress * 100)}%`
: status
? STATUS_LABELS[status.state]
: '正在检测'}
</strong>
<small>
{status
? `版本 ${status.version} · ${formatBytes(status.totalBytes)}`
: '读取模型状态...'}
</small>
{status?.error && <p className="voice-model-error">{status.error}</p>}
<p className="voice-model-license">
<a href={SENSEVOICE_URL} target="_blank" rel="noreferrer">
SenseVoiceMIT
</a>
<span> · </span>
<a href={SHERPA_URL} target="_blank" rel="noreferrer">
sherpa-onnxApache-2.0
</a>
</p>
</div>
<div className="voice-model-actions">
{status?.state === 'downloading' ? (
<button type="button" onClick={() => void cancelDownload()}>
</button>
) : status?.state === 'ready' ? (
<>
<button type="button" onClick={() => void openDirectory()}>
</button>
<button
type="button"
className="settings-danger-button"
disabled={busy}
onClick={() => void removeModel()}
>
</button>
</>
) : (
<button
type="button"
className="settings-primary-button"
disabled={busy || !status?.supported}
onClick={() => void download()}
>
{status?.state === 'invalid' || status?.state === 'error'
? '重新下载模型'
: '下载模型'}
</button>
)}
<button type="button" disabled={busy} onClick={() => void refresh()}>
</button>
</div>
{status?.state === 'downloading' && (
<div className="voice-model-progress">
<div>
<span>{Math.round(status.progress * 100)}%</span>
<small>
{formatBytes(status.downloadedBytes)} / {formatBytes(status.totalBytes)}
</small>
</div>
<progress value={status.progress} max={1} aria-label="语音模型下载进度" />
</div>
)}
</section>
<h2 className="settings-section-heading"></h2>
<section className="settings-card voice-platform-list">
<div>
<strong>Windows</strong>
<span> Windows 10/11 64 </span>
</div>
<div>
<strong>macOS</strong>
<span> Intel Apple </span>
</div>
</section>
<p className="settings-footnote">
</p>
</div>
</div>
</div>
)
}
+4
View File
@@ -764,6 +764,10 @@
color: var(--wxex-text-primary); color: var(--wxex-text-primary);
} }
.export-task-error {
color: var(--wxex-danger, #b42318);
}
button { button {
border: 1px solid var(--wxex-border); border: 1px solid var(--wxex-border);
border-radius: 5px; border-radius: 5px;
+57
View File
@@ -1,9 +1,66 @@
/* Voice Player */ /* Voice Player */
.voice-player {
display: grid;
min-width: 190px;
gap: 7px;
}
.voice-message { .voice-message {
cursor: pointer; cursor: pointer;
user-select: none; user-select: none;
} }
.voice-text-action,
.voice-model-panel button {
border: 0;
background: transparent;
color: var(--wxex-brand);
cursor: pointer;
font-size: 12px;
}
.voice-text-action {
margin-left: auto;
padding: 2px 0 2px 8px;
}
.voice-model-panel {
display: grid;
grid-template-columns: minmax(0, 1fr) auto;
align-items: center;
gap: 6px 10px;
padding-top: 7px;
border-top: 1px solid var(--wxex-border);
color: var(--wxex-text-secondary);
font-size: 12px;
line-height: 18px;
white-space: normal;
}
.voice-model-panel progress {
width: 100%;
height: 5px;
}
.voice-transcript,
.voice-transcript-status,
.voice-transcript-error {
padding-top: 7px;
border-top: 1px solid var(--wxex-border);
font-size: 13px;
line-height: 20px;
white-space: pre-wrap;
word-break: break-word;
}
.voice-transcript-status {
color: var(--wxex-text-muted);
}
.voice-transcript-error {
color: var(--wxex-danger, #c63c3c);
}
.voice-loading { .voice-loading {
opacity: 0.6; opacity: 0.6;
} }
@@ -118,6 +118,185 @@
} }
} }
.voice-runtime-card dl {
display: grid;
grid-template-columns: repeat(3, minmax(0, 1fr));
gap: 20px 32px;
margin: 0;
div {
min-width: 0;
}
dt {
color: var(--wxex-text-muted);
font-size: 11px;
}
dd {
margin: 5px 0 0;
color: var(--wxex-text-primary);
font-size: 13px;
}
}
.voice-header-status {
display: grid;
min-width: 132px;
justify-items: end;
gap: 7px;
progress {
width: 112px;
height: 6px;
accent-color: var(--wxex-brand);
}
}
.voice-status-success {
color: var(--wxex-success, #2e8b68) !important;
}
.voice-status-error,
.voice-model-error {
color: var(--wxex-danger, #c85a5a) !important;
}
.voice-model-license {
margin: 9px 0 0;
color: var(--wxex-text-muted);
font-size: 11px;
line-height: 17px;
a {
color: var(--wxex-brand);
text-decoration: none;
&:hover {
text-decoration: underline;
}
}
}
.voice-model-card {
display: grid;
grid-template-columns: minmax(0, 1fr) auto;
gap: 18px 28px;
align-items: center;
}
.voice-model-summary {
min-width: 0;
strong,
small {
display: block;
}
strong {
color: var(--wxex-text-primary);
font-size: 18px;
}
small {
margin-top: 5px;
color: var(--wxex-text-secondary);
font-size: 11px;
}
}
.voice-model-error {
margin: 8px 0 0;
font-size: 12px;
line-height: 18px;
}
.voice-model-actions {
display: flex;
flex-wrap: wrap;
justify-content: flex-end;
gap: 8px;
> button {
min-height: 34px;
padding: 7px 12px;
border: 1px solid var(--wxex-border);
border-radius: var(--wxex-radius-sm);
background: var(--wxex-bg-elevated);
color: var(--wxex-text-primary);
cursor: pointer;
font: 12px/18px var(--wxex-font);
&:disabled {
cursor: not-allowed;
opacity: 0.55;
}
}
> .settings-primary-button {
border-color: var(--wxex-brand);
background: var(--wxex-brand);
color: #fff;
}
}
.voice-model-progress {
grid-column: 1 / -1;
display: grid;
gap: 8px;
> div {
display: flex;
justify-content: space-between;
color: var(--wxex-text-secondary);
font-size: 12px;
}
progress {
width: 100%;
height: 7px;
accent-color: var(--wxex-brand);
}
}
.voice-platform-list {
padding-top: 8px;
padding-bottom: 8px;
> div {
display: grid;
grid-template-columns: 120px minmax(0, 1fr);
gap: 16px;
padding: 14px 0;
border-bottom: 1px solid var(--wxex-border);
&:last-child {
border-bottom: 0;
}
}
strong {
color: var(--wxex-text-primary);
font-size: 13px;
}
span {
color: var(--wxex-text-secondary);
font-size: 12px;
}
}
@media (max-width: 760px) {
.voice-runtime-card dl,
.voice-model-card {
grid-template-columns: 1fr;
}
.voice-model-actions {
justify-content: flex-start;
}
}
.settings-option-card { .settings-option-card {
padding: 12px; padding: 12px;
} }
@@ -308,7 +487,6 @@
--wxex-nav-width: 68px; --wxex-nav-width: 68px;
--wxex-shell-content-top: 8px; --wxex-shell-content-top: 8px;
} }
} }
.boot-splash.is-quiet { .boot-splash.is-quiet {
+1
View File
@@ -24,6 +24,7 @@ export interface ExportRequest {
endTime?: number endTime?: number
kinds: ExportMessageKind[] kinds: ExportMessageKind[]
includeMedia: boolean includeMedia: boolean
includeVoiceTranscripts?: boolean
preferOriginal?: boolean preferOriginal?: boolean
fallbackThumbnail?: boolean fallbackThumbnail?: boolean
keepMissing?: boolean keepMissing?: boolean
+2
View File
@@ -23,6 +23,8 @@ export interface Message {
contentData?: ParsedContent contentData?: ParsedContent
voiceDataUrl?: string voiceDataUrl?: string
voiceDuration?: number voiceDuration?: number
voiceTranscript?: string
voiceTranscriptError?: string
localId?: number localId?: number
serverId?: string serverId?: string
createTime?: number createTime?: number
+58
View File
@@ -0,0 +1,58 @@
export const DEFAULT_VOICE_MODEL_ID = 'sensevoice-small-int8'
export interface VoiceMessageReference {
sessionId: string
localId: number
createTime: number
svrId?: string | number
}
export type VoiceModelState =
| 'missing'
| 'downloading'
| 'ready'
| 'invalid'
| 'error'
| 'unsupported'
export interface VoiceModelStatus {
modelId: string
version: string
state: VoiceModelState
downloadedBytes: number
totalBytes: number
progress: number
platform: NodeJS.Platform
architecture: string
supported: boolean
error?: string
}
export interface VoiceModelDownloadResult {
success: boolean
status: VoiceModelStatus
error?: string
}
export type VoiceRecognitionErrorCode =
| 'NOT_CONNECTED'
| 'MODEL_NOT_READY'
| 'VOICE_NOT_FOUND'
| 'DECODE_FAILED'
| 'EMPTY_AUDIO'
| 'CANCELLED'
| 'TIMEOUT'
| 'WORKER_FAILED'
| 'RECOGNITION_FAILED'
export interface VoiceRecognitionResult {
success: boolean
transcript?: string
language?: string
durationMs?: number
cached?: boolean
error?: string
code?: VoiceRecognitionErrorCode
}
export interface VoiceModelProgressEvent extends VoiceModelStatus {}
@@ -0,0 +1,52 @@
import { render, screen } from '@testing-library/react'
import userEvent from '@testing-library/user-event'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { ExportTaskCenter } from '../../src/renderer/src/components/export/ExportTaskCenter'
describe('export task center', () => {
const writeText = vi.fn().mockResolvedValue(undefined)
beforeEach(() => {
writeText.mockClear()
Object.defineProperty(navigator, 'clipboard', {
configurable: true,
value: { writeText }
})
})
it('shows the failure reason and copies a diagnostic log', async () => {
render(
<ExportTaskCenter
open
taskCount={0}
tasks={[
{
jobId: 'failed-export',
contactId: 'fixture',
contactName: '脱敏会话',
format: 'html',
status: 'failed',
progress: {
jobId: 'failed-export',
phase: 'failed',
processed: 0,
percent: 15,
error: 'EPERM: operation not permitted, copyfile'
},
createdAt: new Date('2026-08-04T15:00:00.000Z').getTime()
}
]}
onToggle={vi.fn()}
onCancel={vi.fn()}
/>
)
expect(screen.getByText('EPERM: operation not permitted, copyfile')).toBeInTheDocument()
await userEvent.click(screen.getByRole('button', { name: '复制日志' }))
expect(writeText).toHaveBeenCalledOnce()
expect(writeText.mock.calls[0][0]).toContain('会话:脱敏会话')
expect(writeText.mock.calls[0][0]).toContain('EPERM: operation not permitted, copyfile')
expect(screen.getByRole('button', { name: '已复制' })).toBeInTheDocument()
})
})
@@ -0,0 +1,72 @@
import { render, screen, waitFor } from '@testing-library/react'
import userEvent from '@testing-library/user-event'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { ExportWorkspace } from '../../src/renderer/src/components/export/ExportWorkspace'
import type { VoiceModelStatus } from '../../src/shared/voice-recognition'
const readyStatus: VoiceModelStatus = {
modelId: 'sensevoice-small-int8',
version: '2024-07-17',
state: 'ready',
downloadedBytes: 239_549_735,
totalBytes: 239_549_735,
progress: 1,
platform: 'win32',
architecture: 'x64',
supported: true
}
describe('export voice transcripts', () => {
beforeEach(() => {
window.api = {
getVoiceModelStatus: vi.fn().mockResolvedValue(readyStatus),
onExportProgress: vi.fn(() => vi.fn())
} as typeof window.api
})
it('enables voice transcription by default for a ready HTML voice export', async () => {
const onStartExport = vi.fn().mockResolvedValue({
success: true,
messageCount: 1,
outputPath: 'C:\\fixture\\index.html'
})
render(
<ExportWorkspace
contacts={[
{
m_nsUsrName: 'filehelper',
m_nsNickName: '文件传输助手',
md5: 'fixture-contact',
type: 'user'
}
]}
selectedContact={null}
previewMessages={[]}
selfInfo={null}
dbReady
onSelectContact={vi.fn()}
onOpenSettings={vi.fn()}
exportTasks={[]}
onStartExport={onStartExport}
onCancelExport={vi.fn()}
/>
)
await userEvent.click(screen.getAllByRole('button', { name: /HTML/ })[0])
await userEvent.click(screen.getByRole('checkbox', { name: '语音' }))
const transcriptOption = await screen.findByRole('checkbox', {
name: '语音转文字,显示在语音条下方'
})
expect(transcriptOption).toBeEnabled()
expect(transcriptOption).toBeChecked()
await userEvent.click(screen.getByRole('button', { name: '开始导出' }))
await waitFor(() => expect(onStartExport).toHaveBeenCalledOnce())
expect(onStartExport.mock.calls[0][0]).toMatchObject({
format: 'html',
includeVoiceTranscripts: true,
kinds: expect.arrayContaining(['voice'])
})
})
})
+60 -1
View File
@@ -29,7 +29,28 @@ describe('VoicePlayer', () => {
getVoiceData: vi.fn().mockResolvedValue({ getVoiceData: vi.fn().mockResolvedValue({
success: true, success: true,
data: 'UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAEAfAAABAAgAZGF0YQAAAAA=' data: 'UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAEAfAAABAAgAZGF0YQAAAAA='
}) }),
getVoiceModelStatus: vi.fn().mockResolvedValue({
modelId: 'sensevoice-small-int8',
version: 'fixture',
state: 'ready',
downloadedBytes: 10,
totalBytes: 10,
progress: 1,
platform: 'win32',
architecture: 'x64',
supported: true
}),
recognizeVoice: vi.fn().mockResolvedValue({
success: true,
transcript: '这是固定的测试转写',
language: 'zh',
cached: false
}),
downloadVoiceModel: vi.fn(),
cancelVoiceModelDownload: vi.fn(),
cancelVoiceRecognition: vi.fn(),
onVoiceModelProgress: vi.fn(() => vi.fn())
} as typeof window.api } as typeof window.api
}) })
@@ -44,4 +65,42 @@ describe('VoicePlayer', () => {
expect(container.querySelector('.voice-icon')).toHaveClass('playing') expect(container.querySelector('.voice-icon')).toHaveClass('playing')
expect(screen.queryByText('当前版本暂不支持播放')).not.toBeInTheDocument() expect(screen.queryByText('当前版本暂不支持播放')).not.toBeInTheDocument()
}) })
it('recognizes one voice message and renders the transcript', async () => {
render(<VoicePlayer sessionId="filehelper" localId={11} createTime={1785553200} duration={1} />)
await userEvent.click(screen.getByRole('button', { name: '转文字' }))
await waitFor(() =>
expect(window.api.recognizeVoice).toHaveBeenCalledWith({
sessionId: 'filehelper',
localId: 11,
createTime: 1785553200,
svrId: undefined
})
)
expect(await screen.findByText('这是固定的测试转写')).toBeInTheDocument()
})
it('opens centralized settings when recognition assets are missing', async () => {
vi.mocked(window.api.getVoiceModelStatus).mockResolvedValue({
modelId: 'sensevoice-small-int8',
version: 'fixture',
state: 'missing',
downloadedBytes: 0,
totalBytes: 239_549_735,
progress: 0,
platform: 'win32',
architecture: 'x64',
supported: true
})
render(<VoicePlayer sessionId="filehelper" localId={12} createTime={1785553300} duration={2} />)
const openSettings = vi.fn()
window.addEventListener('wxe:open-voice-recognition-settings', openSettings, { once: true })
await userEvent.click(screen.getByRole('button', { name: '转文字' }))
expect(await screen.findByText(/请先在设置中准备离线语音模型/)).toBeInTheDocument()
await userEvent.click(screen.getByRole('button', { name: '前往设置' }))
expect(openSettings).toHaveBeenCalledOnce()
expect(window.api.recognizeVoice).not.toHaveBeenCalled()
})
}) })
@@ -0,0 +1,90 @@
import { render, screen, waitFor } from '@testing-library/react'
import userEvent from '@testing-library/user-event'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { VoiceRecognitionPage } from '../../src/renderer/src/features/settings/pages/VoiceRecognitionPage'
import type { VoiceModelStatus } from '../../src/shared/voice-recognition'
const readyStatus: VoiceModelStatus = {
modelId: 'sensevoice-small-int8',
version: '2024-07-17',
state: 'ready',
downloadedBytes: 239_549_735,
totalBytes: 239_549_735,
progress: 1,
platform: 'win32',
architecture: 'x64',
supported: true
}
describe('voice recognition settings', () => {
beforeEach(() => {
window.api = {
getVoiceModelStatus: vi.fn().mockResolvedValue(readyStatus),
downloadVoiceModel: vi.fn(),
cancelVoiceModelDownload: vi.fn(),
removeVoiceModel: vi.fn().mockResolvedValue({ ...readyStatus, state: 'missing' }),
openVoiceModelDirectory: vi.fn().mockResolvedValue({ success: true }),
onVoiceModelProgress: vi.fn(() => vi.fn())
} as typeof window.api
})
it('shows Windows runtime and installed model actions', async () => {
render(<VoiceRecognitionPage onNotice={vi.fn()} />)
expect(await screen.findByText('Windows 64 位')).toBeInTheDocument()
expect(screen.queryByText('额外环境')).not.toBeInTheDocument()
expect(screen.getByRole('link', { name: 'SenseVoiceMIT' })).toBeInTheDocument()
expect(screen.getByRole('link', { name: 'sherpa-onnxApache-2.0' })).toBeInTheDocument()
expect(screen.getByRole('button', { name: '打开模型目录' })).toBeInTheDocument()
expect(screen.getByRole('button', { name: '删除模型' })).toBeInTheDocument()
})
it('downloads the model from the centralized settings page', async () => {
const missing = { ...readyStatus, state: 'missing' as const, progress: 0, downloadedBytes: 0 }
vi.mocked(window.api.getVoiceModelStatus).mockResolvedValue(missing)
vi.mocked(window.api.downloadVoiceModel).mockResolvedValue({
success: true,
status: readyStatus
})
const notice = vi.fn()
render(<VoiceRecognitionPage onNotice={notice} />)
await userEvent.click(await screen.findByRole('button', { name: '下载模型' }))
await waitFor(() => expect(window.api.downloadVoiceModel).toHaveBeenCalledOnce())
expect(notice).toHaveBeenCalledWith('离线语音模型已准备好')
})
it('shows download percentage in the header and model card', async () => {
const missing = {
...readyStatus,
state: 'missing' as const,
progress: 0,
downloadedBytes: 0
}
let progressListener: ((status: VoiceModelStatus) => void) | undefined
let finishDownload:
| ((value: { success: boolean; status: VoiceModelStatus }) => void)
| undefined
vi.mocked(window.api.getVoiceModelStatus).mockResolvedValue(missing)
vi.mocked(window.api.onVoiceModelProgress).mockImplementation((listener) => {
progressListener = listener
return vi.fn()
})
vi.mocked(window.api.downloadVoiceModel).mockReturnValue(
new Promise((resolve) => {
finishDownload = resolve
})
)
render(<VoiceRecognitionPage onNotice={vi.fn()} />)
await userEvent.click(await screen.findByRole('button', { name: '下载模型' }))
progressListener?.({
...missing,
state: 'downloading',
downloadedBytes: Math.round(missing.totalBytes * 0.42),
progress: 0.42
})
expect(await screen.findByText('下载中 42%')).toBeInTheDocument()
expect(screen.getByText('正在下载 42%')).toBeInTheDocument()
finishDownload?.({ success: true, status: readyStatus })
})
})
+18
View File
@@ -223,6 +223,24 @@ handle('db:getImage', (md5, datName, sessionId, options) =>
} }
) )
handle('db:getVoiceData', () => ({ success: true, data: voiceData })) handle('db:getVoiceData', () => ({ success: true, data: voiceData }))
const voiceModelStatus = (state = 'missing') => ({
modelId: 'sensevoice-small-int8',
version: '2024-07-17',
state,
downloadedBytes: state === 'ready' ? 239549735 : 0,
totalBytes: 239549735,
progress: state === 'ready' ? 1 : 0,
platform: process.platform,
architecture: process.arch,
supported: process.platform === 'win32' || process.platform === 'darwin'
})
handle('voice:getModelStatus', () => voiceModelStatus())
handle('voice:downloadModel', () => ({ success: true, status: voiceModelStatus('ready') }))
handle('voice:cancelModelDownload', () => ({ success: true }))
handle('voice:removeModel', () => voiceModelStatus())
handle('voice:openModelDirectory', () => ({ success: true }))
handle('voice:recognize', () => ({ success: true, transcript: '固定脱敏转写文本', language: 'zh' }))
handle('voice:cancelRecognition', () => ({ success: true }))
handle('db:getSticker', (url) => handle('db:getSticker', (url) =>
String(url || '').includes('403') String(url || '').includes('403')
? { ? {
+58 -1
View File
@@ -1,6 +1,7 @@
import { dirname, join } from 'path' import { dirname, join } from 'path'
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'fs' import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'fs'
import { tmpdir } from 'os' import { tmpdir } from 'os'
import fsExtra from 'fs-extra'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type { Message } from '../../src/shared/types' import type { Message } from '../../src/shared/types'
@@ -207,11 +208,20 @@ describe('media export flow', () => {
outputName: 'fixture', outputName: 'fixture',
kinds: ['voice', 'image', 'video', 'file'], kinds: ['voice', 'image', 'video', 'file'],
includeMedia: true, includeMedia: true,
includeVoiceTranscripts: true,
preferOriginal: true, preferOriginal: true,
fallbackThumbnail: true, fallbackThumbnail: true,
keepMissing: true keepMissing: true
}, },
win as never win as never,
{
recognize: vi.fn().mockResolvedValue({
success: true,
transcript: '这是导出的固定语音转写',
language: 'zh',
cached: true
})
} as never
) )
expect(result.success).toBe(true) expect(result.success).toBe(true)
@@ -231,6 +241,7 @@ describe('media export flow', () => {
expect(readFileSync(join(outputDir, file.exportMediaUrl!), 'utf8')).toBe('附件内容') expect(readFileSync(join(outputDir, file.exportMediaUrl!), 'utf8')).toBe('附件内容')
expect(html).toContain('<script src="data/messages.js"></script>') expect(html).toContain('<script src="data/messages.js"></script>')
expect(voice.voiceDataUrl).toMatch(/^voices\/voice_[0-9a-f]{16}\.wav$/) expect(voice.voiceDataUrl).toMatch(/^voices\/voice_[0-9a-f]{16}\.wav$/)
expect(voice.voiceTranscript).toBe('这是导出的固定语音转写')
expect(video.exportMediaUrl).toMatch(/^media\/video_[0-9a-f]{16}\.mp4$/) expect(video.exportMediaUrl).toMatch(/^media\/video_[0-9a-f]{16}\.mp4$/)
expect(file.exportMediaUrl).toMatch(/^media\/file_[0-9a-f]{16}_测试附件\.txt$/) expect(file.exportMediaUrl).toMatch(/^media\/file_[0-9a-f]{16}_测试附件\.txt$/)
expect(missingVoice.exportMediaError).toBe('语音文件缺失:本地未找到语音数据') expect(missingVoice.exportMediaError).toBe('语音文件缺失:本地未找到语音数据')
@@ -302,6 +313,52 @@ describe('media export flow', () => {
expect(existsSync(join(dirname(second.outputPath!), 'data', 'messages.js.bak'))).toBe(true) expect(existsSync(join(dirname(second.outputPath!), 'data', 'messages.js.bak'))).toBe(true)
}) })
it('reuses existing video and file assets when Windows rejects an overwrite', async () => {
const { runExport } = await import('../../src/main/export-service')
const win = { isDestroyed: () => true, webContents: { send: vi.fn() } }
const request = {
userMd5: 'fixture-user',
name: '媒体复用会话',
format: 'html' as const,
outputName: 'reused-media-fixture',
kinds: ['video', 'file'] as const,
includeMedia: true,
keepMissing: true
}
const first = await runExport(
{ ...request, jobId: 'media-reuse-first', kinds: [...request.kinds] },
win as never
)
expect(first.success).toBe(true)
const originalCopyFile = fsExtra.copyFile.bind(fsExtra)
const copyFile = vi
.spyOn(fsExtra, 'copyFile')
.mockRejectedValueOnce(
Object.assign(new Error('EPERM: operation not permitted, copyfile'), { code: 'EPERM' })
)
.mockRejectedValueOnce(
Object.assign(new Error('EPERM: operation not permitted, copyfile'), { code: 'EPERM' })
)
.mockImplementation(originalCopyFile)
const second = await runExport(
{ ...request, jobId: 'media-reuse-second', kinds: [...request.kinds] },
win as never
)
copyFile.mockRestore()
expect(second.success).toBe(true)
const archive = readArchive(second.outputPath!)
expect(archive.messages.find((item) => item.id === 'video')?.exportMediaUrl).toMatch(
/^media\/video_/
)
expect(archive.messages.find((item) => item.id === 'file')?.exportMediaUrl).toMatch(
/^media\/file_/
)
})
it('refuses to merge a different conversation into an existing named archive', async () => { it('refuses to merge a different conversation into an existing named archive', async () => {
const { runExport } = await import('../../src/main/export-service') const { runExport } = await import('../../src/main/export-service')
const win = { isDestroyed: () => true, webContents: { send: vi.fn() } } const win = { isDestroyed: () => true, webContents: { send: vi.fn() } }
@@ -48,6 +48,23 @@ describe('preload IPC contract', () => {
'fixture-session', 'fixture-session',
{ force: true, priority: 0 } { force: true, priority: 0 }
) )
const voiceReference = {
sessionId: 'filehelper',
localId: 11,
createTime: 1785553200,
svrId: 'server-11'
}
await api.recognizeVoice(voiceReference)
expect(invoke).toHaveBeenLastCalledWith('voice:recognize', voiceReference)
await api.cancelVoiceRecognition(voiceReference)
expect(invoke).toHaveBeenLastCalledWith('voice:cancelRecognition', voiceReference)
await api.downloadVoiceModel()
expect(invoke).toHaveBeenLastCalledWith('voice:downloadModel')
await api.removeVoiceModel()
expect(invoke).toHaveBeenLastCalledWith('voice:removeModel')
await api.openVoiceModelDirectory()
expect(invoke).toHaveBeenLastCalledWith('voice:openModelDirectory')
}) })
it('preserves key API return values without exposing ipcRenderer', async () => { it('preserves key API return values without exposing ipcRenderer', async () => {
+45 -1
View File
@@ -102,11 +102,55 @@ describe('export media', () => {
expect(html).toContain('class="file-attachment" href="') expect(html).toContain('class="file-attachment" href="')
expect(html).toContain('class="quote-reference"') expect(html).toContain('class="quote-reference"')
expect(html).toContain('message.exportMediaError') expect(html).toContain('message.exportMediaError')
expect(html).toContain('.audio-wrap { width: 260px; max-width: 100%; min-width: 0; }') expect(html).toContain('.audio-wrap { width: 380px; max-width: 100%; min-width: 0; }')
expect(html).toContain('.audio { display: block; width: 100%; max-width: 100%; height: 38px; }') expect(html).toContain('.audio { display: block; width: 100%; max-width: 100%; height: 38px; }')
expect(html).toContain('class="voice-transcript"')
expect(html).toContain('message.voiceTranscript')
expect(html).toContain('class="message-stack"')
expect(html).not.toMatch(/(?:src|href)="[A-Za-z]:\\/) expect(html).not.toMatch(/(?:src|href)="[A-Za-z]:\\/)
}) })
it('renders a voice transcript below audio inside the same exported bubble', () => {
const html = renderExportPage('语音转写档案')
const dom = new JSDOM(html, { runScripts: 'outside-only' })
Object.assign(dom.window, {
__WECHAT_EXPORT__: {
version: 1,
sourceId: 'fixture',
name: '语音转写档案',
exportedAt: '2026-08-04T00:00:00.000Z',
messages: [
{
id: 'voice-transcript',
from: 'user',
type: '语音',
datetime: '2026-08-04 14:26',
content: '[语音消息]',
isSender: true,
voiceDataUrl: 'voices/fixture.wav',
voiceTranscript: '试一下',
createTime: 1_785_549_600
}
]
}
})
dom.window.eval(inlineScriptOf(html))
const stack = dom.window.document.querySelector('.message-stack')!
const bubble = stack.querySelector('.bubble')!
const transcript = stack.querySelector('.voice-transcript')!
expect(bubble.querySelector('audio')?.getAttribute('src')).toBe('voices/fixture.wav')
expect(transcript.textContent).toBe('试一下')
expect(bubble.contains(transcript)).toBe(true)
expect(stack.children).toHaveLength(1)
expect(
bubble.querySelector('audio')!.compareDocumentPosition(transcript) &
dom.window.Node.DOCUMENT_POSITION_FOLLOWING
).toBeTruthy()
expect(bubble.textContent).not.toContain('[语音消息]')
dom.window.close()
})
it('renders explicit and keyboard-accessible lightbox closing controls', () => { it('renders explicit and keyboard-accessible lightbox closing controls', () => {
const html = renderExportPage('图片预览') const html = renderExportPage('图片预览')
+29 -1
View File
@@ -5,10 +5,11 @@ import { dirname, join, resolve } from 'path'
import { afterAll, describe, expect, it } from 'vitest' import { afterAll, describe, expect, it } from 'vitest'
const nodeRequire = createRequire(import.meta.url) const nodeRequire = createRequire(import.meta.url)
const { validateFfmpegRuntime, validateSilkWasmRuntime } = nodeRequire( const { validateFfmpegRuntime, validateSherpaRuntime, validateSilkWasmRuntime } = nodeRequire(
'../../scripts/after-pack.cjs' '../../scripts/after-pack.cjs'
) as { ) as {
validateFfmpegRuntime: (runtimeResources: string, platform?: NodeJS.Platform) => void validateFfmpegRuntime: (runtimeResources: string, platform?: NodeJS.Platform) => void
validateSherpaRuntime: (runtimeResources: string, platform: NodeJS.Platform, arch: string) => void
validateSilkWasmRuntime: (runtimeResources: string) => void validateSilkWasmRuntime: (runtimeResources: string) => void
} }
const root = mkdtempSync(join(tmpdir(), 'wxe-runtime-package-')) const root = mkdtempSync(join(tmpdir(), 'wxe-runtime-package-'))
@@ -49,4 +50,31 @@ describe('production runtime packaging', () => {
const config = readFileSync(resolve(__dirname, '../../electron-builder.yml'), 'utf8') const config = readFileSync(resolve(__dirname, '../../electron-builder.yml'), 'utf8')
expect(config).toContain('node_modules/ffmpeg-static/**') expect(config).toContain('node_modules/ffmpeg-static/**')
}) })
it('requires the matching Windows and macOS sherpa native runtime', () => {
const resources = join(root, 'sherpa-resources')
const unpacked = join(resources, 'app.asar.unpacked', 'node_modules')
const base = join(unpacked, 'sherpa-onnx-node')
mkdirSync(base, { recursive: true })
writeFileSync(join(base, 'package.json'), '{}')
writeFileSync(join(base, 'sherpa-onnx.js'), 'module.exports = {}')
expect(() => validateSherpaRuntime(resources, 'win32', 'x64')).toThrow(/win-x64/)
const windows = join(unpacked, 'sherpa-onnx-win-x64')
mkdirSync(windows, { recursive: true })
writeFileSync(join(windows, 'package.json'), '{}')
writeFileSync(join(windows, 'sherpa-onnx.node'), 'fixture')
expect(() => validateSherpaRuntime(resources, 'win32', 'x64')).not.toThrow()
expect(() => validateSherpaRuntime(resources, 'darwin', 'arm64')).toThrow(/darwin-arm64/)
const mac = join(unpacked, 'sherpa-onnx-darwin-arm64')
mkdirSync(mac, { recursive: true })
writeFileSync(join(mac, 'package.json'), '{}')
writeFileSync(join(mac, 'sherpa-onnx.node'), 'fixture')
expect(() => validateSherpaRuntime(resources, 'darwin', 'arm64')).not.toThrow()
const config = readFileSync(resolve(__dirname, '../../electron-builder.yml'), 'utf8')
expect(config).toContain('node_modules/sherpa-onnx-node/**')
expect(config).toContain('node_modules/sherpa-onnx-*/**')
})
}) })
+141
View File
@@ -0,0 +1,141 @@
import { mkdtempSync, rmSync } from 'fs'
import { tmpdir } from 'os'
import { join } from 'path'
import { afterAll, describe, expect, it, vi } from 'vitest'
import { PcmAudioProcessor } from '../../src/main/voice-pipeline/audio-processor'
import { VoiceTaskScheduler } from '../../src/main/voice-pipeline/task-scheduler'
import { SqliteTranscriptRepository } from '../../src/main/voice-pipeline/transcript-repository'
import type { TranscriptRecord } from '../../src/main/voice-pipeline/types'
import { SENSEVOICE_MODEL_FILES } from '../../src/main/voice-pipeline/model-manager'
const root = mkdtempSync(join(tmpdir(), 'wxe-voice-pipeline-'))
describe('SenseVoice model manifest', () => {
it('uses the Git LFS content digest rather than the Hugging Face xet hash', () => {
expect(SENSEVOICE_MODEL_FILES[0]).toMatchObject({
name: 'model.int8.onnx',
size: 239_233_841,
sha256: 'c71f0ce00bec95b07744e116345e33d8cbbe08cef896382cf907bf4b51a2cd51'
})
expect(SENSEVOICE_MODEL_FILES[0].sha256).not.toBe(
'c45ba1d6a13329c4aca1dc118cabdc643ca09cb8192abb979648dd68f9917323'
)
})
})
function pcm16(samples: number[]): Buffer {
const buffer = Buffer.alloc(samples.length * 2)
samples.forEach((sample, index) => buffer.writeInt16LE(sample, index * 2))
return buffer
}
describe('PCM audio processing', () => {
it('really resamples 24 kHz PCM to 16 kHz and trims outer silence', () => {
const silence = Array.from({ length: 2400 }, () => 0)
const tone = Array.from({ length: 24000 }, (_, index) =>
Math.round(Math.sin((index / 24000) * Math.PI * 440 * 2) * 20000)
)
const processor = new PcmAudioProcessor({ silencePaddingMs: 0 })
const output = processor.process({
pcm: pcm16([...silence, ...tone, ...silence]),
sampleRate: 24000,
channels: 1,
sourceHash: 'fixture-audio'
})
expect(output.sampleRate).toBe(16000)
expect(output.samples.length).toBeGreaterThan(15900)
expect(output.samples.length).toBeLessThanOrEqual(16000)
expect(output.durationMs).toBeGreaterThanOrEqual(990)
expect(Math.max(...output.samples)).toBeLessThanOrEqual(0.92)
})
it('returns an empty signal when the source only contains silence', () => {
const output = new PcmAudioProcessor().process({
pcm: pcm16(Array.from({ length: 2400 }, () => 0)),
sampleRate: 24000,
channels: 1,
sourceHash: 'silence'
})
expect(output.samples).toHaveLength(0)
})
})
describe('voice task scheduling', () => {
it('runs recognition tasks serially', async () => {
const scheduler = new VoiceTaskScheduler()
const order: string[] = []
let releaseFirst: (() => void) | undefined
const first = scheduler.schedule('first', async () => {
order.push('first:start')
await new Promise<void>((resolve) => {
releaseFirst = resolve
})
order.push('first:end')
return 1
})
const second = scheduler.schedule('second', async () => {
order.push('second')
return 2
})
await vi.waitFor(() => expect(order).toEqual(['first:start']))
releaseFirst?.()
await expect(Promise.all([first, second])).resolves.toEqual([1, 2])
expect(order).toEqual(['first:start', 'first:end', 'second'])
})
it('cancels a queued task without running it', async () => {
const scheduler = new VoiceTaskScheduler()
let releaseFirst: (() => void) | undefined
const first = scheduler.schedule(
'first',
() =>
new Promise<void>((resolve) => {
releaseFirst = resolve
})
)
const queued = scheduler.schedule('queued', async () => 'should-not-run')
expect(scheduler.cancel('queued')).toBe(true)
await expect(queued).rejects.toMatchObject({ name: 'AbortError' })
releaseFirst?.()
await first
})
})
describe('transcript repository', () => {
afterAll(() => rmSync(root, { recursive: true, force: true }))
it('keeps records isolated by account and model fingerprint', () => {
const repository = new SqliteTranscriptRepository(join(root, 'transcripts.sqlite'))
const record: TranscriptRecord = {
accountId: 'account-a',
messageIdentity: 'message-1',
audioHash: 'audio-1',
processorVersion: 'processor-v1',
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a',
transcript: '固定测试文本',
language: 'zh',
durationMs: 1200,
createdAt: 1,
updatedAt: 1
}
repository.save(record)
const key = {
accountId: record.accountId,
messageIdentity: record.messageIdentity,
audioHash: record.audioHash,
processorVersion: record.processorVersion,
recognizerId: record.recognizerId,
modelVersion: record.modelVersion,
modelFingerprint: record.modelFingerprint
}
expect(repository.find(key)).toMatchObject({ transcript: '固定测试文本' })
expect(repository.find({ ...key, accountId: 'account-b' })).toBeNull()
expect(repository.find({ ...key, modelFingerprint: 'fingerprint-b' })).toBeNull()
repository.close()
})
})