mirror of
https://wget.la/https://github.com/Wxw-Gu/WechatExplorer
synced 2026-10-05 21:05:39 +08:00
fix: 修复语音消息取错、时长精度与转写不刷新
- 时长解析:改取 <voicemsg voicelength>(毫秒),不再误取 length(SILK 字节数) - 时长显示:四舍五入对齐微信口径,有原生时长时不被解码时长覆盖 - 转写:重新识别改为强制重算,跳过身份级缓存短路(音频级缓存仍生效) - 日报:语音累计秒数显示取整,不再出现小数
This commit is contained in:
@@ -1246,10 +1246,15 @@ async function runSingleExport(
|
||||
const wavChannels =
|
||||
audioBuffer.length >= 44 ? audioBuffer.readUInt16LE(22) : 1
|
||||
const pcmBytes = Math.max(0, audioBuffer.length - 44)
|
||||
message.voiceDuration = Math.max(
|
||||
1,
|
||||
Math.round(pcmBytes / (wavSampleRate * wavChannels * 2))
|
||||
)
|
||||
// 口径统一:消息解析阶段已从 <voicemsg voicelength> 拿到微信的原始秒数(带小数),
|
||||
// 它是唯一权威来源,不要覆盖。只有拿不到时才退回用 PCM 字节数估算——
|
||||
// 那份估算是整秒、且下限 1 秒(WAV 缺失头部时的兜底),语义不同。
|
||||
if (message.voiceDuration == null) {
|
||||
message.voiceDuration = Math.max(
|
||||
1,
|
||||
Math.round(pcmBytes / (wavSampleRate * wavChannels * 2))
|
||||
)
|
||||
}
|
||||
} catch (error) {
|
||||
keepMediaError(
|
||||
request,
|
||||
|
||||
@@ -588,7 +588,7 @@ const renderReportHtml = async (request: GroupReportExportRequest): Promise<stri
|
||||
(item, index) => `<div class="rank tm-fragment tm-ranking-item">
|
||||
${renderAvatar(item.sender, 'ranking')}
|
||||
<b>${index + 1}. ${escapeHtml(item.sender)}</b>
|
||||
<span>${item.count} 条 · ${item.durationSec} 秒</span>
|
||||
<span>${item.count} 条 · ${Math.round(item.durationSec)} 秒</span>
|
||||
</div>`
|
||||
)
|
||||
.join('')
|
||||
|
||||
+8
-5
@@ -1853,12 +1853,15 @@ app.whenReady().then(async () => {
|
||||
return error ? { success: false, error } : { success: true }
|
||||
})
|
||||
|
||||
ipcMain.handle('voice:recognize', (_, reference: VoiceMessageReference) => {
|
||||
if (!voiceRecognition) {
|
||||
return { success: false, code: 'NOT_CONNECTED', error: '语音识别服务尚未初始化' }
|
||||
ipcMain.handle(
|
||||
'voice:recognize',
|
||||
(_, reference: VoiceMessageReference, options?: { force?: boolean }) => {
|
||||
if (!voiceRecognition) {
|
||||
return { success: false, code: 'NOT_CONNECTED', error: '语音识别服务尚未初始化' }
|
||||
}
|
||||
return voiceRecognition.recognize(reference, options)
|
||||
}
|
||||
return voiceRecognition.recognize(reference)
|
||||
})
|
||||
)
|
||||
|
||||
ipcMain.handle('voice:getTranscriptSnapshot', (_, reference: VoiceMessageReference) => {
|
||||
return voiceRecognition?.getTranscriptSnapshot(reference) || { state: 'pending' as const }
|
||||
|
||||
@@ -125,7 +125,10 @@ export type ParsedContent =
|
||||
export function parseMessageContent(content: string, messageType: number): ParsedContent {
|
||||
// Voice rows may keep their binary payload outside msgContent, so an empty
|
||||
// content string is still a valid voice message.
|
||||
if (messageType === 34) return { type: 'voice' }
|
||||
if (messageType === 34) {
|
||||
const duration = parseVoiceDurationSeconds(content)
|
||||
return duration === undefined ? { type: 'voice' } : { type: 'voice', duration }
|
||||
}
|
||||
if (!content || typeof content !== 'string') {
|
||||
return { type: 'unknown', raw: content || '' }
|
||||
}
|
||||
@@ -157,6 +160,33 @@ export function parseMessageContent(content: string, messageType: number): Parse
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 语音时长藏在解压后的 message_content 里:`<voicemsg ... voicelength="1600" ...>`,单位毫秒。
|
||||
* Msg_* 表没有 voice_length 列,这是唯一来源。
|
||||
*
|
||||
* `<voicemsg>` 上两个极易混淆的属性(真机实测,同一条 1.6 秒语音,2026-09-20):
|
||||
*
|
||||
* - `voicelength="1600"` → **毫秒时长**。这条语音微信气泡显示 2"(1.6 秒四舍五入)。
|
||||
* **要取的是它。**
|
||||
* - `length="6672"` → **SILK 编码数据的字节数,与时长无关**。
|
||||
* 已验证:`wcdb_get_voice_data` 取出的 SILK 恰好是 6672 字节,
|
||||
* 解码后为 51200 字节 PCM(1.6 秒)。误取它会算出 6.672 秒,把 2" 显示成 0:07。
|
||||
*
|
||||
* 换算成秒后**刻意保留小数**(1600ms → 1.6):在这里取整会把精度永久丢掉,
|
||||
* 后面显示层再怎么四舍五入都对不回微信的口径(微信是四舍五入到整秒)。
|
||||
*
|
||||
* 注:`<videomsg length="...">` 的 `length` 同理是字节数,不是时长。
|
||||
*/
|
||||
function parseVoiceDurationSeconds(content: string): number | undefined {
|
||||
if (!content || typeof content !== 'string') return undefined
|
||||
const decoded = decodeXmlEntities(stripChatroomPrefix(content))
|
||||
const rawLength = extractXmlAttribute(decoded, 'voicemsg', 'voicelength')
|
||||
if (!rawLength) return undefined
|
||||
const milliseconds = Number(rawLength)
|
||||
if (!Number.isFinite(milliseconds) || milliseconds <= 0) return undefined
|
||||
return milliseconds / 1000
|
||||
}
|
||||
|
||||
function parseVideoMessage(content: string): ParsedContent {
|
||||
const decoded = decodeXmlEntities(stripChatroomPrefix(content))
|
||||
const md5 = normalizeMd5(extractXmlAttribute(decoded, 'videomsg', 'md5'))
|
||||
|
||||
@@ -669,6 +669,9 @@ function listSourceMessages(
|
||||
: msg.mesLocalID || Math.random().toString()
|
||||
)
|
||||
const imageContent = contentData?.type === 'image' ? contentData : undefined
|
||||
// 语音时长来自 message_content 的 <voicemsg voicelength>(毫秒)——注意不是 length,
|
||||
// 那是 SILK 数据字节数。已在 parseMessageContent 里换算成秒。
|
||||
const voiceDuration = contentData?.type === 'voice' ? contentData.duration : undefined
|
||||
// Local ids repeat across conversations. Scope media handles to this database
|
||||
// connection and image without changing the message id used by other clients.
|
||||
const mediaId = imageContent
|
||||
@@ -737,7 +740,8 @@ function listSourceMessages(
|
||||
createTime,
|
||||
recoveredFromRecallJournal,
|
||||
contentData,
|
||||
media
|
||||
media,
|
||||
voiceDuration
|
||||
}
|
||||
})
|
||||
|
||||
|
||||
@@ -37,21 +37,29 @@ export class VoicePipeline {
|
||||
async run(
|
||||
accountId: string,
|
||||
reference: VoiceMessageReference,
|
||||
signal?: AbortSignal
|
||||
signal?: AbortSignal,
|
||||
options?: { force?: boolean }
|
||||
): Promise<{ transcript: string; language?: string; durationMs: number; cached: boolean }> {
|
||||
const messageIdentity = voiceMessageIdentity(reference)
|
||||
const compatible = this.transcripts.findCompatible({
|
||||
accountId,
|
||||
messageIdentity,
|
||||
processorVersion: this.audioProcessor.version,
|
||||
...this.recognizer.metadata
|
||||
})
|
||||
if (compatible?.transcript.trim()) {
|
||||
return {
|
||||
transcript: compatible.transcript.trim(),
|
||||
language: compatible.language,
|
||||
durationMs: compatible.durationMs,
|
||||
cached: true
|
||||
/*
|
||||
* findCompatible 只按消息身份匹配,不含 audio_hash:音频被换掉(例如取音频的
|
||||
* 逻辑修好后)时它仍会命中旧记录。用户主动触发的识别必须跳过它,重新取一次
|
||||
* 音频 —— 音频级缓存 find(key) 带 audio_hash,才是正确的失效机制。
|
||||
*/
|
||||
if (!options?.force) {
|
||||
const compatible = this.transcripts.findCompatible({
|
||||
accountId,
|
||||
messageIdentity,
|
||||
processorVersion: this.audioProcessor.version,
|
||||
...this.recognizer.metadata
|
||||
})
|
||||
if (compatible?.transcript.trim()) {
|
||||
return {
|
||||
transcript: compatible.transcript.trim(),
|
||||
language: compatible.language,
|
||||
durationMs: compatible.durationMs,
|
||||
cached: true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -23,6 +23,7 @@ type TranscriptUpdateListener = (update: VoiceTranscriptUpdate) => Promise<void>
|
||||
type RecognitionOptions = {
|
||||
priority?: VoiceRecognitionPriority
|
||||
publishTranscriptUpdate?: boolean
|
||||
force?: boolean
|
||||
}
|
||||
|
||||
export class VoiceRecognitionUseCase {
|
||||
@@ -108,7 +109,9 @@ export class VoiceRecognitionUseCase {
|
||||
error: '请先下载语音识别模型'
|
||||
} as const
|
||||
}
|
||||
const result = await pipeline.run(accountId, reference, signal)
|
||||
const result = await pipeline.run(accountId, reference, signal, {
|
||||
force: options?.force
|
||||
})
|
||||
if (signal.aborted || !this.isCurrentAccount(accountId, generation)) {
|
||||
throw new DOMException('Recognition cancelled', 'AbortError')
|
||||
}
|
||||
|
||||
@@ -73,11 +73,19 @@ export class VoiceService {
|
||||
pcmResult.audio.sampleRate,
|
||||
pcmResult.audio.channels
|
||||
)
|
||||
// duration 由 PCM 字节数反算(wavData 去掉 44 字节头即 PCM),
|
||||
// 用来和用户实际听到的长度对齐、排查「时长显示不对」这类问题。
|
||||
// 注意它**不是权威值**:权威时长在消息 XML 的 <voicemsg voicelength>(毫秒),
|
||||
// 显示层以那个为准;这里只是解码结果的自证。
|
||||
const durationSeconds =
|
||||
pcmData.length / (pcmResult.audio.sampleRate * pcmResult.audio.channels * 2)
|
||||
console.log(
|
||||
'[VoiceService] wavData length:',
|
||||
wavData.length,
|
||||
'base64 length:',
|
||||
wavData.toString('base64').length
|
||||
wavData.toString('base64').length,
|
||||
'duration:',
|
||||
`${durationSeconds.toFixed(2)}s`
|
||||
)
|
||||
|
||||
const base64Data = wavData.toString('base64')
|
||||
|
||||
Vendored
+4
-1
@@ -378,7 +378,10 @@ declare global {
|
||||
cancelVoiceModelDownload: () => Promise<{ success: boolean }>
|
||||
removeVoiceModel: () => Promise<VoiceModelStatus>
|
||||
openVoiceModelDirectory: () => Promise<{ success: boolean; error?: string }>
|
||||
recognizeVoice: (reference: VoiceMessageReference) => Promise<VoiceRecognitionResult>
|
||||
recognizeVoice: (
|
||||
reference: VoiceMessageReference,
|
||||
options?: { force?: boolean }
|
||||
) => Promise<VoiceRecognitionResult>
|
||||
getVoiceTranscriptSnapshot: (
|
||||
reference: VoiceMessageReference
|
||||
) => Promise<VoiceTranscriptSnapshot>
|
||||
|
||||
@@ -285,8 +285,10 @@ const api = {
|
||||
removeVoiceModel: (): Promise<VoiceModelStatus> => ipcRenderer.invoke('voice:removeModel'),
|
||||
openVoiceModelDirectory: (): Promise<{ success: boolean; error?: string }> =>
|
||||
ipcRenderer.invoke('voice:openModelDirectory'),
|
||||
recognizeVoice: (reference: VoiceMessageReference): Promise<VoiceRecognitionResult> =>
|
||||
ipcRenderer.invoke('voice:recognize', reference),
|
||||
recognizeVoice: (
|
||||
reference: VoiceMessageReference,
|
||||
options?: { force?: boolean }
|
||||
): Promise<VoiceRecognitionResult> => ipcRenderer.invoke('voice:recognize', reference, options),
|
||||
getVoiceTranscriptSnapshot: (
|
||||
reference: VoiceMessageReference
|
||||
): Promise<VoiceTranscriptSnapshot> =>
|
||||
|
||||
@@ -73,12 +73,14 @@ export function VoicePlayer({
|
||||
const audio = new Audio()
|
||||
audio.preload = 'auto'
|
||||
audio.src = blobUrl
|
||||
audio.onloadedmetadata = () => {
|
||||
if (Number.isFinite(audio.duration)) setAudioDuration(audio.duration)
|
||||
}
|
||||
audio.ontimeupdate = () => {
|
||||
// Silk 解码出来的时长比微信的 length 系统性偏短(实测少 20–279ms),
|
||||
// 用它覆盖会把刚补回来的精度又弄丢:有原生时长时一律不覆盖,没有才回退。
|
||||
const useDecodedDuration = () => {
|
||||
if (duration !== undefined) return
|
||||
if (Number.isFinite(audio.duration)) setAudioDuration(audio.duration)
|
||||
}
|
||||
audio.onloadedmetadata = useDecodedDuration
|
||||
audio.ontimeupdate = useDecodedDuration
|
||||
audio.onended = () => {
|
||||
setIsPlaying(false)
|
||||
if (globalCurrentAudio === audio) {
|
||||
@@ -89,7 +91,7 @@ export function VoicePlayer({
|
||||
audioRef.current = audio
|
||||
objectUrlRef.current = blobUrl
|
||||
return audio
|
||||
}, [])
|
||||
}, [duration])
|
||||
|
||||
const handlePlayPause = useCallback(async () => {
|
||||
if (loading) return
|
||||
@@ -163,7 +165,8 @@ export function VoicePlayer({
|
||||
|
||||
setTranscribing(true)
|
||||
try {
|
||||
const result = await window.api.recognizeVoice(voiceReference)
|
||||
// 用户主动触发:跳过身份级缓存,重新取音频(音频级缓存仍生效)。
|
||||
const result = await window.api.recognizeVoice(voiceReference, { force: true })
|
||||
if (result.success) {
|
||||
setTranscript(result.transcript?.trim() || '未识别出文字')
|
||||
setModelStatus(null)
|
||||
@@ -204,8 +207,11 @@ export function VoicePlayer({
|
||||
|
||||
const formatDuration = (seconds: number | undefined): string => {
|
||||
if (!seconds || !isFinite(seconds)) return '0:00'
|
||||
const mins = Math.floor(seconds / 60)
|
||||
const secs = Math.floor(seconds % 60)
|
||||
// 微信的口径是四舍五入到整秒(length=4211ms 显示 4"),所以这里也必须 round 而非 floor。
|
||||
// 先整体取整再拆分钟/秒,59.6 → "1:00" 而不会冒出 "0:60"。
|
||||
const rounded = Math.round(seconds)
|
||||
const mins = Math.floor(rounded / 60)
|
||||
const secs = rounded % 60
|
||||
return `${mins}:${secs.toString().padStart(2, '0')}`
|
||||
}
|
||||
|
||||
|
||||
@@ -138,9 +138,13 @@ export const summarySender = (
|
||||
export const summaryContent = (message: Message): string => {
|
||||
const data = message.contentData
|
||||
if (message.type === '语音' || data?.type === 'voice') {
|
||||
// 语音时长来自 <voicemsg voicelength>(毫秒),是带小数的秒(1600ms → 1.6):
|
||||
// 累加时保留精度,只在显示给人的文案里取整。
|
||||
const durationSec = data?.type === 'voice' ? data.duration : undefined
|
||||
const durationLabel = durationSec ? ` ${Math.round(durationSec)}秒` : ''
|
||||
return message.voiceTranscript?.trim()
|
||||
? `[语音${data?.type === 'voice' && data.duration ? ` ${data.duration}秒` : ''}] ${message.voiceTranscript.trim()}`
|
||||
: `[语音${data?.type === 'voice' && data.duration ? ` ${data.duration}秒` : ''}]`
|
||||
? `[语音${durationLabel}] ${message.voiceTranscript.trim()}`
|
||||
: `[语音${durationLabel}]`
|
||||
}
|
||||
if (!data) return message.content?.trim() || `[${message.type || '消息'}]`
|
||||
|
||||
@@ -567,14 +571,14 @@ const buildMediaSection = async (
|
||||
voiceHighlights.push({
|
||||
title: '语音输出王',
|
||||
sender: voiceLeaderboard[0].sender,
|
||||
note: `共发送 ${voiceLeaderboard[0].count} 条语音,累计 ${voiceLeaderboard[0].durationSec} 秒。`
|
||||
note: `共发送 ${voiceLeaderboard[0].count} 条语音,累计 ${Math.round(voiceLeaderboard[0].durationSec)} 秒。`
|
||||
})
|
||||
}
|
||||
if (bestStreak && bestStreak.count >= 2) {
|
||||
voiceHighlights.push({
|
||||
title: '连续发言时刻',
|
||||
sender: bestStreak.sender,
|
||||
note: `${bestStreak.time} 连发 ${bestStreak.count} 条语音,共 ${bestStreak.duration} 秒。`
|
||||
note: `${bestStreak.time} 连发 ${bestStreak.count} 条语音,共 ${Math.round(bestStreak.duration)} 秒。`
|
||||
})
|
||||
}
|
||||
|
||||
@@ -810,7 +814,7 @@ export const buildGroupReportFacts = async (
|
||||
|
||||
const factsPrompt = [
|
||||
`报告模式:${reportMode === 'compact' ? '精简版(30秒可读完)' : '完整版(保留更多上下文)'}`,
|
||||
`消息统计:共 ${transcriptRows.length} 条,活跃成员 ${speakerCounts.size} 人,图片 ${imageCount} 张,表情 ${stickerCount} 条,语音 ${voiceCount} 条(累计 ${voiceDurationSec} 秒)。`,
|
||||
`消息统计:共 ${transcriptRows.length} 条,活跃成员 ${speakerCounts.size} 人,图片 ${imageCount} 张,表情 ${stickerCount} 条,语音 ${voiceCount} 条(累计 ${Math.round(voiceDurationSec)} 秒)。`,
|
||||
activeTimeline ? `活跃时段:${activeTimeline}` : '',
|
||||
// AI 图片理解结果(由 ImageInsightService 提供,缓存命中或已调用 Vision)
|
||||
(media.visionGallery?.length ?? 0) > 0
|
||||
@@ -824,7 +828,7 @@ export const buildGroupReportFacts = async (
|
||||
voiceLeaderboard.length
|
||||
? `语音榜:${voiceLeaderboard
|
||||
.slice(0, 3)
|
||||
.map((item) => `${item.sender} ${item.count} 条 / ${item.durationSec} 秒`)
|
||||
.map((item) => `${item.sender} ${item.count} 条 / ${Math.round(item.durationSec)} 秒`)
|
||||
.join(';')}`
|
||||
: '',
|
||||
collectQuestionCandidates(messages, contact, isGroup).length
|
||||
|
||||
Reference in New Issue
Block a user