fix: 修复语音消息取错、时长精度与转写不刷新

- 时长解析:改取 <voicemsg voicelength>(毫秒),不再误取 length(SILK 字节数)
- 时长显示:四舍五入对齐微信口径,有原生时长时不被解码时长覆盖
- 转写:重新识别改为强制重算,跳过身份级缓存短路(音频级缓存仍生效)
- 日报:语音累计秒数显示取整,不再出现小数
This commit is contained in:
Wxw-Gu
2026-09-20 15:40:01 +08:00
parent 9b82d29037
commit 9593ca0f54
21 changed files with 435 additions and 52 deletions
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+9 -4
View File
@@ -1246,10 +1246,15 @@ async function runSingleExport(
const wavChannels =
audioBuffer.length >= 44 ? audioBuffer.readUInt16LE(22) : 1
const pcmBytes = Math.max(0, audioBuffer.length - 44)
message.voiceDuration = Math.max(
1,
Math.round(pcmBytes / (wavSampleRate * wavChannels * 2))
)
// 口径统一:消息解析阶段已从 <voicemsg voicelength> 拿到微信的原始秒数(带小数),
// 它是唯一权威来源,不要覆盖。只有拿不到时才退回用 PCM 字节数估算——
// 那份估算是整秒、且下限 1 秒(WAV 缺失头部时的兜底),语义不同。
if (message.voiceDuration == null) {
message.voiceDuration = Math.max(
1,
Math.round(pcmBytes / (wavSampleRate * wavChannels * 2))
)
}
} catch (error) {
keepMediaError(
request,
+1 -1
View File
@@ -588,7 +588,7 @@ const renderReportHtml = async (request: GroupReportExportRequest): Promise<stri
(item, index) => `<div class="rank tm-fragment tm-ranking-item">
${renderAvatar(item.sender, 'ranking')}
<b>${index + 1}. ${escapeHtml(item.sender)}</b>
<span>${item.count} 条 · ${item.durationSec} 秒</span>
<span>${item.count} 条 · ${Math.round(item.durationSec)} 秒</span>
</div>`
)
.join('')
+8 -5
View File
@@ -1853,12 +1853,15 @@ app.whenReady().then(async () => {
return error ? { success: false, error } : { success: true }
})
ipcMain.handle('voice:recognize', (_, reference: VoiceMessageReference) => {
if (!voiceRecognition) {
return { success: false, code: 'NOT_CONNECTED', error: '语音识别服务尚未初始化' }
ipcMain.handle(
'voice:recognize',
(_, reference: VoiceMessageReference, options?: { force?: boolean }) => {
if (!voiceRecognition) {
return { success: false, code: 'NOT_CONNECTED', error: '语音识别服务尚未初始化' }
}
return voiceRecognition.recognize(reference, options)
}
return voiceRecognition.recognize(reference)
})
)
ipcMain.handle('voice:getTranscriptSnapshot', (_, reference: VoiceMessageReference) => {
return voiceRecognition?.getTranscriptSnapshot(reference) || { state: 'pending' as const }
+31 -1
View File
@@ -125,7 +125,10 @@ export type ParsedContent =
export function parseMessageContent(content: string, messageType: number): ParsedContent {
// Voice rows may keep their binary payload outside msgContent, so an empty
// content string is still a valid voice message.
if (messageType === 34) return { type: 'voice' }
if (messageType === 34) {
const duration = parseVoiceDurationSeconds(content)
return duration === undefined ? { type: 'voice' } : { type: 'voice', duration }
}
if (!content || typeof content !== 'string') {
return { type: 'unknown', raw: content || '' }
}
@@ -157,6 +160,33 @@ export function parseMessageContent(content: string, messageType: number): Parse
}
}
/**
* 语音时长藏在解压后的 message_content 里:`<voicemsg ... voicelength="1600" ...>`,单位毫秒。
* Msg_* 表没有 voice_length 列,这是唯一来源。
*
* `<voicemsg>` 上两个极易混淆的属性(真机实测,同一条 1.6 秒语音,2026-09-20):
*
* - `voicelength="1600"` → **毫秒时长**。这条语音微信气泡显示 2"(1.6 秒四舍五入)。
* **要取的是它。**
* - `length="6672"` → **SILK 编码数据的字节数,与时长无关**。
* 已验证:`wcdb_get_voice_data` 取出的 SILK 恰好是 6672 字节,
* 解码后为 51200 字节 PCM(1.6 秒)。误取它会算出 6.672 秒,把 2" 显示成 0:07。
*
* 换算成秒后**刻意保留小数**(1600ms → 1.6):在这里取整会把精度永久丢掉,
* 后面显示层再怎么四舍五入都对不回微信的口径(微信是四舍五入到整秒)。
*
* 注:`<videomsg length="...">` 的 `length` 同理是字节数,不是时长。
*/
function parseVoiceDurationSeconds(content: string): number | undefined {
if (!content || typeof content !== 'string') return undefined
const decoded = decodeXmlEntities(stripChatroomPrefix(content))
const rawLength = extractXmlAttribute(decoded, 'voicemsg', 'voicelength')
if (!rawLength) return undefined
const milliseconds = Number(rawLength)
if (!Number.isFinite(milliseconds) || milliseconds <= 0) return undefined
return milliseconds / 1000
}
function parseVideoMessage(content: string): ParsedContent {
const decoded = decodeXmlEntities(stripChatroomPrefix(content))
const md5 = normalizeMd5(extractXmlAttribute(decoded, 'videomsg', 'md5'))
+5 -1
View File
@@ -669,6 +669,9 @@ function listSourceMessages(
: msg.mesLocalID || Math.random().toString()
)
const imageContent = contentData?.type === 'image' ? contentData : undefined
// 语音时长来自 message_content 的 <voicemsg voicelength>(毫秒)——注意不是 length,
// 那是 SILK 数据字节数。已在 parseMessageContent 里换算成秒。
const voiceDuration = contentData?.type === 'voice' ? contentData.duration : undefined
// Local ids repeat across conversations. Scope media handles to this database
// connection and image without changing the message id used by other clients.
const mediaId = imageContent
@@ -737,7 +740,8 @@ function listSourceMessages(
createTime,
recoveredFromRecallJournal,
contentData,
media
media,
voiceDuration
}
})
+21 -13
View File
@@ -37,21 +37,29 @@ export class VoicePipeline {
async run(
accountId: string,
reference: VoiceMessageReference,
signal?: AbortSignal
signal?: AbortSignal,
options?: { force?: boolean }
): Promise<{ transcript: string; language?: string; durationMs: number; cached: boolean }> {
const messageIdentity = voiceMessageIdentity(reference)
const compatible = this.transcripts.findCompatible({
accountId,
messageIdentity,
processorVersion: this.audioProcessor.version,
...this.recognizer.metadata
})
if (compatible?.transcript.trim()) {
return {
transcript: compatible.transcript.trim(),
language: compatible.language,
durationMs: compatible.durationMs,
cached: true
/*
* findCompatible 只按消息身份匹配,不含 audio_hash:音频被换掉(例如取音频的
* 逻辑修好后)时它仍会命中旧记录。用户主动触发的识别必须跳过它,重新取一次
* 音频 —— 音频级缓存 find(key) 带 audio_hash,才是正确的失效机制。
*/
if (!options?.force) {
const compatible = this.transcripts.findCompatible({
accountId,
messageIdentity,
processorVersion: this.audioProcessor.version,
...this.recognizer.metadata
})
if (compatible?.transcript.trim()) {
return {
transcript: compatible.transcript.trim(),
language: compatible.language,
durationMs: compatible.durationMs,
cached: true
}
}
}
@@ -23,6 +23,7 @@ type TranscriptUpdateListener = (update: VoiceTranscriptUpdate) => Promise<void>
type RecognitionOptions = {
priority?: VoiceRecognitionPriority
publishTranscriptUpdate?: boolean
force?: boolean
}
export class VoiceRecognitionUseCase {
@@ -108,7 +109,9 @@ export class VoiceRecognitionUseCase {
error: '请先下载语音识别模型'
} as const
}
const result = await pipeline.run(accountId, reference, signal)
const result = await pipeline.run(accountId, reference, signal, {
force: options?.force
})
if (signal.aborted || !this.isCurrentAccount(accountId, generation)) {
throw new DOMException('Recognition cancelled', 'AbortError')
}
+9 -1
View File
@@ -73,11 +73,19 @@ export class VoiceService {
pcmResult.audio.sampleRate,
pcmResult.audio.channels
)
// duration 由 PCM 字节数反算(wavData 去掉 44 字节头即 PCM),
// 用来和用户实际听到的长度对齐、排查「时长显示不对」这类问题。
// 注意它**不是权威值**:权威时长在消息 XML 的 <voicemsg voicelength>(毫秒),
// 显示层以那个为准;这里只是解码结果的自证。
const durationSeconds =
pcmData.length / (pcmResult.audio.sampleRate * pcmResult.audio.channels * 2)
console.log(
'[VoiceService] wavData length:',
wavData.length,
'base64 length:',
wavData.toString('base64').length
wavData.toString('base64').length,
'duration:',
`${durationSeconds.toFixed(2)}s`
)
const base64Data = wavData.toString('base64')
+4 -1
View File
@@ -378,7 +378,10 @@ declare global {
cancelVoiceModelDownload: () => Promise<{ success: boolean }>
removeVoiceModel: () => Promise<VoiceModelStatus>
openVoiceModelDirectory: () => Promise<{ success: boolean; error?: string }>
recognizeVoice: (reference: VoiceMessageReference) => Promise<VoiceRecognitionResult>
recognizeVoice: (
reference: VoiceMessageReference,
options?: { force?: boolean }
) => Promise<VoiceRecognitionResult>
getVoiceTranscriptSnapshot: (
reference: VoiceMessageReference
) => Promise<VoiceTranscriptSnapshot>
+4 -2
View File
@@ -285,8 +285,10 @@ const api = {
removeVoiceModel: (): Promise<VoiceModelStatus> => ipcRenderer.invoke('voice:removeModel'),
openVoiceModelDirectory: (): Promise<{ success: boolean; error?: string }> =>
ipcRenderer.invoke('voice:openModelDirectory'),
recognizeVoice: (reference: VoiceMessageReference): Promise<VoiceRecognitionResult> =>
ipcRenderer.invoke('voice:recognize', reference),
recognizeVoice: (
reference: VoiceMessageReference,
options?: { force?: boolean }
): Promise<VoiceRecognitionResult> => ipcRenderer.invoke('voice:recognize', reference, options),
getVoiceTranscriptSnapshot: (
reference: VoiceMessageReference
): Promise<VoiceTranscriptSnapshot> =>
+14 -8
View File
@@ -73,12 +73,14 @@ export function VoicePlayer({
const audio = new Audio()
audio.preload = 'auto'
audio.src = blobUrl
audio.onloadedmetadata = () => {
if (Number.isFinite(audio.duration)) setAudioDuration(audio.duration)
}
audio.ontimeupdate = () => {
// Silk 解码出来的时长比微信的 length 系统性偏短(实测少 20–279ms),
// 用它覆盖会把刚补回来的精度又弄丢:有原生时长时一律不覆盖,没有才回退。
const useDecodedDuration = () => {
if (duration !== undefined) return
if (Number.isFinite(audio.duration)) setAudioDuration(audio.duration)
}
audio.onloadedmetadata = useDecodedDuration
audio.ontimeupdate = useDecodedDuration
audio.onended = () => {
setIsPlaying(false)
if (globalCurrentAudio === audio) {
@@ -89,7 +91,7 @@ export function VoicePlayer({
audioRef.current = audio
objectUrlRef.current = blobUrl
return audio
}, [])
}, [duration])
const handlePlayPause = useCallback(async () => {
if (loading) return
@@ -163,7 +165,8 @@ export function VoicePlayer({
setTranscribing(true)
try {
const result = await window.api.recognizeVoice(voiceReference)
// 用户主动触发:跳过身份级缓存,重新取音频(音频级缓存仍生效)。
const result = await window.api.recognizeVoice(voiceReference, { force: true })
if (result.success) {
setTranscript(result.transcript?.trim() || '未识别出文字')
setModelStatus(null)
@@ -204,8 +207,11 @@ export function VoicePlayer({
const formatDuration = (seconds: number | undefined): string => {
if (!seconds || !isFinite(seconds)) return '0:00'
const mins = Math.floor(seconds / 60)
const secs = Math.floor(seconds % 60)
// 微信的口径是四舍五入到整秒(length=4211ms 显示 4"),所以这里也必须 round 而非 floor。
// 先整体取整再拆分钟/秒,59.6 → "1:00" 而不会冒出 "0:60"。
const rounded = Math.round(seconds)
const mins = Math.floor(rounded / 60)
const secs = rounded % 60
return `${mins}:${secs.toString().padStart(2, '0')}`
}
+10 -6
View File
@@ -138,9 +138,13 @@ export const summarySender = (
export const summaryContent = (message: Message): string => {
const data = message.contentData
if (message.type === '语音' || data?.type === 'voice') {
// 语音时长来自 <voicemsg voicelength>(毫秒),是带小数的秒(1600ms → 1.6):
// 累加时保留精度,只在显示给人的文案里取整。
const durationSec = data?.type === 'voice' ? data.duration : undefined
const durationLabel = durationSec ? ` ${Math.round(durationSec)}秒` : ''
return message.voiceTranscript?.trim()
? `[语音${data?.type === 'voice' && data.duration ? ` ${data.duration}秒` : ''}] ${message.voiceTranscript.trim()}`
: `[语音${data?.type === 'voice' && data.duration ? ` ${data.duration}秒` : ''}]`
? `[语音${durationLabel}] ${message.voiceTranscript.trim()}`
: `[语音${durationLabel}]`
}
if (!data) return message.content?.trim() || `[${message.type || '消息'}]`
@@ -567,14 +571,14 @@ const buildMediaSection = async (
voiceHighlights.push({
title: '语音输出王',
sender: voiceLeaderboard[0].sender,
note: `共发送 ${voiceLeaderboard[0].count} 条语音,累计 ${voiceLeaderboard[0].durationSec} 秒。`
note: `共发送 ${voiceLeaderboard[0].count} 条语音,累计 ${Math.round(voiceLeaderboard[0].durationSec)} 秒。`
})
}
if (bestStreak && bestStreak.count >= 2) {
voiceHighlights.push({
title: '连续发言时刻',
sender: bestStreak.sender,
note: `${bestStreak.time} 连发 ${bestStreak.count} 条语音,共 ${bestStreak.duration} 秒。`
note: `${bestStreak.time} 连发 ${bestStreak.count} 条语音,共 ${Math.round(bestStreak.duration)} 秒。`
})
}
@@ -810,7 +814,7 @@ export const buildGroupReportFacts = async (
const factsPrompt = [
`报告模式:${reportMode === 'compact' ? '精简版(30秒可读完)' : '完整版(保留更多上下文)'}`,
`消息统计:共 ${transcriptRows.length} 条,活跃成员 ${speakerCounts.size} 人,图片 ${imageCount} 张,表情 ${stickerCount} 条,语音 ${voiceCount} 条(累计 ${voiceDurationSec} 秒)。`,
`消息统计:共 ${transcriptRows.length} 条,活跃成员 ${speakerCounts.size} 人,图片 ${imageCount} 张,表情 ${stickerCount} 条,语音 ${voiceCount} 条(累计 ${Math.round(voiceDurationSec)} 秒)。`,
activeTimeline ? `活跃时段:${activeTimeline}` : '',
// AI 图片理解结果(由 ImageInsightService 提供,缓存命中或已调用 Vision)
(media.visionGallery?.length ?? 0) > 0
@@ -824,7 +828,7 @@ export const buildGroupReportFacts = async (
voiceLeaderboard.length
? `语音榜:${voiceLeaderboard
.slice(0, 3)
.map((item) => `${item.sender} ${item.count} 条 / ${item.durationSec} 秒`)
.map((item) => `${item.sender} ${item.count} 条 / ${Math.round(item.durationSec)} 秒`)
.join(';')}`
: '',
collectQuestionCandidates(messages, contact, isGroup).length
+80 -7
View File
@@ -1,4 +1,4 @@
import { render, screen, waitFor } from '@testing-library/react'
import { act, render, screen, waitFor } from '@testing-library/react'
import userEvent from '@testing-library/user-event'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { VoicePlayer } from '../../src/renderer/src/components/VoicePlayer'
@@ -6,6 +6,8 @@ import { VoicePlayer } from '../../src/renderer/src/components/VoicePlayer'
const play = vi.fn(() => Promise.resolve())
const pause = vi.fn()
let lastAudio: FakeAudio | null = null
class FakeAudio {
preload = ''
src = ''
@@ -18,12 +20,27 @@ class FakeAudio {
pause = pause
load = vi.fn()
removeAttribute = vi.fn()
constructor() {
lastAudio = this
}
}
const renderPlayer = (duration?: number) =>
render(
<VoicePlayer
sessionId="filehelper"
localId={11}
createTime={1785553200}
duration={duration}
/>
)
describe('VoicePlayer', () => {
beforeEach(() => {
play.mockClear()
pause.mockClear()
lastAudio = null
vi.stubGlobal('Audio', FakeAudio)
window.api = {
getVoiceData: vi.fn().mockResolvedValue({
@@ -70,17 +87,73 @@ describe('VoicePlayer', () => {
render(<VoicePlayer sessionId="filehelper" localId={11} createTime={1785553200} duration={1} />)
await userEvent.click(screen.getByRole('button', { name: '转文字' }))
// 用户主动触发必须带 force:否则会被身份级缓存(不含 audio_hash)短路。
await waitFor(() =>
expect(window.api.recognizeVoice).toHaveBeenCalledWith({
sessionId: 'filehelper',
localId: 11,
createTime: 1785553200,
svrId: undefined
})
expect(window.api.recognizeVoice).toHaveBeenCalledWith(
{
sessionId: 'filehelper',
localId: 11,
createTime: 1785553200,
svrId: undefined
},
{ force: true }
)
)
expect(await screen.findByText('这是固定的测试转写')).toBeInTheDocument()
})
it('rounds the shown duration the way WeChat does', () => {
// 微信 4211ms 显示 4",所以口径是 round 不是 floor;进位要传给分钟位。
const cases: Array<[number | undefined, string]> = [
[1.7, '0:02'],
[4.211, '0:04'],
[7.505, '0:08'],
[59.6, '1:00'],
[119.6, '2:00'],
[3, '0:03'],
[7, '0:07'],
[15, '0:15'],
[0, '0:00'],
[undefined, '0:00'],
[NaN, '0:00']
]
for (const [duration, expected] of cases) {
const { container, unmount } = renderPlayer(duration)
expect(container.querySelector('.voice-duration')?.textContent).toBe(expected)
unmount()
}
})
it('keeps the WeChat duration instead of the systematically short decoded one', async () => {
const { container } = renderPlayer(1.979)
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:02')
await userEvent.click(container.querySelector('.voice-message') as HTMLElement)
await waitFor(() => expect(lastAudio).not.toBeNull())
// Silk 解码时长比微信 length 系统性偏短(实测少 20–279ms),用它覆盖会把精度弄丢。
lastAudio!.duration = 1
await act(async () => {
lastAudio!.onloadedmetadata?.()
lastAudio!.ontimeupdate?.()
})
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:02')
})
it('falls back to the decoded duration when WeChat did not provide one', async () => {
const { container } = renderPlayer()
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:00')
await userEvent.click(container.querySelector('.voice-message') as HTMLElement)
await waitFor(() => expect(lastAudio).not.toBeNull())
lastAudio!.duration = 59.6
lastAudio!.onloadedmetadata?.()
await waitFor(() =>
expect(container.querySelector('.voice-duration')?.textContent).toBe('1:00')
)
})
it('opens centralized settings when recognition assets are missing', async () => {
vi.mocked(window.api.getVoiceModelStatus).mockResolvedValue({
modelId: 'sensevoice-small-int8',
+35 -1
View File
@@ -4,7 +4,10 @@ import {
getSummaryDateRangeAt,
parseGroupDailyReport
} from '../../src/renderer/src/utils/group-report'
import { summaryContent } from '../../src/renderer/src/utils/group-report-facts'
import {
buildGroupReportFacts,
summaryContent
} from '../../src/renderer/src/utils/group-report-facts'
import { summarySender } from '../../src/renderer/src/utils/group-report-facts'
import { selectHeroParticipantNames } from '../../src/shared/group-report'
import type { GroupReportMetadata } from '../../src/shared/group-report'
@@ -62,6 +65,37 @@ describe('group report parsing', () => {
expect(summaryContent(message)).toContain('今晚八点确认发布。')
})
it('rounds accumulated voice seconds in the facts summary', async () => {
// 语音时长是 <voicemsg voicelength>(毫秒)换算来的小数秒(1600ms → 1.6)。
// 累加必须保留原始精度(否则多条累积会越差越多),但展示给用户的文案要取整。
const messages: Message[] = [
{
id: 'voice-a',
from: 'member',
type: '语音',
datetime: '2026-08-06 10:00:00',
content: '[语音]',
isSender: false,
contentData: { type: 'voice', duration: 1.979 }
},
{
id: 'voice-b',
from: 'member',
type: '语音',
datetime: '2026-08-06 10:00:05',
content: '[语音]',
isSender: false,
contentData: { type: 'voice', duration: 5.379 }
}
]
const snapshot = await buildGroupReportFacts(messages, null, true, 'compact')
// 1.979 + 5.379 = 7.358 → 文案必须显示整数
expect(snapshot.factsPrompt).toContain('累计 7 秒')
expect(snapshot.factsPrompt).not.toContain('7.358')
})
it('includes a voice transcript when legacy cached messages have no contentData', () => {
const message: Message = {
id: 'voice-legacy',
+32
View File
@@ -17,6 +17,38 @@ describe('message parser', () => {
).toMatchObject({ type: 'sticker', md5: 'abcdefabcdefabcdefabcdefabcdefab' })
})
it('reads the WeChat voice length in milliseconds and keeps fractional seconds', () => {
// 真机实测(2026-09-20):voicelength 才是毫秒时长,length 是编码数据长度——别取错。
// 属性值取自一条真机采样的语音(1.6 秒,微信气泡显示 2")。
const parsed = parseMessageContent(
'<msg><voicemsg endflag="1" cancelflag="0" forwardflag="0" voiceformat="4" voicelength="1600" length="6672" bufid="0" /></msg>',
34
)
expect(parsed).toEqual({ type: 'voice', duration: 1.6 })
// 误取 length 会得到 6.672 秒(把 2" 的语音显示成 0:07)——这条断言就是防这个回归。
expect(parsed).not.toEqual({ type: 'voice', duration: 6.672 })
// 微信四舍五入到整秒,取整必须在显示层做,不能在解析层丢精度。
expect(parseMessageContent('<msg><voicemsg voicelength="4211" /></msg>', 34)).toEqual({
type: 'voice',
duration: 4.211
})
})
it('leaves the voice duration undefined when the payload is missing or unusable', () => {
expect(parseMessageContent('', 34)).toEqual({ type: 'voice' })
expect(parseMessageContent('voice fixture', 34)).toEqual({ type: 'voice' })
expect(parseMessageContent('<msg><voicemsg voiceformat="4" /></msg>', 34)).toEqual({
type: 'voice'
})
expect(parseMessageContent('<msg><voicemsg voicelength="0" /></msg>', 34)).toEqual({
type: 'voice'
})
expect(parseMessageContent('<msg><voicemsg voicelength="abc" /></msg>', 34)).toEqual({
type: 'voice'
})
})
it('keeps video metadata when WeChat omits every MD5 field', () => {
const parsed = parseMessageContent(
'<msg><videomsg length="6402169" playlength="30" cdnthumbwidth="224" cdnthumbheight="398" aeskey="25201cc658042689d1ad6747cea2b240" rawmd5="" /></msg>',
+125
View File
@@ -361,3 +361,128 @@ describe('voice pipeline cache lookup', () => {
repository.close()
})
})
/*
* force 的语义:用户主动触发时必须绕开 findCompatible(它只按消息身份匹配,
* 不含 audio_hash),重新取一次音频;音频级 find(key) 带 audio_hash,保留。
*/
describe('voice pipeline force recognition', () => {
const forceRoot = mkdtempSync(join(tmpdir(), 'wxe-voice-pipeline-force-'))
afterAll(() => rmSync(forceRoot, { recursive: true, force: true }))
function createPipeline(repository: SqliteTranscriptRepository, transcript: string) {
const resolve = vi.fn().mockResolvedValue({
data: Buffer.from('encoded'),
codec: 'silk',
sourceHash: 'fresh-audio'
})
const decode = vi.fn().mockResolvedValue({
pcm: Buffer.from([1, 0]),
sampleRate: 16000,
channels: 1,
sourceHash: 'fresh-audio'
})
const process = vi.fn().mockReturnValue({
samples: new Float32Array([0.1]),
sampleRate: 16000,
channels: 1,
sourceHash: 'fresh-audio',
processorVersion: 'processor-v1',
durationMs: 1
})
const recognize = vi.fn().mockResolvedValue({ text: transcript })
const pipeline = new VoicePipeline(
{ resolve },
{ decode } as never,
{ version: 'processor-v1', process },
{
metadata: {
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a'
},
recognize,
dispose: vi.fn()
},
repository
)
return { pipeline, resolve, decode, process, recognize }
}
function seedStaleTranscript(repository: SqliteTranscriptRepository) {
repository.save({
accountId: 'account-a',
messageIdentity: voiceMessageIdentity(reference),
audioHash: 'stale-audio',
processorVersion: 'processor-v1',
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a',
transcript: '旧音频的转写',
durationMs: 900,
createdAt: 1,
updatedAt: 1
})
}
const reference = { sessionId: 'session', localId: 1, createTime: 2 }
it('skips the identity-level cache when force is requested', async () => {
const repository = new SqliteTranscriptRepository(join(forceRoot, 'forced.sqlite'))
seedStaleTranscript(repository)
const findCompatible = vi.spyOn(repository, 'findCompatible')
const find = vi.spyOn(repository, 'find')
const { pipeline, resolve, recognize } = createPipeline(repository, '正确音频的转写')
await expect(
pipeline.run('account-a', reference, undefined, { force: true })
).resolves.toMatchObject({ transcript: '正确音频的转写', cached: false })
expect(findCompatible).not.toHaveBeenCalled()
expect(resolve).toHaveBeenCalledOnce()
expect(find).toHaveBeenCalled()
expect(recognize).toHaveBeenCalledOnce()
expect(
repository.find({
accountId: 'account-a',
messageIdentity: voiceMessageIdentity(reference),
audioHash: 'fresh-audio',
processorVersion: 'processor-v1',
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a'
})?.transcript
).toBe('正确音频的转写')
repository.close()
})
it('keeps consulting the identity-level cache without force', async () => {
const repository = new SqliteTranscriptRepository(join(forceRoot, 'unchanged.sqlite'))
seedStaleTranscript(repository)
const findCompatible = vi.spyOn(repository, 'findCompatible')
const { pipeline, resolve, decode, process, recognize } = createPipeline(
repository,
'新转写'
)
await expect(pipeline.run('account-a', reference)).resolves.toMatchObject({
transcript: '旧音频的转写',
cached: true
})
expect(findCompatible).toHaveBeenCalledOnce()
expect(resolve).not.toHaveBeenCalled()
expect(decode).not.toHaveBeenCalled()
expect(process).not.toHaveBeenCalled()
expect(recognize).not.toHaveBeenCalled()
repository.close()
const explicitFalse = new SqliteTranscriptRepository(join(forceRoot, 'explicit-false.sqlite'))
seedStaleTranscript(explicitFalse)
const second = createPipeline(explicitFalse, '新转写')
await expect(
second.pipeline.run('account-a', reference, undefined, { force: false })
).resolves.toMatchObject({ transcript: '旧音频的转写', cached: true })
expect(second.resolve).not.toHaveBeenCalled()
explicitFalse.close()
})
})
@@ -101,6 +101,49 @@ describe('VoiceRecognitionUseCase transcript updates', () => {
await useCase.dispose()
})
it('forwards force to the pipeline so manual re-recognition can bypass the identity cache', async () => {
const useCase = createUseCase()
const state = useCase as unknown as {
pipeline: { run: ReturnType<typeof vi.fn> }
}
state.pipeline.run.mockResolvedValue({
transcript: '重新识别的文字',
durationMs: 900,
cached: false
})
const reference = { sessionId: 'fixture-contact', localId: 12, createTime: 1_785_895_203 }
const result = await useCase.recognize(reference, { force: true })
expect(result).toMatchObject({ success: true, transcript: '重新识别的文字' })
expect(state.pipeline.run).toHaveBeenCalledWith(
'account-a',
reference,
expect.anything(),
{ force: true }
)
await useCase.dispose()
})
it('leaves force undefined when the caller does not ask for a recompute', async () => {
const useCase = createUseCase()
const state = useCase as unknown as {
pipeline: { run: ReturnType<typeof vi.fn> }
}
state.pipeline.run.mockResolvedValue({ transcript: '缓存文字', durationMs: 900, cached: true })
const reference = { sessionId: 'fixture-contact', localId: 13, createTime: 1_785_895_204 }
await useCase.recognize(reference)
expect(state.pipeline.run).toHaveBeenCalledWith(
'account-a',
reference,
expect.anything(),
{ force: undefined }
)
await useCase.dispose()
})
it('publishes an explicit cached transcript for a coalesced export index refresh', async () => {
const useCase = createUseCase()
const listener = vi.fn().mockResolvedValue(undefined)