fix: 修复语音消息取错、时长精度与转写不刷新

- 时长解析:改取 <voicemsg voicelength>(毫秒),不再误取 length(SILK 字节数)
- 时长显示:四舍五入对齐微信口径,有原生时长时不被解码时长覆盖
- 转写:重新识别改为强制重算,跳过身份级缓存短路(音频级缓存仍生效)
- 日报:语音累计秒数显示取整,不再出现小数
This commit is contained in:
Wxw-Gu
2026-09-20 15:40:01 +08:00
parent 9b82d29037
commit 9593ca0f54
21 changed files with 435 additions and 52 deletions
+80 -7
View File
@@ -1,4 +1,4 @@
import { render, screen, waitFor } from '@testing-library/react'
import { act, render, screen, waitFor } from '@testing-library/react'
import userEvent from '@testing-library/user-event'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { VoicePlayer } from '../../src/renderer/src/components/VoicePlayer'
@@ -6,6 +6,8 @@ import { VoicePlayer } from '../../src/renderer/src/components/VoicePlayer'
const play = vi.fn(() => Promise.resolve())
const pause = vi.fn()
let lastAudio: FakeAudio | null = null
class FakeAudio {
preload = ''
src = ''
@@ -18,12 +20,27 @@ class FakeAudio {
pause = pause
load = vi.fn()
removeAttribute = vi.fn()
constructor() {
lastAudio = this
}
}
const renderPlayer = (duration?: number) =>
render(
<VoicePlayer
sessionId="filehelper"
localId={11}
createTime={1785553200}
duration={duration}
/>
)
describe('VoicePlayer', () => {
beforeEach(() => {
play.mockClear()
pause.mockClear()
lastAudio = null
vi.stubGlobal('Audio', FakeAudio)
window.api = {
getVoiceData: vi.fn().mockResolvedValue({
@@ -70,17 +87,73 @@ describe('VoicePlayer', () => {
render(<VoicePlayer sessionId="filehelper" localId={11} createTime={1785553200} duration={1} />)
await userEvent.click(screen.getByRole('button', { name: '转文字' }))
// 用户主动触发必须带 force:否则会被身份级缓存(不含 audio_hash)短路。
await waitFor(() =>
expect(window.api.recognizeVoice).toHaveBeenCalledWith({
sessionId: 'filehelper',
localId: 11,
createTime: 1785553200,
svrId: undefined
})
expect(window.api.recognizeVoice).toHaveBeenCalledWith(
{
sessionId: 'filehelper',
localId: 11,
createTime: 1785553200,
svrId: undefined
},
{ force: true }
)
)
expect(await screen.findByText('这是固定的测试转写')).toBeInTheDocument()
})
it('rounds the shown duration the way WeChat does', () => {
// 微信 4211ms 显示 4",所以口径是 round 不是 floor;进位要传给分钟位。
const cases: Array<[number | undefined, string]> = [
[1.7, '0:02'],
[4.211, '0:04'],
[7.505, '0:08'],
[59.6, '1:00'],
[119.6, '2:00'],
[3, '0:03'],
[7, '0:07'],
[15, '0:15'],
[0, '0:00'],
[undefined, '0:00'],
[NaN, '0:00']
]
for (const [duration, expected] of cases) {
const { container, unmount } = renderPlayer(duration)
expect(container.querySelector('.voice-duration')?.textContent).toBe(expected)
unmount()
}
})
it('keeps the WeChat duration instead of the systematically short decoded one', async () => {
const { container } = renderPlayer(1.979)
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:02')
await userEvent.click(container.querySelector('.voice-message') as HTMLElement)
await waitFor(() => expect(lastAudio).not.toBeNull())
// Silk 解码时长比微信 length 系统性偏短(实测少 20–279ms),用它覆盖会把精度弄丢。
lastAudio!.duration = 1
await act(async () => {
lastAudio!.onloadedmetadata?.()
lastAudio!.ontimeupdate?.()
})
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:02')
})
it('falls back to the decoded duration when WeChat did not provide one', async () => {
const { container } = renderPlayer()
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:00')
await userEvent.click(container.querySelector('.voice-message') as HTMLElement)
await waitFor(() => expect(lastAudio).not.toBeNull())
lastAudio!.duration = 59.6
lastAudio!.onloadedmetadata?.()
await waitFor(() =>
expect(container.querySelector('.voice-duration')?.textContent).toBe('1:00')
)
})
it('opens centralized settings when recognition assets are missing', async () => {
vi.mocked(window.api.getVoiceModelStatus).mockResolvedValue({
modelId: 'sensevoice-small-int8',
+35 -1
View File
@@ -4,7 +4,10 @@ import {
getSummaryDateRangeAt,
parseGroupDailyReport
} from '../../src/renderer/src/utils/group-report'
import { summaryContent } from '../../src/renderer/src/utils/group-report-facts'
import {
buildGroupReportFacts,
summaryContent
} from '../../src/renderer/src/utils/group-report-facts'
import { summarySender } from '../../src/renderer/src/utils/group-report-facts'
import { selectHeroParticipantNames } from '../../src/shared/group-report'
import type { GroupReportMetadata } from '../../src/shared/group-report'
@@ -62,6 +65,37 @@ describe('group report parsing', () => {
expect(summaryContent(message)).toContain('今晚八点确认发布。')
})
it('rounds accumulated voice seconds in the facts summary', async () => {
// 语音时长是 <voicemsg voicelength>(毫秒)换算来的小数秒(1600ms → 1.6)。
// 累加必须保留原始精度(否则多条累积会越差越多),但展示给用户的文案要取整。
const messages: Message[] = [
{
id: 'voice-a',
from: 'member',
type: '语音',
datetime: '2026-08-06 10:00:00',
content: '[语音]',
isSender: false,
contentData: { type: 'voice', duration: 1.979 }
},
{
id: 'voice-b',
from: 'member',
type: '语音',
datetime: '2026-08-06 10:00:05',
content: '[语音]',
isSender: false,
contentData: { type: 'voice', duration: 5.379 }
}
]
const snapshot = await buildGroupReportFacts(messages, null, true, 'compact')
// 1.979 + 5.379 = 7.358 → 文案必须显示整数
expect(snapshot.factsPrompt).toContain('累计 7 秒')
expect(snapshot.factsPrompt).not.toContain('7.358')
})
it('includes a voice transcript when legacy cached messages have no contentData', () => {
const message: Message = {
id: 'voice-legacy',
+32
View File
@@ -17,6 +17,38 @@ describe('message parser', () => {
).toMatchObject({ type: 'sticker', md5: 'abcdefabcdefabcdefabcdefabcdefab' })
})
it('reads the WeChat voice length in milliseconds and keeps fractional seconds', () => {
// 真机实测(2026-09-20):voicelength 才是毫秒时长,length 是编码数据长度——别取错。
// 属性值取自一条真机采样的语音(1.6 秒,微信气泡显示 2")。
const parsed = parseMessageContent(
'<msg><voicemsg endflag="1" cancelflag="0" forwardflag="0" voiceformat="4" voicelength="1600" length="6672" bufid="0" /></msg>',
34
)
expect(parsed).toEqual({ type: 'voice', duration: 1.6 })
// 误取 length 会得到 6.672 秒(把 2" 的语音显示成 0:07)——这条断言就是防这个回归。
expect(parsed).not.toEqual({ type: 'voice', duration: 6.672 })
// 微信四舍五入到整秒,取整必须在显示层做,不能在解析层丢精度。
expect(parseMessageContent('<msg><voicemsg voicelength="4211" /></msg>', 34)).toEqual({
type: 'voice',
duration: 4.211
})
})
it('leaves the voice duration undefined when the payload is missing or unusable', () => {
expect(parseMessageContent('', 34)).toEqual({ type: 'voice' })
expect(parseMessageContent('voice fixture', 34)).toEqual({ type: 'voice' })
expect(parseMessageContent('<msg><voicemsg voiceformat="4" /></msg>', 34)).toEqual({
type: 'voice'
})
expect(parseMessageContent('<msg><voicemsg voicelength="0" /></msg>', 34)).toEqual({
type: 'voice'
})
expect(parseMessageContent('<msg><voicemsg voicelength="abc" /></msg>', 34)).toEqual({
type: 'voice'
})
})
it('keeps video metadata when WeChat omits every MD5 field', () => {
const parsed = parseMessageContent(
'<msg><videomsg length="6402169" playlength="30" cdnthumbwidth="224" cdnthumbheight="398" aeskey="25201cc658042689d1ad6747cea2b240" rawmd5="" /></msg>',
+125
View File
@@ -361,3 +361,128 @@ describe('voice pipeline cache lookup', () => {
repository.close()
})
})
/*
* force 的语义:用户主动触发时必须绕开 findCompatible(它只按消息身份匹配,
* 不含 audio_hash),重新取一次音频;音频级 find(key) 带 audio_hash,保留。
*/
describe('voice pipeline force recognition', () => {
const forceRoot = mkdtempSync(join(tmpdir(), 'wxe-voice-pipeline-force-'))
afterAll(() => rmSync(forceRoot, { recursive: true, force: true }))
function createPipeline(repository: SqliteTranscriptRepository, transcript: string) {
const resolve = vi.fn().mockResolvedValue({
data: Buffer.from('encoded'),
codec: 'silk',
sourceHash: 'fresh-audio'
})
const decode = vi.fn().mockResolvedValue({
pcm: Buffer.from([1, 0]),
sampleRate: 16000,
channels: 1,
sourceHash: 'fresh-audio'
})
const process = vi.fn().mockReturnValue({
samples: new Float32Array([0.1]),
sampleRate: 16000,
channels: 1,
sourceHash: 'fresh-audio',
processorVersion: 'processor-v1',
durationMs: 1
})
const recognize = vi.fn().mockResolvedValue({ text: transcript })
const pipeline = new VoicePipeline(
{ resolve },
{ decode } as never,
{ version: 'processor-v1', process },
{
metadata: {
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a'
},
recognize,
dispose: vi.fn()
},
repository
)
return { pipeline, resolve, decode, process, recognize }
}
function seedStaleTranscript(repository: SqliteTranscriptRepository) {
repository.save({
accountId: 'account-a',
messageIdentity: voiceMessageIdentity(reference),
audioHash: 'stale-audio',
processorVersion: 'processor-v1',
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a',
transcript: '旧音频的转写',
durationMs: 900,
createdAt: 1,
updatedAt: 1
})
}
const reference = { sessionId: 'session', localId: 1, createTime: 2 }
it('skips the identity-level cache when force is requested', async () => {
const repository = new SqliteTranscriptRepository(join(forceRoot, 'forced.sqlite'))
seedStaleTranscript(repository)
const findCompatible = vi.spyOn(repository, 'findCompatible')
const find = vi.spyOn(repository, 'find')
const { pipeline, resolve, recognize } = createPipeline(repository, '正确音频的转写')
await expect(
pipeline.run('account-a', reference, undefined, { force: true })
).resolves.toMatchObject({ transcript: '正确音频的转写', cached: false })
expect(findCompatible).not.toHaveBeenCalled()
expect(resolve).toHaveBeenCalledOnce()
expect(find).toHaveBeenCalled()
expect(recognize).toHaveBeenCalledOnce()
expect(
repository.find({
accountId: 'account-a',
messageIdentity: voiceMessageIdentity(reference),
audioHash: 'fresh-audio',
processorVersion: 'processor-v1',
recognizerId: 'sensevoice',
modelVersion: 'model-v1',
modelFingerprint: 'fingerprint-a'
})?.transcript
).toBe('正确音频的转写')
repository.close()
})
it('keeps consulting the identity-level cache without force', async () => {
const repository = new SqliteTranscriptRepository(join(forceRoot, 'unchanged.sqlite'))
seedStaleTranscript(repository)
const findCompatible = vi.spyOn(repository, 'findCompatible')
const { pipeline, resolve, decode, process, recognize } = createPipeline(
repository,
'新转写'
)
await expect(pipeline.run('account-a', reference)).resolves.toMatchObject({
transcript: '旧音频的转写',
cached: true
})
expect(findCompatible).toHaveBeenCalledOnce()
expect(resolve).not.toHaveBeenCalled()
expect(decode).not.toHaveBeenCalled()
expect(process).not.toHaveBeenCalled()
expect(recognize).not.toHaveBeenCalled()
repository.close()
const explicitFalse = new SqliteTranscriptRepository(join(forceRoot, 'explicit-false.sqlite'))
seedStaleTranscript(explicitFalse)
const second = createPipeline(explicitFalse, '新转写')
await expect(
second.pipeline.run('account-a', reference, undefined, { force: false })
).resolves.toMatchObject({ transcript: '旧音频的转写', cached: true })
expect(second.resolve).not.toHaveBeenCalled()
explicitFalse.close()
})
})
@@ -101,6 +101,49 @@ describe('VoiceRecognitionUseCase transcript updates', () => {
await useCase.dispose()
})
it('forwards force to the pipeline so manual re-recognition can bypass the identity cache', async () => {
const useCase = createUseCase()
const state = useCase as unknown as {
pipeline: { run: ReturnType<typeof vi.fn> }
}
state.pipeline.run.mockResolvedValue({
transcript: '重新识别的文字',
durationMs: 900,
cached: false
})
const reference = { sessionId: 'fixture-contact', localId: 12, createTime: 1_785_895_203 }
const result = await useCase.recognize(reference, { force: true })
expect(result).toMatchObject({ success: true, transcript: '重新识别的文字' })
expect(state.pipeline.run).toHaveBeenCalledWith(
'account-a',
reference,
expect.anything(),
{ force: true }
)
await useCase.dispose()
})
it('leaves force undefined when the caller does not ask for a recompute', async () => {
const useCase = createUseCase()
const state = useCase as unknown as {
pipeline: { run: ReturnType<typeof vi.fn> }
}
state.pipeline.run.mockResolvedValue({ transcript: '缓存文字', durationMs: 900, cached: true })
const reference = { sessionId: 'fixture-contact', localId: 13, createTime: 1_785_895_204 }
await useCase.recognize(reference)
expect(state.pipeline.run).toHaveBeenCalledWith(
'account-a',
reference,
expect.anything(),
{ force: undefined }
)
await useCase.dispose()
})
it('publishes an explicit cached transcript for a coalesced export index refresh', async () => {
const useCase = createUseCase()
const listener = vi.fn().mockResolvedValue(undefined)