mirror of
https://wget.la/https://github.com/Wxw-Gu/WechatExplorer
synced 2026-10-03 18:33:14 +08:00
fix: 修复语音消息取错、时长精度与转写不刷新
- 时长解析:改取 <voicemsg voicelength>(毫秒),不再误取 length(SILK 字节数) - 时长显示:四舍五入对齐微信口径,有原生时长时不被解码时长覆盖 - 转写:重新识别改为强制重算,跳过身份级缓存短路(音频级缓存仍生效) - 日报:语音累计秒数显示取整,不再出现小数
This commit is contained in:
@@ -1,4 +1,4 @@
|
||||
import { render, screen, waitFor } from '@testing-library/react'
|
||||
import { act, render, screen, waitFor } from '@testing-library/react'
|
||||
import userEvent from '@testing-library/user-event'
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { VoicePlayer } from '../../src/renderer/src/components/VoicePlayer'
|
||||
@@ -6,6 +6,8 @@ import { VoicePlayer } from '../../src/renderer/src/components/VoicePlayer'
|
||||
const play = vi.fn(() => Promise.resolve())
|
||||
const pause = vi.fn()
|
||||
|
||||
let lastAudio: FakeAudio | null = null
|
||||
|
||||
class FakeAudio {
|
||||
preload = ''
|
||||
src = ''
|
||||
@@ -18,12 +20,27 @@ class FakeAudio {
|
||||
pause = pause
|
||||
load = vi.fn()
|
||||
removeAttribute = vi.fn()
|
||||
|
||||
constructor() {
|
||||
lastAudio = this
|
||||
}
|
||||
}
|
||||
|
||||
const renderPlayer = (duration?: number) =>
|
||||
render(
|
||||
<VoicePlayer
|
||||
sessionId="filehelper"
|
||||
localId={11}
|
||||
createTime={1785553200}
|
||||
duration={duration}
|
||||
/>
|
||||
)
|
||||
|
||||
describe('VoicePlayer', () => {
|
||||
beforeEach(() => {
|
||||
play.mockClear()
|
||||
pause.mockClear()
|
||||
lastAudio = null
|
||||
vi.stubGlobal('Audio', FakeAudio)
|
||||
window.api = {
|
||||
getVoiceData: vi.fn().mockResolvedValue({
|
||||
@@ -70,17 +87,73 @@ describe('VoicePlayer', () => {
|
||||
render(<VoicePlayer sessionId="filehelper" localId={11} createTime={1785553200} duration={1} />)
|
||||
await userEvent.click(screen.getByRole('button', { name: '转文字' }))
|
||||
|
||||
// 用户主动触发必须带 force:否则会被身份级缓存(不含 audio_hash)短路。
|
||||
await waitFor(() =>
|
||||
expect(window.api.recognizeVoice).toHaveBeenCalledWith({
|
||||
sessionId: 'filehelper',
|
||||
localId: 11,
|
||||
createTime: 1785553200,
|
||||
svrId: undefined
|
||||
})
|
||||
expect(window.api.recognizeVoice).toHaveBeenCalledWith(
|
||||
{
|
||||
sessionId: 'filehelper',
|
||||
localId: 11,
|
||||
createTime: 1785553200,
|
||||
svrId: undefined
|
||||
},
|
||||
{ force: true }
|
||||
)
|
||||
)
|
||||
expect(await screen.findByText('这是固定的测试转写')).toBeInTheDocument()
|
||||
})
|
||||
|
||||
it('rounds the shown duration the way WeChat does', () => {
|
||||
// 微信 4211ms 显示 4",所以口径是 round 不是 floor;进位要传给分钟位。
|
||||
const cases: Array<[number | undefined, string]> = [
|
||||
[1.7, '0:02'],
|
||||
[4.211, '0:04'],
|
||||
[7.505, '0:08'],
|
||||
[59.6, '1:00'],
|
||||
[119.6, '2:00'],
|
||||
[3, '0:03'],
|
||||
[7, '0:07'],
|
||||
[15, '0:15'],
|
||||
[0, '0:00'],
|
||||
[undefined, '0:00'],
|
||||
[NaN, '0:00']
|
||||
]
|
||||
for (const [duration, expected] of cases) {
|
||||
const { container, unmount } = renderPlayer(duration)
|
||||
expect(container.querySelector('.voice-duration')?.textContent).toBe(expected)
|
||||
unmount()
|
||||
}
|
||||
})
|
||||
|
||||
it('keeps the WeChat duration instead of the systematically short decoded one', async () => {
|
||||
const { container } = renderPlayer(1.979)
|
||||
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:02')
|
||||
|
||||
await userEvent.click(container.querySelector('.voice-message') as HTMLElement)
|
||||
await waitFor(() => expect(lastAudio).not.toBeNull())
|
||||
// Silk 解码时长比微信 length 系统性偏短(实测少 20–279ms),用它覆盖会把精度弄丢。
|
||||
lastAudio!.duration = 1
|
||||
await act(async () => {
|
||||
lastAudio!.onloadedmetadata?.()
|
||||
lastAudio!.ontimeupdate?.()
|
||||
})
|
||||
|
||||
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:02')
|
||||
})
|
||||
|
||||
it('falls back to the decoded duration when WeChat did not provide one', async () => {
|
||||
const { container } = renderPlayer()
|
||||
expect(container.querySelector('.voice-duration')?.textContent).toBe('0:00')
|
||||
|
||||
await userEvent.click(container.querySelector('.voice-message') as HTMLElement)
|
||||
await waitFor(() => expect(lastAudio).not.toBeNull())
|
||||
lastAudio!.duration = 59.6
|
||||
lastAudio!.onloadedmetadata?.()
|
||||
|
||||
await waitFor(() =>
|
||||
expect(container.querySelector('.voice-duration')?.textContent).toBe('1:00')
|
||||
)
|
||||
})
|
||||
|
||||
it('opens centralized settings when recognition assets are missing', async () => {
|
||||
vi.mocked(window.api.getVoiceModelStatus).mockResolvedValue({
|
||||
modelId: 'sensevoice-small-int8',
|
||||
|
||||
@@ -4,7 +4,10 @@ import {
|
||||
getSummaryDateRangeAt,
|
||||
parseGroupDailyReport
|
||||
} from '../../src/renderer/src/utils/group-report'
|
||||
import { summaryContent } from '../../src/renderer/src/utils/group-report-facts'
|
||||
import {
|
||||
buildGroupReportFacts,
|
||||
summaryContent
|
||||
} from '../../src/renderer/src/utils/group-report-facts'
|
||||
import { summarySender } from '../../src/renderer/src/utils/group-report-facts'
|
||||
import { selectHeroParticipantNames } from '../../src/shared/group-report'
|
||||
import type { GroupReportMetadata } from '../../src/shared/group-report'
|
||||
@@ -62,6 +65,37 @@ describe('group report parsing', () => {
|
||||
expect(summaryContent(message)).toContain('今晚八点确认发布。')
|
||||
})
|
||||
|
||||
it('rounds accumulated voice seconds in the facts summary', async () => {
|
||||
// 语音时长是 <voicemsg voicelength>(毫秒)换算来的小数秒(1600ms → 1.6)。
|
||||
// 累加必须保留原始精度(否则多条累积会越差越多),但展示给用户的文案要取整。
|
||||
const messages: Message[] = [
|
||||
{
|
||||
id: 'voice-a',
|
||||
from: 'member',
|
||||
type: '语音',
|
||||
datetime: '2026-08-06 10:00:00',
|
||||
content: '[语音]',
|
||||
isSender: false,
|
||||
contentData: { type: 'voice', duration: 1.979 }
|
||||
},
|
||||
{
|
||||
id: 'voice-b',
|
||||
from: 'member',
|
||||
type: '语音',
|
||||
datetime: '2026-08-06 10:00:05',
|
||||
content: '[语音]',
|
||||
isSender: false,
|
||||
contentData: { type: 'voice', duration: 5.379 }
|
||||
}
|
||||
]
|
||||
|
||||
const snapshot = await buildGroupReportFacts(messages, null, true, 'compact')
|
||||
|
||||
// 1.979 + 5.379 = 7.358 → 文案必须显示整数
|
||||
expect(snapshot.factsPrompt).toContain('累计 7 秒')
|
||||
expect(snapshot.factsPrompt).not.toContain('7.358')
|
||||
})
|
||||
|
||||
it('includes a voice transcript when legacy cached messages have no contentData', () => {
|
||||
const message: Message = {
|
||||
id: 'voice-legacy',
|
||||
|
||||
@@ -17,6 +17,38 @@ describe('message parser', () => {
|
||||
).toMatchObject({ type: 'sticker', md5: 'abcdefabcdefabcdefabcdefabcdefab' })
|
||||
})
|
||||
|
||||
it('reads the WeChat voice length in milliseconds and keeps fractional seconds', () => {
|
||||
// 真机实测(2026-09-20):voicelength 才是毫秒时长,length 是编码数据长度——别取错。
|
||||
// 属性值取自一条真机采样的语音(1.6 秒,微信气泡显示 2")。
|
||||
const parsed = parseMessageContent(
|
||||
'<msg><voicemsg endflag="1" cancelflag="0" forwardflag="0" voiceformat="4" voicelength="1600" length="6672" bufid="0" /></msg>',
|
||||
34
|
||||
)
|
||||
expect(parsed).toEqual({ type: 'voice', duration: 1.6 })
|
||||
// 误取 length 会得到 6.672 秒(把 2" 的语音显示成 0:07)——这条断言就是防这个回归。
|
||||
expect(parsed).not.toEqual({ type: 'voice', duration: 6.672 })
|
||||
|
||||
// 微信四舍五入到整秒,取整必须在显示层做,不能在解析层丢精度。
|
||||
expect(parseMessageContent('<msg><voicemsg voicelength="4211" /></msg>', 34)).toEqual({
|
||||
type: 'voice',
|
||||
duration: 4.211
|
||||
})
|
||||
})
|
||||
|
||||
it('leaves the voice duration undefined when the payload is missing or unusable', () => {
|
||||
expect(parseMessageContent('', 34)).toEqual({ type: 'voice' })
|
||||
expect(parseMessageContent('voice fixture', 34)).toEqual({ type: 'voice' })
|
||||
expect(parseMessageContent('<msg><voicemsg voiceformat="4" /></msg>', 34)).toEqual({
|
||||
type: 'voice'
|
||||
})
|
||||
expect(parseMessageContent('<msg><voicemsg voicelength="0" /></msg>', 34)).toEqual({
|
||||
type: 'voice'
|
||||
})
|
||||
expect(parseMessageContent('<msg><voicemsg voicelength="abc" /></msg>', 34)).toEqual({
|
||||
type: 'voice'
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps video metadata when WeChat omits every MD5 field', () => {
|
||||
const parsed = parseMessageContent(
|
||||
'<msg><videomsg length="6402169" playlength="30" cdnthumbwidth="224" cdnthumbheight="398" aeskey="25201cc658042689d1ad6747cea2b240" rawmd5="" /></msg>',
|
||||
|
||||
@@ -361,3 +361,128 @@ describe('voice pipeline cache lookup', () => {
|
||||
repository.close()
|
||||
})
|
||||
})
|
||||
|
||||
/*
|
||||
* force 的语义:用户主动触发时必须绕开 findCompatible(它只按消息身份匹配,
|
||||
* 不含 audio_hash),重新取一次音频;音频级 find(key) 带 audio_hash,保留。
|
||||
*/
|
||||
describe('voice pipeline force recognition', () => {
|
||||
const forceRoot = mkdtempSync(join(tmpdir(), 'wxe-voice-pipeline-force-'))
|
||||
afterAll(() => rmSync(forceRoot, { recursive: true, force: true }))
|
||||
|
||||
function createPipeline(repository: SqliteTranscriptRepository, transcript: string) {
|
||||
const resolve = vi.fn().mockResolvedValue({
|
||||
data: Buffer.from('encoded'),
|
||||
codec: 'silk',
|
||||
sourceHash: 'fresh-audio'
|
||||
})
|
||||
const decode = vi.fn().mockResolvedValue({
|
||||
pcm: Buffer.from([1, 0]),
|
||||
sampleRate: 16000,
|
||||
channels: 1,
|
||||
sourceHash: 'fresh-audio'
|
||||
})
|
||||
const process = vi.fn().mockReturnValue({
|
||||
samples: new Float32Array([0.1]),
|
||||
sampleRate: 16000,
|
||||
channels: 1,
|
||||
sourceHash: 'fresh-audio',
|
||||
processorVersion: 'processor-v1',
|
||||
durationMs: 1
|
||||
})
|
||||
const recognize = vi.fn().mockResolvedValue({ text: transcript })
|
||||
const pipeline = new VoicePipeline(
|
||||
{ resolve },
|
||||
{ decode } as never,
|
||||
{ version: 'processor-v1', process },
|
||||
{
|
||||
metadata: {
|
||||
recognizerId: 'sensevoice',
|
||||
modelVersion: 'model-v1',
|
||||
modelFingerprint: 'fingerprint-a'
|
||||
},
|
||||
recognize,
|
||||
dispose: vi.fn()
|
||||
},
|
||||
repository
|
||||
)
|
||||
return { pipeline, resolve, decode, process, recognize }
|
||||
}
|
||||
|
||||
function seedStaleTranscript(repository: SqliteTranscriptRepository) {
|
||||
repository.save({
|
||||
accountId: 'account-a',
|
||||
messageIdentity: voiceMessageIdentity(reference),
|
||||
audioHash: 'stale-audio',
|
||||
processorVersion: 'processor-v1',
|
||||
recognizerId: 'sensevoice',
|
||||
modelVersion: 'model-v1',
|
||||
modelFingerprint: 'fingerprint-a',
|
||||
transcript: '旧音频的转写',
|
||||
durationMs: 900,
|
||||
createdAt: 1,
|
||||
updatedAt: 1
|
||||
})
|
||||
}
|
||||
|
||||
const reference = { sessionId: 'session', localId: 1, createTime: 2 }
|
||||
|
||||
it('skips the identity-level cache when force is requested', async () => {
|
||||
const repository = new SqliteTranscriptRepository(join(forceRoot, 'forced.sqlite'))
|
||||
seedStaleTranscript(repository)
|
||||
const findCompatible = vi.spyOn(repository, 'findCompatible')
|
||||
const find = vi.spyOn(repository, 'find')
|
||||
const { pipeline, resolve, recognize } = createPipeline(repository, '正确音频的转写')
|
||||
|
||||
await expect(
|
||||
pipeline.run('account-a', reference, undefined, { force: true })
|
||||
).resolves.toMatchObject({ transcript: '正确音频的转写', cached: false })
|
||||
|
||||
expect(findCompatible).not.toHaveBeenCalled()
|
||||
expect(resolve).toHaveBeenCalledOnce()
|
||||
expect(find).toHaveBeenCalled()
|
||||
expect(recognize).toHaveBeenCalledOnce()
|
||||
expect(
|
||||
repository.find({
|
||||
accountId: 'account-a',
|
||||
messageIdentity: voiceMessageIdentity(reference),
|
||||
audioHash: 'fresh-audio',
|
||||
processorVersion: 'processor-v1',
|
||||
recognizerId: 'sensevoice',
|
||||
modelVersion: 'model-v1',
|
||||
modelFingerprint: 'fingerprint-a'
|
||||
})?.transcript
|
||||
).toBe('正确音频的转写')
|
||||
repository.close()
|
||||
})
|
||||
|
||||
it('keeps consulting the identity-level cache without force', async () => {
|
||||
const repository = new SqliteTranscriptRepository(join(forceRoot, 'unchanged.sqlite'))
|
||||
seedStaleTranscript(repository)
|
||||
const findCompatible = vi.spyOn(repository, 'findCompatible')
|
||||
const { pipeline, resolve, decode, process, recognize } = createPipeline(
|
||||
repository,
|
||||
'新转写'
|
||||
)
|
||||
|
||||
await expect(pipeline.run('account-a', reference)).resolves.toMatchObject({
|
||||
transcript: '旧音频的转写',
|
||||
cached: true
|
||||
})
|
||||
expect(findCompatible).toHaveBeenCalledOnce()
|
||||
expect(resolve).not.toHaveBeenCalled()
|
||||
expect(decode).not.toHaveBeenCalled()
|
||||
expect(process).not.toHaveBeenCalled()
|
||||
expect(recognize).not.toHaveBeenCalled()
|
||||
repository.close()
|
||||
|
||||
const explicitFalse = new SqliteTranscriptRepository(join(forceRoot, 'explicit-false.sqlite'))
|
||||
seedStaleTranscript(explicitFalse)
|
||||
const second = createPipeline(explicitFalse, '新转写')
|
||||
await expect(
|
||||
second.pipeline.run('account-a', reference, undefined, { force: false })
|
||||
).resolves.toMatchObject({ transcript: '旧音频的转写', cached: true })
|
||||
expect(second.resolve).not.toHaveBeenCalled()
|
||||
explicitFalse.close()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -101,6 +101,49 @@ describe('VoiceRecognitionUseCase transcript updates', () => {
|
||||
await useCase.dispose()
|
||||
})
|
||||
|
||||
it('forwards force to the pipeline so manual re-recognition can bypass the identity cache', async () => {
|
||||
const useCase = createUseCase()
|
||||
const state = useCase as unknown as {
|
||||
pipeline: { run: ReturnType<typeof vi.fn> }
|
||||
}
|
||||
state.pipeline.run.mockResolvedValue({
|
||||
transcript: '重新识别的文字',
|
||||
durationMs: 900,
|
||||
cached: false
|
||||
})
|
||||
const reference = { sessionId: 'fixture-contact', localId: 12, createTime: 1_785_895_203 }
|
||||
|
||||
const result = await useCase.recognize(reference, { force: true })
|
||||
|
||||
expect(result).toMatchObject({ success: true, transcript: '重新识别的文字' })
|
||||
expect(state.pipeline.run).toHaveBeenCalledWith(
|
||||
'account-a',
|
||||
reference,
|
||||
expect.anything(),
|
||||
{ force: true }
|
||||
)
|
||||
await useCase.dispose()
|
||||
})
|
||||
|
||||
it('leaves force undefined when the caller does not ask for a recompute', async () => {
|
||||
const useCase = createUseCase()
|
||||
const state = useCase as unknown as {
|
||||
pipeline: { run: ReturnType<typeof vi.fn> }
|
||||
}
|
||||
state.pipeline.run.mockResolvedValue({ transcript: '缓存文字', durationMs: 900, cached: true })
|
||||
const reference = { sessionId: 'fixture-contact', localId: 13, createTime: 1_785_895_204 }
|
||||
|
||||
await useCase.recognize(reference)
|
||||
|
||||
expect(state.pipeline.run).toHaveBeenCalledWith(
|
||||
'account-a',
|
||||
reference,
|
||||
expect.anything(),
|
||||
{ force: undefined }
|
||||
)
|
||||
await useCase.dispose()
|
||||
})
|
||||
|
||||
it('publishes an explicit cached transcript for a coalesced export index refresh', async () => {
|
||||
const useCase = createUseCase()
|
||||
const listener = vi.fn().mockResolvedValue(undefined)
|
||||
|
||||
Reference in New Issue
Block a user