mirror of
https://wget.la/https://github.com/Wxw-Gu/WechatExplorer
synced 2026-10-03 18:33:14 +08:00
fix: 日报头像补齐下载重试
This commit is contained in:
@@ -107,47 +107,135 @@ const fallbackAvatar = (name: string): RenderedAvatar => {
|
||||
}
|
||||
}
|
||||
|
||||
const imageMimeType = (contentType: string | null, source: string): string => {
|
||||
if (contentType?.startsWith('image/')) return contentType.split(';')[0]
|
||||
const extension = path.extname(source).toLowerCase()
|
||||
if (extension === '.png') return 'image/png'
|
||||
if (extension === '.webp') return 'image/webp'
|
||||
if (extension === '.gif') return 'image/gif'
|
||||
return 'image/jpeg'
|
||||
/**
|
||||
* 只认真实图片字节。
|
||||
*
|
||||
* `content-type` 与 URL 扩展名都**不可信**:微信 CDN 在限流 / 反盗链时会返回 200 + HTML 正文。
|
||||
* 旧实现按扩展名猜 mime 并默认 `image/jpeg`,会把 HTML 内联成"解码失败的 data URL" ——
|
||||
* 在报告里表现为**空白头像**(比首字占位更糟:用户看不到任何东西,也不知道为什么)。
|
||||
*/
|
||||
const IMAGE_MAGIC: Array<{ mime: string; bytes: number[] }> = [
|
||||
{ mime: 'image/jpeg', bytes: [0xff, 0xd8, 0xff] },
|
||||
{ mime: 'image/png', bytes: [0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a] },
|
||||
{ mime: 'image/gif', bytes: [0x47, 0x49, 0x46, 0x38] },
|
||||
{ mime: 'image/bmp', bytes: [0x42, 0x4d] }
|
||||
]
|
||||
|
||||
export const detectImageMime = (bytes: Buffer): string | undefined => {
|
||||
for (const signature of IMAGE_MAGIC) {
|
||||
if (
|
||||
bytes.length >= signature.bytes.length &&
|
||||
signature.bytes.every((byte, index) => bytes[index] === byte)
|
||||
) {
|
||||
return signature.mime
|
||||
}
|
||||
}
|
||||
if (
|
||||
bytes.length >= 12 &&
|
||||
bytes.toString('ascii', 0, 4) === 'RIFF' &&
|
||||
bytes.toString('ascii', 8, 12) === 'WEBP'
|
||||
) {
|
||||
return 'image/webp'
|
||||
}
|
||||
const head = bytes.subarray(0, 64).toString('utf8').trimStart()
|
||||
if (head.startsWith('<svg')) return 'image/svg+xml'
|
||||
if (head.startsWith('<?xml') && head.includes('<svg')) return 'image/svg+xml'
|
||||
return undefined
|
||||
}
|
||||
|
||||
const embedAvatar = async (source: string | undefined, name: string): Promise<RenderedAvatar> => {
|
||||
const AVATAR_FETCH_TIMEOUT_MS = 8000
|
||||
/** 首次 + 一次重试:单次瞬时失败(限流 / 连接重置 / 超时)不该让一个人永久退回首字。 */
|
||||
const AVATAR_FETCH_ATTEMPTS = 2
|
||||
/** 同一 origin 的并发上限。几十个头像同时打一个 CDN 会显著抬高被限流的概率。 */
|
||||
const AVATAR_FETCH_CONCURRENCY = 6
|
||||
/** 进程内头像缓存条目上限(老报告重渲染 / 连续生成同一群时不必重复下载)。 */
|
||||
const AVATAR_CACHE_LIMIT = 256
|
||||
|
||||
const sleep = (ms: number): Promise<void> => new Promise((resolve) => setTimeout(resolve, ms))
|
||||
|
||||
const avatarEmbedCache = new Map<string, RenderedAvatar>()
|
||||
|
||||
const rememberAvatarEmbed = (source: string, rendered: RenderedAvatar): void => {
|
||||
if (rendered.fallback) return
|
||||
avatarEmbedCache.set(source, rendered)
|
||||
while (avatarEmbedCache.size > AVATAR_CACHE_LIMIT) {
|
||||
const oldest = avatarEmbedCache.keys().next().value
|
||||
if (oldest === undefined) break
|
||||
avatarEmbedCache.delete(oldest)
|
||||
}
|
||||
}
|
||||
|
||||
/** 有界并发:把 N 个任务压到 limit 个同时在飞,结果顺序与输入一致。 */
|
||||
export const mapWithConcurrency = async <T, R>(
|
||||
items: readonly T[],
|
||||
limit: number,
|
||||
task: (item: T) => Promise<R>
|
||||
): Promise<R[]> => {
|
||||
const results: R[] = new Array(items.length)
|
||||
let cursor = 0
|
||||
const workers = Array.from({ length: Math.max(1, Math.min(limit, items.length)) }, async () => {
|
||||
for (;;) {
|
||||
const index = cursor
|
||||
cursor += 1
|
||||
if (index >= items.length) return
|
||||
results[index] = await task(items[index])
|
||||
}
|
||||
})
|
||||
await Promise.all(workers)
|
||||
return results
|
||||
}
|
||||
|
||||
const readAvatarSource = async (source: string): Promise<RenderedAvatar> => {
|
||||
if (/^https?:\/\//i.test(source)) {
|
||||
const response = await fetch(source, {
|
||||
headers: {
|
||||
'User-Agent': 'Mozilla/5.0 TraceMemo',
|
||||
Referer: 'https://weixin.qq.com/'
|
||||
},
|
||||
signal: AbortSignal.timeout(AVATAR_FETCH_TIMEOUT_MS)
|
||||
})
|
||||
if (!response.ok) throw new Error(`HTTP ${response.status}`)
|
||||
const bytes = Buffer.from(await response.arrayBuffer())
|
||||
const mime = detectImageMime(bytes)
|
||||
if (!mime) {
|
||||
throw new Error(
|
||||
`not an image (content-type=${response.headers.get('content-type') || 'unknown'}, ${bytes.length} bytes)`
|
||||
)
|
||||
}
|
||||
return { source: `data:${mime};base64,${bytes.toString('base64')}`, fallback: false }
|
||||
}
|
||||
|
||||
const localPath = source.startsWith('file://') ? new URL(source) : source
|
||||
const bytes = await fs.readFile(localPath)
|
||||
const mime = detectImageMime(bytes)
|
||||
if (!mime) throw new Error(`not an image (${bytes.length} bytes)`)
|
||||
return { source: `data:${mime};base64,${bytes.toString('base64')}`, fallback: false }
|
||||
}
|
||||
|
||||
export const embedAvatar = async (
|
||||
source: string | undefined,
|
||||
name: string
|
||||
): Promise<RenderedAvatar> => {
|
||||
if (!source) return fallbackAvatar(name)
|
||||
if (/^data:image\/[a-z0-9.+/-]+;base64,[a-z0-9+/=]+$/i.test(source))
|
||||
return { source, fallback: false }
|
||||
|
||||
try {
|
||||
if (/^https?:\/\//i.test(source)) {
|
||||
const response = await fetch(source, {
|
||||
headers: {
|
||||
'User-Agent': 'Mozilla/5.0 TraceMemo',
|
||||
Referer: 'https://weixin.qq.com/'
|
||||
},
|
||||
signal: AbortSignal.timeout(8000)
|
||||
})
|
||||
if (!response.ok) throw new Error(`HTTP ${response.status}`)
|
||||
const mime = imageMimeType(response.headers.get('content-type'), source)
|
||||
return {
|
||||
source: `data:${mime};base64,${Buffer.from(await response.arrayBuffer()).toString('base64')}`,
|
||||
fallback: false
|
||||
}
|
||||
}
|
||||
const cached = avatarEmbedCache.get(source)
|
||||
if (cached) return cached
|
||||
|
||||
const localPath = source.startsWith('file://') ? new URL(source) : source
|
||||
const buffer = await fs.readFile(localPath)
|
||||
return {
|
||||
source: `data:${imageMimeType(null, source)};base64,${buffer.toString('base64')}`,
|
||||
fallback: false
|
||||
let lastError: unknown
|
||||
for (let attempt = 1; attempt <= AVATAR_FETCH_ATTEMPTS; attempt += 1) {
|
||||
try {
|
||||
const embedded = await readAvatarSource(source)
|
||||
rememberAvatarEmbed(source, embedded)
|
||||
return embedded
|
||||
} catch (error) {
|
||||
lastError = error
|
||||
if (attempt < AVATAR_FETCH_ATTEMPTS) await sleep(150 * attempt)
|
||||
}
|
||||
} catch (error) {
|
||||
console.warn(`[GroupReport] avatar fallback for ${name}:`, error)
|
||||
return fallbackAvatar(name)
|
||||
}
|
||||
console.warn(`[GroupReport] avatar fallback for ${name}:`, lastError)
|
||||
return fallbackAvatar(name)
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -288,12 +376,22 @@ const renderReportHtml = async (request: GroupReportExportRequest): Promise<stri
|
||||
report.media?.voiceHighlights?.forEach((item) => avatarNames.add(item.sender))
|
||||
report.media?.funBadges?.forEach((item) => avatarNames.add(item.owner))
|
||||
|
||||
const avatars = new Map<string, RenderedAvatar>()
|
||||
await Promise.all(
|
||||
Array.from(avatarNames).map(async (name) => {
|
||||
avatars.set(name, await embedAvatar(metadata.avatars[name], name))
|
||||
})
|
||||
// 有界并发 + 单条重试:几十个头像同时打同一个 CDN 会被限流,瞬时失败会让一个人
|
||||
// 在整份报告里永久退化成首字占位(实测同一天三次生成:0% / 0% / 32% 失败)。
|
||||
const renderedAvatars = await mapWithConcurrency(
|
||||
Array.from(avatarNames),
|
||||
AVATAR_FETCH_CONCURRENCY,
|
||||
async (name) => [name, await embedAvatar(metadata.avatars[name], name)] as const
|
||||
)
|
||||
const avatars = new Map(renderedAvatars)
|
||||
const fallbackCount = renderedAvatars.filter(([, item]) => item.fallback).length
|
||||
if (fallbackCount > 0) {
|
||||
// 只记数量,不记人名:报告本身已经有名字,这里只需要一个可诊断的信号。
|
||||
metadata.warnings = metadata.warnings ?? []
|
||||
metadata.warnings.push(
|
||||
`avatar fallback ${fallbackCount}/${renderedAvatars.length}: 未取到真实头像,已用首字占位`
|
||||
)
|
||||
}
|
||||
const avatar = (name: string): RenderedAvatar => avatars.get(name) || fallbackAvatar(name)
|
||||
const renderAvatar = (
|
||||
name: string,
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
|
||||
vi.mock('electron', () => ({ app: { getPath: () => '/tmp' } }))
|
||||
|
||||
import {
|
||||
detectImageMime,
|
||||
embedAvatar,
|
||||
mapWithConcurrency
|
||||
} from '../../src/main/group-report-service'
|
||||
|
||||
/**
|
||||
* 「日报有的头像没生成出来」的回归。
|
||||
*
|
||||
* 实测根因(同一天三次生成同一份报告):真实头像失败率 0% / 0% / 32%,
|
||||
* 且失败的那 4 个人稍后重新请求全部 200 + 合法 JPEG —— 说明是**瞬时**网络失败。
|
||||
* 旧实现对每个头像只发一次请求、失败即永久退回首字占位,于是少数人整份报告都是占位图。
|
||||
*
|
||||
* 这里锁死三件事:
|
||||
* 1. 只信真实图片字节(200 + HTML 反盗链页不得内联成"解码失败"的空白头像);
|
||||
* 2. 瞬时失败要重试一次;
|
||||
* 3. 成功的头像要进缓存(老报告重渲染 / 连续生成不再重复下载)。
|
||||
*/
|
||||
|
||||
const JPEG = Buffer.concat([Buffer.from([0xff, 0xd8, 0xff, 0xe0]), Buffer.alloc(64, 7)])
|
||||
const PNG = Buffer.concat([
|
||||
Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]),
|
||||
Buffer.alloc(32, 3)
|
||||
])
|
||||
const HTML = Buffer.from('<!DOCTYPE html><html><body>anti hotlink</body></html>')
|
||||
|
||||
const imageResponse = (bytes: Buffer, contentType = 'image/jpeg'): Response =>
|
||||
new Response(new Uint8Array(bytes), {
|
||||
status: 200,
|
||||
headers: { 'content-type': contentType }
|
||||
})
|
||||
|
||||
describe('detectImageMime — 只认真实图片字节', () => {
|
||||
it('识别常见图片格式', () => {
|
||||
expect(detectImageMime(JPEG)).toBe('image/jpeg')
|
||||
expect(detectImageMime(PNG)).toBe('image/png')
|
||||
expect(detectImageMime(Buffer.from('GIF89a----'))).toBe('image/gif')
|
||||
expect(detectImageMime(Buffer.from([0x42, 0x4d, 0, 0]))).toBe('image/bmp')
|
||||
const webp = Buffer.concat([
|
||||
Buffer.from('RIFF'),
|
||||
Buffer.alloc(4),
|
||||
Buffer.from('WEBP'),
|
||||
Buffer.alloc(8)
|
||||
])
|
||||
expect(detectImageMime(webp)).toBe('image/webp')
|
||||
expect(detectImageMime(Buffer.from('<svg xmlns="http://www.w3.org/2000/svg"></svg>'))).toBe(
|
||||
'image/svg+xml'
|
||||
)
|
||||
})
|
||||
|
||||
it('HTML 反盗链页 / 空响应 一律不认(否则会内联成空白头像)', () => {
|
||||
expect(detectImageMime(HTML)).toBeUndefined()
|
||||
expect(detectImageMime(Buffer.alloc(0))).toBeUndefined()
|
||||
expect(detectImageMime(Buffer.from('{"error":"forbidden"}'))).toBeUndefined()
|
||||
// XML 前缀但正文不是 svg(HTML 页面带 xml 声明)也不许蒙混过关
|
||||
expect(
|
||||
detectImageMime(Buffer.from('<?xml version="1.0"?><html><body>x</body></html>'))
|
||||
).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('mapWithConcurrency — 有界并发且保序', () => {
|
||||
it('同时在飞的任务数不超过上限,结果顺序与输入一致', async () => {
|
||||
let inFlight = 0
|
||||
let peak = 0
|
||||
const items = Array.from({ length: 20 }, (_, index) => index)
|
||||
const results = await mapWithConcurrency(items, 6, async (item) => {
|
||||
inFlight += 1
|
||||
peak = Math.max(peak, inFlight)
|
||||
await new Promise((resolve) => setTimeout(resolve, 5))
|
||||
inFlight -= 1
|
||||
return item * 2
|
||||
})
|
||||
expect(peak).toBeLessThanOrEqual(6)
|
||||
expect(peak).toBeGreaterThan(1)
|
||||
expect(results).toEqual(items.map((item) => item * 2))
|
||||
})
|
||||
|
||||
it('空输入与单元素输入都安全', async () => {
|
||||
expect(await mapWithConcurrency([], 4, async (item: number) => item)).toEqual([])
|
||||
expect(await mapWithConcurrency([1], 4, async (item) => item + 1)).toEqual([2])
|
||||
})
|
||||
})
|
||||
|
||||
describe('embedAvatar — 不因单次瞬时失败永久退化', () => {
|
||||
it('200 + 非图片正文(反盗链页)→ 首字占位,绝不内联成解码失败的空白头像', async () => {
|
||||
const fetchMock = vi.fn(async () => imageResponse(HTML, 'text/html'))
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const result = await embedAvatar('https://wx.qlogo.cn/mmhead/ver_1/antihotlink', '张三')
|
||||
|
||||
expect(result.fallback).toBe(true)
|
||||
expect(result.source.startsWith('data:image/svg+xml;base64,')).toBe(true)
|
||||
// 两次尝试(首次 + 重试)都拿到 HTML → 都判失败
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2)
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
|
||||
it('首次失败、重试成功 → 拿到真实头像(不是首字占位)', async () => {
|
||||
const fetchMock = vi
|
||||
.fn<() => Promise<Response>>()
|
||||
.mockResolvedValueOnce(new Response('rate limited', { status: 429 }))
|
||||
.mockResolvedValueOnce(imageResponse(JPEG))
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const result = await embedAvatar('https://wx.qlogo.cn/mmhead/ver_1/retry-once', '李四')
|
||||
|
||||
expect(result.fallback).toBe(false)
|
||||
expect(result.source.startsWith('data:image/jpeg;base64,')).toBe(true)
|
||||
expect(Buffer.from(result.source.split(',')[1], 'base64').subarray(0, 3)).toEqual(
|
||||
Buffer.from([0xff, 0xd8, 0xff])
|
||||
)
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2)
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
|
||||
it('两次都失败 → 首字占位,且不再继续请求', async () => {
|
||||
const fetchMock = vi.fn(async () => new Response('boom', { status: 500 }))
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const result = await embedAvatar('https://wx.qlogo.cn/mmhead/ver_1/always-down', '王五')
|
||||
|
||||
expect(result.fallback).toBe(true)
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2)
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
|
||||
it('同一 URL 成功后进缓存:重渲染 / 连续生成不重复下载', async () => {
|
||||
const fetchMock = vi.fn(async () => imageResponse(PNG, 'image/png'))
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const first = await embedAvatar('https://wx.qlogo.cn/mmhead/ver_1/cached', '赵六')
|
||||
const second = await embedAvatar('https://wx.qlogo.cn/mmhead/ver_1/cached', '赵六')
|
||||
|
||||
expect(first.fallback).toBe(false)
|
||||
expect(second).toEqual(first)
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1)
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
|
||||
it('没有来源时直接首字占位,不发请求', async () => {
|
||||
const fetchMock = vi.fn(async () => imageResponse(JPEG))
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const result = await embedAvatar(undefined, '无名')
|
||||
|
||||
expect(result.fallback).toBe(true)
|
||||
expect(fetchMock).not.toHaveBeenCalled()
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
|
||||
it('已是 data URL 时直接复用,不发请求', async () => {
|
||||
const fetchMock = vi.fn(async () => imageResponse(JPEG))
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const dataUrl = `data:image/png;base64,${PNG.toString('base64')}`
|
||||
const result = await embedAvatar(dataUrl, '已内联')
|
||||
|
||||
expect(result).toEqual({ source: dataUrl, fallback: false })
|
||||
expect(fetchMock).not.toHaveBeenCalled()
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
})
|
||||
Reference in New Issue
Block a user