diff --git a/electron.vite.config.1789066213093.mjs b/electron.vite.config.1789066213093.mjs new file mode 100644 index 0000000..1d0f7b7 --- /dev/null +++ b/electron.vite.config.1789066213093.mjs @@ -0,0 +1,35 @@ +// electron.vite.config.ts +import { resolve } from "path"; +import { defineConfig } from "electron-vite"; +import react from "@vitejs/plugin-react"; +var electron_vite_config_default = defineConfig({ + main: { + build: { + rollupOptions: { + input: { + index: resolve("src/main/index.ts"), + reportTemplateTest: resolve("src/main/report-template-test-entry.ts"), + queryAgentPoc: resolve("src/main/query-agent-poc-entry.ts"), + voiceRecognitionWorker: resolve("src/main/voice-pipeline/voice-recognition-worker.ts"), + knowledgeWorker: resolve("src/main/knowledge/knowledge-worker.ts") + }, + output: { + entryFileNames: "[name].js" + }, + external: ["koffi", "sherpa-onnx-node"] + } + } + }, + preload: {}, + renderer: { + resolve: { + alias: { + "@renderer": resolve("src/renderer/src") + } + }, + plugins: [react()] + } +}); +export { + electron_vite_config_default as default +}; diff --git a/src/main/services/query-agent-poc-service.ts b/src/main/services/query-agent-poc-service.ts index bcd90c4..a121163 100644 --- a/src/main/services/query-agent-poc-service.ts +++ b/src/main/services/query-agent-poc-service.ts @@ -1,5 +1,5 @@ import type { AIChatToolCall, AIChatToolDefinition } from './ai-provider-service' -import { LOCAL_QUERY_TOOL_DEFINITIONS } from '../../shared/local-query-api' +import { LOCAL_QUERY_TOOL_DEFINITIONS, type QueryTemporalBasisKind } from '../../shared/local-query-api' const MAX_TOOL_CALLS = 5 const FORBIDDEN_INPUT_KEYS = new Set(['apiKey', 'authorization', 'token', 'databasePath', 'sql', 'wxid', 'md5']) @@ -70,6 +70,17 @@ export interface QueryAgentTraceItem { status: string resultCount?: number evidenceCount?: number + /** LLM 声明的 temporalBasis。Host 消费它决定 policy,但不会传给 Local Query API。 */ + temporalBasis?: { kind: QueryTemporalBasisKind; sourceText?: string } + /** Host 自动执行的扩大查询(当前仅 temporalBasis.kind=recall_hint + 有界范围 + 0 结果)。 */ + autoFallback?: { + reason: 'soft_temporal_hint_zero_result' + timeRange: Record + status: string + durationMs: number + resultCount?: number + evidenceCount?: number + } } export interface QueryAgentModelCallDiagnostic { @@ -112,7 +123,7 @@ const SYSTEM_PROMPT = `你是 TraceMemo 的本地聊天查询助手,只能使 规划原则: - 先判断问题需要哪种证据,再调用最少的 Tool。每次收到 Tool Result 后都判断“当前 Evidence 是否已经足以给出有边界的回答”;足够就立即回答,不为追求绝对完整继续调查。 -- query_messages 是精确事实查询,适用于能用联系人、时间、方向、消息类型、顺序等结构条件表达的问题。earliest/latest 等时间边界也是结构条件,必须使用 order 与 limit 精确查询,不能使用抽样 overview。结果已经回答问题时,不要追加 conversation_overview。若返回 0 条且你判断是时间范围或结构条件不合适,允许再查一次并合理扩大或更换条件,但必须与上一次实质不同。 +- query_messages 是精确事实查询,适用于能用联系人、时间、方向、消息类型、顺序等结构条件表达的问题。earliest/latest 等时间边界也是结构条件,必须使用 order 与 limit 精确查询,不能使用抽样 overview。每次调用都必须如实声明 temporalBasis。结果已经回答问题时,不要追加 conversation_overview。 - 需要绝对时间范围时,startTime/endTime 必须使用带时区偏移的 ISO-8601 字符串(例如 2026-08-01T00:00:00+08:00 或 2026-07-31T16:00:00Z)。不要传 epoch 数字,也不要传没有时区的裸本地时间。 - search_messages 是关键词检索,适用于结构条件无法确定答案的问题。queries 的每一项都是一次独立的字面检索:一项只放一个简短关键词,不要把多个近义词或整句话塞进同一项。首次最多 4 项。检索到 Evidence 后直接判断;只有本次完全没有 Evidence 时,才允许再检索一次,且每一项都必须与上一次实质不同。 - conversation_overview 只用于真正需要理解一个时间范围内整体聊了什么、主要话题或整体互动的 broad summary。它返回 temporal coverage sample,不代表完整聊天,也不是检索不足时的默认 fallback。 @@ -281,18 +292,64 @@ function canonicalizeTimeRange(timeRange: Record, now: Date): { return { value: { kind: 'absolute', startTime: Math.floor(startMs / 1000), endTime: Math.floor(endMs / 1000) } } } +/** canonical(已剥离、已校验)的 temporalBasis。 */ +interface CanonicalTemporalBasis { + kind: QueryTemporalBasisKind + sourceText?: string +} + /** * LLM Tool Adapter 的 canonicalization: + * - temporalBasis 是 LLM-facing 元数据:Host 用它决定 policy,但**剥离**后不传给 Local Query API * - absolute 的 ISO-8601 → Local Query API 的 epoch seconds * - search_messages 的 queries[] → Local Query API 的 query + variants + * + * 这里只做**纯 lexical** 校验(sourceText 是否为用户问题子串 + kind 与 timeRange 是否自洽), + * 不解释时间短语的意思。 */ -function canonicalizeToolInput(name: string, input: Record, now: Date): { input?: Record; error?: ToolArgumentValidationError } { +function canonicalizeToolInput( + name: string, + input: Record, + now: Date, + question: string +): { input?: Record; temporalBasis?: CanonicalTemporalBasis; error?: ToolArgumentValidationError } { const output: Record = { ...input } + const rawBasis = output.temporalBasis + delete output.temporalBasis + + let temporalBasis: CanonicalTemporalBasis | undefined + if (rawBasis && typeof rawBasis === 'object' && !Array.isArray(rawBasis)) { + const record = rawBasis as Record + const kind = record.kind + if (kind === 'constraint' || kind === 'recall_hint' || kind === 'none') { + const sourceText = typeof record.sourceText === 'string' ? record.sourceText.trim() : undefined + if (kind === 'none') { + if (sourceText) { + return { error: argError('temporalBasis.sourceText', 'forbidden_for_none', null, sourceText, 'kind=none 表示问题里没有任何时间表达,此时不要提供 sourceText。') } + } + } else if (!sourceText) { + return { error: argError('temporalBasis.sourceText', 'required', '用户原问题中的时间原文片段', undefined, `kind=${kind} 时必须给出用户原问题中实际出现的时间片段。`) } + } else if (!question.includes(sourceText)) { + return { error: argError('temporalBasis.sourceText', 'source_not_in_question', question, sourceText, 'sourceText 必须是用户原问题中逐字出现的片段,不要改写、翻译或补全。') } + } + temporalBasis = { kind, ...(sourceText ? { sourceText } : {}) } + } + } + if (output.timeRange && typeof output.timeRange === 'object' && !Array.isArray(output.timeRange)) { const canonical = canonicalizeTimeRange(output.timeRange as Record, now) if (canonical.error) return { error: canonical.error } output.timeRange = canonical.value } + + // 用户没给任何时间表达时,不得凭空造一个有界范围(earliest/latest 用 order/limit 表达)。 + if (temporalBasis?.kind === 'none') { + const kind = rangeKind(output) + if (kind && kind !== 'all') { + return { error: argError('timeRange', 'temporal_basis_mismatch', 'timeRange.kind=all', output.timeRange, '用户没有给出任何时间表达(temporalBasis.kind=none)。请改用 timeRange.kind=all;如果需要 earliest/latest 这类边界,用 order 与 limit 表达。') } + } + } + if (name === 'search_messages') { const raw = Array.isArray(output.queries) ? (output.queries as unknown[]) : [] const probes = raw.filter((value): value is string => typeof value === 'string').map((value) => value.trim()).filter(Boolean) @@ -302,7 +359,7 @@ function canonicalizeToolInput(name: string, input: Record, now output.query = first if (rest.length) output.variants = rest } - return { input: output } + return { input: output, temporalBasis } } interface ZeroResultRetryState { @@ -310,12 +367,63 @@ interface ZeroResultRetryState { searchSignatures: string[] queryAttempts: number querySignatures: string[] + /** + * temporalBasis.kind=constraint 的 query_messages 一旦执行,就锁定其 canonical timeRange。 + * 后续 retry 若替换时间范围会被拒绝 —— 用户明确给出的时间边界不得扩大。 + */ + lockedConstraintRange?: { signature: string; timeRange: Record } } function newRetryState(): ZeroResultRetryState { return { searchAttempts: 0, searchSignatures: [], queryAttempts: 0, querySignatures: [] } } +/** Host 自动执行的扩大查询原因(当前只有一种)。 */ +const AUTO_FALLBACK_REASON = 'soft_temporal_hint_zero_result' as const + +function rangeKind(input: Record): string | undefined { + const range = input.timeRange + if (!range || typeof range !== 'object' || Array.isArray(range)) return undefined + const kind = (range as Record).kind + return typeof kind === 'string' ? kind : undefined +} + +/** + * constraint 时间边界锁定:时间来源由 LLM 判断(temporalBasis), + * 但"不得扩大"由 Host 结构性保证,不依赖 prompt。 + */ +function lockConstraintTimeRange( + name: string, + temporalBasis: CanonicalTemporalBasis | undefined, + input: Record, + state: ZeroResultRetryState +): ToolArgumentValidationError | undefined { + if (name !== 'query_messages' || temporalBasis?.kind !== 'constraint') return undefined + const signature = JSON.stringify(input.timeRange ?? null) + if (!state.lockedConstraintRange) { + state.lockedConstraintRange = { signature, timeRange: (input.timeRange as Record) ?? {} } + return undefined + } + if (state.lockedConstraintRange.signature === signature) return undefined + return argError('timeRange', 'constraint_time_range_immutable', state.lockedConstraintRange.timeRange, input.timeRange, '用户明确给出的时间范围不得改变;只能放宽 direction / messageTypes 等非时间条件。') +} + +/** + * 只有"LLM 自己推断的近似时间范围(recall_hint)+ 有界范围 + 首次 0 结果"才自动扩大。 + * 已经是 all 时无需扩大;constraint 一律不扩大(由 lockConstraintTimeRange 保证)。 + */ +function shouldAutoBroaden( + name: string, + temporalBasis: CanonicalTemporalBasis | undefined, + input: Record, + result: QueryAgentToolResult +): boolean { + if (name !== 'query_messages') return false + if (temporalBasis?.kind !== 'recall_hint') return false + if (resultCount(result).resultCount !== 0) return false + return rangeKind(input) !== 'all' +} + function normalizedTarget(input: Record): string { const target = input.target const query = target && typeof target === 'object' ? (target as Record).query : undefined @@ -351,11 +459,16 @@ function retryNote(name: string, result: QueryAgentToolResult, state: ZeroResult if (result.status !== 'completed') return undefined const counts = resultCount(result) if (name === 'search_messages' && !counts.evidenceCount && state.searchAttempts <= ZERO_RESULT_RETRY_LIMIT) return '本次检索没有任何 Evidence。允许再执行一次 search_messages,但每一项都必须与上一次实质不同;完全相同的检索会被拒绝。' - if (name === 'query_messages' && counts.resultCount === 0 && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT) return '本次精确查询返回 0 条。允许再执行一次 query_messages,用于合理扩大或更换时间范围、方向或消息类型;完全相同的条件会被拒绝。' + if (name === 'query_messages' && counts.resultCount === 0 && !result.fallbackLookup && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT) return '本次精确查询返回 0 条。允许再执行一次 query_messages,用于放宽 direction 或 messageTypes 等非时间条件;改变时间范围会被拒绝。' return undefined } -export function validateToolArguments(name: string, value: unknown, now: Date = new Date()): { input?: Record; error?: ToolArgumentValidationError } { +export function validateToolArguments( + name: string, + value: unknown, + now: Date = new Date(), + question = '' +): { input?: Record; temporalBasis?: CanonicalTemporalBasis; error?: ToolArgumentValidationError } { const definition = LOCAL_QUERY_TOOL_DEFINITIONS.find((tool) => tool.name === name) if (!definition) return { error: argError('$', 'tool', 'supported tool', name) } if (!value || typeof value !== 'object' || Array.isArray(value)) { @@ -366,7 +479,7 @@ export function validateToolArguments(name: string, value: unknown, now: Date = } const error = validateSchema(value, definition.parameters as ToolSchema) if (error) return { error } - return canonicalizeToolInput(name, value as Record, now) + return canonicalizeToolInput(name, value as Record, now, question) } function resultCount(result: QueryAgentToolResult): { resultCount?: number; evidenceCount?: number } { @@ -389,6 +502,13 @@ function toolResultForModel(name: string, result: QueryAgentToolResult, callsUse if (result.anchor) visible.anchor = messageRecordForModel(result.anchor) if (Array.isArray(result.before)) visible.before = result.before.map(messageRecordForModel) if (Array.isArray(result.after)) visible.after = result.after.map(messageRecordForModel) + if (result.fallbackLookup && typeof result.fallbackLookup === 'object') { + const fallback = result.fallbackLookup as Record + visible.fallbackLookup = { + ...fallback, + ...(Array.isArray(fallback.messages) ? { messages: fallback.messages.map(messageRecordForModel) } : {}) + } + } visible._agent = { toolName: name, toolCallsUsed: callsUsed, @@ -399,13 +519,16 @@ function toolResultForModel(name: string, result: QueryAgentToolResult, callsUse nextTools.length ? '先判断当前 Evidence 是否足以回答;足够就立即回答,只在含义仍有明确歧义时使用当前可用 Tool。' : '工具阶段已经结束。必须直接给出有边界的最终回答,不得再调用 Tool。', + result.fallbackLookup + ? '本次结果有两个 scope:顶层是原查询,fallbackLookup 是系统自动扩大到全部历史后的结果。回答时必须分别说明这两个范围,不要让用户以为原问题就是按“全部历史”提出的。' + : undefined, note ].filter(Boolean).join(' ') } return visible } -function nextToolDefinitions(name: string, result: QueryAgentToolResult, state: ZeroResultRetryState): AIChatToolDefinition[] { +function nextToolDefinitions(name: string, result: QueryAgentToolResult, state: ZeroResultRetryState, rangeWasAll = false): AIChatToolDefinition[] { // 重复重试已被拒绝,不再开放工具,避免用有限的 tool budget 反复试同一条件。 if (result.constraint === 'duplicate_retry') return [] if (result.status === 'invalid_tool_arguments') return toolDefinition(name) @@ -418,6 +541,10 @@ function nextToolDefinitions(name: string, result: QueryAgentToolResult, state: return state.searchAttempts <= ZERO_RESULT_RETRY_LIMIT ? toolDefinition('search_messages') : [] } if (name === 'query_messages') { + // Host 已经自动执行过一次扩大查询:不再开放 retry,避免出现第三次查询。 + if (result.fallbackLookup) return [] + // 已经查了全部历史且 0 结果:再换时间范围毫无意义(更窄只会更少)。 + if (rangeWasAll && counts.resultCount === 0) return [] // 只有 0 结果才开放一次重试;有结果时保持原有 stopping。 return counts.resultCount === 0 && state.queryAttempts <= ZERO_RESULT_RETRY_LIMIT ? toolDefinition('query_messages') : [] } @@ -485,6 +612,8 @@ export class QueryAgentPocService { const inputStartedAt = Date.now() let toolResult: QueryAgentToolResult | undefined let traceInput: Record = {} + let temporalBasis: CanonicalTemporalBasis | undefined + let autoFallback: QueryAgentTraceItem['autoFallback'] try { if (!tools.some((tool) => tool.function.name === call.name)) { toolResult = { status: 'invalid_tool_arguments', field: '$', constraint: 'tool_availability', expected: tools.map((tool) => tool.function.name), actual: call.name } @@ -495,17 +624,44 @@ export class QueryAgentPocService { toolResult = { status: 'invalid_tool_arguments', field: '$', constraint: 'json' } } if (!toolResult) { - const validated = validateToolArguments(call.name, parsed, now) + const validated = validateToolArguments(call.name, parsed, now, trimmed) if (validated.error) { toolResult = validated.error } else { traceInput = validated.input || {} - if (duplicateRetry(call.name, traceInput, retry)) { + temporalBasis = validated.temporalBasis + // constraint 时间边界由 Host 结构性锁定:不依赖 prompt,也不静默改写用户问题。 + const constraintViolation = lockConstraintTimeRange(call.name, temporalBasis, traceInput, retry) + if (constraintViolation) { + toolResult = constraintViolation + } else if (duplicateRetry(call.name, traceInput, retry)) { // 明确拒绝“换关键词重搜”里的 identical retry,让模型改用实质不同的条件。 toolResult = { status: 'invalid_tool_arguments', field: '$', constraint: 'duplicate_retry', expected: '与上一次实质不同的条件', actual: '与上一次完全相同的条件' } } else { recordAttempt(call.name, traceInput, retry) - toolResult = await this.executeTool(call.name, traceInput) + const primary = await this.executeTool(call.name, traceInput) + toolResult = primary + // recall_hint + 有界范围 + 0 结果 → Host 自动做一次“全部历史”corrective lookup。 + // 这是一次本地 Query API 调用:不增加 LLM 往返,也不占用 MAX_TOOL_CALLS。 + if (shouldAutoBroaden(call.name, temporalBasis, traceInput, primary)) { + const fallbackStartedAt = Date.now() + const fallbackResult = await this.executeTool('query_messages', { ...traceInput, timeRange: { kind: 'all' } }) + const fallbackCounts = resultCount(fallbackResult) + autoFallback = { reason: AUTO_FALLBACK_REASON, timeRange: { kind: 'all' }, status: fallbackResult.status, durationMs: Date.now() - fallbackStartedAt, ...fallbackCounts } + toolResult = { + ...primary, + fallbackLookup: { + reason: AUTO_FALLBACK_REASON, + explanation: '你的 temporalBasis.kind=recall_hint 表示该时间范围只是你推断的回忆线索,并非用户给出的硬边界;首次查询为 0 条,系统已自动在全部历史中再查一次。请分别说明这两个范围的结果。', + timeRange: { kind: 'all' }, + status: fallbackResult.status, + resolvedTimeRange: fallbackResult.resolvedTimeRange, + coverage: fallbackResult.coverage, + returnedCount: fallbackResult.returnedCount, + messages: fallbackResult.messages + } + } + } } } } @@ -518,10 +674,10 @@ export class QueryAgentPocService { result.toolCallCount += 1 result.toolTotalMs += durationMs const counts = resultCount(completedToolResult) - result.traces.push({ toolName: call.name, input: sanitizeInput(traceInput), durationMs, status: completedToolResult.status, ...counts }) + result.traces.push({ toolName: call.name, input: sanitizeInput(traceInput), durationMs, status: completedToolResult.status, ...counts, ...(temporalBasis ? { temporalBasis } : {}), ...(autoFallback ? { autoFallback } : {}) }) const nextTools = completedToolResult.constraint === 'tool_availability' ? tools - : nextToolDefinitions(call.name, completedToolResult, retry) + : nextToolDefinitions(call.name, completedToolResult, retry, rangeKind(traceInput) === 'all') const note = retryNote(call.name, completedToolResult, retry) messages.push({ role: 'tool', tool_call_id: call.id, name: call.name, content: JSON.stringify(toolResultForModel(call.name, completedToolResult, result.toolCallCount, nextTools, note)) }) tools = nextTools diff --git a/src/shared/local-query-api.ts b/src/shared/local-query-api.ts index c226746..5a4550c 100644 --- a/src/shared/local-query-api.ts +++ b/src/shared/local-query-api.ts @@ -30,8 +30,33 @@ const timeRangeSchema = { required: ['kind'], additionalProperties: false } +/** + * LLM-facing 时间来源语义。只由 Query Agent orchestration 消费: + * Host 在执行 Local Query API 之前会剥离它 —— 它不进入公共 Query API contract。 + * kind 的分类由 LLM 决定;Host 只读取分类结果并执行确定性 policy, + * 不做中文关键词 / 正则 / 时间短语映射。 + */ +export type QueryTemporalBasisKind = 'constraint' | 'recall_hint' | 'none' +const temporalBasisSchema = { + type: 'object', + required: ['kind'], + additionalProperties: false, + properties: { + kind: { + enum: ['constraint', 'recall_hint', 'none'], + description: + '用户问题中的时间表达扮演什么角色,必填。constraint:问题里存在能够归一化成具体时间范围的表达(具体日期、具体年月、月份、上个月、今年、过去 N 天、某日到某日等)。判断看“这个表达本身能不能确定查询边界”,与用户是否确信事情发生过无关——例如“我记得他上个月好像发过文件,是不是”里的“上个月”仍是 constraint;没写年份但按当前时间能正常归一化的月份(如“八月份”)同样是 constraint。recall_hint:用户确实提到时间感觉,但该表达无法确定唯一的查询边界,只能当回忆线索(模糊的近情感、“以前某阵子”)。此时你可以自己挑一个合理的有界范围做首次查询,但它是搜索启发式,不是用户约束。none:问题里没有任何时间信息,此时必须用 timeRange.kind=all,不要凭空造时间范围。' + }, + sourceText: { + type: 'string', + minLength: 1, + description: + 'kind 为 constraint 或 recall_hint 时必填:用户原问题中实际出现的时间相关原文片段,必须逐字摘录(不要改写、翻译、补全或加标点)。kind 为 none 时不要提供该字段。' + } + } +} export const LOCAL_QUERY_TOOL_DEFINITIONS: LocalQueryToolDefinition[] = [ - { name: 'query_messages', description: '精确读取符合联系人、时间、方向、消息类型、顺序等结构条件的消息;适合具体事实和 earliest/latest 等时间边界查询,边界查询使用 order 与 limit。若结果为 0 且条件明显不合适,可再查一次并合理扩大或更换条件,但条件必须与上一次实质不同。', parameters: { type: 'object', required: ['target', 'timeRange'], additionalProperties: false, properties: { target: targetSchema, timeRange: timeRangeSchema, direction: { enum: ['any', 'from_target', 'to_target'] }, messageTypes: { type: 'array', items: { enum: ['text', 'image', 'voice', 'video', 'file', 'link', 'sticker', 'system', 'other'] } }, order: { enum: ['asc', 'desc'] }, limit: { type: 'integer', minimum: 1, maximum: 200 }, excludeSystem: { type: 'boolean' } } } }, + { name: 'query_messages', description: '精确读取符合联系人、时间、方向、消息类型、顺序等结构条件的消息;适合具体事实和 earliest/latest 等时间边界查询,边界查询使用 order 与 limit。每次调用都必须声明 temporalBasis,说明这个时间范围来自用户的明确约束、模糊回忆线索,还是用户根本没给时间信息。', parameters: { type: 'object', required: ['target', 'timeRange', 'temporalBasis'], additionalProperties: false, properties: { target: targetSchema, timeRange: timeRangeSchema, temporalBasis: temporalBasisSchema, direction: { enum: ['any', 'from_target', 'to_target'] }, messageTypes: { type: 'array', items: { enum: ['text', 'image', 'voice', 'video', 'file', 'link', 'sticker', 'system', 'other'] } }, order: { enum: ['asc', 'desc'] }, limit: { type: 'integer', minimum: 1, maximum: 200 }, excludeSystem: { type: 'boolean' } } } }, { name: 'search_messages', description: '在指定联系人和时间范围内做关键词检索并返回相关 Evidence。queries 的每一项都是一次独立的字面检索:一项只放一个简短关键词,不要把多个近义词或整句话放进同一项,也不要指望一项内部被拆词理解。首次最多 4 项;只有在本次检索完全没有 Evidence 时,才允许再检索一次,且每一项都必须与上一次实质不同。', parameters: { type: 'object', required: ['target', 'timeRange', 'queries'], additionalProperties: false, properties: { target: targetSchema, timeRange: timeRangeSchema, queries: { type: 'array', description: '独立检索项列表,每项一个简短关键词,最多 4 项;每一项单独检索,不会组合成一句话理解。', minItems: 1, maxItems: 4, items: { type: 'string', minLength: 1 } }, limit: { type: 'integer', minimum: 1, maximum: 200 } } } }, { name: 'message_context', description: '补充已找到的单条有价值 Evidence 的前后消息;仅在该 Evidence 缺少语境、无法判断含义时使用,不是默认确认步骤。', parameters: { type: 'object', required: ['messageRef'], additionalProperties: false, properties: { messageRef: { type: 'string', minLength: 1 }, before: { type: 'integer', minimum: 0, maximum: 50 }, after: { type: 'integer', minimum: 0, maximum: 50 } } } }, { name: 'conversation_overview', description: '提取指定联系人和时间范围的整体聊天覆盖样本;只用于 broad summary,不是语义搜索 fallback,也不能确定 earliest/latest 等精确时间边界。', parameters: { type: 'object', required: ['target', 'timeRange'], additionalProperties: false, properties: { target: targetSchema, timeRange: timeRangeSchema } } } diff --git a/tests/unit/query-agent-poc-service.test.ts b/tests/unit/query-agent-poc-service.test.ts index 85e583e..a7cf292 100644 --- a/tests/unit/query-agent-poc-service.test.ts +++ b/tests/unit/query-agent-poc-service.test.ts @@ -8,11 +8,15 @@ function provider(responses: Array { it('runs a bounded model -> tool -> model loop and records sanitized trace', async () => { const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 1, messages: [{ messageRef: 'secret-ref' }] })) const service = new QueryAgentPocService(provider([ - { success: true, toolCalls: [{ id: 'call-1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, limit: 1 }) }] }, + { success: true, toolCalls: [{ id: 'call-1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: { kind: 'none' }, limit: 1 }) }] }, { success: true, data: '第一条消息是图片。' } ]), execute) const result = await service.run('我和 BOBO 最开始聊了什么') @@ -49,7 +53,7 @@ describe('QueryAgentPocService', () => { it('removes tools after a sufficient exact result', async () => { const configuredProvider = provider([ - { success: true, toolCalls: [{ id: 'call-1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, limit: 1 }) }] }, + { success: true, toolCalls: [{ id: 'call-1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: { kind: 'none' }, limit: 1 }) }] }, { success: true, data: '完成' } ]) await new QueryAgentPocService(configuredProvider, vi.fn(async () => ({ status: 'completed', returnedCount: 1 }))).run('第一条消息') @@ -59,7 +63,7 @@ describe('QueryAgentPocService', () => { it('does not execute a tool that is unavailable after the stopping boundary', async () => { const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 1 })) const configuredProvider = provider([ - { success: true, toolCalls: [{ id: 'call-1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, limit: 1 }) }] }, + { success: true, toolCalls: [{ id: 'call-1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: { kind: 'none' }, limit: 1 }) }] }, { success: true, toolCalls: [{ id: 'call-2', name: 'conversation_overview', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' } }) }] }, { success: true, data: '完成' } ]) @@ -120,9 +124,14 @@ describe('QueryAgentPocService', () => { expect(validateToolArguments('search_messages', { ...base, query: 'x' }).error).toMatchObject({ field: 'query', constraint: 'additionalProperties' }) // 漏掉 queries 时按 required 拒绝。 expect(validateToolArguments('search_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' } }).error).toMatchObject({ field: 'queries', constraint: 'required' }) - expect(validateToolArguments('query_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' }, direction: 'sideways' }).error).toMatchObject({ field: 'direction', constraint: 'enum' }) - expect(validateToolArguments('query_messages', { target: { query: '' }, timeRange: { kind: 'all' } }).error).toMatchObject({ field: 'target.query', constraint: 'minLength' }) - expect(validateToolArguments('query_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' }, limit: 0 }).error).toMatchObject({ field: 'limit', constraint: 'minimum' }) + const q = '上个月 BOBO 有没有给我发过文件' + const basis = { kind: 'constraint', sourceText: '上个月' } + expect(validateToolArguments('query_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: basis, direction: 'sideways' }, undefined, q).error).toMatchObject({ field: 'direction', constraint: 'enum' }) + expect(validateToolArguments('query_messages', { target: { query: '' }, timeRange: { kind: 'all' }, temporalBasis: basis }, undefined, q).error).toMatchObject({ field: 'target.query', constraint: 'minLength' }) + expect(validateToolArguments('query_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: basis, limit: 0 }, undefined, q).error).toMatchObject({ field: 'limit', constraint: 'minimum' }) + // temporalBasis 是 query_messages 的必填字段,kind 只接受 constraint / recall_hint / none。 + expect(validateToolArguments('query_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' } }, undefined, q).error).toMatchObject({ field: 'temporalBasis', constraint: 'required' }) + expect(validateToolArguments('query_messages', { target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: { kind: 'maybe' } }, undefined, q).error).toMatchObject({ field: 'temporalBasis.kind', constraint: 'enum' }) expect(validateToolArguments('message_context', { messageRef: 'opaque', before: 51 }).error).toMatchObject({ field: 'before', constraint: 'maximum' }) }) @@ -136,8 +145,8 @@ describe('QueryAgent absolute time contract', () => { const now = new Date('2026-09-10T00:00:00Z') const target = { query: 'BOBO' } - function canonical(timeRange: Record) { - return validateToolArguments('query_messages', { target, timeRange }, now) + function canonical(timeRange: Record, basis: { kind: string; sourceText?: string } = { kind: 'constraint', sourceText: '8 月' }) { + return validateToolArguments('query_messages', { target, timeRange, temporalBasis: basis }, now, '8 月 BOBO 有没有给我发过文件') } it('canonicalizes an ISO-8601 absolute range with offset into Local Query API epoch seconds', () => { @@ -210,8 +219,13 @@ describe('QueryAgent zero-result limited retry', () => { function searchCall(id: string, queries: string[], timeRange: Record = { kind: 'all' }) { return { success: true as const, toolCalls: [{ id, name: 'search_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange, queries }) }] } } - function queryCall(id: string, timeRange: Record = { kind: 'all' }) { - return { success: true as const, toolCalls: [{ id, name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange }) }] } + function queryCall( + id: string, + timeRange: Record = { kind: 'all' }, + basis: { kind: string; sourceText?: string } = { kind: 'constraint', sourceText: MONTH_TEXT }, + extra: Record = {} + ) { + return { success: true as const, toolCalls: [{ id, name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange, temporalBasis: basis, ...extra }) }] } } const emptySearch = async () => ({ status: 'completed', evidenceCount: 0, evidence: [] }) const emptyQuery = async () => ({ status: 'completed', returnedCount: 0, messages: [] }) @@ -263,21 +277,10 @@ describe('QueryAgent zero-result limited retry', () => { expect(vi.mocked(configured.chatWithTools).mock.calls[1]?.[1].map((tool) => tool.function.name)).toEqual(['message_context']) }) - it('allows one query_messages retry after a zero-result exact query', async () => { - const execute = vi.fn() - .mockResolvedValueOnce({ status: 'completed', returnedCount: 0, messages: [] }) - .mockResolvedValueOnce({ status: 'completed', returnedCount: 1, messages: [{ messageRef: 'ref' }] }) - const configured = provider([queryCall('c1', { kind: 'last_7_days' }), queryCall('c2', { kind: 'all' }), { success: true, data: '找到。' }]) - const result = await new QueryAgentPocService(configured, execute).run('找文件') - expect(execute).toHaveBeenCalledTimes(2) - expect(execute.mock.calls[1][1]).toMatchObject({ timeRange: { kind: 'all' } }) - expect(result.traces[1].status).toBe('completed') - }) - it('rejects an identical query_messages retry', async () => { const execute = vi.fn(emptyQuery) const configured = provider([queryCall('c1', { kind: 'previous_month' }), queryCall('c2', { kind: 'previous_month' }), { success: true, data: 'x' }]) - const result = await new QueryAgentPocService(configured, execute).run('找文件') + const result = await new QueryAgentPocService(configured, execute).run(MONTH_QUESTION) expect(execute).toHaveBeenCalledTimes(1) expect(result.traces[1].status).toBe('invalid_tool_arguments') }) @@ -285,7 +288,7 @@ describe('QueryAgent zero-result limited retry', () => { it('keeps the efficient stop when the exact query already returned messages', async () => { const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 3, messages: [] })) const configured = provider([queryCall('c1'), { success: true, data: 'ok' }]) - await new QueryAgentPocService(configured, execute).run('找') + await new QueryAgentPocService(configured, execute).run(MONTH_QUESTION) expect(vi.mocked(configured.chatWithTools).mock.calls[1]?.[1]).toEqual([]) }) @@ -299,15 +302,229 @@ describe('QueryAgent zero-result limited retry', () => { queryCall('c5', { kind: 'this_year' }), queryCall('c6', { kind: 'this_month' }) ] - const result = await new QueryAgentPocService(provider(responses), execute).run('找') + const result = await new QueryAgentPocService(provider(responses), execute).run(MONTH_QUESTION) expect(result.toolCallCount).toBe(5) expect(result.error).toContain('最大工具调用次数') }) + + it('allows a corrective retry that keeps an explicit absolute range and only relaxes other conditions', async () => { + const execute = vi.fn() + .mockResolvedValueOnce({ status: 'completed', returnedCount: 0, messages: [] }) + .mockResolvedValueOnce({ status: 'completed', returnedCount: 2, messages: [{ messageRef: 'ref' }] }) + const range = { kind: 'absolute', startTime: '2026-08-01T00:00:00+08:00', endTime: '2026-08-31T23:59:59+08:00' } + const call = (id: string, messageTypes: string[]) => ({ + success: true as const, + toolCalls: [{ id, name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: range, temporalBasis: { kind: 'constraint', sourceText: '8 月' }, messageTypes }) }] + }) + const configured = provider([call('c1', ['file']), call('c2', ['text']), { success: true, data: '找到。' }]) + const result = await new QueryAgentPocService(configured, execute).run('今年 8 月有没有给我发过文件') + expect(execute).toHaveBeenCalledTimes(2) + expect(result.traces.map((trace) => trace.status)).toEqual(['completed', 'completed']) + // the explicit range is preserved on the retry — only the non-temporal condition changed + const firstRange = (execute.mock.calls[0][1] as Record).timeRange + const secondRange = (execute.mock.calls[1][1] as Record).timeRange + expect(secondRange).toEqual(firstRange) + expect(secondRange.kind).toBe('absolute') + }) +}) + +describe('QueryAgent temporal basis policy', () => { + const emptyQuery = async () => ({ status: 'completed', returnedCount: 0, messages: [] }) + const CONSTRAINT_Q = '上个月 BOBO 有没有给我发过文件' + const HINT_Q = 'BOBO 前阵子发我的文件在哪' + const CONSTRAINT = { kind: 'constraint', sourceText: '上个月' } + const HINT = { kind: 'recall_hint', sourceText: '前阵子' } + const NONE = { kind: 'none' } + function queryCall(id: string, timeRange: Record, temporalBasis: { kind: string; sourceText?: string }, extra: Record = {}) { + return { success: true as const, toolCalls: [{ id, name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange, temporalBasis, ...extra }) }] } + } + function toolMessageAt(configured: QueryAgentProvider, callIndex: number): Record { + const message = vi.mocked(configured.chatWithTools).mock.calls[callIndex]?.[0].findLast((item) => item.role === 'tool') + return JSON.parse(String(message?.content)) as Record + } + + it('requires temporalBasis on query_messages', async () => { + const execute = vi.fn(emptyQuery) + const call = { success: true as const, toolCalls: [{ id: 'c1', name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' } }) }] } + const result = await new QueryAgentPocService(provider([call, { success: true, data: 'x' }]), execute).run(CONSTRAINT_Q) + expect(execute).not.toHaveBeenCalled() + expect(result.traces[0]).toMatchObject({ status: 'invalid_tool_arguments' }) + }) + + it('exposes the declared temporalBasis on the trace and never forwards it to the Local Query API', async () => { + const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 1, messages: [] })) + const configured = provider([queryCall('c1', { kind: 'last_7_days' }, HINT), { success: true, data: 'ok' }]) + const result = await new QueryAgentPocService(configured, execute).run(HINT_Q) + expect(result.traces[0].temporalBasis).toEqual({ kind: 'recall_hint', sourceText: '前阵子' }) + expect(result.traces[0].input).not.toHaveProperty('temporalBasis') + expect(execute.mock.calls[0][1]).not.toHaveProperty('temporalBasis') + }) + + it('rejects a sourceText that is not literally in the user question', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'previous_month' }, { kind: 'constraint', sourceText: '去年冬天' }), { success: true, data: 'x' }]) + const result = await new QueryAgentPocService(configured, execute).run(CONSTRAINT_Q) + expect(execute).not.toHaveBeenCalled() + expect(result.traces[0]).toMatchObject({ status: 'invalid_tool_arguments' }) + expect(JSON.stringify(toolMessageAt(configured, 1))).toContain('source_not_in_question') + }) + + it('requires sourceText for constraint and recall_hint', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'previous_month' }, { kind: 'constraint' }), { success: true, data: 'x' }]) + const result = await new QueryAgentPocService(configured, execute).run(CONSTRAINT_Q) + expect(execute).not.toHaveBeenCalled() + expect(result.traces[0]).toMatchObject({ status: 'invalid_tool_arguments' }) + }) + + it('rejects sourceText when kind is none', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'all' }, { kind: 'none', sourceText: '上个月' }), { success: true, data: 'x' }]) + const result = await new QueryAgentPocService(configured, execute).run(CONSTRAINT_Q) + expect(execute).not.toHaveBeenCalled() + expect(result.traces[0]).toMatchObject({ status: 'invalid_tool_arguments' }) + expect(JSON.stringify(toolMessageAt(configured, 1))).toContain('forbidden_for_none') + }) + + it('accepts a Unicode sourceText verbatim without parsing it', async () => { + const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 1, messages: [] })) + const question = '8 月 1 日到 9 月 1 日之间 BOBO 有没有发过文件?' + const sourceText = '8 月 1 日到 9 月 1 日' + const range = { kind: 'absolute', startTime: '2026-08-01T00:00:00+08:00', endTime: '2026-09-01T00:00:00+08:00' } + const configured = provider([queryCall('c1', range, { kind: 'constraint', sourceText }), { success: true, data: 'ok' }]) + const result = await new QueryAgentPocService(configured, execute).run(question) + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces[0].temporalBasis).toEqual({ kind: 'constraint', sourceText }) + }) + + it('never broadens a constraint relative range to all when it returns zero', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'previous_month' }, CONSTRAINT), { success: true, data: '上个月没有。' }]) + const result = await new QueryAgentPocService(configured, execute).run(CONSTRAINT_Q) + expect(execute).toHaveBeenCalledTimes(1) + expect(execute.mock.calls[0][1]).toMatchObject({ timeRange: { kind: 'previous_month' } }) + expect(result.traces[0].autoFallback).toBeUndefined() + }) + + it('never broadens a constraint absolute range to all when it returns zero', async () => { + const execute = vi.fn(emptyQuery) + const range = { kind: 'absolute', startTime: '2026-08-01T00:00:00+08:00', endTime: '2026-08-31T23:59:59+08:00' } + const configured = provider([queryCall('c1', range, { kind: 'constraint', sourceText: '8 月' }), { success: true, data: '没有。' }]) + const result = await new QueryAgentPocService(configured, execute).run('2026 年 8 月有没有给我发过文件') + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces[0].autoFallback).toBeUndefined() + expect(result.traces[0].input.timeRange).toMatchObject({ kind: 'absolute' }) + }) + + it('rejects a constraint retry that replaces the user time range', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'previous_month' }, CONSTRAINT), queryCall('c2', { kind: 'all' }, CONSTRAINT), { success: true, data: 'x' }]) + const result = await new QueryAgentPocService(configured, execute).run(CONSTRAINT_Q) + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces[1]).toMatchObject({ status: 'invalid_tool_arguments' }) + expect(JSON.stringify(toolMessageAt(configured, 2))).toContain('constraint_time_range_immutable') + }) + + it('locks a constraint range so later retries cannot spend the budget on another range', async () => { + const execute = vi.fn(emptyQuery) + const responses = [ + queryCall('c1', { kind: 'previous_month' }, CONSTRAINT), + queryCall('c2', { kind: 'this_month' }, CONSTRAINT), + queryCall('c3', { kind: 'all' }, CONSTRAINT), + { success: true, data: 'x' } + ] + const result = await new QueryAgentPocService(provider(responses), execute).run(CONSTRAINT_Q) + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces.map((trace) => trace.status)).toEqual(['completed', 'invalid_tool_arguments', 'invalid_tool_arguments']) + }) + + it('allows a constraint retry that keeps the range and only relaxes messageTypes', async () => { + const execute = vi.fn() + .mockResolvedValueOnce({ status: 'completed', returnedCount: 0, messages: [] }) + .mockResolvedValueOnce({ status: 'completed', returnedCount: 2, messages: [{ messageRef: 'ref' }] }) + const configured = provider([ + queryCall('c1', { kind: 'previous_month' }, CONSTRAINT, { messageTypes: ['file'] }), + queryCall('c2', { kind: 'previous_month' }, CONSTRAINT, { messageTypes: ['text'] }), + { success: true, data: '找到。' } + ]) + const result = await new QueryAgentPocService(configured, execute).run(CONSTRAINT_Q) + expect(execute).toHaveBeenCalledTimes(2) + expect(result.traces.map((trace) => trace.status)).toEqual(['completed', 'completed']) + }) + + it('does not broaden when a recall_hint range already returned messages', async () => { + const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 3, messages: [] })) + const configured = provider([queryCall('c1', { kind: 'last_7_days' }, HINT), { success: true, data: 'ok' }]) + const result = await new QueryAgentPocService(configured, execute).run(HINT_Q) + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces[0].autoFallback).toBeUndefined() + expect(vi.mocked(configured.chatWithTools).mock.calls[1]?.[1]).toEqual([]) + }) + + it('automatically runs one all-history corrective lookup for a zero-result recall_hint range', async () => { + const execute = vi.fn() + .mockResolvedValueOnce({ status: 'completed', returnedCount: 0, messages: [] }) + .mockResolvedValueOnce({ status: 'completed', returnedCount: 1, messages: [{ messageRef: 'ref', sourceKind: 'file' }], resolvedTimeRange: { kind: 'all', label: '全部历史' } }) + const configured = provider([queryCall('c1', { kind: 'last_7_days' }, HINT), { success: true, data: '找到了。' }]) + const result = await new QueryAgentPocService(configured, execute).run(HINT_Q) + expect(execute).toHaveBeenCalledTimes(2) + expect(execute.mock.calls[1][1]).toMatchObject({ timeRange: { kind: 'all' } }) + // the corrective lookup is Host orchestration, not a model tool call + expect(result.toolCallCount).toBe(1) + expect(result.traces[0].autoFallback).toMatchObject({ reason: 'soft_temporal_hint_zero_result', resultCount: 1 }) + }) + + it('does not repeat an all-history lookup when a recall_hint query already used all', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'all' }, HINT), queryCall('c2', { kind: 'all' }, HINT), { success: true, data: 'x' }]) + const result = await new QueryAgentPocService(configured, execute).run(HINT_Q) + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces[0].autoFallback).toBeUndefined() + expect(result.traces[1]).toMatchObject({ status: 'invalid_tool_arguments' }) + }) + + it('presents primary and fallback scopes separately to the model', async () => { + const execute = vi.fn() + .mockResolvedValueOnce({ status: 'completed', returnedCount: 0, messages: [], query: { resolvedTimeRange: { kind: 'last_7_days', label: '近 7 天' } } }) + .mockResolvedValueOnce({ status: 'completed', returnedCount: 1, messages: [{ messageRef: 'ref', text: 'x' }], resolvedTimeRange: { kind: 'all', label: '全部历史' } }) + const configured = provider([queryCall('c1', { kind: 'last_7_days' }, HINT), { success: true, data: 'ok' }]) + await new QueryAgentPocService(configured, execute).run(HINT_Q) + const presented = toolMessageAt(configured, 1) + expect(presented.returnedCount).toBe(0) + expect(presented.fallbackLookup).toMatchObject({ reason: 'soft_temporal_hint_zero_result', timeRange: { kind: 'all' }, returnedCount: 1 }) + expect(String(presented._agent.instruction)).toContain('fallbackLookup') + }) + + it('stops after a corrective lookup that also finds nothing', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'last_7_days' }, HINT), queryCall('c2', { kind: 'this_month' }, HINT), { success: true, data: '都没找到。' }]) + const result = await new QueryAgentPocService(configured, execute).run(HINT_Q) + // primary + one corrective lookup only; the third attempt is refused because tools are closed + expect(execute).toHaveBeenCalledTimes(2) + expect(result.traces[1]).toMatchObject({ status: 'invalid_tool_arguments' }) + }) + + it('accepts none with timeRange all', async () => { + const execute = vi.fn(async () => ({ status: 'completed', returnedCount: 2, messages: [] })) + const configured = provider([queryCall('c1', { kind: 'all' }, NONE), { success: true, data: 'ok' }]) + const result = await new QueryAgentPocService(configured, execute).run('BOBO 给我发过文件吗') + expect(execute).toHaveBeenCalledTimes(1) + expect(result.traces[0].temporalBasis).toEqual({ kind: 'none' }) + }) + + it('rejects none combined with a bounded time range', async () => { + const execute = vi.fn(emptyQuery) + const configured = provider([queryCall('c1', { kind: 'last_7_days' }, NONE), { success: true, data: 'x' }]) + const result = await new QueryAgentPocService(configured, execute).run('BOBO 给我发过文件吗') + expect(execute).not.toHaveBeenCalled() + expect(result.traces[0]).toMatchObject({ status: 'invalid_tool_arguments' }) + expect(JSON.stringify(toolMessageAt(configured, 1))).toContain('temporal_basis_mismatch') + }) }) describe('QueryAgent 耗时与请求级诊断记录', () => { function queryCall(id: string) { - return { success: true as const, toolCalls: [{ id, name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, limit: 1 }) }] } + return { success: true as const, toolCalls: [{ id, name: 'query_messages', arguments: JSON.stringify({ target: { query: 'BOBO' }, timeRange: { kind: 'all' }, temporalBasis: { kind: 'none' }, limit: 1 }) }] } } it('记录每次模型调用耗时,包括失败的那次(首次调用失败)', async () => {