From 081e1f50e2b2c393a2ffd9ac72f652df3be0dfd2 Mon Sep 17 00:00:00 2001 From: leokun Date: Tue, 25 Aug 2026 20:50:57 +0800 Subject: [PATCH] feat: NormalizedProvider --- server/prompt/cursor/tools.json | 2 +- server/src/model/inference.rs | 107 ++++++++++++++++++++++++++++++- server/src/provider/mod.rs | 1 + server/src/provider/normalize.rs | 28 ++++++++ server/src/provider/router.rs | 9 +-- server/src/run/model_cycle.rs | 3 +- server/src/store/providers.rs | 5 +- server/tests/prefix_stability.rs | 28 ++++++-- 8 files changed, 169 insertions(+), 14 deletions(-) create mode 100644 server/src/provider/normalize.rs diff --git a/server/prompt/cursor/tools.json b/server/prompt/cursor/tools.json index 6e07082..021ee41 100644 --- a/server/prompt/cursor/tools.json +++ b/server/prompt/cursor/tools.json @@ -705,7 +705,7 @@ "type": "function", "function": { "name": "Task", - "description": "启动一个新代理,自主处理边界清晰且适合委派的任务。\n\nTask 工具会启动专用子代理。每种 subagent_type 都有特定的能力和可用工具。使用 Task 时,必须通过 subagent_type 选择代理类型。\n\n默认行为\n\n默认由当前代理直接处理用户请求,并优先使用 Read、Glob、Grep、Shell、MCP 等直接工具。任务范围较大、步骤较多、需要探索代码库、暂时不确定答案,或者理论上可以并行,都不能单独构成调用 Task 的理由。\n\n只有符合以下至少一种情况时,才可以使用 Task:\n- 用户明确要求启动代理、子代理、worker,或者明确要求并行委派。\n- 存在一项工作量实质、边界清晰、可以独立完成的工作流,将它委派出去能够明显帮助当前任务。\n- 任务确实需要某个专用 subagent_type 才具备的能力。\n\n如果当前代理通过一次或少量直接工具调用就能完成任务,不得使用 Task。不要把整个用户请求交给子代理后直接返回它的结果;当前代理仍然对理解用户意图、整合结果和最终答复负责。\n\n并发规则\n\n- 默认启动一到三个子代理,具体数量应与彼此独立且确有必要委派的工作流数量一致。\n- 只有用户明确要求并行代理,或者确实存在两到三个彼此独立且工作量实质的工作流时,才可以同时启动多个子代理。\n- 单条回复最多启动三个子代理,即使可以构造出更多并行方向也不得超过三个。\n- 不得为了并行而人为拆分同一项调查、同一条执行链或可以由一个代理连续完成的工作。\n- 确需同时启动多个子代理时,在同一条消息中发出多个 Task 调用。\n\n示例\n\n- 用户问“ClientError 类定义在哪里?”:直接使用 Grep 或 Glob,不调用 Task。\n- 用户要求读取一个已知文件:直接使用 Read,不调用 Task。\n- 用户要求在两三个指定文件中搜索代码:直接使用 Read、Grep 或 Glob,不调用 Task。\n- 用户要求通过数据库 API 执行查询:直接调用对应 MCP,不调用 Task。\n- 用户泛泛询问代码库结构:先使用直接工具调查;问题范围广本身不等于必须委派。\n- 用户明确要求“分别启动两个代理独立调查客户端和服务端”:可以同时启动两个边界清晰的 Task。\n\n使用要求\n\n- description 必须是简短、具体、便于用户识别的标题。\n- prompt 必须明确说明子代理要完成的工作、范围、限制和最终应返回的信息。\n- 子代理看不到用户原始消息和父代理此前的步骤,因此 prompt 必须包含完成任务所必需的上下文;但不要复制无关上下文。\n- 子代理返回的内容是供父代理使用的工作结果。父代理应根据任务风险进行必要核对,而不是无条件接受。\n- 子代理类型的描述只说明其能力,不能推翻“默认由当前代理直接处理”的规则。不得仅仅因为某个类型声称可主动使用,就主动调用它。\n- 如果用户明确要求并行运行子代理,仍然必须遵守单条回复最多三个的限制。\n\n恢复与打断\n\n- 使用 resume 并传入已有代理 ID,可以在保留其上下文的情况下继续该代理。\n- 如果目标代理仍在运行,除非 interrupt=true,否则恢复请求会失败。\n- 只有用户明确要求打断或改变正在运行的代理时,才能设置 interrupt=true。\n- resume=\"self\" 表示从当前父代理的完整对话上下文分叉出一个新子代理。\n- 未使用 resume 时,每次 Task 调用都会启动一个全新的代理,因此 prompt 必须自包含。\n\n展示规则\n\n如果在面向用户的答复中提到代理或子代理,必须使用 `[名称](id)` 链接,不得使用 `[agent]`、`[worker]`、`[subagent]` 等泛化标签。云端代理修改代码后,应链接 `[Review](bc-id#changes)`;如果确切知道新增和删除行数,可使用 `[Review +A −D](bc-id#changes)`,并将 A、D 替换为真实数字。只有代理使用了 computer use 时才能使用 `[Try Live](bc-id#desktop)`。\n\n可用的 subagent_type\n\n- generalPurpose:处理已经确定适合委派的、工作量实质且边界清晰的通用任务。仅仅因为搜索结果不确定,不足以调用它。\n- explore:处理已经确定适合委派的、边界清晰且工作量实质的代码库探索。可用于按模式查找文件、搜索关键词或梳理代码结构。调用时应说明调查范围以及所需深度:quick、medium 或 very thorough。\n- shell:专门执行命令、Git 操作及其他终端任务。只有当这本身构成适合独立委派的工作流时才使用。\n- cursor-guide:阅读 Cursor 产品文档,回答 Cursor Desktop、IDE、CLI、Cloud Agents、Bugbot 等产品问题。\n- ci-investigator:调查单个失败的 PR CI 检查,并返回简短的根因总结。\n- bugbot:只有用户明确要求对本地代码改动进行 Bugbot 式审查时才能使用。description 必须严格为 `Bugbot`。除非用户明确要求后台运行,否则 run_in_background=false。prompt 使用固定格式:`Full Repository Path: ...\\nDiff: \\nChange Description: ...\\nCustom Instructions: ...`。默认使用 `Diff: branch changes`。只有常规 diff 无法生成时,才把 natural language 作为最后选择。该类型不支持 resume,每次都启动新代理。\n- security-review:只有用户明确要求安全审查本地代码改动时才能使用。description 必须严格为 `Security Review`。除非用户明确要求后台运行,否则 run_in_background=false。prompt 使用固定格式:`Full Repository Path: ...\\nDiff: \\nCustom Instructions: ...`。默认使用 `Diff: branch changes`。该类型不支持 resume,每次都启动新代理。\n- best-of-n-runner:在隔离的 Git worktree 中执行任务,用于用户明确要求的 Best-of-N 并行尝试或隔离实验。\n- test-subagent:仅在该类型自身的具体说明与当前任务明确匹配,并且任务已经满足委派条件时使用。\n\n子代理模型\n\n只有用户明确要求指定子代理模型时,才可以从以下列表选择:\n- inherit\n- claude-opus-5-thinking-high\n- composer-2.5-fast\n- cursor-grok-4.5-low\n- cursor-grok-4.6-high-fast\n- gpt-5.6-sol-medium\n\n用户没有明确指定模型时,使用 inherit。用户请求的模型不在列表中时,不得擅自替换或猜测;应跳过该模型的子代理调用,并告诉用户该模型不可用以及当前可用的模型。面向用户说明所选模型时,除非用户本来就使用 kebab-case 名称,否则不要直接展示 kebab-case slug。\n\n后台代理\n\n后台代理会在当前回复结束后自动发送完成通知。不得使用 AwaitShell 等待、轮询或主动检查 Task 启动的后台代理。可以继续处理其他工作,或者结束当前回复。", + "description": "Launch a new agent to autonomously handle a clearly bounded task that is suitable for delegation.\n\nThe Task tool starts a dedicated subagent. Each subagent_type has specific capabilities and available tools. When using Task, select the agent type through subagent_type.\n\nDefault behavior\n\nHandle the user's request directly by default, preferring direct tools such as Read, Glob, Grep, Shell, and MCP. A task being broad, multi-step, requiring codebase exploration, having an uncertain answer, or theoretically parallelizable is not by itself a reason to use Task.\n\nUse Task only when at least one of the following applies:\n- The user explicitly asks to start an agent, subagent, or worker, or explicitly asks for parallel delegation.\n- There is a substantial, clearly bounded workflow that can be completed independently and delegating it would materially help the current task.\n- The task genuinely requires capabilities provided by a specialized subagent_type.\n\nIf the current agent can complete the work with one or a few direct tool calls, do not use Task. Do not hand the entire user request to a subagent and simply return its result; the current agent remains responsible for understanding the user's intent, integrating results, and producing the final response.\n\nConcurrency rules\n\n- Launch one to three subagents by default, matching the number of independent workflows that genuinely need delegation.\n- Launch multiple subagents at the same time only when the user explicitly requests parallel agents or when there are two or three independent, substantial workflows.\n- When the user does not specify a number, launch at most three subagents in a single response. If the user explicitly requests more, you may launch the requested number.\n- Do not artificially split one investigation, one execution chain, or work that one agent can complete sequentially merely to create parallelism.\n- When multiple subagents should start together, issue multiple Task calls in the same message.\n\nExamples\n\n- The user asks, \"Where is the ClientError class defined?\": use Grep or Glob directly; do not use Task.\n- The user asks to read a known file: use Read directly; do not use Task.\n- The user asks to search two or three specified files: use Read, Grep, or Glob directly; do not use Task.\n- The user asks to run a query through a database API: call the relevant MCP directly; do not use Task.\n- The user broadly asks about the repository structure: investigate with direct tools first; broad scope alone does not require delegation.\n- The user explicitly asks to \"start two agents to investigate the client and server separately\": start two clearly bounded Tasks in parallel.\n\nUsage requirements\n\n- description must be a short, specific title that users can easily recognize.\n- prompt must clearly state the work the subagent should complete, its scope, constraints, and the information it should return.\n- Subagents cannot see the user's original message or the parent's previous steps, so prompt must include the context required to complete the task without copying unrelated context.\n- A subagent's response is working material for the parent. Verify it as appropriate for the task's risk instead of accepting it unconditionally.\n- Descriptions of subagent types explain their capabilities but do not override the rule that the current agent handles work directly by default. Do not call a type proactively merely because its description says it can be used proactively.\n- If the user explicitly requests parallel subagents, follow the number requested by the user.\n\nResume and interruption\n\n- Use resume with an existing agent ID to continue that agent while preserving its context.\n- If the target agent is still running, a resume request fails unless interrupt=true.\n- Set interrupt=true only when the user explicitly asks to interrupt or change a running agent.\n- resume=\"self\" forks a new subagent from the current parent's full conversation context.\n- Without resume, each Task call starts a new agent, so prompt must be self-contained.\n\nDisplay rules\n\nIf you mention an agent or subagent in a user-facing response, link it as `[Name](id)`. Do not use generic labels such as `[agent]`, `[worker]`, or `[subagent]`. When a cloud subagent edits code, link to `[Review](bc-id#changes)`, or use `[Review +A −D](bc-id#changes)` when the exact added and deleted line counts are known, replacing A and D with the real numbers. Use `[Try Live](bc-id#desktop)` only when the agent used computer use.\n\nAvailable subagent_type values\n\n- generalPurpose: handles substantial, clearly bounded general work that has already been determined suitable for delegation. Uncertain search results alone are not enough reason to use it.\n- explore: handles clearly bounded, substantial codebase exploration that has already been determined suitable for delegation. It can find files by patterns, search keywords, or map code structure. State the scope and desired depth: quick, medium, or very thorough.\n- shell: executes commands, Git operations, and other terminal work. Use it only when that work itself forms an independently delegable workflow.\n- cursor-guide: reads Cursor product documentation and answers questions about Cursor Desktop, IDE, CLI, Cloud Agents, Bugbot, and related products.\n- ci-investigator: investigates one failing PR CI check and returns a concise root-cause summary.\n- bugbot: use only when the user explicitly requests a Bugbot-style review of local code changes. description must be exactly `Bugbot`. Unless the user explicitly asks for background execution, set run_in_background=false. Use this exact prompt format: `Full Repository Path: ...\\nDiff: \\nChange Description: ...\\nCustom Instructions: ...`. Default to `Diff: branch changes`. Use natural language only as a last resort when a normal diff cannot be generated. This type does not support resume; each call starts a new agent.\n- security-review: use only when the user explicitly requests a security review of local code changes. description must be exactly `Security Review`. Unless the user explicitly asks for background execution, set run_in_background=false. Use this exact prompt format: `Full Repository Path: ...\\nDiff: \\nCustom Instructions: ...`. Default to `Diff: branch changes`. This type does not support resume; each call starts a new agent.\n- best-of-n-runner: performs tasks in isolated Git worktrees for user-requested Best-of-N parallel attempts or isolated experiments.\n- test-subagent: use only when the type's own specific instructions clearly match the current task and the task already satisfies the delegation conditions.\n\nSubagent model\n\nChoose from the following list only when the user explicitly requests a subagent model:\n- inherit\n- claude-opus-5-thinking-high\n- composer-2.5-fast\n- cursor-grok-4.5-low\n- cursor-grok-4.6-high-fast\n- gpt-5.6-sol-medium\n\nWhen the user does not explicitly specify a model, use inherit. If the requested model is not in the list, do not substitute or guess. Skip that subagent call and tell the user that the model is unavailable and which models are available. When describing the selected model to the user, do not show the kebab-case slug unless the user already used it.\n\nBackground agents\n\nBackground agents automatically send a completion notification after the current response ends. Do not use AwaitShell to wait for, poll, or actively check a subagent started with Task. Continue other work or end the current response.", "parameters": { "type": "object", "properties": { diff --git a/server/src/model/inference.rs b/server/src/model/inference.rs index e460dd8..07c3daf 100644 --- a/server/src/model/inference.rs +++ b/server/src/model/inference.rs @@ -1,6 +1,8 @@ use serde::{Deserialize, Serialize}; -use super::{ModelSpec, ProjectedMessage, ToolDefinition}; +use super::{ModelSpec, ProjectedContent, ProjectedMessage, ToolDefinition}; + +const PROVIDER_TOOL_CALL_ID_MAX_CHARS: usize = 64; #[derive(Clone, Debug, Serialize, Deserialize, PartialEq)] pub struct PromptSpec { @@ -23,3 +25,106 @@ pub struct ModelInvocation { pub provider_call_index: u64, pub request: ModelRequest, } + +pub(crate) fn normalize_provider_tool_call_ids(history: &mut [ProjectedMessage]) { + for message in history { + match &mut message.content { + ProjectedContent::Assistant { calls, .. } => { + for call in calls { + truncate_tool_call_id(&mut call.call_id); + } + } + ProjectedContent::ToolResult(result) => { + truncate_tool_call_id(&mut result.call_id); + } + ProjectedContent::Parts(_) => {} + } + } +} + +fn truncate_tool_call_id(call_id: &mut String) { + if let Some((end, _)) = call_id.char_indices().nth(PROVIDER_TOOL_CALL_ID_MAX_CHARS) { + call_id.truncate(end); + } +} + +#[cfg(test)] +mod tests { + use super::normalize_provider_tool_call_ids; + use crate::model::{ + ProjectedContent, ProjectedMessage, Role, ToolCallContent, ToolResultContent, + }; + + #[test] + fn provider_tool_call_ids_are_truncated_once_for_every_provider() { + let call_id = format!("cursor-tool-call:{}", "x".repeat(68)); + assert_eq!(call_id.len(), 85); + let expected = call_id[..64].to_string(); + let mut history = vec![ + ProjectedMessage { + message_id: "assistant".into(), + role: Role::Assistant, + content: ProjectedContent::Assistant { + text: String::new(), + thinking: String::new(), + replay_state: None, + calls: vec![ToolCallContent { + index: 0, + call_id: call_id.clone(), + name: "Shell".into(), + arguments: serde_json::json!({}), + }], + }, + }, + ProjectedMessage { + message_id: "result".into(), + role: Role::Tool, + content: ProjectedContent::ToolResult(ToolResultContent { + call_id, + name: "Shell".into(), + content: "done".into(), + is_error: false, + image: None, + provider_parts: Vec::new(), + }), + }, + ]; + + normalize_provider_tool_call_ids(&mut history); + + let ProjectedContent::Assistant { calls, .. } = &history[0].content else { + panic!("expected assistant message"); + }; + let ProjectedContent::ToolResult(result) = &history[1].content else { + panic!("expected tool result"); + }; + assert_eq!(calls[0].call_id, expected); + assert_eq!(result.call_id, expected); + } + + #[test] + fn provider_tool_call_id_truncation_counts_unicode_characters() { + let mut history = vec![ProjectedMessage { + message_id: "assistant".into(), + role: Role::Assistant, + content: ProjectedContent::Assistant { + text: String::new(), + thinking: String::new(), + replay_state: None, + calls: vec![ToolCallContent { + index: 0, + call_id: format!("{}界y", "x".repeat(63)), + name: "Read".into(), + arguments: serde_json::json!({}), + }], + }, + }]; + + normalize_provider_tool_call_ids(&mut history); + + let ProjectedContent::Assistant { calls, .. } = &history[0].content else { + panic!("expected assistant message"); + }; + assert_eq!(calls[0].call_id, format!("{}界", "x".repeat(63))); + } +} diff --git a/server/src/provider/mod.rs b/server/src/provider/mod.rs index 2d99398..b9fc7f7 100644 --- a/server/src/provider/mod.rs +++ b/server/src/provider/mod.rs @@ -1,5 +1,6 @@ mod anthropic; mod event; +mod normalize; mod openai_chat; mod openai_responses; mod recorder; diff --git a/server/src/provider/normalize.rs b/server/src/provider/normalize.rs new file mode 100644 index 0000000..ff85711 --- /dev/null +++ b/server/src/provider/normalize.rs @@ -0,0 +1,28 @@ +use std::sync::Arc; + +use tokio_util::sync::CancellationToken; + +use crate::model::{normalize_provider_tool_call_ids, ModelInvocation}; + +use super::{Provider, ProviderStream}; + +pub(super) struct NormalizedProvider { + inner: Arc, +} + +impl NormalizedProvider { + pub(super) fn new(inner: Arc) -> Self { + Self { inner } + } +} + +impl Provider for NormalizedProvider { + fn stream( + &self, + mut invocation: ModelInvocation, + cancellation: CancellationToken, + ) -> ProviderStream { + normalize_provider_tool_call_ids(&mut invocation.request.history); + self.inner.stream(invocation, cancellation) + } +} diff --git a/server/src/provider/router.rs b/server/src/provider/router.rs index 711d38a..097597a 100644 --- a/server/src/provider/router.rs +++ b/server/src/provider/router.rs @@ -12,8 +12,8 @@ use crate::{ }; use super::{ - AnthropicProvider, CallRecorder, OpenAiChatProvider, OpenAiResponsesProvider, Provider, - ProviderStream, + normalize::NormalizedProvider, AnthropicProvider, CallRecorder, OpenAiChatProvider, + OpenAiResponsesProvider, Provider, ProviderStream, }; pub struct ProviderRouter { @@ -160,7 +160,7 @@ fn build_inner( .timeout(config.request_timeout) .build()?, }; - Ok(match config.kind { + let provider: Arc = match config.kind { ProviderKind::OpenAiChat => { Arc::new(OpenAiChatProvider::new(client, config.clone()).with_recorder(recorder)) } @@ -170,5 +170,6 @@ fn build_inner( ProviderKind::Anthropic => { Arc::new(AnthropicProvider::new(client, config.clone()).with_recorder(recorder)) } - }) + }; + Ok(Arc::new(NormalizedProvider::new(provider))) } diff --git a/server/src/run/model_cycle.rs b/server/src/run/model_cycle.rs index 738a11a..33e122e 100644 --- a/server/src/run/model_cycle.rs +++ b/server/src/run/model_cycle.rs @@ -308,7 +308,8 @@ pub async fn consume_model_cycle( usage, )); } - if matches!(finish_reason, FinishReason::ToolUse) != !calls.is_empty() { + let has_tool_calls = !calls.is_empty(); + if matches!(finish_reason, FinishReason::ToolUse) != has_tool_calls { return Err(failure( RunFailure::Protocol("finish reason and tool calls disagree".into()), text, diff --git a/server/src/store/providers.rs b/server/src/store/providers.rs index 442eca5..00876f5 100644 --- a/server/src/store/providers.rs +++ b/server/src/store/providers.rs @@ -908,7 +908,10 @@ mod tests { assert_eq!(listed[0].api_key.as_deref(), Some("secret")); assert!(listed[0].has_api_key); - let without_key = ProviderEndpointInput { api_key: None, ..provider() }; + let without_key = ProviderEndpointInput { + api_key: None, + ..provider() + }; let empty = store.create_provider(&without_key).await.unwrap(); assert_eq!(empty.api_key, None); assert!(!empty.has_api_key); diff --git a/server/tests/prefix_stability.rs b/server/tests/prefix_stability.rs index da3d7e4..f1e7172 100644 --- a/server/tests/prefix_stability.rs +++ b/server/tests/prefix_stability.rs @@ -172,7 +172,7 @@ fn every_prompt_mode_loads_the_captured_tool_set() { "SembleSearch", "SembleFindRelated", ], - "cefa1800d7440611b6c3e922fa59f1fe262eca4cee52ea71043fe1f29e52c659", + "ec10becac85819cda321298762892852194c78601db66cc0b4ce74bc1213e29e", ); assert_mode( &assets, @@ -194,7 +194,7 @@ fn every_prompt_mode_loads_the_captured_tool_set() { "SembleSearch", "SembleFindRelated", ], - "235a2a9a7785844eb5186f1c8f2294a36a04bbf103887d05ec386f8c7cc52abc", + "e2eb8a1ebd70d53b1b2eb6bedabdce62ff070a05a6168216013d0a1144ed8bb5", ); assert_mode( &assets, @@ -218,7 +218,7 @@ fn every_prompt_mode_loads_the_captured_tool_set() { "SembleSearch", "SembleFindRelated", ], - "cefa1800d7440611b6c3e922fa59f1fe262eca4cee52ea71043fe1f29e52c659", + "ec10becac85819cda321298762892852194c78601db66cc0b4ce74bc1213e29e", ); assert_mode( &assets, @@ -244,7 +244,7 @@ fn every_prompt_mode_loads_the_captured_tool_set() { "SembleSearch", "SembleFindRelated", ], - "04c5fb238eb3695936ceed610b481caf7507f934efb88cb7130a8756e12959e3", + "25f7b559941baabfc9b1046455b04ca812fc41a6878ad55a43d83f0bd18cd92f", ); assert_mode( &assets, @@ -272,7 +272,7 @@ fn every_prompt_mode_loads_the_captured_tool_set() { "SembleSearch", "SembleFindRelated", ], - "f88c55fdbb53be377e64cc6280ebb23c75e2d244b5c3463ed18e90752c3b7ff5", + "48c8e0fe825f9c2450307ca5e70cde7077c4282c135b2cd15338bd4bd0c43636", ); assert_mode( &assets, @@ -282,8 +282,24 @@ fn every_prompt_mode_loads_the_captured_tool_set() { ); assert_eq!( schema_digest(&assets.mode(Mode::Agent).tools), - "4324c36fa047fbe4c93a5d5f0b736c559a942e097266c5bae057804289f8b359" + "e53a72c1d131ff3f65c619799232440b064e90e99f5d3fcceb63e32598d3a0fc" ); + let task = assets + .mode(Mode::Agent) + .tools + .iter() + .find(|tool| tool.name == "Task") + .unwrap(); + assert!(task.description.contains( + "When the user does not specify a number, launch at most three subagents in a single response. If the user explicitly requests more, you may launch the requested number." + )); + assert!(task.description.contains( + "If the user explicitly requests parallel subagents, follow the number requested by the user." + )); + assert!(!task + .description + .chars() + .any(|character| ('\u{4e00}'..='\u{9fff}').contains(&character))); let shell = assets .mode(Mode::Agent) .tools