fix: plugin effort compress

This commit is contained in:
leokun
2026-08-31 15:38:02 +08:00
parent 9120b90be7
commit 8bd0d70add
13 changed files with 40 additions and 74 deletions
-2
View File
@@ -235,9 +235,7 @@ export interface PluginModelDescriptor {
description: string | null;
icon: string;
providerType: string;
contextWindowTokens: number | null;
maxOutputTokens: number | null;
thinking: boolean;
images: boolean;
}
@@ -164,7 +164,10 @@ Deno.test("official model discovery excludes hidden models and puts the default
display_name: "GPT First",
supported_in_api: true,
visibility: "list",
supported_reasoning_efforts: ["low", "medium"],
supported_reasoning_levels: [
{ effort: "low", description: "Fast responses" },
{ effort: "medium", description: "Balanced" },
],
},
{ slug: "gpt-second", supported_in_api: true, visibility: "list" },
{ slug: "gpt-hidden", supported_in_api: true, visibility: "hidden" },
@@ -172,7 +175,7 @@ Deno.test("official model discovery excludes hidden models and puts the default
],
});
assertEquals(models.map((model) => model.id), ["gpt-second", "gpt-first"]);
assertEquals(models[1].capabilities, { thinking: true, images: true });
assertEquals(models[1].capabilities, { images: true });
assertEquals(models[1].privateData, { reasoningEfforts: ["low", "medium"] });
});
+6 -7
View File
@@ -24,7 +24,11 @@ function positiveInteger(value: unknown): number | null {
}
function parseReasoningEfforts(model: Record<string, unknown>): string[] {
const source = model.supported_reasoning_efforts ??
const source = model.supported_reasoning_levels ??
model.supportedReasoningLevels ??
model.reasoning_levels ??
model.reasoningLevels ??
model.supported_reasoning_efforts ??
model.supportedReasoningEfforts ??
model.reasoning_efforts ??
model.reasoningEfforts;
@@ -61,10 +65,6 @@ export function parseOfficialModels(body: unknown): ModelDefinition[] {
seen.add(id);
const efforts = parseReasoningEfforts(model);
const description = text(model.description);
const contextWindowTokens = positiveInteger(
model.context_window_tokens ?? model.contextWindowTokens ?? model.context_window ??
model.contextWindow,
);
const maxOutputTokens = positiveInteger(
model.max_output_tokens ?? model.maxOutputTokens ?? model.max_completion_tokens ??
model.maxCompletionTokens,
@@ -74,9 +74,8 @@ export function parseOfficialModels(body: unknown): ModelDefinition[] {
displayName: text(model.display_name ?? model.displayName ?? model.title ?? model.name) ??
id,
...(description ? { description } : {}),
...(contextWindowTokens !== null ? { contextWindowTokens } : {}),
...(maxOutputTokens !== null ? { maxOutputTokens } : {}),
capabilities: { thinking: efforts.length > 0, images: true },
capabilities: { images: true },
privateData: { reasoningEfforts: efforts },
});
}
@@ -160,9 +160,8 @@ Deno.test("model discovery parses both language-models and standard list shapes"
});
assertEquals(richModels.map((model) => model.id), ["grok-4", "grok-3-mini"]);
assertEquals(richModels[0].displayName, "Grok 4");
assertEquals(richModels[0].contextWindowTokens, 256_000);
assertEquals(richModels[0].capabilities, { thinking: false, images: true });
assertEquals(richModels[1].capabilities, { thinking: false, images: false });
assertEquals(richModels[0].capabilities, { images: true });
assertEquals(richModels[1].capabilities, { images: false });
const plainModels = parseGrokModels({ data: [{ id: "grok-4-fast" }] });
assertEquals(plainModels.map((model) => model.id), ["grok-4-fast"]);
+2 -16
View File
@@ -9,12 +9,12 @@ export const FALLBACK_MODELS: ModelDefinition[] = [
{
id: "grok-4.6",
displayName: "Grok 4.6",
capabilities: { thinking: false, images: true },
capabilities: { images: true },
},
{
id: "grok-4.5",
displayName: "Grok 4.5",
capabilities: { thinking: false, images: true },
capabilities: { images: true },
},
];
@@ -28,15 +28,6 @@ function text(value: unknown): string | null {
return typeof value === "string" && value.trim() ? value.trim() : null;
}
function positiveInteger(value: unknown): number | null {
const parsed = typeof value === "number"
? value
: typeof value === "string"
? Number(value)
: NaN;
return Number.isFinite(parsed) && parsed > 0 ? Math.floor(parsed) : null;
}
function modalities(value: unknown): string[] {
return Array.isArray(value)
? value.flatMap((item) => (typeof item === "string" ? [item.toLowerCase()] : []))
@@ -66,15 +57,10 @@ export function parseGrokModels(body: unknown): ModelDefinition[] {
if (!id || seen.has(id)) continue;
seen.add(id);
const inputs = modalities(model?.input_modalities ?? model?.inputModalities);
const contextWindowTokens = positiveInteger(
model?.context_window ?? model?.contextWindow ?? model?.max_prompt_length,
);
models.push({
id,
displayName: displayName(id),
...(contextWindowTokens !== null ? { contextWindowTokens } : {}),
capabilities: {
thinking: false,
images: inputs.length === 0 || inputs.includes("image"),
},
});
-1
View File
@@ -392,7 +392,6 @@ impl ControlService {
if model_hash.starts_with(crate::plugin::ADAPTER_ID_PREFIX) {
let descriptor = self.plugins.model_descriptor(model_hash).await?;
model.display_name = Some(descriptor.display_name);
model.context_window_tokens = descriptor.context_window_tokens;
model.max_output_tokens = Some(descriptor.max_output_tokens.unwrap_or(65_536));
} else {
let configured = self
+6 -11
View File
@@ -580,14 +580,9 @@ fn available_plugin_model(model: &PluginModelDescriptor) -> AvailableModel {
let tooltip = TooltipData {
markdown_content: model.description.clone(),
};
let contexts = context_options(model.context_window_tokens);
let variants = model_variants(
&model.id,
&model.display_name,
&tooltip,
&contexts,
model.thinking,
);
// Effort 与上下文档位由宿主统一提供,与内置模型一致;插件不再声明这两项。
let contexts = context_options(None);
let variants = model_variants(&model.id, &model.display_name, &tooltip, &contexts, true);
let legacy_slugs = variants
.iter()
.filter_map(|variant| variant.legacy_slug.clone())
@@ -598,7 +593,7 @@ fn available_plugin_model(model: &PluginModelDescriptor) -> AvailableModel {
supports_agent: Some(true),
degradation_status: Some(0),
tooltip_data: Some(tooltip.clone()),
supports_thinking: Some(model.thinking),
supports_thinking: Some(true),
supports_images: Some(model.images),
supports_max_mode: Some(false),
client_display_name: Some(model.display_name.clone()),
@@ -610,7 +605,7 @@ fn available_plugin_model(model: &PluginModelDescriptor) -> AvailableModel {
inputbox_short_model_name: Some(model.display_name.clone()),
supports_sandboxing: Some(true),
supports_cmd_k: Some(false),
parameter_definitions: model_parameters(&contexts, model.thinking),
parameter_definitions: model_parameters(&contexts, true),
variants,
legacy_slugs,
named_model_section_index: Some(1),
@@ -633,7 +628,7 @@ fn usable_plugin_model(model: &PluginModelDescriptor) -> agent::ModelDetails {
display_model_id: model.id.clone(),
display_name: model.display_name.clone(),
display_name_short: model.display_name.clone(),
thinking_details: model.thinking.then(agent::ThinkingDetails::default),
thinking_details: Some(agent::ThinkingDetails::default()),
..Default::default()
}
}
-4
View File
@@ -107,9 +107,7 @@ pub struct PluginModelDescriptor {
pub description: Option<String>,
pub icon: String,
pub provider_type: String,
pub context_window_tokens: Option<u64>,
pub max_output_tokens: Option<u64>,
pub thinking: bool,
pub images: bool,
}
@@ -209,9 +207,7 @@ impl PluginModelDescriptor {
description: model.description.clone(),
icon: icon.to_owned(),
provider_type: provider.provider_type.clone(),
context_window_tokens: model.context_window_tokens,
max_output_tokens: model.max_output_tokens,
thinking: model.thinking,
images: model.images,
}
}
-2
View File
@@ -2,7 +2,6 @@ import type { JsonValue, PluginContext } from "./plugin.ts";
import type { ResourceSnapshot } from "./resource.ts";
export type ModelCapabilities = {
thinking?: boolean;
images?: boolean;
};
@@ -10,7 +9,6 @@ export type ModelDefinition = {
id: string;
displayName: string;
description?: string;
contextWindowTokens?: number;
maxOutputTokens?: number;
capabilities?: ModelCapabilities;
/** 之后的调用原样传回;永远不会展示给用户。 */
+3 -12
View File
@@ -135,12 +135,8 @@ pub struct StoredModel {
#[serde(default)]
pub description: Option<String>,
#[serde(default)]
pub context_window_tokens: Option<u64>,
#[serde(default)]
pub max_output_tokens: Option<u64>,
#[serde(default)]
pub thinking: bool,
#[serde(default)]
pub images: bool,
#[serde(default)]
pub private_data: serde_json::Value,
@@ -179,13 +175,9 @@ impl StoredModel {
.get("description")
.and_then(serde_json::Value::as_str)
.map(str::to_owned),
context_window_tokens: object
.get("contextWindowTokens")
.and_then(serde_json::Value::as_u64),
max_output_tokens: object
.get("maxOutputTokens")
.and_then(serde_json::Value::as_u64),
thinking: capability("thinking"),
images: capability("images"),
private_data: object
.get("privateData")
@@ -200,9 +192,8 @@ impl StoredModel {
"id": self.id,
"displayName": self.display_name,
"description": self.description,
"contextWindowTokens": self.context_window_tokens,
"maxOutputTokens": self.max_output_tokens,
"capabilities": { "thinking": self.thinking, "images": self.images },
"capabilities": { "images": self.images },
"privateData": self.private_data,
})
}
@@ -450,7 +441,7 @@ mod tests {
let model = StoredModel::from_definition(&serde_json::json!({
"id": "gpt-test",
"displayName": "GPT Test",
"capabilities": {"thinking": true},
"capabilities": {"images": true},
"privateData": {"reasoningEfforts": ["low"]},
}))
.unwrap();
@@ -460,7 +451,7 @@ mod tests {
.unwrap();
let models = store.models("dev.example", "codex").await.unwrap();
assert_eq!(models.len(), 1);
assert!(models[0].thinking);
assert!(models[0].images);
assert_eq!(models[0].private_data["reasoningEfforts"][0], "low");
}
}
-3
View File
@@ -67,9 +67,6 @@ impl Provider for ProviderRouter {
recorder.request(serde_json::json!({}), &crate::plugin::plugin_llm_request(&invocation)?).await?;
let mut routed = invocation.clone();
routed.request.model.display_name = Some(plan.model.display_name.clone());
if let Some(tokens) = plan.model.context_window_tokens {
routed.request.model.context_window_tokens.get_or_insert(tokens);
}
if let Some(tokens) = plan.model.max_output_tokens {
routed.request.model.max_output_tokens.get_or_insert(tokens);
}
+15 -10
View File
@@ -2,9 +2,7 @@
use std::collections::HashSet;
use crate::model::{
CanonicalMessage, LlmCallUsageAnchor, PreparedRun, ProjectedMessage, RunAction,
};
use crate::model::{CanonicalMessage, LlmCallUsageAnchor, PreparedRun, ProjectedMessage};
const FALLBACK_CHARS: usize = 12_000;
@@ -34,9 +32,6 @@ pub(super) fn should_compact(
projected_messages: &[ProjectedMessage],
anchor: Option<ContextUsageAnchor>,
) -> bool {
if prepared.action != RunAction::Start {
return false;
}
let Some(context_window) = prepared.model.context_window_tokens else {
return false;
};
@@ -113,15 +108,15 @@ fn estimate_serialized_tokens(serialized: &str) -> u64 {
mod tests {
use super::*;
use crate::model::{
project_messages, CheckpointId, ConversationId, ModelSpec, Origin, PromptSpec, Role, RunId,
RunKind,
project_messages, CheckpointId, ConversationId, ModelSpec, Origin, PromptSpec, Role,
RunAction, RunId, RunKind,
};
#[test]
fn automatic_compaction_starts_only_after_the_context_window_is_exceeded() {
fn automatic_compaction_runs_for_start_and_resume_actions_after_the_limit() {
let mut model = ModelSpec::new("model");
model.context_window_tokens = Some(200_000);
let prepared = PreparedRun {
let mut prepared = PreparedRun {
run_id: RunId::new("run"),
cursor_request_id: None,
conversation_id: ConversationId::new("conversation"),
@@ -169,5 +164,15 @@ mod tests {
&projected,
anchor(200_001)
));
prepared.action = RunAction::Resume {
pending_tool_round: None,
};
assert!(should_compact(
&prepared,
&messages,
&projected,
anchor(200_001)
));
}
}
+1 -1
View File
@@ -170,7 +170,7 @@ impl RunEngine {
Ok(messages) => messages,
Err(error) => return (RunOutcome::Failed(error.into()), usage),
};
let context_anchor = if !auto_compacted && prepared.action == RunAction::Start {
let context_anchor = if !auto_compacted {
match self
.store
.latest_llm_call_usage_anchor(