fix(run): make automatic compaction recover an over-limit conversation (#426)

Automatic compaction cannot do its job once a conversation crosses the
context window, so the conversation stays there permanently. Observed
against a 1M-token Anthropic window:

1. The summarize call replays the full history. It runs precisely
   because that history is too large, so the request is itself over the
   limit ("prompt is too long"), or it ends with an assistant/tool
   message that Anthropic refuses as a prefill. Either way the run falls
   back to the 12K truncated JSON summary, which discards the context.
   In one trace the summarizer received 771 messages (2.78 MB) and
   returned a single token.

2. The compaction check uses a 10K fixed reserve. The estimate trails
   the provider's own count by the request context and provider-side
   overhead that the message-tail estimate does not model; a 948K
   estimate passed the check and Anthropic counted 1,017,628.

3. When the provider does refuse the prompt, the run retries the same
   prompt eight times at 5s intervals and then fails. Nothing compacts.

Fixes, all in server/src/run:

- compaction_history trims the summarizer input to the context budget
  at user-turn boundaries (never splitting a tool call from its
  results) and guarantees it ends with a user message.
- context_budget keeps 10% of the window free instead of a fixed 10K,
  so the reserve scales with the model and absorbs the drift.
- A provider refusal matching is_context_overflow compacts once and
  retries the turn instead of failing it.
- 4xx responses other than 408/425/429 are terminal. A rejected request
  fails identically every time, so retrying only delays the error.

Separately, Cursor can resume a finished turn whose checkpoint already
ends with the assistant, which Anthropic also rejects as a prefill.
run/history.rs appends a transient user tail to every provider request
that would otherwise end with the assistant. The tail is never
persisted, so committed checkpoints stay an exact prefix of the next
turn and the usage anchor still counts persisted messages only.
This commit is contained in:
The Gru
2026-09-06 20:33:03 +08:00
committed by GitHub
parent 9637efd795
commit fdae9c41c7
6 changed files with 663 additions and 23 deletions
+81 -4
View File
@@ -25,6 +25,12 @@ pub struct RunEngine {
provider: Arc<dyn Provider>,
}
/// Provider-visible tail appended when the committed history ends with the
/// assistant, which happens when Cursor resumes a turn that had already
/// finished (for example after a cancelled follow-up).
const CONTINUE_MESSAGE_ID: &str = "runtime:continue";
const CONTINUE_INSTRUCTION: &str = "Continue from where you left off.";
impl RunEngine {
pub fn new(store: Store, provider: Arc<dyn Provider>) -> Self {
Self { store, provider }
@@ -171,6 +177,10 @@ impl RunEngine {
};
}
// A provider refusal for an over-limit prompt triggers one compaction
// per run. A second refusal after compacting means the current input
// itself does not fit, and compacting again would only destroy history.
let mut overflow_compacted = false;
'model: loop {
if cancellation.is_cancelled() {
return (RunOutcome::Cancelled, usage);
@@ -226,6 +236,18 @@ impl RunEngine {
if let Err(error) = hydrate_tool_images(&self.store, &mut history).await {
return (RunOutcome::Failed(error.into()), usage);
}
// The usage anchor counts persisted messages only; a transient
// tail is provider-visible but never committed.
let anchored_messages = history.len();
let history = if prepared.action == RunAction::Compact {
super::history::user_terminated(
history,
"compaction:instruction",
super::compaction::INSTRUCTIONS,
)
} else {
super::history::user_terminated(history, CONTINUE_MESSAGE_ID, CONTINUE_INSTRUCTION)
};
let request = crate::model::ModelRequest {
prompt: prepared.prompt.clone(),
model: prepared.model.clone(),
@@ -296,7 +318,7 @@ impl RunEngine {
update_context_usage_anchor(
&mut context_usage_anchor,
cycle_usage,
request.history.len(),
anchored_messages,
);
accumulate_usage(&mut usage, cycle_usage);
}
@@ -306,7 +328,7 @@ impl RunEngine {
update_context_usage_anchor(
&mut context_usage_anchor,
cycle_usage,
request.history.len(),
anchored_messages,
);
accumulate_usage(&mut usage, cycle_usage);
}
@@ -353,7 +375,7 @@ impl RunEngine {
update_context_usage_anchor(
&mut context_usage_anchor,
cycle_usage,
request.history.len(),
anchored_messages,
);
accumulate_usage(&mut usage, cycle_usage);
}
@@ -361,6 +383,58 @@ impl RunEngine {
let _ = emit(client, RunEvent::CycleInterrupted).await;
return (RunOutcome::Cancelled, usage);
}
// The estimate that cleared the compaction check can
// still land over the real limit: it trails the
// provider's own count by whatever the request adds
// after the anchor was taken. The provider is the
// authority, so treat its refusal as the trigger the
// estimate missed. Without this a conversation that
// crosses the line is wedged: every retry rebuilds the
// same prompt and gets the same refusal.
if !overflow_compacted
&& prepared.action != RunAction::Compact
&& matches!(&cycle_failure.failure, RunFailure::Provider(message)
if super::compaction::is_context_overflow(message))
{
tracing::warn!(
provider_call_index,
checkpoint_id = checkpoint.0,
"provider rejected the prompt as over-limit; compacting and retrying"
);
overflow_compacted = true;
checkpoint = match super::messages::append_batches(
&self.store,
prepared,
client,
cancellation,
checkpoint,
std::mem::take(&mut pending_insertions),
)
.await
{
Ok((checkpoint, _)) => checkpoint,
Err(outcome) => return (outcome, usage),
};
let messages =
match self.store.load_checkpoint_messages(checkpoint).await {
Ok(messages) => messages,
Err(error) => return (RunOutcome::Failed(error.into()), usage),
};
match self
.auto_compact(prepared, checkpoint, &messages, client, cancellation)
.await
{
Ok((next_checkpoint, compaction_usage)) => {
checkpoint = next_checkpoint;
context_usage_anchor = None;
if let Some(compaction_usage) = compaction_usage {
accumulate_usage(&mut usage, compaction_usage);
}
continue 'model;
}
Err(outcome) => return (outcome, usage),
}
}
if !should_retry(&cycle_failure, retries) {
return (RunOutcome::Failed(cycle_failure.failure), usage);
}
@@ -459,7 +533,7 @@ impl RunEngine {
update_context_usage_anchor(
&mut context_usage_anchor,
cycle_usage,
request.history.len(),
anchored_messages,
);
accumulate_usage(&mut usage, cycle_usage);
}
@@ -721,6 +795,9 @@ impl RunEngine {
.await
.map_err(|error| RunOutcome::Failed(error.into()))?;
let history = crate::model::project_messages(&compactable)
.map(|history| {
super::compaction::compaction_history(history, prepared.model.context_window_tokens)
})
.map_err(|error| RunOutcome::Failed(error.into()))?;
let mut model = prepared.model.clone();
model.max_output_tokens = Some(super::compaction::OUTPUT_TOKENS);