返回 CodeWhale
context.rs
根目录 / crates / tui / src / core / engine / context.rs
1 //! Context budgeting and prompt-shaping helpers for the engine.
2 //!
3 //! These functions are shared by the streaming turn loop, capacity flow, and
4 //! engine session maintenance code. Keeping them here prevents the top-level
5 //! engine module from accumulating unrelated context-policy details.
6
7 use crate::config::ProviderKind;
8 use crate::context_budget::ContextBudget;
9 #[cfg(test)]
10 pub(super) use crate::route_budget::effective_max_output_tokens;
11 pub(super) use crate::route_budget::effective_max_output_tokens_for_route;
12 use crate::tools::spec::ToolResult;
13 use codewhale_config::route::RouteLimits;
14 use codewhale_models::SystemPrompt;
15 use serde_json::Value;
16 /// Allow a few emergency recovery attempts before failing the turn.
17 pub(super) const MAX_CONTEXT_RECOVERY_ATTEMPTS: u8 = 2;
18 /// Max chars to keep from metadata-provided output summaries.
19 const TOOL_RESULT_METADATA_SUMMARY_CHARS: usize = 320;
20
21 #[cfg(test)]
22 pub(super) use crate::compaction::COMPACTION_SUMMARY_MARKER;
23
24 /// What the model sees of one tool result (#6508).
25 #[derive(Debug, Clone, PartialEq, Eq)]
26 pub(crate) struct ToolResultContextView {
27 pub(crate) text: String,
28 /// The view leaves out bytes of the result and no session artifact holds
29 /// them yet. The engine saves the full output before the result fans out,
30 /// so the view it builds afterwards can name a ref instead.
31 pub(crate) needs_full_output_artifact: bool,
32 }
33
34 impl ToolResultContextView {
35 fn whole(text: impl Into<String>) -> Self {
36 Self {
37 text: text.into(),
38 needs_full_output_artifact: false,
39 }
40 }
41 }
42
43 /// A structured summary and whether it left anything of the raw result out.
44 struct ContextSummary {
45 text: String,
46 lossy: bool,
47 }
48
49 /// Where the full output of a tool call can be read back, from its metadata.
50 struct RecoveryRef<'a> {
51 path: &'a str,
52 /// The `art_<id>` ref `retrieve_tool_result` resolves. `None` for a
53 /// legacy spill, which no tool call reaches.
54 artifact_id: Option<&'a str>,
55 }
56
57 fn recovery_ref(metadata: Option<&Value>) -> Option<RecoveryRef<'_>> {
58 let obj = metadata?.as_object()?;
59 let text = |key: &str| {
60 obj.get(key)
61 .and_then(Value::as_str)
62 .filter(|s| !s.is_empty())
63 };
64 if let Some(artifact_id) = text("artifact_id").filter(|id| id.starts_with("art_")) {
65 let path = text("artifact_path").or_else(|| text("spillover_path"))?;
66 return Some(RecoveryRef {
67 path,
68 artifact_id: Some(artifact_id),
69 });
70 }
71 text("spillover_path").map(|path| RecoveryRef {
72 path,
73 artifact_id: None,
74 })
75 }
76
77 pub(super) fn summarize_text(text: &str, limit: usize) -> String {
78 if text.chars().count() <= limit {
79 return text.to_string();
80 }
81 let take = limit.saturating_sub(3);
82 let mut out: String = text.chars().take(take).collect();
83 out.push_str("...");
84 out
85 }
86
87 /// [`summarize_text`] that records whether it cut anything.
88 fn keep_text(text: &str, limit: usize, lossy: &mut bool) -> String {
89 let kept = summarize_text(text, limit);
90 *lossy |= kept.len() != text.len();
91 kept
92 }
93
94 /// [`summarize_text_head_tail`] that records whether it cut anything.
95 fn keep_head_tail(text: &str, limit: usize, lossy: &mut bool) -> String {
96 let kept = summarize_text_head_tail(text, limit);
97 *lossy |= kept.len() != text.len();
98 kept
99 }
100
101 fn summarize_text_head_tail(text: &str, limit: usize) -> String {
102 let total = text.chars().count();
103 if total <= limit {
104 return text.to_string();
105 }
106 if limit <= 20 {
107 return summarize_text(text, limit);
108 }
109
110 let marker = "\n\n[... output truncated for context ...]\n\n";
111 let marker_len = marker.chars().count();
112 if limit <= marker_len + 20 {
113 return summarize_text(text, limit);
114 }
115
116 let remaining = limit - marker_len;
117 let head_len = remaining.saturating_mul(2) / 3;
118 let tail_len = remaining.saturating_sub(head_len);
119 let head: String = text.chars().take(head_len).collect();
120 let tail_vec: Vec<char> = text.chars().rev().take(tail_len).collect();
121 let tail: String = tail_vec.into_iter().rev().collect();
122 format!("{head}{marker}{tail}")
123 }
124
125 fn tool_result_metadata_summary(metadata: Option<&serde_json::Value>) -> Option<String> {
126 let obj = metadata?.as_object()?;
127 for key in ["summary", "stdout_summary", "stderr_summary", "message"] {
128 if let Some(text) = obj.get(key).and_then(serde_json::Value::as_str) {
129 let trimmed = text.trim();
130 if !trimmed.is_empty() {
131 return Some(summarize_text(trimmed, TOOL_RESULT_METADATA_SUMMARY_CHARS));
132 }
133 }
134 }
135 None
136 }
137
138 fn summarize_subagent_status(status: &serde_json::Value, lossy: &mut bool) -> String {
139 if let Some(raw) = status.as_str() {
140 return raw.to_string();
141 }
142 if let Some(obj) = status.as_object()
143 && let Some((kind, value)) = obj.iter().next()
144 {
145 if let Some(reason) = value.as_str().filter(|s| !s.trim().is_empty()) {
146 return format!("{kind}({})", keep_text(reason.trim(), 120, lossy));
147 }
148 return kind.to_string();
149 }
150 status.to_string()
151 }
152
153 fn summarize_subagent_snapshot(
154 snapshot: &serde_json::Value,
155 index: usize,
156 result_limit: usize,
157 lossy: &mut bool,
158 ) -> String {
159 if let Some(inner) = snapshot.get("snapshot") {
160 return summarize_subagent_snapshot(inner, index, result_limit, lossy);
161 }
162
163 let Some(obj) = snapshot.as_object() else {
164 return format!(
165 "- item {index}: {}",
166 keep_text(&snapshot.to_string(), result_limit, lossy)
167 );
168 };
169
170 let agent_id = obj
171 .get("agent_id")
172 .and_then(serde_json::Value::as_str)
173 .unwrap_or("unknown");
174 let agent_type = obj
175 .get("agent_type")
176 .and_then(serde_json::Value::as_str)
177 .unwrap_or("agent");
178 let status = obj
179 .get("status")
180 .map(|status| summarize_subagent_status(status, lossy))
181 .unwrap_or_else(|| "unknown".to_string());
182 let objective = obj
183 .get("assignment")
184 .and_then(|assignment| assignment.get("objective"))
185 .and_then(serde_json::Value::as_str)
186 .map(str::trim)
187 .filter(|s| !s.is_empty())
188 .map(|s| keep_text(s, 220, lossy));
189 let result = obj
190 .get("result")
191 .and_then(serde_json::Value::as_str)
192 .map(str::trim)
193 .filter(|s| !s.is_empty())
194 .map(|s| keep_text(s, result_limit, lossy));
195 let steps = obj.get("steps_taken").and_then(serde_json::Value::as_u64);
196 let duration_ms = obj.get("duration_ms").and_then(serde_json::Value::as_u64);
197
198 let mut lines = vec![format!("- {agent_id} ({agent_type}) status={status}")];
199 if let Some(objective) = objective {
200 lines.push(format!(" objective: {objective}"));
201 }
202 match result {
203 Some(result) => lines.push(format!(" result: {result}")),
204 None => lines.push(" result: not available yet".to_string()),
205 }
206 if steps.is_some() || duration_ms.is_some() {
207 let steps = steps
208 .map(|n| n.to_string())
209 .unwrap_or_else(|| "?".to_string());
210 let duration_ms = duration_ms
211 .map(|n| n.to_string())
212 .unwrap_or_else(|| "?".to_string());
213 lines.push(format!(" stats: steps={steps}, duration_ms={duration_ms}"));
214 }
215 lines.join("\n")
216 }
217
218 /// Guidance heading every summarized sub-agent result. It names only tools
219 /// the model can call: `read` and `bash` are eager, and `handle_read` is
220 /// deferred, so it carries its activation path (#6747).
221 pub(crate) fn subagent_summary_guidance() -> String {
222 format!(
223 "Child results are self-reports; verify side effects with `read` (or `bash`, e.g. `git status`, where available) before claiming success.\n\
224 Use `handle_read` on `transcript_handle` for bounded transcript slices when the returned summary is not enough; {}.\n",
225 crate::tools::handle::HANDLE_READ_ACTIVATION_HINT
226 )
227 }
228
229 /// A payload is a sub-agent snapshot when it carries the identity/status shape
230 /// this summarizer knows how to render (`agent_id`/`agent_type`, optionally
231 /// wrapped in a `snapshot` field).
232 fn looks_like_subagent_snapshot(value: &serde_json::Value) -> bool {
233 let value = value.get("snapshot").unwrap_or(value);
234 value
235 .as_object()
236 .is_some_and(|obj| obj.contains_key("agent_id") || obj.contains_key("agent_type"))
237 }
238
239 /// Sub-agent snapshots shown in full before the rest are only counted.
240 const SUBAGENT_SNAPSHOTS_SHOWN: usize = 8;
241 /// Floor for one sub-agent's result text on the smallest routes.
242 const SUBAGENT_RESULT_MIN_CHARS: usize = 1_600;
243
244 fn compact_subagent_tool_result_for_context(
245 tool_name: &str,
246 raw: &str,
247 budget: usize,
248 ) -> Option<ContextSummary> {
249 if tool_name != "agent" {
250 return None;
251 }
252
253 let parsed: serde_json::Value = serde_json::from_str(raw).ok()?;
254 let snapshots: Vec<&serde_json::Value> = match &parsed {
255 serde_json::Value::Array(items) => items.iter().collect(),
256 serde_json::Value::Object(_) => vec![&parsed],
257 _ => return None,
258 };
259
260 // Coordination envelopes (`wait`, `status`, `claim`, ...) carry typed
261 // fields the parent needs verbatim: `settled`, `still_running`,
262 // `timed_out`, `waited_ms`, `note`. Projecting them through the snapshot
263 // renderer replaced every one with `unknown (agent) status=unknown` and
264 // dropped the real payload. Summarize only snapshot-shaped results; let
265 // anything else fall through to the generic bounded path.
266 if snapshots.is_empty()
267 || !snapshots
268 .iter()
269 .all(|value| looks_like_subagent_snapshot(value))
270 {
271 return None;
272 }
273
274 // Each shown result gets an equal share of half the route budget, so a
275 // child's report is not cut at a fixed size on a large route.
276 let shown = snapshots.len().min(SUBAGENT_SNAPSHOTS_SHOWN);
277 let result_limit = (budget / 2 / shown.max(1)).max(SUBAGENT_RESULT_MIN_CHARS);
278 let mut lossy = snapshots.len() > SUBAGENT_SNAPSHOTS_SHOWN;
279 let mut out = String::from("[sub-agent result summarized for parent context]\n");
280 out.push_str(&subagent_summary_guidance());
281 for (idx, snapshot) in snapshots.iter().enumerate() {
282 if idx >= SUBAGENT_SNAPSHOTS_SHOWN {
283 out.push_str(&format!(
284 "- ... {} more sub-agent result(s) omitted from context summary\n",
285 snapshots.len().saturating_sub(idx)
286 ));
287 break;
288 }
289 out.push_str(&summarize_subagent_snapshot(
290 snapshot,
291 idx + 1,
292 result_limit,
293 &mut lossy,
294 ));
295 out.push('\n');
296 }
297 Some(ContextSummary {
298 text: out.trim_end().to_string(),
299 lossy,
300 })
301 }
302
303 fn json_text<'a>(value: &'a Value, key: &str) -> Option<&'a str> {
304 value
305 .get(key)
306 .and_then(Value::as_str)
307 .map(str::trim)
308 .filter(|s| !s.is_empty())
309 }
310
311 fn json_number_text(value: &Value, key: &str) -> Option<String> {
312 value
313 .get(key)
314 .and_then(|value| {
315 value
316 .as_i64()
317 .map(|n| n.to_string())
318 .or_else(|| value.as_u64().map(|n| n.to_string()))
319 })
320 .or_else(|| {
321 value
322 .get(key)
323 .and_then(Value::as_str)
324 .map(str::trim)
325 .filter(|s| !s.is_empty())
326 .map(ToString::to_string)
327 })
328 }
329
330 /// Characters a structured summary keeps for its own header lines and the
331 /// recovery footer, before the rest of the budget goes to output streams.
332 const STRUCTURED_SUMMARY_FRAME_CHARS: usize = 1_000;
333 /// Floor for one stream or gate detail on the smallest routes.
334 const STRUCTURED_DETAIL_MIN_CHARS: usize = 600;
335
336 fn compact_run_tests_result_for_context(
337 raw: &str,
338 metadata: Option<&Value>,
339 budget: usize,
340 ) -> Option<ContextSummary> {
341 let parsed: Value = serde_json::from_str(raw).ok()?;
342 let success = parsed.get("success")?.as_bool()?;
343 let exit_code = json_number_text(&parsed, "exit_code").unwrap_or_else(|| "?".to_string());
344 let command = json_text(&parsed, "command").unwrap_or("(unknown command)");
345 let stdout = json_text(&parsed, "stdout");
346 let stderr = json_text(&parsed, "stderr");
347 let mut lossy = false;
348
349 let mut lines = vec![
350 "[run_tests result summarized for context]".to_string(),
351 format!(
352 "status: {}, exit_code: {exit_code}",
353 if success { "passed" } else { "failed" }
354 ),
355 format!("command: {}", keep_text(command, 300, &mut lossy)),
356 ];
357 // The cargo failure summary names the failing tests. It leads, so the
358 // names stay inline however long the streams are.
359 if let Some(summary) = metadata.and_then(|metadata| json_text(metadata, "summary")) {
360 lines.push(format!("failure summary: {summary}"));
361 }
362 let header_chars: usize = lines.iter().map(|line| line.chars().count() + 1).sum();
363 let streams = usize::from(stderr.is_some()) + usize::from(stdout.is_some());
364 let stream_limit = (budget.saturating_sub(header_chars + STRUCTURED_SUMMARY_FRAME_CHARS)
365 / streams.max(1))
366 .max(STRUCTURED_DETAIL_MIN_CHARS);
367 if let Some(stderr) = stderr {
368 lines.push(format!(
369 "stderr: {}",
370 keep_head_tail(stderr, stream_limit, &mut lossy)
371 ));
372 }
373 if let Some(stdout) = stdout {
374 lines.push(format!(
375 "stdout: {}",
376 keep_head_tail(stdout, stream_limit, &mut lossy)
377 ));
378 }
379 Some(ContextSummary {
380 text: lines.join("\n"),
381 lossy,
382 })
383 }
384
385 fn run_verifier_status_rank(status: Option<&str>) -> u8 {
386 match status.unwrap_or_default() {
387 "failed" | "timeout" => 0,
388 "skipped" => 1,
389 "passed" => 2,
390 _ => 3,
391 }
392 }
393
394 /// Gates listed one per line before the rest are only counted.
395 const VERIFIER_GATES_SHOWN: usize = 12;
396
397 fn compact_run_verifiers_result_for_context(raw: &str, budget: usize) -> Option<ContextSummary> {
398 let parsed: Value = serde_json::from_str(raw).ok()?;
399 let gates = parsed.get("gates")?.as_array()?;
400 let summary = json_text(&parsed, "summary")
401 .map(ToString::to_string)
402 .unwrap_or_else(|| {
403 let passed = json_number_text(&parsed, "passed").unwrap_or_else(|| "?".to_string());
404 let failed = json_number_text(&parsed, "failed").unwrap_or_else(|| "?".to_string());
405 let skipped = json_number_text(&parsed, "skipped").unwrap_or_else(|| "?".to_string());
406 format!("{passed} passed, {failed} failed, {skipped} skipped")
407 });
408
409 let mut ordered: Vec<&Value> = gates.iter().collect();
410 ordered.sort_by(|a, b| {
411 run_verifier_status_rank(json_text(a, "status"))
412 .cmp(&run_verifier_status_rank(json_text(b, "status")))
413 .then_with(|| json_text(a, "name").cmp(&json_text(b, "name")))
414 });
415
416 let mut lossy = ordered.len() > VERIFIER_GATES_SHOWN;
417 let mut lines = vec![
418 "[run_verifiers result summarized for context]".to_string(),
419 format!("summary: {summary}"),
420 ];
421 let profile = json_text(&parsed, "profile");
422 let level = json_text(&parsed, "level");
423 if profile.is_some() || level.is_some() {
424 lines.push(format!(
425 "selection: profile={}, level={}",
426 profile.unwrap_or("?"),
427 level.unwrap_or("?")
428 ));
429 }
430
431 let shown = ordered.iter().take(VERIFIER_GATES_SHOWN);
432 let detailed = shown
433 .clone()
434 .filter(|gate| json_text(gate, "status") != Some("passed"))
435 .count();
436 let detail_limit = (budget.saturating_sub(STRUCTURED_SUMMARY_FRAME_CHARS * 2)
437 / detailed.max(1))
438 .max(STRUCTURED_DETAIL_MIN_CHARS);
439
440 for gate in shown {
441 let name = json_text(gate, "name").unwrap_or("gate");
442 let ecosystem = json_text(gate, "ecosystem").unwrap_or("unknown");
443 let status = json_text(gate, "status").unwrap_or("unknown");
444 let exit = json_number_text(gate, "exit_code")
445 .map(|code| format!(" exit={code}"))
446 .unwrap_or_default();
447 lines.push(format!("- {name} ({ecosystem}): {status}{exit}"));
448 // A stream longer than the verifier keeps in memory is saved whole.
449 for stream in ["stdout", "stderr"] {
450 if let Some(reference) = json_text(gate, &format!("{stream}_log_ref")) {
451 lines.push(format!(
452 " full {stream}: retrieve_tool_result ref=\"{reference}\""
453 ));
454 }
455 }
456
457 let stdout = json_text(gate, "stdout");
458 let stderr = json_text(gate, "stderr");
459 if status == "passed" {
460 // A passing gate's output stays out of context; the saved full
461 // output keeps it readable.
462 lossy |= stdout.is_some() || stderr.is_some();
463 continue;
464 }
465 if let Some(command) = json_text(gate, "command") {
466 lines.push(format!(
467 " command: {}",
468 keep_text(command, 240, &mut lossy)
469 ));
470 }
471 // One detail per gate: the skip reason, else stderr, else stdout.
472 let (detail, other_streams) = match json_text(gate, "skipped_reason") {
473 Some(reason) => (Some(reason), stderr.is_some() || stdout.is_some()),
474 None => match stderr {
475 Some(stderr) => (Some(stderr), stdout.is_some()),
476 None => (stdout, false),
477 },
478 };
479 lossy |= other_streams;
480 if let Some(detail) = detail {
481 lines.push(format!(
482 " detail: {}",
483 keep_head_tail(detail, detail_limit, &mut lossy)
484 ));
485 }
486 }
487 if ordered.len() > VERIFIER_GATES_SHOWN {
488 lines.push(format!(
489 "- ... {} more gate(s) omitted from context summary",
490 ordered.len() - VERIFIER_GATES_SHOWN
491 ));
492 }
493
494 Some(ContextSummary {
495 text: lines.join("\n"),
496 lossy,
497 })
498 }
499
500 fn compact_task_gate_run_result_for_context(raw: &str) -> Option<ContextSummary> {
501 let parsed: Value = serde_json::from_str(raw).ok()?;
502 let gate = parsed.get("gate")?;
503 let gate_name = json_text(gate, "gate").unwrap_or("gate");
504 let status = json_text(gate, "status").unwrap_or("unknown");
505 let command = json_text(gate, "command").unwrap_or("(unknown command)");
506 let summary = json_text(gate, "summary")
507 .or_else(|| json_text(&parsed, "stderr_summary"))
508 .or_else(|| json_text(&parsed, "stdout_summary"));
509 let exit = json_number_text(gate, "exit_code")
510 .map(|code| format!(", exit_code: {code}"))
511 .unwrap_or_default();
512 let mut lossy = false;
513
514 let mut lines = vec![
515 "[task_gate_run result summarized for context]".to_string(),
516 format!("gate: {gate_name}, status: {status}{exit}"),
517 format!("command: {}", keep_text(command, 300, &mut lossy)),
518 ];
519 if let Some(summary) = summary {
520 lines.push(format!("summary: {}", keep_text(summary, 800, &mut lossy)));
521 }
522 if let Some(log_path) = json_text(gate, "log_path") {
523 lines.push(format!("log_path: {log_path}"));
524 }
525 Some(ContextSummary {
526 text: lines.join("\n"),
527 lossy,
528 })
529 }
530
531 fn compact_structured_tool_result_for_context(
532 tool_name: &str,
533 raw: &str,
534 metadata: Option<&Value>,
535 budget: usize,
536 ) -> Option<ContextSummary> {
537 match tool_name {
538 "run_tests" => compact_run_tests_result_for_context(raw, metadata, budget),
539 "run_verifiers" => compact_run_verifiers_result_for_context(raw, budget),
540 // `tasks` is the unified durable-task tool (piagent phase B); its
541 // gate_run action emits the same gate payload as the legacy
542 // `task_gate_run` alias. The compactor returns None unless the
543 // content actually parses as a gate result, so non-gate `tasks`
544 // results fall through to the generic path unchanged.
545 "task_gate_run" | "tasks" => compact_task_gate_run_result_for_context(raw),
546 _ => None,
547 }
548 }
549
550 #[cfg(test)]
551 pub(crate) fn compact_tool_result_for_context(
552 model: &str,
553 tool_name: &str,
554 output: &ToolResult,
555 ) -> String {
556 compact_tool_result_for_route(ProviderKind::Deepseek, model, None, tool_name, output)
557 }
558
559 pub(crate) fn compact_tool_result_for_route(
560 provider: ProviderKind,
561 model: &str,
562 route_limits: Option<RouteLimits>,
563 tool_name: &str,
564 output: &ToolResult,
565 ) -> String {
566 tool_result_context_view(provider, model, route_limits, tool_name, output).text
567 }
568
569 /// The model's view of one tool result (#6508).
570 ///
571 /// There is one size authority: [`crate::route_budget::route_inline_char_budget`].
572 /// A result within it reaches the model whole, whatever the tool. A larger
573 /// one is cut to a head and tail around a footer that names where the full
574 /// output lives and the `art_<id>` ref `retrieve_tool_result` reads it back
575 /// with. The view never writes anything: when it would leave bytes out and no
576 /// artifact holds them, it says so through `needs_full_output_artifact`, and
577 /// the engine saves the full output and builds the view again.
578 ///
579 /// Structured summaries (`run_tests`, `run_verifiers`, gate results,
580 /// sub-agent snapshots) keep their shape, scale their detail with the same
581 /// budget, and follow the same rule whenever they leave something out.
582 pub(crate) fn tool_result_context_view(
583 provider: ProviderKind,
584 model: &str,
585 route_limits: Option<RouteLimits>,
586 tool_name: &str,
587 output: &ToolResult,
588 ) -> ToolResultContextView {
589 let raw = output.content.trim();
590 if raw.is_empty() {
591 return ToolResultContextView::whole(String::new());
592 }
593 let metadata = output.metadata.as_ref();
594
595 // A result already bounded by the adaptive evidence envelope is an
596 // honest, context-sized preview whose footer names the artifact path and
597 // a recovery instruction. Re-compacting it would strip that recovery
598 // contract and double-truncate the output, so pass it through unchanged.
599 if metadata
600 .and_then(|metadata| metadata.get("evidence_available"))
601 .and_then(Value::as_bool)
602 .unwrap_or(false)
603 {
604 return ToolResultContextView::whole(raw);
605 }
606
607 // The `read` primitive already bounds itself to an explicit per-call byte
608 // budget and, when that budget truncates the file, ends with a footer
609 // naming the exact offset to continue from. Compacting it a second time
610 // would drop content the caller deliberately budgeted for *and* delete the
611 // continuation contract. A result that stayed inside its declared budget
612 // therefore passes through; one that exceeded it takes the path below.
613 if metadata
614 .and_then(|metadata| metadata.get("read_budget_bytes"))
615 .and_then(Value::as_u64)
616 .is_some_and(|budget| raw.len() as u64 <= budget)
617 {
618 return ToolResultContextView::whole(raw);
619 }
620
621 let budget =
622 crate::route_budget::route_inline_char_budget_for_route(provider, model, route_limits);
623 let recovery = recovery_ref(metadata);
624 let recovery_path = recovery.as_ref().map(|recovery| recovery.path);
625 let retrieval_ref = recovery.as_ref().and_then(|recovery| recovery.artifact_id);
626
627 let summary = compact_subagent_tool_result_for_context(tool_name, raw, budget)
628 .or_else(|| compact_structured_tool_result_for_context(tool_name, raw, metadata, budget));
629 if let Some(summary) = summary {
630 return summary_view(summary, budget, recovery.as_ref());
631 }
632
633 // Spillover already saved the full output and left a preview. Re-fit that
634 // preview to the budget and keep its ref; never save the preview itself.
635 if let Some(metadata) =
636 metadata.filter(|metadata| metadata.get("retained_head_bytes").is_some())
637 && let Some(text) = crate::tools::truncate::refit_spilled_preview(
638 &output.content,
639 metadata,
640 budget,
641 recovery_path,
642 retrieval_ref,
643 )
644 {
645 return ToolResultContextView::whole(text.trim().to_string());
646 }
647
648 if raw.chars().count() <= budget {
649 return ToolResultContextView::whole(raw);
650 }
651
652 let lead = tool_result_metadata_summary(metadata)
653 .map(|summary| format!("Summary: {summary}\n"))
654 .unwrap_or_default();
655 let body = crate::tools::truncate::fit_to_inline_budget(
656 raw,
657 budget.saturating_sub(lead.chars().count()),
658 recovery_path,
659 retrieval_ref,
660 );
661 ToolResultContextView {
662 text: format!("{lead}{body}"),
663 needs_full_output_artifact: recovery.is_none(),
664 }
665 }
666
667 /// Finish a structured summary: when it left anything out, end it with the
668 /// same recovery instruction every other cut carries.
669 fn summary_view(
670 summary: ContextSummary,
671 budget: usize,
672 recovery: Option<&RecoveryRef<'_>>,
673 ) -> ToolResultContextView {
674 if !summary.lossy {
675 return ToolResultContextView::whole(summary.text);
676 }
677 let recovery_path = recovery.map(|recovery| recovery.path);
678 let retrieval_ref = recovery.and_then(|recovery| recovery.artifact_id);
679 let footer = match recovery_path {
680 Some(path) => format!(
681 "[full output at {path}; {}]",
682 crate::tools::truncate::spillover_recovery_instruction(retrieval_ref)
683 ),
684 None => format!(
685 "[the full output could not be saved; {}]",
686 crate::tools::truncate::spillover_recovery_instruction(None)
687 ),
688 };
689 let text = crate::tools::truncate::fit_to_inline_budget(
690 &summary.text,
691 budget.saturating_sub(footer.chars().count() + 1),
692 recovery_path,
693 retrieval_ref,
694 );
695 ToolResultContextView {
696 text: format!("{text}\n{footer}"),
697 needs_full_output_artifact: recovery.is_none(),
698 }
699 }
700
701 pub(super) fn extract_compaction_summary_prompt(
702 prompt: Option<SystemPrompt>,
703 ) -> Option<SystemPrompt> {
704 crate::compaction::extract_compaction_summary(prompt.as_ref())
705 }
706
707 /// Internal input-side token budget for a provider/model route:
708 /// `window - reserved_output - headroom`. Used by the preflight check,
709 /// emergency recovery, and capacity trimming to decide when to compact.
710 /// Unknown model ids fall back to the provider's conservative default instead
711 /// of disabling preflight; custom long-context deployments can still advertise
712 /// their window with a `-256k`/`-1024k` model suffix.
713 ///
714 /// The reserved-output term is the route-effective request cap: exactly what
715 /// the API can receive after explicit overrides, compatibility/route ceilings,
716 /// and the route window are intersected. A second hidden reasoning reserve
717 /// would make preflight disagree with the wire request and can cause premature
718 /// compaction on otherwise valid large-window inputs.
719 #[cfg(test)]
720 pub(super) fn context_input_budget_for_provider(
721 provider: ProviderKind,
722 model: &str,
723 ) -> Option<usize> {
724 context_input_budget_for_route(provider, model, None, 0)
725 }
726
727 /// Public so external callers (e.g. a host/bridge deriving its own compaction
728 /// trigger line) can reuse the *exact* same internal input-budget math — window
729 /// minus the route-effective output reservation
730 /// (`route_output_reservation`) minus headroom —
731 /// instead of re-deriving those constants and silently drifting from the engine.
732 /// Pass `input_tokens = 0` to get the full emergency input budget for the route.
733 pub fn context_input_budget_for_route(
734 provider: ProviderKind,
735 model: &str,
736 route_limits: Option<RouteLimits>,
737 input_tokens: usize,
738 ) -> Option<usize> {
739 route_context_budget_for_route(provider, model, route_limits, input_tokens)
740 .and_then(|budget| usize::try_from(budget.available_input_tokens).ok())
741 }
742
743 #[cfg(test)]
744 pub(super) fn route_context_budget_for_provider(
745 provider: ProviderKind,
746 model: &str,
747 input_tokens: usize,
748 ) -> Option<ContextBudget> {
749 route_context_budget_for_route(provider, model, None, input_tokens)
750 }
751
752 pub(super) fn route_context_budget_for_route(
753 provider: ProviderKind,
754 model: &str,
755 route_limits: Option<RouteLimits>,
756 input_tokens: usize,
757 ) -> Option<ContextBudget> {
758 crate::route_budget::route_context_budget(provider, model, route_limits, input_tokens)
759 }
760
761 pub(super) fn is_context_length_error_message(message: &str) -> bool {
762 // Only genuine context-length rejections may drive the bounded
763 // context-recovery retry. The broader `InvalidInput` bucket also holds
764 // wrong-model rejections ("Model not exist."), malformed requests, and
765 // truncated-output terminations, where re-sending a compacted history
766 // cannot help and would hide the real error.
767 let lower = message.to_lowercase();
768 lower.contains("model output truncated")
769 || lower.contains("model response incomplete")
770 || crate::llm_client::is_context_length_message(&lower)
771 || (lower.contains("requested") && lower.contains("tokens") && lower.contains("maximum"))
772 }
773
774 /// The turn is over: the input still exceeds the route's input budget after
775 /// the bounded recovery. Say what ran and name the levers that exist where
776 /// the message is read — an interactive session has `/compact` and `/clear`;
777 /// a headless host (`exec`, app-server, CI) has neither (#6374).
778 pub(super) fn context_overflow_exhausted_message(
779 interactive: bool,
780 emergency_compactions: u32,
781 estimated_input: usize,
782 input_budget: usize,
783 ) -> String {
784 let passes = match emergency_compactions {
785 1 => "1 emergency compaction pass".to_string(),
786 n => format!("{n} emergency compaction passes"),
787 };
788 let levers = if interactive {
789 "Run /compact to summarize further or /clear to start over; a larger context route or a lower output cap also raises the input budget."
790 } else {
791 "Shorten the input or choose a larger context route; a lower output cap (CODEWHALE_MAX_OUTPUT_TOKENS) or a lower [compaction] retained_user_message_tokens raises the usable input budget."
792 };
793 format!(
794 "Context is still above this route's input budget after {passes} \
795 (~{estimated_input} tokens estimated, ~{input_budget} budget). {levers}"
796 )
797 }
798
799 /// The single error line for a request that cannot fit the route and has
800 /// too little earlier conversation to summarize (experience mark 2). It names the real
801 /// cause and one next step instead of blaming a compaction that never had
802 /// anything to work with.
803 pub(super) fn context_does_not_fit_message(
804 interactive: bool,
805 local_ollama: bool,
806 model: &str,
807 estimated_input: usize,
808 input_budget: usize,
809 prefix_tokens: usize,
810 ) -> String {
811 let pick = |what: &str| {
812 if interactive {
813 format!("Pick {what}: /model.")
814 } else {
815 format!("Choose {what}.")
816 }
817 };
818 if local_ollama && crate::local_ollama::looks_like_non_chat_tag(model) {
819 return format!("{model} can't chat. {}", pick("a chat model"));
820 }
821 let larger = if local_ollama {
822 "a larger model, or raise num_ctx"
823 } else {
824 "a larger model"
825 };
826 if prefix_tokens >= input_budget {
827 format!(
828 "{model}'s context window (~{input_budget} tokens usable) is smaller than \
829 Codewhale's working instructions (~{prefix_tokens} tokens). {}",
830 pick(larger)
831 )
832 } else {
833 format!(
834 "This message (~{estimated_input} tokens with Codewhale's instructions) does not \
835 fit {model}'s window (~{input_budget} tokens usable), and there is not enough \
836 earlier conversation to summarize. Shorten it, or {}",
837 pick(larger).to_lowercase()
838 )
839 }
840 }
841
842 pub(super) fn is_image_input_rejection_message(message: &str) -> bool {
843 let lower = message.to_lowercase();
844 let image_signal = lower.contains("image_url")
845 || lower.contains("content.type")
846 || lower.contains("content type")
847 || lower.contains("does not support image")
848 || lower.contains("image input")
849 || lower.contains("unsupported modality")
850 || lower
851 .split(|character: char| !character.is_alphanumeric())
852 .any(|term| term == "vision");
853 let rejection_signal = lower.contains("400")
854 || lower.contains("invalid")
855 || lower.contains("unsupported")
856 || lower.contains("not support");
857 image_signal && rejection_signal
858 }
859
860 #[cfg(test)]
861 mod tests {
862 use super::is_image_input_rejection_message;
863
864 #[test]
865 fn image_rejection_classifier_matches_provider_400s() {
866 assert!(is_image_input_rejection_message(
867 r#"request (400): {"error":{"code":"1214","message":"messages.content.type 参数非法, 取值范围 ['text']"}}"#
868 ));
869 assert!(is_image_input_rejection_message(
870 "Invalid content type. image_url is only supported by certain models."
871 ));
872 assert!(!is_image_input_rejection_message("Model not exist."));
873 assert!(!is_image_input_rejection_message("invalid revision id"));
874 assert!(!is_image_input_rejection_message(
875 "This model's maximum context length is 131072 tokens."
876 ));
877 }
878 }
879
879 lines RUST