| 1 | //! Context budgeting and prompt-shaping helpers for the engine. |
| 2 | //! |
| 3 | //! These functions are shared by the streaming turn loop, capacity flow, and |
| 4 | //! engine session maintenance code. Keeping them here prevents the top-level |
| 5 | //! engine module from accumulating unrelated context-policy details. |
| 6 | |
| 7 | use crate::config::ProviderKind; |
| 8 | use crate::context_budget::ContextBudget; |
| 9 | #[cfg(test)] |
| 10 | pub(super) use crate::route_budget::effective_max_output_tokens; |
| 11 | pub(super) use crate::route_budget::effective_max_output_tokens_for_route; |
| 12 | use crate::tools::spec::ToolResult; |
| 13 | use codewhale_config::route::RouteLimits; |
| 14 | use codewhale_models::SystemPrompt; |
| 15 | use serde_json::Value; |
| 16 | /// Allow a few emergency recovery attempts before failing the turn. |
| 17 | pub(super) const MAX_CONTEXT_RECOVERY_ATTEMPTS: u8 = 2; |
| 18 | /// Max chars to keep from metadata-provided output summaries. |
| 19 | const TOOL_RESULT_METADATA_SUMMARY_CHARS: usize = 320; |
| 20 | |
| 21 | #[cfg(test)] |
| 22 | pub(super) use crate::compaction::COMPACTION_SUMMARY_MARKER; |
| 23 | |
| 24 | /// What the model sees of one tool result (#6508). |
| 25 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 26 | pub(crate) struct ToolResultContextView { |
| 27 | pub(crate) text: String, |
| 28 | /// The view leaves out bytes of the result and no session artifact holds |
| 29 | /// them yet. The engine saves the full output before the result fans out, |
| 30 | /// so the view it builds afterwards can name a ref instead. |
| 31 | pub(crate) needs_full_output_artifact: bool, |
| 32 | } |
| 33 | |
| 34 | impl ToolResultContextView { |
| 35 | fn whole(text: impl Into<String>) -> Self { |
| 36 | Self { |
| 37 | text: text.into(), |
| 38 | needs_full_output_artifact: false, |
| 39 | } |
| 40 | } |
| 41 | } |
| 42 | |
| 43 | /// A structured summary and whether it left anything of the raw result out. |
| 44 | struct ContextSummary { |
| 45 | text: String, |
| 46 | lossy: bool, |
| 47 | } |
| 48 | |
| 49 | /// Where the full output of a tool call can be read back, from its metadata. |
| 50 | struct RecoveryRef<'a> { |
| 51 | path: &'a str, |
| 52 | /// The `art_<id>` ref `retrieve_tool_result` resolves. `None` for a |
| 53 | /// legacy spill, which no tool call reaches. |
| 54 | artifact_id: Option<&'a str>, |
| 55 | } |
| 56 | |
| 57 | fn recovery_ref(metadata: Option<&Value>) -> Option<RecoveryRef<'_>> { |
| 58 | let obj = metadata?.as_object()?; |
| 59 | let text = |key: &str| { |
| 60 | obj.get(key) |
| 61 | .and_then(Value::as_str) |
| 62 | .filter(|s| !s.is_empty()) |
| 63 | }; |
| 64 | if let Some(artifact_id) = text("artifact_id").filter(|id| id.starts_with("art_")) { |
| 65 | let path = text("artifact_path").or_else(|| text("spillover_path"))?; |
| 66 | return Some(RecoveryRef { |
| 67 | path, |
| 68 | artifact_id: Some(artifact_id), |
| 69 | }); |
| 70 | } |
| 71 | text("spillover_path").map(|path| RecoveryRef { |
| 72 | path, |
| 73 | artifact_id: None, |
| 74 | }) |
| 75 | } |
| 76 | |
| 77 | pub(super) fn summarize_text(text: &str, limit: usize) -> String { |
| 78 | if text.chars().count() <= limit { |
| 79 | return text.to_string(); |
| 80 | } |
| 81 | let take = limit.saturating_sub(3); |
| 82 | let mut out: String = text.chars().take(take).collect(); |
| 83 | out.push_str("..."); |
| 84 | out |
| 85 | } |
| 86 | |
| 87 | /// [`summarize_text`] that records whether it cut anything. |
| 88 | fn keep_text(text: &str, limit: usize, lossy: &mut bool) -> String { |
| 89 | let kept = summarize_text(text, limit); |
| 90 | *lossy |= kept.len() != text.len(); |
| 91 | kept |
| 92 | } |
| 93 | |
| 94 | /// [`summarize_text_head_tail`] that records whether it cut anything. |
| 95 | fn keep_head_tail(text: &str, limit: usize, lossy: &mut bool) -> String { |
| 96 | let kept = summarize_text_head_tail(text, limit); |
| 97 | *lossy |= kept.len() != text.len(); |
| 98 | kept |
| 99 | } |
| 100 | |
| 101 | fn summarize_text_head_tail(text: &str, limit: usize) -> String { |
| 102 | let total = text.chars().count(); |
| 103 | if total <= limit { |
| 104 | return text.to_string(); |
| 105 | } |
| 106 | if limit <= 20 { |
| 107 | return summarize_text(text, limit); |
| 108 | } |
| 109 | |
| 110 | let marker = "\n\n[... output truncated for context ...]\n\n"; |
| 111 | let marker_len = marker.chars().count(); |
| 112 | if limit <= marker_len + 20 { |
| 113 | return summarize_text(text, limit); |
| 114 | } |
| 115 | |
| 116 | let remaining = limit - marker_len; |
| 117 | let head_len = remaining.saturating_mul(2) / 3; |
| 118 | let tail_len = remaining.saturating_sub(head_len); |
| 119 | let head: String = text.chars().take(head_len).collect(); |
| 120 | let tail_vec: Vec<char> = text.chars().rev().take(tail_len).collect(); |
| 121 | let tail: String = tail_vec.into_iter().rev().collect(); |
| 122 | format!("{head}{marker}{tail}") |
| 123 | } |
| 124 | |
| 125 | fn tool_result_metadata_summary(metadata: Option<&serde_json::Value>) -> Option<String> { |
| 126 | let obj = metadata?.as_object()?; |
| 127 | for key in ["summary", "stdout_summary", "stderr_summary", "message"] { |
| 128 | if let Some(text) = obj.get(key).and_then(serde_json::Value::as_str) { |
| 129 | let trimmed = text.trim(); |
| 130 | if !trimmed.is_empty() { |
| 131 | return Some(summarize_text(trimmed, TOOL_RESULT_METADATA_SUMMARY_CHARS)); |
| 132 | } |
| 133 | } |
| 134 | } |
| 135 | None |
| 136 | } |
| 137 | |
| 138 | fn summarize_subagent_status(status: &serde_json::Value, lossy: &mut bool) -> String { |
| 139 | if let Some(raw) = status.as_str() { |
| 140 | return raw.to_string(); |
| 141 | } |
| 142 | if let Some(obj) = status.as_object() |
| 143 | && let Some((kind, value)) = obj.iter().next() |
| 144 | { |
| 145 | if let Some(reason) = value.as_str().filter(|s| !s.trim().is_empty()) { |
| 146 | return format!("{kind}({})", keep_text(reason.trim(), 120, lossy)); |
| 147 | } |
| 148 | return kind.to_string(); |
| 149 | } |
| 150 | status.to_string() |
| 151 | } |
| 152 | |
| 153 | fn summarize_subagent_snapshot( |
| 154 | snapshot: &serde_json::Value, |
| 155 | index: usize, |
| 156 | result_limit: usize, |
| 157 | lossy: &mut bool, |
| 158 | ) -> String { |
| 159 | if let Some(inner) = snapshot.get("snapshot") { |
| 160 | return summarize_subagent_snapshot(inner, index, result_limit, lossy); |
| 161 | } |
| 162 | |
| 163 | let Some(obj) = snapshot.as_object() else { |
| 164 | return format!( |
| 165 | "- item {index}: {}", |
| 166 | keep_text(&snapshot.to_string(), result_limit, lossy) |
| 167 | ); |
| 168 | }; |
| 169 | |
| 170 | let agent_id = obj |
| 171 | .get("agent_id") |
| 172 | .and_then(serde_json::Value::as_str) |
| 173 | .unwrap_or("unknown"); |
| 174 | let agent_type = obj |
| 175 | .get("agent_type") |
| 176 | .and_then(serde_json::Value::as_str) |
| 177 | .unwrap_or("agent"); |
| 178 | let status = obj |
| 179 | .get("status") |
| 180 | .map(|status| summarize_subagent_status(status, lossy)) |
| 181 | .unwrap_or_else(|| "unknown".to_string()); |
| 182 | let objective = obj |
| 183 | .get("assignment") |
| 184 | .and_then(|assignment| assignment.get("objective")) |
| 185 | .and_then(serde_json::Value::as_str) |
| 186 | .map(str::trim) |
| 187 | .filter(|s| !s.is_empty()) |
| 188 | .map(|s| keep_text(s, 220, lossy)); |
| 189 | let result = obj |
| 190 | .get("result") |
| 191 | .and_then(serde_json::Value::as_str) |
| 192 | .map(str::trim) |
| 193 | .filter(|s| !s.is_empty()) |
| 194 | .map(|s| keep_text(s, result_limit, lossy)); |
| 195 | let steps = obj.get("steps_taken").and_then(serde_json::Value::as_u64); |
| 196 | let duration_ms = obj.get("duration_ms").and_then(serde_json::Value::as_u64); |
| 197 | |
| 198 | let mut lines = vec![format!("- {agent_id} ({agent_type}) status={status}")]; |
| 199 | if let Some(objective) = objective { |
| 200 | lines.push(format!(" objective: {objective}")); |
| 201 | } |
| 202 | match result { |
| 203 | Some(result) => lines.push(format!(" result: {result}")), |
| 204 | None => lines.push(" result: not available yet".to_string()), |
| 205 | } |
| 206 | if steps.is_some() || duration_ms.is_some() { |
| 207 | let steps = steps |
| 208 | .map(|n| n.to_string()) |
| 209 | .unwrap_or_else(|| "?".to_string()); |
| 210 | let duration_ms = duration_ms |
| 211 | .map(|n| n.to_string()) |
| 212 | .unwrap_or_else(|| "?".to_string()); |
| 213 | lines.push(format!(" stats: steps={steps}, duration_ms={duration_ms}")); |
| 214 | } |
| 215 | lines.join("\n") |
| 216 | } |
| 217 | |
| 218 | /// Guidance heading every summarized sub-agent result. It names only tools |
| 219 | /// the model can call: `read` and `bash` are eager, and `handle_read` is |
| 220 | /// deferred, so it carries its activation path (#6747). |
| 221 | pub(crate) fn subagent_summary_guidance() -> String { |
| 222 | format!( |
| 223 | "Child results are self-reports; verify side effects with `read` (or `bash`, e.g. `git status`, where available) before claiming success.\n\ |
| 224 | Use `handle_read` on `transcript_handle` for bounded transcript slices when the returned summary is not enough; {}.\n", |
| 225 | crate::tools::handle::HANDLE_READ_ACTIVATION_HINT |
| 226 | ) |
| 227 | } |
| 228 | |
| 229 | /// A payload is a sub-agent snapshot when it carries the identity/status shape |
| 230 | /// this summarizer knows how to render (`agent_id`/`agent_type`, optionally |
| 231 | /// wrapped in a `snapshot` field). |
| 232 | fn looks_like_subagent_snapshot(value: &serde_json::Value) -> bool { |
| 233 | let value = value.get("snapshot").unwrap_or(value); |
| 234 | value |
| 235 | .as_object() |
| 236 | .is_some_and(|obj| obj.contains_key("agent_id") || obj.contains_key("agent_type")) |
| 237 | } |
| 238 | |
| 239 | /// Sub-agent snapshots shown in full before the rest are only counted. |
| 240 | const SUBAGENT_SNAPSHOTS_SHOWN: usize = 8; |
| 241 | /// Floor for one sub-agent's result text on the smallest routes. |
| 242 | const SUBAGENT_RESULT_MIN_CHARS: usize = 1_600; |
| 243 | |
| 244 | fn compact_subagent_tool_result_for_context( |
| 245 | tool_name: &str, |
| 246 | raw: &str, |
| 247 | budget: usize, |
| 248 | ) -> Option<ContextSummary> { |
| 249 | if tool_name != "agent" { |
| 250 | return None; |
| 251 | } |
| 252 | |
| 253 | let parsed: serde_json::Value = serde_json::from_str(raw).ok()?; |
| 254 | let snapshots: Vec<&serde_json::Value> = match &parsed { |
| 255 | serde_json::Value::Array(items) => items.iter().collect(), |
| 256 | serde_json::Value::Object(_) => vec![&parsed], |
| 257 | _ => return None, |
| 258 | }; |
| 259 | |
| 260 | // Coordination envelopes (`wait`, `status`, `claim`, ...) carry typed |
| 261 | // fields the parent needs verbatim: `settled`, `still_running`, |
| 262 | // `timed_out`, `waited_ms`, `note`. Projecting them through the snapshot |
| 263 | // renderer replaced every one with `unknown (agent) status=unknown` and |
| 264 | // dropped the real payload. Summarize only snapshot-shaped results; let |
| 265 | // anything else fall through to the generic bounded path. |
| 266 | if snapshots.is_empty() |
| 267 | || !snapshots |
| 268 | .iter() |
| 269 | .all(|value| looks_like_subagent_snapshot(value)) |
| 270 | { |
| 271 | return None; |
| 272 | } |
| 273 | |
| 274 | // Each shown result gets an equal share of half the route budget, so a |
| 275 | // child's report is not cut at a fixed size on a large route. |
| 276 | let shown = snapshots.len().min(SUBAGENT_SNAPSHOTS_SHOWN); |
| 277 | let result_limit = (budget / 2 / shown.max(1)).max(SUBAGENT_RESULT_MIN_CHARS); |
| 278 | let mut lossy = snapshots.len() > SUBAGENT_SNAPSHOTS_SHOWN; |
| 279 | let mut out = String::from("[sub-agent result summarized for parent context]\n"); |
| 280 | out.push_str(&subagent_summary_guidance()); |
| 281 | for (idx, snapshot) in snapshots.iter().enumerate() { |
| 282 | if idx >= SUBAGENT_SNAPSHOTS_SHOWN { |
| 283 | out.push_str(&format!( |
| 284 | "- ... {} more sub-agent result(s) omitted from context summary\n", |
| 285 | snapshots.len().saturating_sub(idx) |
| 286 | )); |
| 287 | break; |
| 288 | } |
| 289 | out.push_str(&summarize_subagent_snapshot( |
| 290 | snapshot, |
| 291 | idx + 1, |
| 292 | result_limit, |
| 293 | &mut lossy, |
| 294 | )); |
| 295 | out.push('\n'); |
| 296 | } |
| 297 | Some(ContextSummary { |
| 298 | text: out.trim_end().to_string(), |
| 299 | lossy, |
| 300 | }) |
| 301 | } |
| 302 | |
| 303 | fn json_text<'a>(value: &'a Value, key: &str) -> Option<&'a str> { |
| 304 | value |
| 305 | .get(key) |
| 306 | .and_then(Value::as_str) |
| 307 | .map(str::trim) |
| 308 | .filter(|s| !s.is_empty()) |
| 309 | } |
| 310 | |
| 311 | fn json_number_text(value: &Value, key: &str) -> Option<String> { |
| 312 | value |
| 313 | .get(key) |
| 314 | .and_then(|value| { |
| 315 | value |
| 316 | .as_i64() |
| 317 | .map(|n| n.to_string()) |
| 318 | .or_else(|| value.as_u64().map(|n| n.to_string())) |
| 319 | }) |
| 320 | .or_else(|| { |
| 321 | value |
| 322 | .get(key) |
| 323 | .and_then(Value::as_str) |
| 324 | .map(str::trim) |
| 325 | .filter(|s| !s.is_empty()) |
| 326 | .map(ToString::to_string) |
| 327 | }) |
| 328 | } |
| 329 | |
| 330 | /// Characters a structured summary keeps for its own header lines and the |
| 331 | /// recovery footer, before the rest of the budget goes to output streams. |
| 332 | const STRUCTURED_SUMMARY_FRAME_CHARS: usize = 1_000; |
| 333 | /// Floor for one stream or gate detail on the smallest routes. |
| 334 | const STRUCTURED_DETAIL_MIN_CHARS: usize = 600; |
| 335 | |
| 336 | fn compact_run_tests_result_for_context( |
| 337 | raw: &str, |
| 338 | metadata: Option<&Value>, |
| 339 | budget: usize, |
| 340 | ) -> Option<ContextSummary> { |
| 341 | let parsed: Value = serde_json::from_str(raw).ok()?; |
| 342 | let success = parsed.get("success")?.as_bool()?; |
| 343 | let exit_code = json_number_text(&parsed, "exit_code").unwrap_or_else(|| "?".to_string()); |
| 344 | let command = json_text(&parsed, "command").unwrap_or("(unknown command)"); |
| 345 | let stdout = json_text(&parsed, "stdout"); |
| 346 | let stderr = json_text(&parsed, "stderr"); |
| 347 | let mut lossy = false; |
| 348 | |
| 349 | let mut lines = vec![ |
| 350 | "[run_tests result summarized for context]".to_string(), |
| 351 | format!( |
| 352 | "status: {}, exit_code: {exit_code}", |
| 353 | if success { "passed" } else { "failed" } |
| 354 | ), |
| 355 | format!("command: {}", keep_text(command, 300, &mut lossy)), |
| 356 | ]; |
| 357 | // The cargo failure summary names the failing tests. It leads, so the |
| 358 | // names stay inline however long the streams are. |
| 359 | if let Some(summary) = metadata.and_then(|metadata| json_text(metadata, "summary")) { |
| 360 | lines.push(format!("failure summary: {summary}")); |
| 361 | } |
| 362 | let header_chars: usize = lines.iter().map(|line| line.chars().count() + 1).sum(); |
| 363 | let streams = usize::from(stderr.is_some()) + usize::from(stdout.is_some()); |
| 364 | let stream_limit = (budget.saturating_sub(header_chars + STRUCTURED_SUMMARY_FRAME_CHARS) |
| 365 | / streams.max(1)) |
| 366 | .max(STRUCTURED_DETAIL_MIN_CHARS); |
| 367 | if let Some(stderr) = stderr { |
| 368 | lines.push(format!( |
| 369 | "stderr: {}", |
| 370 | keep_head_tail(stderr, stream_limit, &mut lossy) |
| 371 | )); |
| 372 | } |
| 373 | if let Some(stdout) = stdout { |
| 374 | lines.push(format!( |
| 375 | "stdout: {}", |
| 376 | keep_head_tail(stdout, stream_limit, &mut lossy) |
| 377 | )); |
| 378 | } |
| 379 | Some(ContextSummary { |
| 380 | text: lines.join("\n"), |
| 381 | lossy, |
| 382 | }) |
| 383 | } |
| 384 | |
| 385 | fn run_verifier_status_rank(status: Option<&str>) -> u8 { |
| 386 | match status.unwrap_or_default() { |
| 387 | "failed" | "timeout" => 0, |
| 388 | "skipped" => 1, |
| 389 | "passed" => 2, |
| 390 | _ => 3, |
| 391 | } |
| 392 | } |
| 393 | |
| 394 | /// Gates listed one per line before the rest are only counted. |
| 395 | const VERIFIER_GATES_SHOWN: usize = 12; |
| 396 | |
| 397 | fn compact_run_verifiers_result_for_context(raw: &str, budget: usize) -> Option<ContextSummary> { |
| 398 | let parsed: Value = serde_json::from_str(raw).ok()?; |
| 399 | let gates = parsed.get("gates")?.as_array()?; |
| 400 | let summary = json_text(&parsed, "summary") |
| 401 | .map(ToString::to_string) |
| 402 | .unwrap_or_else(|| { |
| 403 | let passed = json_number_text(&parsed, "passed").unwrap_or_else(|| "?".to_string()); |
| 404 | let failed = json_number_text(&parsed, "failed").unwrap_or_else(|| "?".to_string()); |
| 405 | let skipped = json_number_text(&parsed, "skipped").unwrap_or_else(|| "?".to_string()); |
| 406 | format!("{passed} passed, {failed} failed, {skipped} skipped") |
| 407 | }); |
| 408 | |
| 409 | let mut ordered: Vec<&Value> = gates.iter().collect(); |
| 410 | ordered.sort_by(|a, b| { |
| 411 | run_verifier_status_rank(json_text(a, "status")) |
| 412 | .cmp(&run_verifier_status_rank(json_text(b, "status"))) |
| 413 | .then_with(|| json_text(a, "name").cmp(&json_text(b, "name"))) |
| 414 | }); |
| 415 | |
| 416 | let mut lossy = ordered.len() > VERIFIER_GATES_SHOWN; |
| 417 | let mut lines = vec![ |
| 418 | "[run_verifiers result summarized for context]".to_string(), |
| 419 | format!("summary: {summary}"), |
| 420 | ]; |
| 421 | let profile = json_text(&parsed, "profile"); |
| 422 | let level = json_text(&parsed, "level"); |
| 423 | if profile.is_some() || level.is_some() { |
| 424 | lines.push(format!( |
| 425 | "selection: profile={}, level={}", |
| 426 | profile.unwrap_or("?"), |
| 427 | level.unwrap_or("?") |
| 428 | )); |
| 429 | } |
| 430 | |
| 431 | let shown = ordered.iter().take(VERIFIER_GATES_SHOWN); |
| 432 | let detailed = shown |
| 433 | .clone() |
| 434 | .filter(|gate| json_text(gate, "status") != Some("passed")) |
| 435 | .count(); |
| 436 | let detail_limit = (budget.saturating_sub(STRUCTURED_SUMMARY_FRAME_CHARS * 2) |
| 437 | / detailed.max(1)) |
| 438 | .max(STRUCTURED_DETAIL_MIN_CHARS); |
| 439 | |
| 440 | for gate in shown { |
| 441 | let name = json_text(gate, "name").unwrap_or("gate"); |
| 442 | let ecosystem = json_text(gate, "ecosystem").unwrap_or("unknown"); |
| 443 | let status = json_text(gate, "status").unwrap_or("unknown"); |
| 444 | let exit = json_number_text(gate, "exit_code") |
| 445 | .map(|code| format!(" exit={code}")) |
| 446 | .unwrap_or_default(); |
| 447 | lines.push(format!("- {name} ({ecosystem}): {status}{exit}")); |
| 448 | // A stream longer than the verifier keeps in memory is saved whole. |
| 449 | for stream in ["stdout", "stderr"] { |
| 450 | if let Some(reference) = json_text(gate, &format!("{stream}_log_ref")) { |
| 451 | lines.push(format!( |
| 452 | " full {stream}: retrieve_tool_result ref=\"{reference}\"" |
| 453 | )); |
| 454 | } |
| 455 | } |
| 456 | |
| 457 | let stdout = json_text(gate, "stdout"); |
| 458 | let stderr = json_text(gate, "stderr"); |
| 459 | if status == "passed" { |
| 460 | // A passing gate's output stays out of context; the saved full |
| 461 | // output keeps it readable. |
| 462 | lossy |= stdout.is_some() || stderr.is_some(); |
| 463 | continue; |
| 464 | } |
| 465 | if let Some(command) = json_text(gate, "command") { |
| 466 | lines.push(format!( |
| 467 | " command: {}", |
| 468 | keep_text(command, 240, &mut lossy) |
| 469 | )); |
| 470 | } |
| 471 | // One detail per gate: the skip reason, else stderr, else stdout. |
| 472 | let (detail, other_streams) = match json_text(gate, "skipped_reason") { |
| 473 | Some(reason) => (Some(reason), stderr.is_some() || stdout.is_some()), |
| 474 | None => match stderr { |
| 475 | Some(stderr) => (Some(stderr), stdout.is_some()), |
| 476 | None => (stdout, false), |
| 477 | }, |
| 478 | }; |
| 479 | lossy |= other_streams; |
| 480 | if let Some(detail) = detail { |
| 481 | lines.push(format!( |
| 482 | " detail: {}", |
| 483 | keep_head_tail(detail, detail_limit, &mut lossy) |
| 484 | )); |
| 485 | } |
| 486 | } |
| 487 | if ordered.len() > VERIFIER_GATES_SHOWN { |
| 488 | lines.push(format!( |
| 489 | "- ... {} more gate(s) omitted from context summary", |
| 490 | ordered.len() - VERIFIER_GATES_SHOWN |
| 491 | )); |
| 492 | } |
| 493 | |
| 494 | Some(ContextSummary { |
| 495 | text: lines.join("\n"), |
| 496 | lossy, |
| 497 | }) |
| 498 | } |
| 499 | |
| 500 | fn compact_task_gate_run_result_for_context(raw: &str) -> Option<ContextSummary> { |
| 501 | let parsed: Value = serde_json::from_str(raw).ok()?; |
| 502 | let gate = parsed.get("gate")?; |
| 503 | let gate_name = json_text(gate, "gate").unwrap_or("gate"); |
| 504 | let status = json_text(gate, "status").unwrap_or("unknown"); |
| 505 | let command = json_text(gate, "command").unwrap_or("(unknown command)"); |
| 506 | let summary = json_text(gate, "summary") |
| 507 | .or_else(|| json_text(&parsed, "stderr_summary")) |
| 508 | .or_else(|| json_text(&parsed, "stdout_summary")); |
| 509 | let exit = json_number_text(gate, "exit_code") |
| 510 | .map(|code| format!(", exit_code: {code}")) |
| 511 | .unwrap_or_default(); |
| 512 | let mut lossy = false; |
| 513 | |
| 514 | let mut lines = vec![ |
| 515 | "[task_gate_run result summarized for context]".to_string(), |
| 516 | format!("gate: {gate_name}, status: {status}{exit}"), |
| 517 | format!("command: {}", keep_text(command, 300, &mut lossy)), |
| 518 | ]; |
| 519 | if let Some(summary) = summary { |
| 520 | lines.push(format!("summary: {}", keep_text(summary, 800, &mut lossy))); |
| 521 | } |
| 522 | if let Some(log_path) = json_text(gate, "log_path") { |
| 523 | lines.push(format!("log_path: {log_path}")); |
| 524 | } |
| 525 | Some(ContextSummary { |
| 526 | text: lines.join("\n"), |
| 527 | lossy, |
| 528 | }) |
| 529 | } |
| 530 | |
| 531 | fn compact_structured_tool_result_for_context( |
| 532 | tool_name: &str, |
| 533 | raw: &str, |
| 534 | metadata: Option<&Value>, |
| 535 | budget: usize, |
| 536 | ) -> Option<ContextSummary> { |
| 537 | match tool_name { |
| 538 | "run_tests" => compact_run_tests_result_for_context(raw, metadata, budget), |
| 539 | "run_verifiers" => compact_run_verifiers_result_for_context(raw, budget), |
| 540 | // `tasks` is the unified durable-task tool (piagent phase B); its |
| 541 | // gate_run action emits the same gate payload as the legacy |
| 542 | // `task_gate_run` alias. The compactor returns None unless the |
| 543 | // content actually parses as a gate result, so non-gate `tasks` |
| 544 | // results fall through to the generic path unchanged. |
| 545 | "task_gate_run" | "tasks" => compact_task_gate_run_result_for_context(raw), |
| 546 | _ => None, |
| 547 | } |
| 548 | } |
| 549 | |
| 550 | #[cfg(test)] |
| 551 | pub(crate) fn compact_tool_result_for_context( |
| 552 | model: &str, |
| 553 | tool_name: &str, |
| 554 | output: &ToolResult, |
| 555 | ) -> String { |
| 556 | compact_tool_result_for_route(ProviderKind::Deepseek, model, None, tool_name, output) |
| 557 | } |
| 558 | |
| 559 | pub(crate) fn compact_tool_result_for_route( |
| 560 | provider: ProviderKind, |
| 561 | model: &str, |
| 562 | route_limits: Option<RouteLimits>, |
| 563 | tool_name: &str, |
| 564 | output: &ToolResult, |
| 565 | ) -> String { |
| 566 | tool_result_context_view(provider, model, route_limits, tool_name, output).text |
| 567 | } |
| 568 | |
| 569 | /// The model's view of one tool result (#6508). |
| 570 | /// |
| 571 | /// There is one size authority: [`crate::route_budget::route_inline_char_budget`]. |
| 572 | /// A result within it reaches the model whole, whatever the tool. A larger |
| 573 | /// one is cut to a head and tail around a footer that names where the full |
| 574 | /// output lives and the `art_<id>` ref `retrieve_tool_result` reads it back |
| 575 | /// with. The view never writes anything: when it would leave bytes out and no |
| 576 | /// artifact holds them, it says so through `needs_full_output_artifact`, and |
| 577 | /// the engine saves the full output and builds the view again. |
| 578 | /// |
| 579 | /// Structured summaries (`run_tests`, `run_verifiers`, gate results, |
| 580 | /// sub-agent snapshots) keep their shape, scale their detail with the same |
| 581 | /// budget, and follow the same rule whenever they leave something out. |
| 582 | pub(crate) fn tool_result_context_view( |
| 583 | provider: ProviderKind, |
| 584 | model: &str, |
| 585 | route_limits: Option<RouteLimits>, |
| 586 | tool_name: &str, |
| 587 | output: &ToolResult, |
| 588 | ) -> ToolResultContextView { |
| 589 | let raw = output.content.trim(); |
| 590 | if raw.is_empty() { |
| 591 | return ToolResultContextView::whole(String::new()); |
| 592 | } |
| 593 | let metadata = output.metadata.as_ref(); |
| 594 | |
| 595 | // A result already bounded by the adaptive evidence envelope is an |
| 596 | // honest, context-sized preview whose footer names the artifact path and |
| 597 | // a recovery instruction. Re-compacting it would strip that recovery |
| 598 | // contract and double-truncate the output, so pass it through unchanged. |
| 599 | if metadata |
| 600 | .and_then(|metadata| metadata.get("evidence_available")) |
| 601 | .and_then(Value::as_bool) |
| 602 | .unwrap_or(false) |
| 603 | { |
| 604 | return ToolResultContextView::whole(raw); |
| 605 | } |
| 606 | |
| 607 | // The `read` primitive already bounds itself to an explicit per-call byte |
| 608 | // budget and, when that budget truncates the file, ends with a footer |
| 609 | // naming the exact offset to continue from. Compacting it a second time |
| 610 | // would drop content the caller deliberately budgeted for *and* delete the |
| 611 | // continuation contract. A result that stayed inside its declared budget |
| 612 | // therefore passes through; one that exceeded it takes the path below. |
| 613 | if metadata |
| 614 | .and_then(|metadata| metadata.get("read_budget_bytes")) |
| 615 | .and_then(Value::as_u64) |
| 616 | .is_some_and(|budget| raw.len() as u64 <= budget) |
| 617 | { |
| 618 | return ToolResultContextView::whole(raw); |
| 619 | } |
| 620 | |
| 621 | let budget = |
| 622 | crate::route_budget::route_inline_char_budget_for_route(provider, model, route_limits); |
| 623 | let recovery = recovery_ref(metadata); |
| 624 | let recovery_path = recovery.as_ref().map(|recovery| recovery.path); |
| 625 | let retrieval_ref = recovery.as_ref().and_then(|recovery| recovery.artifact_id); |
| 626 | |
| 627 | let summary = compact_subagent_tool_result_for_context(tool_name, raw, budget) |
| 628 | .or_else(|| compact_structured_tool_result_for_context(tool_name, raw, metadata, budget)); |
| 629 | if let Some(summary) = summary { |
| 630 | return summary_view(summary, budget, recovery.as_ref()); |
| 631 | } |
| 632 | |
| 633 | // Spillover already saved the full output and left a preview. Re-fit that |
| 634 | // preview to the budget and keep its ref; never save the preview itself. |
| 635 | if let Some(metadata) = |
| 636 | metadata.filter(|metadata| metadata.get("retained_head_bytes").is_some()) |
| 637 | && let Some(text) = crate::tools::truncate::refit_spilled_preview( |
| 638 | &output.content, |
| 639 | metadata, |
| 640 | budget, |
| 641 | recovery_path, |
| 642 | retrieval_ref, |
| 643 | ) |
| 644 | { |
| 645 | return ToolResultContextView::whole(text.trim().to_string()); |
| 646 | } |
| 647 | |
| 648 | if raw.chars().count() <= budget { |
| 649 | return ToolResultContextView::whole(raw); |
| 650 | } |
| 651 | |
| 652 | let lead = tool_result_metadata_summary(metadata) |
| 653 | .map(|summary| format!("Summary: {summary}\n")) |
| 654 | .unwrap_or_default(); |
| 655 | let body = crate::tools::truncate::fit_to_inline_budget( |
| 656 | raw, |
| 657 | budget.saturating_sub(lead.chars().count()), |
| 658 | recovery_path, |
| 659 | retrieval_ref, |
| 660 | ); |
| 661 | ToolResultContextView { |
| 662 | text: format!("{lead}{body}"), |
| 663 | needs_full_output_artifact: recovery.is_none(), |
| 664 | } |
| 665 | } |
| 666 | |
| 667 | /// Finish a structured summary: when it left anything out, end it with the |
| 668 | /// same recovery instruction every other cut carries. |
| 669 | fn summary_view( |
| 670 | summary: ContextSummary, |
| 671 | budget: usize, |
| 672 | recovery: Option<&RecoveryRef<'_>>, |
| 673 | ) -> ToolResultContextView { |
| 674 | if !summary.lossy { |
| 675 | return ToolResultContextView::whole(summary.text); |
| 676 | } |
| 677 | let recovery_path = recovery.map(|recovery| recovery.path); |
| 678 | let retrieval_ref = recovery.and_then(|recovery| recovery.artifact_id); |
| 679 | let footer = match recovery_path { |
| 680 | Some(path) => format!( |
| 681 | "[full output at {path}; {}]", |
| 682 | crate::tools::truncate::spillover_recovery_instruction(retrieval_ref) |
| 683 | ), |
| 684 | None => format!( |
| 685 | "[the full output could not be saved; {}]", |
| 686 | crate::tools::truncate::spillover_recovery_instruction(None) |
| 687 | ), |
| 688 | }; |
| 689 | let text = crate::tools::truncate::fit_to_inline_budget( |
| 690 | &summary.text, |
| 691 | budget.saturating_sub(footer.chars().count() + 1), |
| 692 | recovery_path, |
| 693 | retrieval_ref, |
| 694 | ); |
| 695 | ToolResultContextView { |
| 696 | text: format!("{text}\n{footer}"), |
| 697 | needs_full_output_artifact: recovery.is_none(), |
| 698 | } |
| 699 | } |
| 700 | |
| 701 | pub(super) fn extract_compaction_summary_prompt( |
| 702 | prompt: Option<SystemPrompt>, |
| 703 | ) -> Option<SystemPrompt> { |
| 704 | crate::compaction::extract_compaction_summary(prompt.as_ref()) |
| 705 | } |
| 706 | |
| 707 | /// Internal input-side token budget for a provider/model route: |
| 708 | /// `window - reserved_output - headroom`. Used by the preflight check, |
| 709 | /// emergency recovery, and capacity trimming to decide when to compact. |
| 710 | /// Unknown model ids fall back to the provider's conservative default instead |
| 711 | /// of disabling preflight; custom long-context deployments can still advertise |
| 712 | /// their window with a `-256k`/`-1024k` model suffix. |
| 713 | /// |
| 714 | /// The reserved-output term is the route-effective request cap: exactly what |
| 715 | /// the API can receive after explicit overrides, compatibility/route ceilings, |
| 716 | /// and the route window are intersected. A second hidden reasoning reserve |
| 717 | /// would make preflight disagree with the wire request and can cause premature |
| 718 | /// compaction on otherwise valid large-window inputs. |
| 719 | #[cfg(test)] |
| 720 | pub(super) fn context_input_budget_for_provider( |
| 721 | provider: ProviderKind, |
| 722 | model: &str, |
| 723 | ) -> Option<usize> { |
| 724 | context_input_budget_for_route(provider, model, None, 0) |
| 725 | } |
| 726 | |
| 727 | /// Public so external callers (e.g. a host/bridge deriving its own compaction |
| 728 | /// trigger line) can reuse the *exact* same internal input-budget math — window |
| 729 | /// minus the route-effective output reservation |
| 730 | /// (`route_output_reservation`) minus headroom — |
| 731 | /// instead of re-deriving those constants and silently drifting from the engine. |
| 732 | /// Pass `input_tokens = 0` to get the full emergency input budget for the route. |
| 733 | pub fn context_input_budget_for_route( |
| 734 | provider: ProviderKind, |
| 735 | model: &str, |
| 736 | route_limits: Option<RouteLimits>, |
| 737 | input_tokens: usize, |
| 738 | ) -> Option<usize> { |
| 739 | route_context_budget_for_route(provider, model, route_limits, input_tokens) |
| 740 | .and_then(|budget| usize::try_from(budget.available_input_tokens).ok()) |
| 741 | } |
| 742 | |
| 743 | #[cfg(test)] |
| 744 | pub(super) fn route_context_budget_for_provider( |
| 745 | provider: ProviderKind, |
| 746 | model: &str, |
| 747 | input_tokens: usize, |
| 748 | ) -> Option<ContextBudget> { |
| 749 | route_context_budget_for_route(provider, model, None, input_tokens) |
| 750 | } |
| 751 | |
| 752 | pub(super) fn route_context_budget_for_route( |
| 753 | provider: ProviderKind, |
| 754 | model: &str, |
| 755 | route_limits: Option<RouteLimits>, |
| 756 | input_tokens: usize, |
| 757 | ) -> Option<ContextBudget> { |
| 758 | crate::route_budget::route_context_budget(provider, model, route_limits, input_tokens) |
| 759 | } |
| 760 | |
| 761 | pub(super) fn is_context_length_error_message(message: &str) -> bool { |
| 762 | // Only genuine context-length rejections may drive the bounded |
| 763 | // context-recovery retry. The broader `InvalidInput` bucket also holds |
| 764 | // wrong-model rejections ("Model not exist."), malformed requests, and |
| 765 | // truncated-output terminations, where re-sending a compacted history |
| 766 | // cannot help and would hide the real error. |
| 767 | let lower = message.to_lowercase(); |
| 768 | lower.contains("model output truncated") |
| 769 | || lower.contains("model response incomplete") |
| 770 | || crate::llm_client::is_context_length_message(&lower) |
| 771 | || (lower.contains("requested") && lower.contains("tokens") && lower.contains("maximum")) |
| 772 | } |
| 773 | |
| 774 | /// The turn is over: the input still exceeds the route's input budget after |
| 775 | /// the bounded recovery. Say what ran and name the levers that exist where |
| 776 | /// the message is read — an interactive session has `/compact` and `/clear`; |
| 777 | /// a headless host (`exec`, app-server, CI) has neither (#6374). |
| 778 | pub(super) fn context_overflow_exhausted_message( |
| 779 | interactive: bool, |
| 780 | emergency_compactions: u32, |
| 781 | estimated_input: usize, |
| 782 | input_budget: usize, |
| 783 | ) -> String { |
| 784 | let passes = match emergency_compactions { |
| 785 | 1 => "1 emergency compaction pass".to_string(), |
| 786 | n => format!("{n} emergency compaction passes"), |
| 787 | }; |
| 788 | let levers = if interactive { |
| 789 | "Run /compact to summarize further or /clear to start over; a larger context route or a lower output cap also raises the input budget." |
| 790 | } else { |
| 791 | "Shorten the input or choose a larger context route; a lower output cap (CODEWHALE_MAX_OUTPUT_TOKENS) or a lower [compaction] retained_user_message_tokens raises the usable input budget." |
| 792 | }; |
| 793 | format!( |
| 794 | "Context is still above this route's input budget after {passes} \ |
| 795 | (~{estimated_input} tokens estimated, ~{input_budget} budget). {levers}" |
| 796 | ) |
| 797 | } |
| 798 | |
| 799 | /// The single error line for a request that cannot fit the route and has |
| 800 | /// too little earlier conversation to summarize (experience mark 2). It names the real |
| 801 | /// cause and one next step instead of blaming a compaction that never had |
| 802 | /// anything to work with. |
| 803 | pub(super) fn context_does_not_fit_message( |
| 804 | interactive: bool, |
| 805 | local_ollama: bool, |
| 806 | model: &str, |
| 807 | estimated_input: usize, |
| 808 | input_budget: usize, |
| 809 | prefix_tokens: usize, |
| 810 | ) -> String { |
| 811 | let pick = |what: &str| { |
| 812 | if interactive { |
| 813 | format!("Pick {what}: /model.") |
| 814 | } else { |
| 815 | format!("Choose {what}.") |
| 816 | } |
| 817 | }; |
| 818 | if local_ollama && crate::local_ollama::looks_like_non_chat_tag(model) { |
| 819 | return format!("{model} can't chat. {}", pick("a chat model")); |
| 820 | } |
| 821 | let larger = if local_ollama { |
| 822 | "a larger model, or raise num_ctx" |
| 823 | } else { |
| 824 | "a larger model" |
| 825 | }; |
| 826 | if prefix_tokens >= input_budget { |
| 827 | format!( |
| 828 | "{model}'s context window (~{input_budget} tokens usable) is smaller than \ |
| 829 | Codewhale's working instructions (~{prefix_tokens} tokens). {}", |
| 830 | pick(larger) |
| 831 | ) |
| 832 | } else { |
| 833 | format!( |
| 834 | "This message (~{estimated_input} tokens with Codewhale's instructions) does not \ |
| 835 | fit {model}'s window (~{input_budget} tokens usable), and there is not enough \ |
| 836 | earlier conversation to summarize. Shorten it, or {}", |
| 837 | pick(larger).to_lowercase() |
| 838 | ) |
| 839 | } |
| 840 | } |
| 841 | |
| 842 | pub(super) fn is_image_input_rejection_message(message: &str) -> bool { |
| 843 | let lower = message.to_lowercase(); |
| 844 | let image_signal = lower.contains("image_url") |
| 845 | || lower.contains("content.type") |
| 846 | || lower.contains("content type") |
| 847 | || lower.contains("does not support image") |
| 848 | || lower.contains("image input") |
| 849 | || lower.contains("unsupported modality") |
| 850 | || lower |
| 851 | .split(|character: char| !character.is_alphanumeric()) |
| 852 | .any(|term| term == "vision"); |
| 853 | let rejection_signal = lower.contains("400") |
| 854 | || lower.contains("invalid") |
| 855 | || lower.contains("unsupported") |
| 856 | || lower.contains("not support"); |
| 857 | image_signal && rejection_signal |
| 858 | } |
| 859 | |
| 860 | #[cfg(test)] |
| 861 | mod tests { |
| 862 | use super::is_image_input_rejection_message; |
| 863 | |
| 864 | #[test] |
| 865 | fn image_rejection_classifier_matches_provider_400s() { |
| 866 | assert!(is_image_input_rejection_message( |
| 867 | r#"request (400): {"error":{"code":"1214","message":"messages.content.type 参数非法, 取值范围 ['text']"}}"# |
| 868 | )); |
| 869 | assert!(is_image_input_rejection_message( |
| 870 | "Invalid content type. image_url is only supported by certain models." |
| 871 | )); |
| 872 | assert!(!is_image_input_rejection_message("Model not exist.")); |
| 873 | assert!(!is_image_input_rejection_message("invalid revision id")); |
| 874 | assert!(!is_image_input_rejection_message( |
| 875 | "This model's maximum context length is 131072 tokens." |
| 876 | )); |
| 877 | } |
| 878 | } |
| 879 |