返回 CodeWhale
truncate.rs
根目录 / crates / tui / src / tools / truncate.rs
1 //! Tool-output spillover writer (#422).
2 //!
3 //! When a tool produces output that's too large to land in the model's
4 //! context budget, we want two things at once:
5 //!
6 //! 1. The transcript / tool-cell renders a bounded preview so the UI
7 //! stays scannable.
8 //! 2. The full router input is preserved under its origin session so bounded
9 //! retrieval and the raw-detail pager can inspect it without leaking a
10 //! process-global filesystem path.
11 //!
12 //! The opt-in adaptive path writes immutable artifacts under
13 //! `~/.codewhale/sessions/<session>/artifacts/`. The historical
14 //! `~/.codewhale/tool_outputs/<sanitised-id>.txt` directory remains only for
15 //! classic-routing compatibility, protected by a digest-bound origin sidecar.
16 //!
17 //! Boot prune drops files whose mtime is older than [`SPILLOVER_MAX_AGE`]
18 //! (7 days). Prune failures are logged and never fatal — the user
19 //! shouldn't see startup wedge because of a stale tool-output file.
20 //!
21 //! ## Live callers
22 //!
23 //! * [`apply_spillover`] — invoked from the engine's tool-execution
24 //! path (`turn_loop.rs`) so any successful tool result over
25 //! [`SPILLOVER_THRESHOLD_BYTES`] spills to disk and the model
26 //! receives a bounded plain preview: a [`SPILLOVER_HEAD_BYTES`] head,
27 //! a short retained tail, and an honest footer naming the on-disk
28 //! path of the full output plus a one-line recovery instruction.
29 //! * Boot prune in `main.rs` deletes files older than
30 //! [`SPILLOVER_MAX_AGE`].
31 //!
32 //! UI-side rendering is owned by `tui/history.rs::render_spillover_annotation`;
33 //! it exposes a calm expand affordance and the tool-details shortcut opens the
34 //! full retained output.
35
36 use std::fs;
37 use std::io;
38 use std::path::{Path, PathBuf};
39 use std::time::{Duration, SystemTime};
40
41 use crate::tools::spec::ToolResult;
42
43 /// Name of the spillover directory under the CodeWhale home.
44 pub const SPILLOVER_DIR_NAME: &str = "tool_outputs";
45
46 const LEGACY_SPILLOVER_OWNER_SCHEMA_VERSION: u32 = 1;
47
48 /// Session proof for compatibility payloads kept in the historical global
49 /// `tool_outputs/` directory.
50 ///
51 /// The payload remains in its legacy location so classic-routing rollback and
52 /// existing detail pagers keep working, but model retrieval is authorized only
53 /// when this sidecar names the active origin session and still matches the
54 /// immutable bytes being returned.
55 #[derive(Debug, Clone, serde::Serialize, serde::Deserialize, PartialEq, Eq)]
56 pub(crate) struct LegacySpilloverOwnership {
57 pub schema_version: u32,
58 pub origin_session: String,
59 pub digest: String,
60 pub size_bytes: u64,
61 }
62
63 /// Default threshold above which a tool result is a candidate for
64 /// spillover. Mirrors the `MAX_MEMORY_SIZE` ceiling we use elsewhere
65 /// for "too large to inline" so the rules feel consistent. Wired
66 /// callers can pass a different value if a tool family has different
67 /// economics.
68 pub const SPILLOVER_THRESHOLD_BYTES: usize = 100 * 1024; // 100 KiB
69
70 /// Default boot-prune age. Older spillover files are deleted on
71 /// startup to keep `~/.codewhale/tool_outputs/` from growing without
72 /// bound. Mirrors the workspace-snapshot 7-day default.
73 pub const SPILLOVER_MAX_AGE: Duration = Duration::from_secs(7 * 24 * 60 * 60);
74
75 #[cfg(test)]
76 static TEST_SPILLOVER_ROOT: std::sync::Mutex<Option<PathBuf>> = std::sync::Mutex::new(None);
77
78 #[cfg(test)]
79 pub(crate) static TEST_SPILLOVER_GUARD: std::sync::Mutex<()> = std::sync::Mutex::new(());
80
81 /// Resolve `~/.codewhale/tool_outputs/`. Returns `None` if the home
82 /// directory can't be determined (CI containers occasionally hit
83 /// this). Callers should treat `None` as "spillover unavailable" and
84 /// degrade gracefully rather than fail the tool call.
85 #[must_use]
86 pub fn spillover_root() -> Option<PathBuf> {
87 #[cfg(test)]
88 if let Some(root) = TEST_SPILLOVER_ROOT
89 .lock()
90 .unwrap_or_else(|err| err.into_inner())
91 .clone()
92 {
93 return Some(root);
94 }
95
96 let home = crate::config::effective_home_dir()?;
97 let primary = home.join(".codewhale").join(SPILLOVER_DIR_NAME);
98 let legacy = home.join(".deepseek").join(SPILLOVER_DIR_NAME);
99 if primary.exists() || !legacy.exists() {
100 return Some(primary);
101 }
102 Some(legacy)
103 }
104
105 /// Override the spillover root for tests without mutating `$HOME`.
106 #[cfg(test)]
107 pub(crate) fn set_test_spillover_root(root: Option<PathBuf>) -> Option<PathBuf> {
108 let mut guard = TEST_SPILLOVER_ROOT
109 .lock()
110 .unwrap_or_else(|err| err.into_inner());
111 std::mem::replace(&mut *guard, root)
112 }
113
114 /// Resolve the spillover-file path for a tool call id. Sanitises the
115 /// id so that a hostile value can't escape the storage directory.
116 /// Returns `None` for empty / fully-invalid ids; the caller should
117 /// treat that as "spillover unavailable" and skip the write.
118 #[must_use]
119 pub fn spillover_path(id: &str) -> Option<PathBuf> {
120 let sanitised = sanitise_id(id)?;
121 Some(spillover_root()?.join(format!("{sanitised}.txt")))
122 }
123
124 #[must_use]
125 pub(crate) fn legacy_spillover_ownership_path(payload_path: &Path) -> PathBuf {
126 payload_path.with_extension("owner.json")
127 }
128
129 /// Publish the proof needed to retrieve a legacy-global spillover safely.
130 ///
131 /// Payload publication happens first. If this atomic sidecar write fails, the
132 /// payload is deliberately left unowned and therefore inaccessible through
133 /// `retrieve_tool_result`; callers must not advertise a retrieval hint.
134 pub(crate) fn publish_legacy_spillover_ownership(
135 payload_path: &Path,
136 session_id: &str,
137 bytes: &[u8],
138 ) -> io::Result<PathBuf> {
139 if session_id.trim().is_empty() {
140 return Err(io::Error::new(
141 io::ErrorKind::InvalidInput,
142 "legacy spillover ownership requires a session id",
143 ));
144 }
145 let ownership = LegacySpilloverOwnership {
146 schema_version: LEGACY_SPILLOVER_OWNER_SCHEMA_VERSION,
147 origin_session: session_id.to_string(),
148 digest: crate::hashing::sha256_hex(bytes),
149 size_bytes: bytes.len().try_into().unwrap_or(u64::MAX),
150 };
151 let sidecar = legacy_spillover_ownership_path(payload_path);
152 let encoded = serde_json::to_vec_pretty(&ownership)
153 .map_err(|error| io::Error::new(io::ErrorKind::InvalidData, error))?;
154 crate::utils::write_atomic(&sidecar, &encoded)?;
155 Ok(sidecar)
156 }
157
158 pub(crate) fn read_legacy_spillover_ownership(
159 payload_path: &Path,
160 ) -> io::Result<LegacySpilloverOwnership> {
161 let sidecar = legacy_spillover_ownership_path(payload_path);
162 if std::fs::symlink_metadata(&sidecar)?
163 .file_type()
164 .is_symlink()
165 {
166 return Err(io::Error::new(
167 io::ErrorKind::PermissionDenied,
168 "legacy spillover ownership sidecar must not be a symlink",
169 ));
170 }
171 let ownership = serde_json::from_slice::<LegacySpilloverOwnership>(&std::fs::read(sidecar)?)
172 .map_err(|error| io::Error::new(io::ErrorKind::InvalidData, error))?;
173 if ownership.schema_version != LEGACY_SPILLOVER_OWNER_SCHEMA_VERSION {
174 return Err(io::Error::new(
175 io::ErrorKind::InvalidData,
176 "unsupported legacy spillover ownership schema",
177 ));
178 }
179 Ok(ownership)
180 }
181
182 /// Resolve the spillover-file path for a SHA256 content hash. Separate
183 /// namespace (`sha_<hex>.txt`) from the tool-call-id files so legacy
184 /// SHA-addressed evidence can be recognized without colliding with
185 /// tool-call references. Retrieval still requires matching ownership
186 /// metadata. `sha` must be the raw 64-char lowercase hex digest —
187 /// case-insensitive matching is done by the caller.
188 #[must_use]
189 pub fn sha_spillover_path(sha: &str) -> Option<PathBuf> {
190 let sha = sha.trim().to_ascii_lowercase();
191 if !is_valid_sha256(&sha) {
192 return None;
193 }
194 Some(spillover_root()?.join(format!("sha_{sha}.txt")))
195 }
196
197 /// True when `s` is a 64-character lowercase ASCII hex string. Used
198 /// to detect bare SHA refs the model might pass to retrieval and to
199 /// validate input to [`sha_spillover_path`].
200 #[must_use]
201 pub fn is_valid_sha256(s: &str) -> bool {
202 s.len() == 64
203 && s.chars()
204 .all(|c| c.is_ascii_hexdigit() && !c.is_ascii_uppercase())
205 }
206
207 /// Write a legacy SHA-addressed spillover fixture for ownership tests.
208 #[cfg(test)]
209 pub fn write_sha_spillover(sha: &str, content: &str) -> io::Result<PathBuf> {
210 let path = sha_spillover_path(sha).ok_or_else(|| {
211 io::Error::new(
212 io::ErrorKind::InvalidInput,
213 "sha must be a 64-char lowercase hex digest",
214 )
215 })?;
216 if path.exists() {
217 return Ok(path);
218 }
219 if let Some(parent) = path.parent() {
220 fs::create_dir_all(parent)?;
221 }
222 crate::utils::write_atomic(&path, content.as_bytes())?;
223 Ok(path)
224 }
225
226 /// Write `content` to the spillover file for `id`. Creates the
227 /// parent directory if needed. Returns the resolved path on success.
228 ///
229 /// Atomic via `write` + filesystem rename guarantees from the
230 /// underlying OS — the file is created at a temp name first and
231 /// then renamed into place. Failures bubble up as `io::Error` so the
232 /// caller can decide whether to surface them.
233 pub fn write_spillover(id: &str, content: &str) -> io::Result<PathBuf> {
234 let path = spillover_path(id).ok_or_else(|| {
235 io::Error::new(
236 io::ErrorKind::InvalidInput,
237 "could not resolve spillover path (empty/invalid id or missing home directory)",
238 )
239 })?;
240 if let Some(parent) = path.parent() {
241 fs::create_dir_all(parent)?;
242 }
243 crate::utils::write_atomic(&path, content.as_bytes())?;
244 Ok(path)
245 }
246
247 /// Drop spillover files older than `max_age`. Returns the number of
248 /// files removed. Non-fatal: directory-missing returns 0; per-file
249 /// errors are logged and skipped. Mirrors
250 /// [`crate::session_manager::prune_workspace_snapshots`].
251 pub fn prune_older_than(max_age: Duration) -> io::Result<usize> {
252 let Some(root) = spillover_root() else {
253 return Ok(0);
254 };
255 if !root.exists() {
256 return Ok(0);
257 }
258 let cutoff = SystemTime::now()
259 .checked_sub(max_age)
260 .unwrap_or(SystemTime::UNIX_EPOCH);
261 let mut pruned = 0usize;
262 for entry in fs::read_dir(&root)? {
263 let entry = match entry {
264 Ok(e) => e,
265 Err(err) => {
266 tracing::warn!(target: "spillover", ?err, "skipping unreadable dir entry");
267 continue;
268 }
269 };
270 let path = entry.path();
271 if !path.is_file() {
272 continue;
273 }
274 let modified = match entry.metadata().and_then(|m| m.modified()) {
275 Ok(t) => t,
276 Err(err) => {
277 tracing::warn!(target: "spillover", ?err, ?path, "skipping unreadable mtime");
278 continue;
279 }
280 };
281 if modified < cutoff {
282 if let Err(err) = fs::remove_file(&path) {
283 tracing::warn!(target: "spillover", ?err, ?path, "spillover prune skipped a file");
284 continue;
285 }
286 pruned += 1;
287 }
288 }
289 Ok(pruned)
290 }
291
292 /// Convenience for the common "too long? spill it." pattern. If
293 /// `content` is at or below `threshold` bytes, returns `None` and the
294 /// caller keeps the inline content. Above the threshold, writes the
295 /// full content to the spillover file and returns
296 /// `Some((head, path))` where `head` is the leading slice the caller
297 /// can show inline. The trailing tail isn't returned — `path` is the
298 /// canonical reference.
299 ///
300 /// `head_bytes` controls how much inline content the caller wants to
301 /// keep. Pass `threshold` for "preserve as much as fits inline" or
302 /// a smaller value (e.g. `4 * 1024`) for "show a peek".
303 pub fn maybe_spillover(
304 id: &str,
305 content: &str,
306 threshold: usize,
307 head_bytes: usize,
308 ) -> io::Result<Option<(String, PathBuf)>> {
309 if content.len() <= threshold {
310 return Ok(None);
311 }
312 let path = write_spillover(id, content)?;
313 // Don't slice mid-utf8: walk back to a char boundary if needed.
314 let cut = head_bytes.min(content.len());
315 let cut = (0..=cut)
316 .rev()
317 .find(|&i| content.is_char_boundary(i))
318 .unwrap_or(0);
319 Ok(Some((content[..cut].to_string(), path)))
320 }
321
322 /// Inline head retained when [`apply_spillover`] truncates a tool
323 /// result. 32 KiB is large enough for the model to keep meaningful
324 /// context (a long stack trace, a `git diff` head, a directory
325 /// listing of typical depth) without consuming the lion's share of
326 /// the per-turn context budget. The full output is preserved
327 /// internally and opens in the tool details view.
328 pub const SPILLOVER_HEAD_BYTES: usize = 32 * 1024;
329 /// Inline tail retained alongside the head so compiler summaries and final
330 /// test failures are not systematically hidden by truncation.
331 pub const SPILLOVER_TAIL_BYTES: usize = 8 * 1024;
332
333 /// Inline head/tail budgets for the adaptive evidence bands. Hybrid results
334 /// keep a generous 32 KiB head + 8 KiB tail so mid-size outputs stay mostly
335 /// readable; handle-only results keep a 16 KiB head + 4 KiB tail. The head
336 /// and tail windows never overlap ([`head_tail_windows`]).
337 const HYBRID_HEAD_BYTES: usize = 32 * 1024;
338 const HYBRID_TAIL_BYTES: usize = 8 * 1024;
339 const HANDLE_ONLY_HEAD_BYTES: usize = 16 * 1024;
340 const HANDLE_ONLY_TAIL_BYTES: usize = 4 * 1024;
341
342 /// Phrase used only by the TUI expand affordance and the UI-side detection of
343 /// historical truncated previews. Never emitted into model-facing content:
344 /// the model cannot open the tool details view, so the model-facing footer
345 /// carries the artifact path and a recovery instruction instead.
346 pub const SPILLOVER_PREVIEW_HINT: &str = "view full output in the tool details view";
347
348 /// Sentinel phrase the TUI matches on to recognise a current-format truncated
349 /// preview. It must stay a literal substring of every footer variant.
350 pub const SPILLOVER_RECOVERY_HINT: &str = "omitted range recovery:";
351
352 /// Model-facing recovery instruction for a truncated tool result.
353 ///
354 /// The previous text — "read it back with the read_file tool or with sed line
355 /// ranges" — named `read_file`, which is not model-visible at all (only `File`
356 /// is), and otherwise leaned on reaching the artifact by path. Reaching it by
357 /// path is *conditional*: `ToolContext::resolve_path` short-circuits under
358 /// trust mode, so `File action="read"` on an artifact under
359 /// `~/.codewhale/sessions/` succeeds in a trusted/auto session and is refused
360 /// as a path escape otherwise — and even when it succeeds it pages the file
361 /// rather than seeking the omitted range. Meanwhile `retrieve_tool_result` —
362 /// model-visible, purpose-built, unconditional, and already named correctly by
363 /// the web overflow path in `tools/web/overflow.rs` — went unmentioned.
364 /// `tests/adaptive_evidence_acceptance.rs` proves end to end that a model
365 /// handed one of these receipts can take the named ref and get the omitted
366 /// bytes back.
367 ///
368 /// The distinction that matters is retrievability, not tidiness. An adaptive
369 /// session artifact carries an `art_<id>` the retrieval tool resolves, so name
370 /// it. A legacy global spillover is authorized by an ownership sidecar whose
371 /// write is allowed to fail (see [`publish_legacy_spillover_ownership`]), so
372 /// promising retrieval there would just be a fourth dead route; say plainly
373 /// that there is no tool call for it and name what does work instead.
374 pub(crate) fn spillover_recovery_instruction(retrieval_ref: Option<&str>) -> String {
375 match retrieval_ref {
376 Some(reference) => format!(
377 "{SPILLOVER_RECOVERY_HINT} call retrieve_tool_result with ref=\"{reference}\" \
378 (mode=\"tail\" for the end, mode=\"lines\" with lines=\"120-160\" for a range, \
379 mode=\"query\" with query=\"…\" to search it)"
380 ),
381 None => format!(
382 "{SPILLOVER_RECOVERY_HINT} no tool call reaches this copy — re-run the command \
383 with narrower output (a tighter filter, or head/tail) if you need the rest"
384 ),
385 }
386 }
387
388 /// Model-facing footer for a truncated tool result. Names how much was
389 /// omitted (bytes and lines), where the complete output lives on disk, and
390 /// how the model can read the omitted range back.
391 fn spillover_preview_footer(
392 omitted_bytes: usize,
393 omitted_lines: usize,
394 recovery_path: &str,
395 retrieval_ref: Option<&str>,
396 ) -> String {
397 format!(
398 "… {} of output omitted ({omitted_lines} lines) — full output at {recovery_path}; {}",
399 crate::artifacts::format_byte_size(omitted_bytes.try_into().unwrap_or(u64::MAX)),
400 spillover_recovery_instruction(retrieval_ref)
401 )
402 }
403
404 /// Split `content` into a head of at most `head_bytes` and a tail of at most
405 /// `tail_bytes` that never overlap: the tail window always starts at or after
406 /// the head window ends, so no byte of the output appears twice and the
407 /// omitted count is exact.
408 fn head_tail_windows(content: &str, head_bytes: usize, tail_bytes: usize) -> (&str, &str) {
409 let head_end = (0..=head_bytes.min(content.len()))
410 .rev()
411 .find(|&index| content.is_char_boundary(index))
412 .unwrap_or(0);
413 let tail_floor = content.len().saturating_sub(tail_bytes).max(head_end);
414 let tail_start = (tail_floor..=content.len())
415 .find(|&index| content.is_char_boundary(index))
416 .unwrap_or(content.len());
417 (&content[..head_end], &content[tail_start..])
418 }
419
420 /// Build the model-facing preview for a truncated tool result: the head, an
421 /// honest footer naming how much was omitted and where the full output can be
422 /// read back, and a short retained tail. When the head and tail windows cover
423 /// the whole output (nothing was actually omitted), the content is returned
424 /// unchanged — the preview never claims a truncation that did not happen.
425 fn truncated_preview(
426 head: &str,
427 tail: &str,
428 original: &str,
429 recovery_path: &str,
430 retrieval_ref: Option<&str>,
431 ) -> String {
432 let omitted = original.len().saturating_sub(head.len() + tail.len());
433 if omitted == 0 {
434 return original.to_string();
435 }
436 let omitted_lines = original[head.len()..original.len() - tail.len()]
437 .lines()
438 .count();
439 format!(
440 "{head}\n\n{}\n\n…\n{tail}",
441 spillover_preview_footer(omitted, omitted_lines, recovery_path, retrieval_ref)
442 )
443 }
444
445 /// Apply spillover to a tool result in place. If the result's
446 /// content exceeds [`SPILLOVER_THRESHOLD_BYTES`], writes the full
447 /// content to a sibling file under `~/.codewhale/tool_outputs/`,
448 /// replaces `result.content` with a [`SPILLOVER_HEAD_BYTES`] head
449 /// plus a footer naming the spillover path and how to read the
450 /// omitted range back, and stamps `metadata.spillover_path` so the
451 /// UI can render its expand annotation.
452 ///
453 /// Returns the spillover path on success, `None` if no spillover
454 /// happened (content small enough, error result, write failure).
455 /// Failures are logged but never bubble up — a tool that produced a
456 /// result shouldn't be marked failed because the spillover writer
457 /// couldn't reach disk; a bounded preview then says the full output could
458 /// not be saved, without advertising a recovery path or artifact.
459 ///
460 /// Error results (`success == false`) are skipped: error messages
461 /// are typically short, and turning them into a truncated preview
462 /// would just hide the error from the model's reasoning.
463 #[cfg_attr(not(test), expect(dead_code))]
464 pub fn apply_spillover(result: &mut ToolResult, tool_id: &str) -> Option<PathBuf> {
465 apply_spillover_inner(result, tool_id, None, false)
466 }
467
468 /// Apply spillover and publish session-scoped exact evidence.
469 ///
470 /// The default (classic) path writes the full bytes under the origin session
471 /// and replaces oversized content with a bounded head/tail preview whose
472 /// footer names the artifact path and how to read the omitted range back.
473 /// The adaptive evidence lane is reachable only through the explicit
474 /// `CODEWHALE_ADAPTIVE_OUTPUT_ROUTING` opt-in.
475 pub fn apply_spillover_with_artifact(
476 result: &mut ToolResult,
477 tool_id: &str,
478 tool_name: &str,
479 session_id: &str,
480 ) -> Option<PathBuf> {
481 apply_spillover_inner(
482 result,
483 tool_id,
484 Some(ArtifactSpilloverContext {
485 tool_name,
486 session_id,
487 }),
488 false,
489 )
490 }
491
492 /// [`apply_spillover_with_artifact`] for callers whose error payloads are
493 /// routinely as large as their successes — sub-agent tool output is often a
494 /// full build log, so the root loop's pass-errors-through rationale does not
495 /// hold there.
496 pub(crate) fn apply_spillover_with_artifact_including_errors(
497 result: &mut ToolResult,
498 tool_id: &str,
499 tool_name: &str,
500 session_id: &str,
501 ) -> Option<PathBuf> {
502 apply_spillover_inner(
503 result,
504 tool_id,
505 Some(ArtifactSpilloverContext {
506 tool_name,
507 session_id,
508 }),
509 true,
510 )
511 }
512
513 #[derive(Clone, Copy)]
514 struct ArtifactSpilloverContext<'a> {
515 tool_name: &'a str,
516 session_id: &'a str,
517 }
518
519 fn apply_spillover_inner(
520 result: &mut ToolResult,
521 tool_id: &str,
522 artifact_context: Option<ArtifactSpilloverContext<'_>>,
523 bound_errors: bool,
524 ) -> Option<PathBuf> {
525 if crate::tools::large_output_router::adaptive_output_routing_enabled()
526 && let Some(context) = artifact_context
527 {
528 return apply_adaptive_evidence_inner(result, tool_id, context);
529 }
530 if !result.success && !bound_errors {
531 return None;
532 }
533 if result.content.len() <= SPILLOVER_THRESHOLD_BYTES {
534 return None;
535 }
536 let original_content = result.content.clone();
537 let outcome = match maybe_spillover(
538 tool_id,
539 &original_content,
540 SPILLOVER_THRESHOLD_BYTES,
541 SPILLOVER_HEAD_BYTES,
542 ) {
543 Ok(Some(pair)) => pair,
544 Ok(None) => return None,
545 Err(err) => {
546 tracing::warn!(
547 target: "spillover",
548 ?err,
549 tool_id,
550 "spillover write failed; retaining a bounded unsaved preview"
551 );
552 bound_unpersisted_output(result, SPILLOVER_HEAD_BYTES, SPILLOVER_TAIL_BYTES);
553 return None;
554 }
555 };
556 let (_head, path) = outcome;
557 let (head, tail) = head_tail_windows(
558 &original_content,
559 SPILLOVER_HEAD_BYTES,
560 SPILLOVER_TAIL_BYTES,
561 );
562 let digest = crate::hashing::sha256_hex(original_content.as_bytes());
563 let path_str = path.display().to_string();
564
565 // Keep publishing the legacy ownership proof even though the model-facing
566 // footer no longer mentions retrieval: the tool-details pager authorizes
567 // legacy spillover reads through this sidecar.
568 if let Some(context) = artifact_context
569 && let Err(err) = publish_legacy_spillover_ownership(
570 &path,
571 context.session_id,
572 original_content.as_bytes(),
573 )
574 {
575 tracing::warn!(
576 target: "spillover",
577 ?err,
578 tool_id,
579 "legacy spillover ownership publication failed"
580 );
581 }
582
583 let mut artifact_path = None;
584 if let Some(context) = artifact_context {
585 let artifact_id = crate::artifacts::artifact_id_for_tool_call(tool_id);
586 // Publish immutably, like adaptive evidence: a turn's artifact
587 // reference names these bytes by digest, so a later call that reuses
588 // the id must fail closed (legacy footer) instead of rewriting them.
589 match crate::artifacts::write_session_artifact_immutable(
590 context.session_id,
591 &artifact_id,
592 original_content.as_bytes(),
593 ) {
594 Ok((absolute_path, relative_path)) => {
595 let record = crate::artifacts::record_tool_output_artifact(
596 context.session_id,
597 tool_id,
598 context.tool_name,
599 relative_path.clone(),
600 &original_content,
601 );
602 result.content = truncated_preview(
603 head,
604 tail,
605 &original_content,
606 &crate::artifacts::format_artifact_relative_path(&absolute_path),
607 Some(artifact_id.as_str()),
608 );
609 artifact_path = Some((absolute_path, relative_path, record));
610 }
611 Err(err) => {
612 tracing::warn!(
613 target: "spillover",
614 ?err,
615 tool_id,
616 "session artifact write failed; falling back to legacy spillover footer"
617 );
618 }
619 }
620 }
621
622 if artifact_path.is_none() {
623 // Legacy fallback: no session artifact was written, so there is no
624 // `art_<id>` ref to hand over — only the on-disk path.
625 result.content = truncated_preview(head, tail, &original_content, &path_str, None);
626 }
627
628 let obj = metadata_object_mut(result);
629 if let Some((absolute_path, relative_path, record)) = artifact_path.as_ref() {
630 stamp_artifact_metadata(obj, absolute_path, relative_path, record);
631 obj.insert(
632 "legacy_spillover_path".into(),
633 serde_json::Value::String(path_str),
634 );
635 } else {
636 obj.insert("spillover_path".into(), serde_json::Value::String(path_str));
637 }
638 if let Some(obj) = result
639 .metadata
640 .as_mut()
641 .and_then(serde_json::Value::as_object_mut)
642 {
643 obj.insert("truncated".into(), serde_json::Value::Bool(true));
644 obj.insert(
645 "content_digest".into(),
646 serde_json::Value::String(format!("sha256:{digest}")),
647 );
648 if artifact_path.is_some() {
649 // Plain hex, the same form the adaptive path and the workspace
650 // read `revision` use; the turn artifact ref carries it verbatim.
651 obj.insert(
652 "artifact_digest".into(),
653 serde_json::Value::String(digest.clone()),
654 );
655 }
656 obj.insert(
657 "original_byte_count".into(),
658 serde_json::Value::Number(serde_json::Number::from(original_content.len() as u64)),
659 );
660 obj.insert(
661 "retained_head_bytes".into(),
662 serde_json::Value::Number(serde_json::Number::from(head.len() as u64)),
663 );
664 obj.insert(
665 "retained_tail_bytes".into(),
666 serde_json::Value::Number(serde_json::Number::from(tail.len() as u64)),
667 );
668 obj.insert(
669 "original_line_count".into(),
670 serde_json::Value::Number(serde_json::Number::from(
671 original_content.lines().count() as u64
672 )),
673 );
674 }
675 artifact_path
676 .map(|(absolute_path, _, _)| absolute_path)
677 .or(Some(path))
678 }
679
680 /// The result's metadata as a JSON object, created when absent. Metadata that
681 /// is not an object is kept under `_prior` rather than dropped.
682 fn metadata_object_mut(result: &mut ToolResult) -> &mut serde_json::Map<String, serde_json::Value> {
683 let metadata = result.metadata.get_or_insert_with(|| serde_json::json!({}));
684 if !metadata.is_object() {
685 let prior = std::mem::replace(metadata, serde_json::json!({}));
686 if let Some(obj) = metadata.as_object_mut() {
687 obj.insert("_prior".into(), prior);
688 }
689 }
690 metadata
691 .as_object_mut()
692 .expect("metadata was just made an object")
693 }
694
695 /// Bound a failed persistence attempt without turning its preview into exact
696 /// evidence. The same windows and footer authority serve both routing modes.
697 fn bound_unpersisted_output(result: &mut ToolResult, head_bytes: usize, tail_bytes: usize) {
698 let original = &result.content;
699 let (head, tail) = head_tail_windows(original, head_bytes, tail_bytes);
700 let omitted = original.len().saturating_sub(head.len() + tail.len());
701 if omitted == 0 {
702 return;
703 }
704 let original_bytes = original.len();
705 let original_lines = original.lines().count();
706 let head_len = head.len();
707 let tail_len = tail.len();
708 let omitted_lines = original[head_len..original_bytes - tail_len]
709 .lines()
710 .count();
711 let digest = crate::hashing::sha256_hex(original.as_bytes());
712 result.content = format!(
713 "{head}\n\n{}\n\n…\n{tail}",
714 preview_footer(omitted, omitted_lines, None, None)
715 );
716 let object = metadata_object_mut(result);
717 // A prior metadata envelope cannot claim that it retained these bytes.
718 for key in [
719 "spillover_path",
720 "legacy_spillover_path",
721 "artifact_path",
722 "artifact_id",
723 "artifact_session_id",
724 "artifact_relative_path",
725 "artifact_byte_size",
726 "artifact_digest",
727 "artifact_generation",
728 "artifact_encoding",
729 "artifact_retention_state",
730 "artifact_preview",
731 "artifact_record",
732 ] {
733 object.remove(key);
734 }
735 object.insert("evidence_available".into(), false.into());
736 object.insert("output_persistence_failed".into(), true.into());
737 object.insert("truncated".into(), true.into());
738 object.insert("content_digest".into(), format!("sha256:{digest}").into());
739 object.insert("original_byte_count".into(), original_bytes.into());
740 object.insert("original_line_count".into(), original_lines.into());
741 object.insert("retained_head_bytes".into(), head_len.into());
742 object.insert("retained_tail_bytes".into(), tail_len.into());
743 }
744
745 /// Stamp the keys the TUI, receipts and `retrieve_tool_result` use to find a
746 /// session artifact that holds a tool call's full output.
747 fn stamp_artifact_metadata(
748 obj: &mut serde_json::Map<String, serde_json::Value>,
749 absolute_path: &Path,
750 relative_path: &Path,
751 record: &crate::artifacts::ArtifactRecord,
752 ) {
753 let absolute = serde_json::Value::String(absolute_path.display().to_string());
754 obj.insert("spillover_path".into(), absolute.clone());
755 obj.insert("artifact_path".into(), absolute);
756 obj.insert("artifact_id".into(), record.id.clone().into());
757 obj.insert(
758 "artifact_session_id".into(),
759 record.session_id.clone().into(),
760 );
761 obj.insert(
762 "artifact_relative_path".into(),
763 crate::artifacts::format_artifact_relative_path(relative_path).into(),
764 );
765 obj.insert("artifact_byte_size".into(), record.byte_size.into());
766 obj.insert("artifact_preview".into(), record.preview.clone().into());
767 }
768
769 /// Save a tool call's full output as a session artifact before the model's
770 /// view of it is cut to the inline budget (#6508).
771 ///
772 /// The result's content is left as it is, so the UI cell still shows it; only
773 /// the metadata gains the artifact keys, so the model-context view can name a
774 /// ref `retrieve_tool_result` resolves. A result that already has an artifact
775 /// (spillover wrote one) is left alone. Returns whether an artifact exists
776 /// afterwards. Publication is immutable: a reused ID may name the same bytes,
777 /// but never replace earlier output. A failed write is logged and reported as
778 /// `false`; the caller's footer then says no tool call reaches the rest instead
779 /// of promising a ref.
780 pub(crate) fn preserve_full_output_for_model_context(
781 result: &mut ToolResult,
782 tool_id: &str,
783 tool_name: &str,
784 session_id: &str,
785 ) -> bool {
786 if result
787 .metadata
788 .as_ref()
789 .and_then(|metadata| metadata.get("output_persistence_failed"))
790 .and_then(serde_json::Value::as_bool)
791 == Some(true)
792 {
793 // Persistence already lost the omitted bytes. Saving this preview as
794 // the full output would manufacture an exact-evidence receipt.
795 return false;
796 }
797 if result
798 .metadata
799 .as_ref()
800 .and_then(|metadata| metadata.get("artifact_id"))
801 .is_some()
802 {
803 return true;
804 }
805 let artifact_id = crate::artifacts::artifact_id_for_tool_call(tool_id);
806 match crate::artifacts::write_session_artifact_immutable(
807 session_id,
808 &artifact_id,
809 result.content.as_bytes(),
810 ) {
811 Ok((absolute_path, relative_path)) => {
812 let record = crate::artifacts::record_tool_output_artifact(
813 session_id,
814 tool_id,
815 tool_name,
816 relative_path.clone(),
817 &result.content,
818 );
819 let digest = crate::hashing::sha256_hex(result.content.as_bytes());
820 let obj = metadata_object_mut(result);
821 stamp_artifact_metadata(obj, &absolute_path, &relative_path, &record);
822 obj.insert("content_digest".into(), format!("sha256:{digest}").into());
823 true
824 }
825 Err(err) => {
826 tracing::warn!(
827 target: "spillover",
828 ?err,
829 tool_id,
830 "could not save full tool output before cutting it for model context"
831 );
832 false
833 }
834 }
835 }
836
837 /// Bytes reserved beside the footer for the preview's blank lines and the
838 /// `…` separator, plus slack for the omitted-size text changing length.
839 const PREVIEW_FRAME_BYTES: usize = 32;
840
841 /// Footer for a preview whose full output lives at `recovery_path`, or, when
842 /// it is `None`, one that says plainly the full output could not be saved.
843 fn preview_footer(
844 omitted_bytes: usize,
845 omitted_lines: usize,
846 recovery_path: Option<&str>,
847 retrieval_ref: Option<&str>,
848 ) -> String {
849 match recovery_path {
850 Some(path) => spillover_preview_footer(omitted_bytes, omitted_lines, path, retrieval_ref),
851 None => format!(
852 "… {} of output omitted ({omitted_lines} lines) — the full output could not be saved; {}",
853 crate::artifacts::format_byte_size(omitted_bytes.try_into().unwrap_or(u64::MAX)),
854 spillover_recovery_instruction(None)
855 ),
856 }
857 }
858
859 /// Drop the path and preview before sacrificing the retrieval reference.
860 /// Budgets smaller than this complete instruction cannot contain a usable
861 /// receipt: keep it intact and let the wire's recovery-marker guard preserve it.
862 fn compact_recovery_footer(retrieval_ref: Option<&str>) -> String {
863 match retrieval_ref {
864 Some(reference) => {
865 format!("{SPILLOVER_RECOVERY_HINT} retrieve_tool_result ref=\"{reference}\"")
866 }
867 None => format!("{SPILLOVER_RECOVERY_HINT} re-run with narrower output"),
868 }
869 }
870
871 /// Fit `content` into `budget` characters for the model: nothing changes when
872 /// it already fits; otherwise a head (two thirds) and a tail (one third)
873 /// around one recovery footer. The cut is measured in bytes, so the result
874 /// never exceeds `budget` characters either, unless the budget cannot hold
875 /// even a compact recovery instruction (which must remain intact).
876 pub(crate) fn fit_to_inline_budget(
877 content: &str,
878 budget: usize,
879 recovery_path: Option<&str>,
880 retrieval_ref: Option<&str>,
881 ) -> String {
882 if content.chars().count() <= budget {
883 return content.to_string();
884 }
885 let footer = preview_footer(
886 content.len(),
887 content.lines().count(),
888 recovery_path,
889 retrieval_ref,
890 );
891 if footer.len() + PREVIEW_FRAME_BYTES > budget {
892 return compact_recovery_footer(retrieval_ref);
893 }
894 let room = budget.saturating_sub(footer.len() + PREVIEW_FRAME_BYTES);
895 let head_bytes = room * 2 / 3;
896 let (head, tail) = head_tail_windows(content, head_bytes, room - head_bytes);
897 let omitted = content.len().saturating_sub(head.len() + tail.len());
898 let omitted_lines = content[head.len()..content.len() - tail.len()]
899 .lines()
900 .count();
901 format!(
902 "{head}\n\n{}\n\n…\n{tail}",
903 preview_footer(omitted, omitted_lines, recovery_path, retrieval_ref)
904 )
905 }
906
907 /// Re-fit a spillover preview (head, footer, tail) to `budget` characters.
908 ///
909 /// Spillover already saved the full output, so this writes nothing: it keeps
910 /// a shorter head and tail from the preview's own windows (found through the
911 /// `retained_head_bytes`/`retained_tail_bytes` metadata spillover stamped) and
912 /// re-emits one footer that names the same recovery ref. Returns `None` when
913 /// the metadata does not describe this preview, so the caller can fall back.
914 pub(crate) fn refit_spilled_preview(
915 preview: &str,
916 metadata: &serde_json::Value,
917 budget: usize,
918 recovery_path: Option<&str>,
919 retrieval_ref: Option<&str>,
920 ) -> Option<String> {
921 if preview.chars().count() <= budget {
922 return Some(preview.to_string());
923 }
924 let field = |key: &str| {
925 metadata
926 .get(key)
927 .and_then(serde_json::Value::as_u64)
928 .and_then(|value| usize::try_from(value).ok())
929 };
930 let head_len = field("retained_head_bytes")?;
931 let tail_len = field("retained_tail_bytes")?;
932 let original_len = field("original_byte_count")?;
933 if head_len + tail_len > preview.len()
934 || !preview.is_char_boundary(head_len)
935 || !preview.is_char_boundary(preview.len() - tail_len)
936 {
937 return None;
938 }
939 let head_window = &preview[..head_len];
940 let tail_window = &preview[preview.len() - tail_len..];
941
942 let footer = preview_footer(
943 original_len,
944 field("original_line_count").unwrap_or(0),
945 recovery_path,
946 retrieval_ref,
947 );
948 if footer.len() + PREVIEW_FRAME_BYTES > budget {
949 return Some(compact_recovery_footer(retrieval_ref));
950 }
951 let room = budget.saturating_sub(footer.len() + PREVIEW_FRAME_BYTES);
952 let head_bytes = (room * 2 / 3).min(head_len);
953 let tail_bytes = (room - head_bytes).min(tail_len);
954 let (head, _) = head_tail_windows(head_window, head_bytes, 0);
955 let (_, tail) = head_tail_windows(tail_window, 0, tail_bytes);
956 let omitted = original_len.saturating_sub(head.len() + tail.len());
957 let kept_lines = head.lines().count() + tail.lines().count();
958 let omitted_lines = match field("original_line_count") {
959 Some(total) => total.saturating_sub(kept_lines),
960 // Older spills did not record a line count; count only the lines this
961 // preview can see, which understates rather than invents.
962 None => {
963 head_window[head.len()..].lines().count()
964 + tail_window[..tail_window.len() - tail.len()]
965 .lines()
966 .count()
967 }
968 };
969 Some(format!(
970 "{head}\n\n{}\n\n…\n{tail}",
971 preview_footer(omitted, omitted_lines, recovery_path, retrieval_ref)
972 ))
973 }
974
975 fn apply_adaptive_evidence_inner(
976 result: &mut ToolResult,
977 tool_id: &str,
978 context: ArtifactSpilloverContext<'_>,
979 ) -> Option<PathBuf> {
980 use crate::tools::large_output_router::{
981 DEFAULT_LARGE_OUTPUT_THRESHOLD_TOKENS, EVIDENCE_RETENTION_SECS, EvidenceArtifact,
982 EvidenceRetentionState, EvidenceRouting, estimate_tokens, publish_evidence_metadata,
983 unix_millis_now,
984 };
985
986 let estimated_tokens = estimate_tokens(&result.content);
987 let threshold = result
988 .metadata
989 .as_ref()
990 .and_then(|metadata| metadata.get("evidence_threshold_tokens"))
991 .and_then(serde_json::Value::as_u64)
992 .and_then(|value| usize::try_from(value).ok())
993 .unwrap_or(DEFAULT_LARGE_OUTPUT_THRESHOLD_TOKENS);
994 let routing = result
995 .metadata
996 .as_ref()
997 .and_then(|metadata| metadata.get("evidence_routing"))
998 .cloned()
999 .and_then(|value| serde_json::from_value::<EvidenceRouting>(value).ok())
1000 .unwrap_or_else(|| EvidenceRouting::from_token_estimate(estimated_tokens, threshold));
1001 if routing == EvidenceRouting::Inline {
1002 return None;
1003 }
1004
1005 let original = result.content.clone();
1006 let (head_bytes, tail_bytes) = if routing == EvidenceRouting::Hybrid {
1007 (HYBRID_HEAD_BYTES, HYBRID_TAIL_BYTES)
1008 } else {
1009 (HANDLE_ONLY_HEAD_BYTES, HANDLE_ONLY_TAIL_BYTES)
1010 };
1011 let (head, tail) = head_tail_windows(&original, head_bytes, tail_bytes);
1012 let omitted = original.len().saturating_sub(head.len() + tail.len());
1013 if omitted == 0 {
1014 // The whole output fits inside the preview budget: there is nothing
1015 // to recover, so publishing an artifact and claiming a truncation
1016 // would both be dishonest. Pass the content through unchanged.
1017 return None;
1018 }
1019 let head_len = head.len();
1020 let tail_len = tail.len();
1021
1022 let artifact_id = crate::artifacts::artifact_id_for_tool_call(tool_id);
1023 let relative_path = crate::artifacts::session_artifact_relative_path(&artifact_id);
1024 let digest = crate::hashing::sha256_hex(original.as_bytes());
1025 let now_ms = unix_millis_now();
1026 let proposed_artifact = EvidenceArtifact {
1027 handle: artifact_id.clone(),
1028 digest: digest.clone(),
1029 size_bytes: original.len().try_into().unwrap_or(u64::MAX),
1030 content_type: if serde_json::from_str::<serde_json::Value>(&original).is_ok() {
1031 "application/json".to_string()
1032 } else {
1033 "text/plain".to_string()
1034 },
1035 tool_name: context.tool_name.to_string(),
1036 call_id: tool_id.to_string(),
1037 origin_session: context.session_id.to_string(),
1038 generation: 1,
1039 redacted: false,
1040 encoding: "utf-8".to_string(),
1041 retention_state: EvidenceRetentionState::Live,
1042 created_at_unix_ms: now_ms,
1043 retain_until_unix_ms: now_ms.saturating_add(EVIDENCE_RETENTION_SECS * 1_000),
1044 storage_path: relative_path.clone(),
1045 };
1046 let artifact = match crate::tools::large_output_router::read_evidence_metadata(
1047 context.session_id,
1048 &artifact_id,
1049 ) {
1050 Ok(existing)
1051 if existing.digest == proposed_artifact.digest
1052 && existing.size_bytes == proposed_artifact.size_bytes
1053 && existing.call_id == proposed_artifact.call_id
1054 && existing.origin_session == proposed_artifact.origin_session =>
1055 {
1056 existing
1057 }
1058 Ok(_) => {
1059 tracing::warn!(target: "evidence", tool_id, "adaptive evidence replay conflicts with immutable metadata");
1060 bound_unpersisted_output(result, head_bytes, tail_bytes);
1061 return None;
1062 }
1063 Err(err) if err.kind() == std::io::ErrorKind::NotFound => {
1064 if let Err(err) = publish_evidence_metadata(context.session_id, &proposed_artifact) {
1065 tracing::warn!(target: "evidence", ?err, tool_id, "adaptive evidence metadata publication failed");
1066 bound_unpersisted_output(result, head_bytes, tail_bytes);
1067 return None;
1068 }
1069 proposed_artifact
1070 }
1071 Err(err) => {
1072 tracing::warn!(target: "evidence", ?err, tool_id, "adaptive evidence metadata validation failed");
1073 bound_unpersisted_output(result, head_bytes, tail_bytes);
1074 return None;
1075 }
1076 };
1077
1078 // Seal the ownership/integrity record before publishing predictable
1079 // `art_<call>.txt` bytes. If metadata publication fails, no payload exists
1080 // for a guessed handle to retrieve without the generation, redaction,
1081 // retention, size, and digest checks above. A metadata-only interruption
1082 // is safe: the handle is never advertised and a retry can idempotently
1083 // publish the matching bytes.
1084 let (absolute_path, relative_path) = match crate::artifacts::write_session_artifact_immutable(
1085 context.session_id,
1086 &artifact_id,
1087 original.as_bytes(),
1088 ) {
1089 Ok(paths) => paths,
1090 Err(err) => {
1091 tracing::warn!(target: "evidence", ?err, tool_id, "adaptive evidence content publication failed");
1092 bound_unpersisted_output(result, head_bytes, tail_bytes);
1093 return None;
1094 }
1095 };
1096
1097 let record = crate::artifacts::record_tool_output_artifact(
1098 context.session_id,
1099 tool_id,
1100 context.tool_name,
1101 relative_path.clone(),
1102 &original,
1103 );
1104 result.content = truncated_preview(
1105 head,
1106 tail,
1107 &original,
1108 &crate::artifacts::format_artifact_relative_path(&absolute_path),
1109 Some(artifact_id.as_str()),
1110 );
1111 let metadata = result.metadata.get_or_insert_with(|| serde_json::json!({}));
1112 if let Some(object) = metadata.as_object_mut() {
1113 object.insert(
1114 "spillover_path".into(),
1115 absolute_path.display().to_string().into(),
1116 );
1117 object.insert("artifact_id".into(), artifact_id.into());
1118 object.insert("artifact_session_id".into(), context.session_id.into());
1119 object.insert(
1120 "artifact_relative_path".into(),
1121 crate::artifacts::format_artifact_relative_path(&relative_path).into(),
1122 );
1123 object.insert("artifact_byte_size".into(), artifact.size_bytes.into());
1124 object.insert("artifact_digest".into(), digest.into());
1125 object.insert("artifact_generation".into(), artifact.generation.into());
1126 object.insert("artifact_encoding".into(), artifact.encoding.into());
1127 object.insert("artifact_retention_state".into(), "live".into());
1128 object.insert("evidence_available".into(), true.into());
1129 object.insert("truncated".into(), true.into());
1130 object.insert("original_byte_count".into(), artifact.size_bytes.into());
1131 object.insert("retained_head_bytes".into(), head_len.into());
1132 object.insert("retained_tail_bytes".into(), tail_len.into());
1133 object.insert(
1134 "artifact_preview".into(),
1135 original.chars().take(200).collect::<String>().into(),
1136 );
1137 object.insert(
1138 "artifact_record".into(),
1139 serde_json::to_value(record).unwrap_or(serde_json::Value::Null),
1140 );
1141 }
1142 Some(absolute_path)
1143 }
1144
1145 /// Sanitise a tool call id for use as a filename. Keeps ASCII
1146 /// alphanumerics, `-`, and `_`; rejects `.` to keep `..` traversal
1147 /// out, rejects empty results. Returns `None` if the input contains
1148 /// no acceptable characters.
1149 fn sanitise_id(id: &str) -> Option<String> {
1150 let cleaned: String = id
1151 .chars()
1152 .filter(|c| c.is_ascii_alphanumeric() || *c == '-' || *c == '_')
1153 .collect();
1154 if cleaned.is_empty() {
1155 None
1156 } else {
1157 Some(cleaned)
1158 }
1159 }
1160
1161 /// Override the storage roots for tests so they don't pollute the
1162 /// user's real `~/.codewhale/` directory. This uses explicit test hooks instead
1163 /// of `$HOME` because Windows home-dir resolution can ignore environment
1164 /// overrides and return the runner profile directory.
1165 #[cfg(test)]
1166 pub(crate) fn with_test_home<F, R>(home: &Path, f: F) -> R
1167 where
1168 F: FnOnce() -> R,
1169 {
1170 let _artifact_guard = crate::artifacts::TEST_ARTIFACT_SESSIONS_GUARD
1171 .lock()
1172 .unwrap_or_else(|err| err.into_inner());
1173
1174 struct StorageRootOverride {
1175 prior_spillover: Option<PathBuf>,
1176 prior_artifacts: Option<PathBuf>,
1177 }
1178
1179 impl Drop for StorageRootOverride {
1180 fn drop(&mut self) {
1181 set_test_spillover_root(self.prior_spillover.take());
1182 crate::artifacts::set_test_artifact_sessions_root(self.prior_artifacts.take());
1183 }
1184 }
1185
1186 // Tests in this module serialize spillover through `TEST_GUARD`; the
1187 // artifact guard above protects the session-artifact root shared with
1188 // artifacts.rs tests.
1189 let prior_spillover =
1190 set_test_spillover_root(Some(home.join(".codewhale").join(SPILLOVER_DIR_NAME)));
1191 let prior_artifacts = crate::artifacts::set_test_artifact_sessions_root(Some(
1192 home.join(".codewhale").join("sessions"),
1193 ));
1194 let _restore = StorageRootOverride {
1195 prior_spillover,
1196 prior_artifacts,
1197 };
1198 f()
1199 }
1200
1201 #[cfg(test)]
1202 mod tests {
1203 use super::*;
1204 use tempfile::tempdir;
1205
1206 #[test]
1207 fn tiny_inline_budgets_keep_the_complete_retrieval_reference() {
1208 let content = "你好 output\n".repeat(500);
1209 let path = format!(
1210 "/very-long-workspace/{}/art_call-1.txt",
1211 "nested/".repeat(100)
1212 );
1213 let reference = "art_12345678-1234-1234-1234-123456789abc";
1214 let metadata = serde_json::json!({
1215 "retained_head_bytes": 1000,
1216 "retained_tail_bytes": 1000,
1217 "original_byte_count": content.len(),
1218 "original_line_count": content.lines().count(),
1219 });
1220 let preview = format!(
1221 "{}\n\n{}\n\n…\n{}",
1222 "h".repeat(1000),
1223 spillover_preview_footer(content.len() - 2000, 300, &path, Some(reference)),
1224 "t".repeat(1000),
1225 );
1226 for budget in [120, 160, 200, 1200] {
1227 for fitted in [
1228 fit_to_inline_budget(&content, budget, Some(&path), Some(reference)),
1229 refit_spilled_preview(&preview, &metadata, budget, Some(&path), Some(reference))
1230 .unwrap(),
1231 ] {
1232 assert!(fitted.len() <= budget, "{budget}: {fitted}");
1233 assert!(fitted.contains(&format!("ref=\"{reference}\"")), "{fitted}");
1234 assert!(fitted.contains("retrieve_tool_result"), "{fitted}");
1235 assert!(fitted.contains(SPILLOVER_RECOVERY_HINT), "{fitted}");
1236 }
1237 }
1238 // Even an impossible allowance must not amputate a reference. The
1239 // marker tells the wire backstop to preserve this complete receipt.
1240 let fitted = fit_to_inline_budget(&content, 1, Some(&path), Some(reference));
1241 assert!(fitted.contains(reference));
1242 assert!(fitted.contains(SPILLOVER_RECOVERY_HINT));
1243 let unsaved = fit_to_inline_budget(&content, 120, None, None);
1244 assert!(unsaved.len() <= 120);
1245 assert!(!unsaved.contains("retrieve_tool_result"));
1246 }
1247
1248 #[test]
1249 fn model_context_archive_reuse_keeps_original_bytes_and_reports_conflicts() {
1250 let _guard = setup();
1251 let home = tempdir().unwrap();
1252 with_test_home(home.path(), || {
1253 let provider = crate::config::ProviderKind::Deepseek;
1254 let model = "deepseek-v3.2-128k";
1255 let budget =
1256 crate::route_budget::route_inline_char_budget_for_route(provider, model, None);
1257 let original = "a".repeat(budget + 1_000);
1258 let prior = serde_json::json!({"prior_key": "prior_value"});
1259 let mut first = ToolResult::success(original.clone()).with_metadata(prior.clone());
1260 assert!(preserve_full_output_for_model_context(
1261 &mut first,
1262 "reused-call",
1263 "exec_shell",
1264 "session-inline",
1265 ));
1266 let metadata = first.metadata.clone().unwrap();
1267 let path = PathBuf::from(metadata["artifact_path"].as_str().unwrap());
1268 assert_eq!(fs::read(&path).unwrap(), original.as_bytes());
1269 assert_eq!(first.content, original);
1270 assert_eq!(metadata["prior_key"], prior["prior_key"]);
1271
1272 // An identical replay still names the original immutable bytes.
1273 let mut identical = ToolResult::success(original.clone()).with_metadata(prior.clone());
1274 assert!(preserve_full_output_for_model_context(
1275 &mut identical,
1276 "reused-call",
1277 "exec_shell",
1278 "session-inline",
1279 ));
1280 assert_eq!(identical.metadata.as_ref(), Some(&metadata));
1281
1282 let changed = "b".repeat(original.len());
1283 let mut conflict = ToolResult::success(changed.clone()).with_metadata(prior.clone());
1284 assert!(!preserve_full_output_for_model_context(
1285 &mut conflict,
1286 "reused-call",
1287 "exec_shell",
1288 "session-inline",
1289 ));
1290 assert_eq!(fs::read(&path).unwrap(), original.as_bytes());
1291 assert_eq!(first.metadata.as_ref(), Some(&metadata));
1292 assert_eq!(conflict.content, changed);
1293 assert_eq!(conflict.metadata.as_ref(), Some(&prior));
1294
1295 // Use the real model-context projection, not a fabricated footer.
1296 let view = crate::core::engine::compact_tool_result_for_route(
1297 provider,
1298 model,
1299 None,
1300 "exec_shell",
1301 &conflict,
1302 );
1303 assert!(view.contains("the full output could not be saved"));
1304 assert!(view.contains("no tool call reaches this copy"));
1305 assert!(!view.contains("retrieve_tool_result"));
1306 assert!(!view.contains("art_reused-call"));
1307 });
1308 }
1309
1310 /// Tests in this module serialize through this guard because they mutate
1311 /// process-global test storage roots. Without it, cargo's parallel runner
1312 /// would observe interleaved overrides.
1313 fn setup() -> std::sync::MutexGuard<'static, ()> {
1314 super::TEST_SPILLOVER_GUARD
1315 .lock()
1316 .unwrap_or_else(|e| e.into_inner())
1317 }
1318
1319 /// Run the adaptive evidence lane directly. These cases exercise it
1320 /// without the `CODEWHALE_ADAPTIVE_OUTPUT_ROUTING` process opt-in so
1321 /// parallel tests keep a deterministic routing decision.
1322 fn adaptive_spillover(
1323 result: &mut ToolResult,
1324 tool_id: &str,
1325 tool_name: &str,
1326 session_id: &str,
1327 ) -> Option<PathBuf> {
1328 apply_adaptive_evidence_inner(
1329 result,
1330 tool_id,
1331 ArtifactSpilloverContext {
1332 tool_name,
1333 session_id,
1334 },
1335 )
1336 }
1337
1338 fn assert_unsaved_preview(result: &mut ToolResult, original: &str, success: bool) {
1339 assert_eq!(result.success, success);
1340 assert!(result.content.len() < SPILLOVER_HEAD_BYTES + SPILLOVER_TAIL_BYTES + 1024);
1341 assert!(result.content.starts_with("first 🐳\n"));
1342 assert!(result.content.ends_with("last 🐳\n"));
1343 assert!(
1344 result
1345 .content
1346 .contains("the full output could not be saved")
1347 );
1348 assert!(
1349 result
1350 .content
1351 .contains("re-run the command with narrower output")
1352 );
1353 assert!(!result.content.contains("full output at"));
1354 assert!(!result.content.contains("retrieve_tool_result"));
1355 let metadata = result.metadata.as_ref().unwrap();
1356 assert_eq!(metadata["original_byte_count"], original.len());
1357 assert_eq!(metadata["evidence_available"], false);
1358 assert_eq!(metadata["output_persistence_failed"], true);
1359 assert_eq!(metadata["truncated"], true);
1360 assert_eq!(metadata["fixture"], "preserved");
1361 assert_eq!(
1362 metadata["content_digest"],
1363 format!("sha256:{}", crate::hashing::sha256_hex(original.as_bytes()))
1364 );
1365 assert!(metadata.get("artifact_id").is_none());
1366 assert!(metadata.get("spillover_path").is_none());
1367 assert!(!preserve_full_output_for_model_context(
1368 result,
1369 "retry",
1370 "exec_shell",
1371 "retry-session"
1372 ));
1373 }
1374
1375 #[test]
1376 fn classic_spillover_write_failure_bounds_success_and_error_without_fake_evidence() {
1377 let _g = setup();
1378 let tmp = tempdir().unwrap();
1379 with_test_home(tmp.path(), || {
1380 let root = spillover_root().unwrap();
1381 fs::create_dir_all(root.parent().unwrap()).unwrap();
1382 fs::write(&root, "keep this file").unwrap();
1383 let original = format!("first 🐳\n{}last 🐳\n", "middle 🐳\n".repeat(20_000));
1384 for success in [true, false] {
1385 let mut result = ToolResult::success(original.clone());
1386 result.success = success;
1387 result.metadata = Some(serde_json::json!({"fixture": "preserved"}));
1388 assert!(apply_spillover_inner(&mut result, "call-failed", None, true).is_none());
1389 assert_unsaved_preview(&mut result, &original, success);
1390 }
1391 assert_eq!(fs::read_to_string(root).unwrap(), "keep this file");
1392 });
1393 }
1394
1395 #[test]
1396 fn adaptive_persistence_failures_bound_output_and_keep_existing_evidence_immutable() {
1397 let _g = setup();
1398 let tmp = tempdir().unwrap();
1399 with_test_home(tmp.path(), || {
1400 let original = format!("first 🐳\n{}last 🐳\n", "middle 🐳\n".repeat(20_000));
1401 for failure in [
1402 "metadata-write",
1403 "metadata-invalid",
1404 "metadata-conflict",
1405 "content-conflict",
1406 ] {
1407 for success in [true, false] {
1408 let session = format!("{failure}-{success}");
1409 let directory = tmp
1410 .path()
1411 .join(".codewhale/sessions")
1412 .join(&session)
1413 .join("artifacts");
1414 fs::create_dir_all(directory.parent().unwrap()).unwrap();
1415 let canary = match failure {
1416 "metadata-write" => directory.clone(),
1417 "metadata-invalid" | "metadata-conflict" => {
1418 directory.join("art_call-failed.evidence.json")
1419 }
1420 _ => directory.join("art_call-failed.txt"),
1421 };
1422 if failure != "metadata-write" {
1423 fs::create_dir_all(&directory).unwrap();
1424 }
1425 if failure == "metadata-conflict" {
1426 let mut retained =
1427 ToolResult::success("retained evidence\n".repeat(10_000));
1428 adaptive_spillover(&mut retained, "call-failed", "exec_shell", &session)
1429 .unwrap();
1430 } else {
1431 fs::write(&canary, "keep this file").unwrap();
1432 }
1433 let canary_before = fs::read(&canary).unwrap();
1434 let mut result = ToolResult::success(original.clone());
1435 result.success = success;
1436 result.metadata = Some(serde_json::json!({
1437 "fixture": "preserved",
1438 "artifact_id": "stale",
1439 "spillover_path": "/stale",
1440 "evidence_available": true,
1441 }));
1442 assert!(
1443 adaptive_spillover(&mut result, "call-failed", "exec_shell", &session)
1444 .is_none()
1445 );
1446 assert_unsaved_preview(&mut result, &original, success);
1447 assert_eq!(fs::read(&canary).unwrap(), canary_before);
1448 if failure == "metadata-conflict" {
1449 assert_eq!(
1450 fs::read_to_string(directory.join("art_call-failed.txt")).unwrap(),
1451 "retained evidence\n".repeat(10_000)
1452 );
1453 }
1454 }
1455 }
1456 assert!(
1457 !tmp.path()
1458 .join(".codewhale/sessions/retry-session")
1459 .exists()
1460 );
1461 });
1462 }
1463
1464 /// The old hint named `read_file`, which is not registered for the model
1465 /// at all, and otherwise pointed at routes that only reach the artifact
1466 /// under trust mode (see [`spillover_recovery_instruction`]), while never
1467 /// naming `retrieve_tool_result` — model-visible, unconditional, and built
1468 /// for exactly this.
1469 #[test]
1470 fn truncation_footer_names_a_recovery_route_that_works() {
1471 let footer = spillover_preview_footer(
1472 4096,
1473 120,
1474 "/tmp/artifacts/art_call-1.txt",
1475 Some("art_call-1"),
1476 );
1477
1478 assert!(footer.contains("retrieve_tool_result"), "{footer}");
1479 assert!(footer.contains("ref=\"art_call-1\""), "{footer}");
1480 assert!(!footer.contains("read_file"), "{footer}");
1481 assert!(!footer.contains("sed"), "{footer}");
1482 }
1483
1484 /// The legacy global copy has no guaranteed authorization sidecar, so the
1485 /// footer must not invent a fourth dead route — but it still must not name
1486 /// the three it used to.
1487 #[test]
1488 fn truncation_footer_without_an_artifact_promises_nothing_it_cannot_deliver() {
1489 let footer = spillover_preview_footer(4096, 120, "/tmp/tool_outputs/call-1.txt", None);
1490
1491 assert!(
1492 footer.contains("no tool call reaches this copy"),
1493 "{footer}"
1494 );
1495 assert!(!footer.contains("retrieve_tool_result"), "{footer}");
1496 assert!(!footer.contains("read_file"), "{footer}");
1497 assert!(!footer.contains("sed"), "{footer}");
1498 }
1499
1500 /// The TUI keys its "this preview was truncated" detection off the shared
1501 /// constant, so it has to stay a literal substring of every variant.
1502 #[test]
1503 fn every_footer_variant_carries_the_ui_detection_marker() {
1504 for reference in [Some("art_call-1"), None] {
1505 let footer = spillover_preview_footer(4096, 120, "/tmp/x.txt", reference);
1506 assert!(footer.contains(SPILLOVER_RECOVERY_HINT), "{footer}");
1507 }
1508 }
1509
1510 #[test]
1511 fn with_test_home_overrides_storage_roots_without_home_resolution() {
1512 let _g = setup();
1513 let tmp = tempdir().unwrap();
1514
1515 with_test_home(tmp.path(), || {
1516 assert_eq!(
1517 spillover_root().as_deref(),
1518 Some(tmp.path().join(".codewhale").join("tool_outputs").as_path())
1519 );
1520 assert_eq!(
1521 crate::artifacts::session_artifact_absolute_path(
1522 "session-123",
1523 &PathBuf::from("artifacts").join("art_call-big.txt")
1524 )
1525 .as_deref(),
1526 Some(
1527 tmp.path()
1528 .join(".codewhale")
1529 .join("sessions")
1530 .join("session-123")
1531 .join("artifacts")
1532 .join("art_call-big.txt")
1533 .as_path()
1534 )
1535 );
1536 });
1537 }
1538
1539 #[test]
1540 fn sanitise_id_keeps_safe_chars_and_drops_dangerous() {
1541 assert_eq!(super::sanitise_id("abc-123_x"), Some("abc-123_x".into()));
1542 // `.` is dropped to keep `..` out of the path.
1543 assert_eq!(super::sanitise_id("../etc"), Some("etc".into()));
1544 assert_eq!(super::sanitise_id("/etc/passwd"), Some("etcpasswd".into()));
1545 // Empty-after-sanitise → None.
1546 assert!(super::sanitise_id("...").is_none());
1547 assert!(super::sanitise_id("").is_none());
1548 }
1549
1550 #[test]
1551 fn write_spillover_creates_directory_and_writes_file() {
1552 let _g = setup();
1553 let tmp = tempdir().unwrap();
1554 with_test_home(tmp.path(), || {
1555 let path = write_spillover("call-abc", "hello world").expect("write");
1556 assert!(path.exists(), "{path:?} missing");
1557 let body = fs::read_to_string(&path).unwrap();
1558 assert_eq!(body, "hello world");
1559 // Directory landed under `<HOME>/.codewhale/tool_outputs/`.
1560 // Compare components instead of a substring on `to_string_lossy`
1561 // — Windows uses `\` as the separator so a `/` substring match
1562 // would falsely fail there.
1563 let components: Vec<&str> = path
1564 .components()
1565 .filter_map(|c| c.as_os_str().to_str())
1566 .collect();
1567 assert!(
1568 components.contains(&".codewhale") && components.contains(&"tool_outputs"),
1569 "spillover path missing expected `.codewhale/tool_outputs/...` segments: {path:?}"
1570 );
1571 });
1572 }
1573
1574 #[test]
1575 fn write_spillover_rejects_empty_id() {
1576 let _g = setup();
1577 let tmp = tempdir().unwrap();
1578 with_test_home(tmp.path(), || {
1579 let err = write_spillover("...", "x").unwrap_err();
1580 assert_eq!(err.kind(), io::ErrorKind::InvalidInput);
1581 });
1582 }
1583
1584 #[test]
1585 fn maybe_spillover_returns_none_below_threshold() {
1586 let _g = setup();
1587 let tmp = tempdir().unwrap();
1588 with_test_home(tmp.path(), || {
1589 let out = maybe_spillover("call-1", "tiny content", 100 * 1024, 4 * 1024).expect("ok");
1590 assert!(out.is_none());
1591 });
1592 }
1593
1594 #[test]
1595 fn maybe_spillover_writes_and_returns_head_above_threshold() {
1596 let _g = setup();
1597 let tmp = tempdir().unwrap();
1598 with_test_home(tmp.path(), || {
1599 // Content larger than the threshold.
1600 let big = "A".repeat(2_000);
1601 let (head, path) = maybe_spillover("call-2", &big, 1_000, 256)
1602 .expect("ok")
1603 .expect("should have spilled");
1604 // Head is bounded.
1605 assert_eq!(head.len(), 256);
1606 // Full content on disk.
1607 let body = fs::read_to_string(&path).unwrap();
1608 assert_eq!(body.len(), 2_000);
1609 });
1610 }
1611
1612 #[test]
1613 fn maybe_spillover_does_not_split_inside_a_codepoint() {
1614 let _g = setup();
1615 let tmp = tempdir().unwrap();
1616 with_test_home(tmp.path(), || {
1617 // 4 byte chars; ask for 3 bytes of head → walks back to
1618 // the previous char boundary (0).
1619 let s = "🐳🐳🐳🐳"; // 4 × 4-byte codepoints
1620 assert_eq!(s.len(), 16);
1621 let (head, _) = maybe_spillover("call-3", s, 1, 3)
1622 .expect("ok")
1623 .expect("spilled");
1624 // 3 isn't a char boundary in this string; walk back → 0.
1625 assert_eq!(head, "");
1626 // Asking for 4 bytes lands on the first char boundary.
1627 let (head, _) = maybe_spillover("call-3b", s, 1, 4)
1628 .expect("ok")
1629 .expect("spilled");
1630 assert_eq!(head, "🐳");
1631 });
1632 }
1633
1634 #[test]
1635 fn prune_older_than_handles_missing_root() {
1636 let _g = setup();
1637 let tmp = tempdir().unwrap();
1638 with_test_home(tmp.path(), || {
1639 // Nothing has ever written; root doesn't exist; that's fine.
1640 let count = prune_older_than(SPILLOVER_MAX_AGE).expect("ok");
1641 assert_eq!(count, 0);
1642 });
1643 }
1644
1645 // The mtime backdate uses utimensat (Unix-only). On Windows the
1646 // filetime_set_modified helper is a no-op, so the prune wouldn't see
1647 // any stale files. Gate the whole test on `cfg(unix)` instead of
1648 // testing a no-op path that can't fail meaningfully.
1649 #[test]
1650 #[cfg(unix)]
1651 fn prune_older_than_keeps_fresh_files_drops_stale_ones() {
1652 let _g = setup();
1653 let tmp = tempdir().unwrap();
1654 with_test_home(tmp.path(), || {
1655 let fresh = write_spillover("fresh", "x").unwrap();
1656 let stale = write_spillover("stale", "y").unwrap();
1657
1658 // Backdate `stale` to 30 days ago.
1659 let thirty_days = SystemTime::now() - Duration::from_secs(30 * 24 * 60 * 60);
1660 filetime_set_modified(&stale, thirty_days);
1661
1662 let pruned = prune_older_than(SPILLOVER_MAX_AGE).unwrap();
1663 assert_eq!(pruned, 1);
1664 assert!(fresh.exists());
1665 assert!(!stale.exists());
1666 });
1667 }
1668
1669 /// Set the mtime on a file. The workspace doesn't pull the
1670 /// `filetime` crate, so we reach for `utimensat` directly on
1671 /// Unix. Windows is a no-op — the prune semantics are the same
1672 /// and the per-cycle stress test lives on the Unix path.
1673 #[cfg(unix)]
1674 fn filetime_set_modified(path: &Path, when: SystemTime) {
1675 let secs = when
1676 .duration_since(SystemTime::UNIX_EPOCH)
1677 .unwrap_or_default()
1678 .as_secs() as libc::time_t;
1679 let times = [
1680 libc::timespec {
1681 tv_sec: secs,
1682 tv_nsec: 0,
1683 },
1684 libc::timespec {
1685 tv_sec: secs,
1686 tv_nsec: 0,
1687 },
1688 ];
1689 let path_c = std::ffi::CString::new(path.as_os_str().as_encoded_bytes()).unwrap();
1690 // SAFETY: path_c is a valid CString; times is a 2-element array
1691 // matching utimensat's signature.
1692 let rc = unsafe { libc::utimensat(libc::AT_FDCWD, path_c.as_ptr(), times.as_ptr(), 0) };
1693 assert_eq!(
1694 rc,
1695 0,
1696 "utimensat failed: {}",
1697 std::io::Error::last_os_error()
1698 );
1699 }
1700
1701 // Windows stub removed in v0.8.8 — the only caller of
1702 // `filetime_set_modified` is `prune_older_than_keeps_fresh_files_drops_stale_ones`,
1703 // which is now `#[cfg(unix)]` because mtime backdating requires
1704 // `utimensat` and a Windows no-op stub can't make the assertion pass
1705 // anyway. Keeping the stub triggered `-D dead-code` on Windows builds
1706 // (the prune test was the only caller) and broke `Test (windows-latest)`.
1707
1708 #[test]
1709 fn apply_spillover_is_noop_below_threshold() {
1710 let _g = setup();
1711 let tmp = tempdir().unwrap();
1712 with_test_home(tmp.path(), || {
1713 let mut result = ToolResult::success("small payload");
1714 let path = apply_spillover(&mut result, "call-small");
1715 assert!(path.is_none());
1716 assert_eq!(result.content, "small payload");
1717 assert!(result.metadata.is_none());
1718 });
1719 }
1720
1721 #[test]
1722 fn apply_spillover_is_noop_for_error_results() {
1723 let _g = setup();
1724 let tmp = tempdir().unwrap();
1725 with_test_home(tmp.path(), || {
1726 // Even very large error messages are passed through —
1727 // truncating an error would hide it from the model.
1728 let big_err = "boom\n".repeat(50_000);
1729 let mut result = ToolResult::error(big_err.clone());
1730 let path = apply_spillover(&mut result, "call-err");
1731 assert!(path.is_none());
1732 assert_eq!(result.content, big_err);
1733 });
1734 }
1735
1736 #[test]
1737 fn apply_spillover_truncates_and_stamps_metadata_above_threshold() {
1738 let _g = setup();
1739 let tmp = tempdir().unwrap();
1740 with_test_home(tmp.path(), || {
1741 // 200 KiB body — well above the 100 KiB threshold.
1742 let big = "X".repeat(200 * 1024);
1743 let mut result = ToolResult::success(big.clone());
1744 let path = apply_spillover(&mut result, "call-big").expect("should spill");
1745
1746 // Inline content shrunk to head + honest preview footer.
1747 assert!(result.content.len() < big.len());
1748 assert!(
1749 !result.content.contains(SPILLOVER_PREVIEW_HINT),
1750 "the tool-details phrase is a UI affordance, not model-facing"
1751 );
1752 assert!(
1753 result.content.contains("of output omitted"),
1754 "footer missing: {}",
1755 &result.content[result.content.len().saturating_sub(200)..]
1756 );
1757 // The footer tells the model where the full output lives and how
1758 // to read the omitted range back.
1759 assert!(result.content.contains("full output at"));
1760 assert!(result.content.contains(&path.display().to_string()));
1761 assert!(result.content.contains(SPILLOVER_RECOVERY_HINT));
1762 assert!(
1763 !result.content.contains("retrieve_tool_result"),
1764 "legacy spillover ownership can fail to publish; promising \
1765 retrieval here would be another dead route"
1766 );
1767 assert!(!result.content.contains("read_file"));
1768 assert!(!result.content.contains("sed"));
1769
1770 // Full bytes are on disk at the returned path.
1771 assert!(path.exists(), "spillover file missing: {path:?}");
1772 let body = fs::read_to_string(&path).unwrap();
1773 assert_eq!(body.len(), 200 * 1024);
1774
1775 // metadata.spillover_path stamped for the UI to find.
1776 let metadata = result.metadata.expect("metadata stamped");
1777 let stamped = metadata
1778 .get("spillover_path")
1779 .and_then(serde_json::Value::as_str)
1780 .expect("spillover_path key present");
1781 assert_eq!(stamped, path.display().to_string());
1782 assert_eq!(metadata["truncated"], true);
1783 assert_eq!(metadata["original_byte_count"], 200 * 1024);
1784 assert_eq!(metadata["retained_head_bytes"], SPILLOVER_HEAD_BYTES);
1785 assert_eq!(metadata["retained_tail_bytes"], SPILLOVER_TAIL_BYTES);
1786 assert!(
1787 metadata["content_digest"]
1788 .as_str()
1789 .is_some_and(|digest| digest.starts_with("sha256:"))
1790 );
1791 });
1792 }
1793
1794 #[test]
1795 fn apply_spillover_with_artifact_writes_session_file_and_plain_preview() {
1796 let _g = setup();
1797 let tmp = tempdir().unwrap();
1798 with_test_home(tmp.path(), || {
1799 let big = "checking crate ... error[E0425]: cannot find value\n".repeat(4_000);
1800 let mut result = ToolResult::success(big.clone());
1801 let path = adaptive_spillover(&mut result, "call-big", "exec_shell", "session-123")
1802 .expect("should spill");
1803
1804 let session_artifact = tmp
1805 .path()
1806 .join(".codewhale")
1807 .join("sessions")
1808 .join("session-123")
1809 .join("artifacts")
1810 .join("art_call-big.txt");
1811 assert_eq!(path, session_artifact);
1812 assert_eq!(fs::read_to_string(&session_artifact).unwrap(), big);
1813 assert!(
1814 !tmp.path()
1815 .join(".codewhale/tool_outputs/call-big.txt")
1816 .exists(),
1817 "adaptive evidence stores one exact origin-session copy"
1818 );
1819 // The model sees a bounded preview with an honest footer: the
1820 // artifact path plus the retrieval call that actually resolves it.
1821 assert!(!result.content.contains(SPILLOVER_PREVIEW_HINT));
1822 assert!(result.content.contains("\n…\n"));
1823 assert!(result.content.contains("of output omitted"));
1824 assert!(result.content.contains("full output at"));
1825 assert!(result.content.contains(SPILLOVER_RECOVERY_HINT));
1826 assert!(
1827 result.content.contains("art_call-big.txt"),
1828 "footer must name the artifact path so the model can recover the output"
1829 );
1830 assert!(!result.content.contains("Exact evidence retained"));
1831 assert!(
1832 result.content.contains("retrieve_tool_result"),
1833 "a session artifact is retrievable; the footer must say so: {}",
1834 result.content
1835 );
1836 assert!(
1837 result.content.contains("ref=\"art_call-big\""),
1838 "the footer must hand over a ref that resolves: {}",
1839 result.content
1840 );
1841 assert!(
1842 session_artifact
1843 .with_file_name("art_call-big.evidence.json")
1844 .exists()
1845 );
1846
1847 let metadata = result.metadata.expect("metadata stamped");
1848 assert_eq!(
1849 metadata
1850 .get("artifact_id")
1851 .and_then(serde_json::Value::as_str),
1852 Some("art_call-big")
1853 );
1854 assert_eq!(
1855 metadata
1856 .get("artifact_relative_path")
1857 .and_then(serde_json::Value::as_str),
1858 Some("artifacts/art_call-big.txt")
1859 );
1860 assert_eq!(
1861 metadata
1862 .get("artifact_session_id")
1863 .and_then(serde_json::Value::as_str),
1864 Some("session-123")
1865 );
1866 assert_eq!(metadata["original_byte_count"], big.len());
1867 assert!(metadata["retained_head_bytes"].as_u64().unwrap_or(0) <= 16 * 1024);
1868 assert!(metadata["retained_tail_bytes"].as_u64().unwrap_or(0) <= 4 * 1024);
1869 });
1870 }
1871
1872 #[test]
1873 fn adaptive_evidence_keeps_success_and_failure_exact_distinct_and_out_of_context() {
1874 let _g = setup();
1875 let tmp = tempdir().unwrap();
1876 with_test_home(tmp.path(), || {
1877 let sentinel = "DEEP_RAW_SENTINEL";
1878 // Payloads must exceed the 32_768-token (≈96 KiB) handle-only
1879 // threshold so adaptive routing actually spills them.
1880 let success_raw = format!(
1881 "{}{}{}",
1882 "head\n".repeat(30_000),
1883 sentinel,
1884 "tail\n".repeat(30_000)
1885 );
1886 let failure_raw = format!("{}{}", "failure\n".repeat(30_000), "FAILURE_END");
1887 let mut success = ToolResult::success(success_raw.clone());
1888 let mut failure = ToolResult::error(failure_raw.clone());
1889
1890 let success_path =
1891 adaptive_spillover(&mut success, "call-success", "exec_shell", "session-a")
1892 .expect("success evidence");
1893 let failure_path =
1894 adaptive_spillover(&mut failure, "call-failure", "mcp_fixture", "session-a")
1895 .expect("failure evidence");
1896
1897 assert_ne!(success_path, failure_path);
1898 assert_eq!(
1899 std::fs::read(&success_path).unwrap(),
1900 success_raw.as_bytes()
1901 );
1902 assert_eq!(
1903 std::fs::read(&failure_path).unwrap(),
1904 failure_raw.as_bytes()
1905 );
1906 assert!(!success.content.contains(sentinel));
1907 // Handle-only preview: 16 KiB head + 4 KiB tail + footer.
1908 assert!(success.content.len() < 21 * 1024);
1909 let success_meta = success.metadata.as_ref().unwrap();
1910 let failure_meta = failure.metadata.as_ref().unwrap();
1911 assert_ne!(
1912 success_meta["artifact_digest"],
1913 failure_meta["artifact_digest"]
1914 );
1915 assert_eq!(success_meta["artifact_session_id"], "session-a");
1916 assert_eq!(failure_meta["artifact_session_id"], "session-a");
1917
1918 let mut replay = ToolResult::success(success_raw);
1919 let replay_path =
1920 adaptive_spillover(&mut replay, "call-success", "exec_shell", "session-a")
1921 .expect("idempotent replay");
1922 assert_eq!(replay_path, success_path);
1923 });
1924 }
1925
1926 #[test]
1927 fn adaptive_evidence_publication_failure_emits_no_handle_or_details_hint() {
1928 let _g = setup();
1929 let tmp = tempdir().unwrap();
1930 with_test_home(tmp.path(), || {
1931 let session_dir = tmp
1932 .path()
1933 .join(".codewhale")
1934 .join("sessions")
1935 .join("session-blocked");
1936 std::fs::create_dir_all(&session_dir).unwrap();
1937 std::fs::write(session_dir.join("artifacts"), b"block artifact directory").unwrap();
1938
1939 let raw = format!(
1940 "{}{}{}",
1941 "publication failure head\n".repeat(1_500),
1942 "DEEP_FAILURE_SENTINEL",
1943 "publication failure tail\n".repeat(1_500),
1944 );
1945 let mut result = ToolResult::error(raw.clone());
1946 let path = adaptive_spillover(
1947 &mut result,
1948 "call-failed-publish",
1949 "mcp_fixture",
1950 "session-blocked",
1951 );
1952
1953 assert!(path.is_none());
1954 assert!(!result.success);
1955 assert!(result.content.len() < SPILLOVER_HEAD_BYTES + SPILLOVER_TAIL_BYTES + 1024);
1956 assert!(!result.content.contains("DEEP_FAILURE_SENTINEL"));
1957 assert!(
1958 result
1959 .content
1960 .contains("the full output could not be saved")
1961 );
1962 assert!(!result.content.contains(SPILLOVER_PREVIEW_HINT));
1963 assert!(!result.content.contains("retrieve_tool_result"));
1964 assert_eq!(
1965 result
1966 .metadata
1967 .as_ref()
1968 .and_then(|metadata| metadata.get("evidence_available")),
1969 Some(&serde_json::Value::Bool(false))
1970 );
1971 assert!(
1972 !session_dir
1973 .join("artifacts/art_call-failed-publish.txt")
1974 .exists()
1975 );
1976 });
1977 }
1978
1979 #[test]
1980 fn adaptive_evidence_metadata_atomic_failure_leaves_payload_unadvertised() {
1981 let _g = setup();
1982 let tmp = tempdir().unwrap();
1983 with_test_home(tmp.path(), || {
1984 let artifact_dir = tmp
1985 .path()
1986 .join(".codewhale")
1987 .join("sessions")
1988 .join("session-metadata-blocked")
1989 .join("artifacts");
1990 std::fs::create_dir_all(artifact_dir.join("art_call-failed-metadata.evidence.json"))
1991 .unwrap();
1992
1993 let raw = format!(
1994 "{}{}{}",
1995 "metadata failure head\n".repeat(1_500),
1996 "DEEP_METADATA_FAILURE_SENTINEL",
1997 "metadata failure tail\n".repeat(1_500),
1998 );
1999 let mut result = ToolResult::success(raw.clone());
2000 let path = adaptive_spillover(
2001 &mut result,
2002 "call-failed-metadata",
2003 "exec_shell",
2004 "session-metadata-blocked",
2005 );
2006
2007 assert!(path.is_none());
2008 assert!(result.success);
2009 assert!(result.content.len() < SPILLOVER_HEAD_BYTES + SPILLOVER_TAIL_BYTES + 1024);
2010 assert!(!result.content.contains("DEEP_METADATA_FAILURE_SENTINEL"));
2011 assert!(
2012 result
2013 .content
2014 .contains("the full output could not be saved")
2015 );
2016 assert_eq!(
2017 result.metadata.as_ref().unwrap()["evidence_available"],
2018 false
2019 );
2020 assert!(!result.content.contains(SPILLOVER_PREVIEW_HINT));
2021 assert!(!result.content.contains("retrieve_tool_result"));
2022 assert!(
2023 !artifact_dir.join("art_call-failed-metadata.txt").exists(),
2024 "metadata failure must leave no payload behind a guessable handle"
2025 );
2026 });
2027 }
2028
2029 #[test]
2030 fn registry_results_spill_to_artifacts_like_every_other_tool() {
2031 // The Registry bypass is gone: an oversized payload (which today can
2032 // only be a bug, since the tool caps model-visible matches at eight)
2033 // takes the same artifact path as any other tool result.
2034 let _g = setup();
2035 let tmp = tempdir().unwrap();
2036 with_test_home(tmp.path(), || {
2037 let original = "registry-entry\n".repeat(10_000);
2038 assert!(original.len() > SPILLOVER_THRESHOLD_BYTES);
2039 let mut result = ToolResult::success(original);
2040
2041 let path = apply_spillover_with_artifact(
2042 &mut result,
2043 "call-registry",
2044 "registry_sync",
2045 "session-registry",
2046 );
2047
2048 assert!(path.is_some(), "oversized registry payload must spill");
2049 });
2050 }
2051
2052 #[test]
2053 fn apply_spillover_preserves_existing_metadata() {
2054 let _g = setup();
2055 let tmp = tempdir().unwrap();
2056 with_test_home(tmp.path(), || {
2057 let big = "Y".repeat(200 * 1024);
2058 let mut result = ToolResult::success(big)
2059 .with_metadata(serde_json::json!({"prior_key": "prior_value"}));
2060 let path = apply_spillover(&mut result, "call-meta").expect("should spill");
2061
2062 let metadata = result.metadata.expect("metadata present");
2063 // Prior keys survive.
2064 assert_eq!(
2065 metadata
2066 .get("prior_key")
2067 .and_then(serde_json::Value::as_str),
2068 Some("prior_value")
2069 );
2070 // New key added alongside.
2071 assert_eq!(
2072 metadata
2073 .get("spillover_path")
2074 .and_then(serde_json::Value::as_str),
2075 Some(path.display().to_string().as_str())
2076 );
2077 });
2078 }
2079
2080 #[test]
2081 fn apply_spillover_wraps_non_object_metadata_under_prior_key() {
2082 // Defends against a tool whose `metadata` is something
2083 // other than a JSON object (rare — most use the `json!({})`
2084 // pattern — but legal per `serde_json::Value`). The
2085 // spillover writer must add `spillover_path` without losing
2086 // the prior payload.
2087 let _g = setup();
2088 let tmp = tempdir().unwrap();
2089 with_test_home(tmp.path(), || {
2090 let big = "Z".repeat(200 * 1024);
2091 let mut result = ToolResult::success(big).with_metadata(serde_json::json!([
2092 "unexpected",
2093 "array",
2094 "payload"
2095 ]));
2096 let path = apply_spillover(&mut result, "call-arr").expect("should spill");
2097
2098 let metadata = result.metadata.expect("metadata stamped");
2099 // Prior payload re-homed under `_prior`.
2100 let prior = metadata.get("_prior").expect("_prior wrap key present");
2101 assert_eq!(
2102 prior,
2103 &serde_json::json!(["unexpected", "array", "payload"]),
2104 "prior array should round-trip under _prior"
2105 );
2106 // New key alongside.
2107 assert_eq!(
2108 metadata
2109 .get("spillover_path")
2110 .and_then(serde_json::Value::as_str),
2111 Some(path.display().to_string().as_str())
2112 );
2113 });
2114 }
2115
2116 // ── Honest-truncation regressions (v0.9.4) ─────────────────────────────
2117
2118 #[test]
2119 fn truncated_preview_returns_content_unchanged_when_nothing_omitted() {
2120 let original = "line one\nline two\nline three\n";
2121 let preview = truncated_preview(original, "", original, "/tmp/artifact.txt", None);
2122 assert_eq!(preview, original);
2123 assert!(
2124 !preview.contains("of output omitted"),
2125 "must never claim a truncation that did not happen"
2126 );
2127 }
2128
2129 #[test]
2130 fn head_tail_windows_never_overlap() {
2131 // Content smaller than head + tail budgets: the tail window shrinks
2132 // so it starts exactly where the head ends — no byte appears twice.
2133 let content = "x".repeat(10_000);
2134 let (head, tail) = head_tail_windows(&content, 8 * 1024, 4 * 1024);
2135 assert_eq!(head.len(), 8 * 1024);
2136 assert_eq!(tail.len(), 10_000 - 8 * 1024);
2137 assert!(head.len() + tail.len() <= content.len());
2138
2139 // Content larger than both budgets: full windows, exact omission.
2140 let big = "y".repeat(100_000);
2141 let (head, tail) = head_tail_windows(&big, 32 * 1024, 8 * 1024);
2142 assert_eq!(head.len(), 32 * 1024);
2143 assert_eq!(tail.len(), 8 * 1024);
2144
2145 // UTF-8 codepoints are never split at either window edge.
2146 let emoji = "🐳".repeat(5_000); // 20_000 bytes, 4 per codepoint
2147 let (head, tail) = head_tail_windows(&emoji, 8 * 1024 + 1, 4 * 1024 + 2);
2148 assert!(emoji.is_char_boundary(head.len()));
2149 assert!(emoji.is_char_boundary(emoji.len() - tail.len()));
2150 assert!(head.len() + tail.len() <= emoji.len());
2151 }
2152
2153 #[test]
2154 fn adaptive_evidence_passes_through_when_preview_budget_covers_output() {
2155 let _g = setup();
2156 let tmp = tempdir().unwrap();
2157 with_test_home(tmp.path(), || {
2158 // 30_000 bytes → 10_000 estimated tokens → Hybrid band under the
2159 // 32_768-token default, but the 32 KiB + 8 KiB preview budget
2160 // covers the whole output, so nothing is actually omitted.
2161 let raw = "mid\n".repeat(7_500);
2162 assert_eq!(raw.len(), 30_000);
2163 let mut result = ToolResult::success(raw.clone());
2164 let path =
2165 adaptive_spillover(&mut result, "call-covered", "exec_shell", "session-covered");
2166 assert!(path.is_none(), "no artifact when nothing is omitted");
2167 assert_eq!(result.content, raw);
2168 assert!(!result.content.contains("of output omitted"));
2169 assert!(
2170 !tmp.path()
2171 .join(".codewhale/sessions/session-covered/artifacts/art_call-covered.txt")
2172 .exists()
2173 );
2174 });
2175 }
2176
2177 #[test]
2178 fn adaptive_evidence_footer_names_artifact_path_and_recovery() {
2179 let _g = setup();
2180 let tmp = tempdir().unwrap();
2181 with_test_home(tmp.path(), || {
2182 // 120_000 bytes → 40_000 estimated tokens → handle-only band.
2183 let raw = "entry\n".repeat(20_000);
2184 assert_eq!(raw.len(), 120_000);
2185 let mut result = ToolResult::success(raw);
2186 let path =
2187 adaptive_spillover(&mut result, "call-honest", "exec_shell", "session-honest")
2188 .expect("should spill");
2189
2190 // Footer: omitted size + line count, artifact path, recovery line.
2191 assert!(result.content.contains("of output omitted ("));
2192 assert!(result.content.contains(" lines)"));
2193 assert!(result.content.contains("full output at"));
2194 assert!(
2195 result
2196 .content
2197 .contains(&crate::artifacts::format_artifact_relative_path(&path))
2198 );
2199 assert!(result.content.contains(SPILLOVER_RECOVERY_HINT));
2200 assert!(!result.content.contains(SPILLOVER_PREVIEW_HINT));
2201
2202 // Head and tail do not overlap: 16 KiB + 4 KiB handle-only
2203 // windows over a 120_000-byte output.
2204 let metadata = result.metadata.expect("metadata stamped");
2205 assert_eq!(metadata["retained_head_bytes"], 16 * 1024);
2206 assert_eq!(metadata["retained_tail_bytes"], 4 * 1024);
2207 });
2208 }
2209
2210 #[test]
2211 fn apply_spillover_with_artifact_defaults_to_classic_head_tail_spillover() {
2212 // 120_000 bytes exceeds the 100 KiB classic threshold. Without the
2213 // `CODEWHALE_ADAPTIVE_OUTPUT_ROUTING` opt-in the public entry point
2214 // keeps a 32 KiB head + 8 KiB tail and writes the session artifact —
2215 // the adaptive handle-only windows and evidence metadata are opt-in.
2216 let _g = setup();
2217 let tmp = tempdir().unwrap();
2218 with_test_home(tmp.path(), || {
2219 let raw = "entry\n".repeat(20_000);
2220 assert_eq!(raw.len(), 120_000);
2221 let mut result = ToolResult::success(raw.clone());
2222 let path = apply_spillover_with_artifact(
2223 &mut result,
2224 "call-classic",
2225 "exec_shell",
2226 "session-classic",
2227 )
2228 .expect("should spill");
2229
2230 let metadata = result.metadata.expect("metadata stamped");
2231 assert_eq!(metadata["retained_head_bytes"], 32 * 1024);
2232 assert_eq!(metadata["retained_tail_bytes"], 8 * 1024);
2233 assert!(result.content.contains(SPILLOVER_RECOVERY_HINT));
2234 assert!(
2235 !tmp.path()
2236 .join(
2237 ".codewhale/sessions/session-classic/artifacts/art_call-classic.evidence.json"
2238 )
2239 .exists(),
2240 "classic lane publishes no adaptive evidence metadata"
2241 );
2242 assert_eq!(std::fs::read_to_string(&path).unwrap(), raw);
2243 // The classic lane names its bytes by digest, like the adaptive
2244 // lane, so a turn artifact ref can carry a revision.
2245 assert_eq!(
2246 metadata["artifact_digest"],
2247 crate::hashing::sha256_hex(raw.as_bytes())
2248 );
2249 let artifact = tmp
2250 .path()
2251 .join(".codewhale/sessions/session-classic/artifacts/art_call-classic.txt");
2252 assert_eq!(std::fs::read_to_string(&artifact).unwrap(), raw);
2253
2254 // The published handle is immutable: a different payload under
2255 // the same call id never rewrites the bytes a ref names.
2256 let mut replay = ToolResult::success("other\n".repeat(20_000));
2257 apply_spillover_with_artifact(
2258 &mut replay,
2259 "call-classic",
2260 "exec_shell",
2261 "session-classic",
2262 )
2263 .expect("replay still spills to the legacy footer");
2264 assert_eq!(std::fs::read_to_string(&artifact).unwrap(), raw);
2265 let replay_metadata = replay.metadata.expect("metadata stamped");
2266 assert!(replay_metadata.get("artifact_id").is_none());
2267 assert!(replay_metadata.get("artifact_digest").is_none());
2268 });
2269 }
2270 }
2271
2271 lines RUST