返回 CodeWhale
file.rs
根目录 / crates / tui / src / tools / file.rs
1 //! File system engines for the lowercase `read`, `write`, and `edit` primitives
2 //! plus deferred workspace helpers such as `list_dir`. The older `File`,
3 //! `read_file`, `write_file`, and `edit_file` names remain registered but hidden
4 //! so saved sessions can replay their original schemas and behavior.
5 //!
6 //! These tools provide safe file system operations within the workspace,
7 //! with path validation to prevent escaping the workspace boundary.
8
9 use super::diff_format::make_unified_diff;
10 use super::rust_format::{NORMALIZED_NOTE, normalize_edit};
11 use super::spec::{
12 ApprovalRequirement, RichToolResult, ToolCapability, ToolContext, ToolError, ToolResult,
13 ToolSpec, lsp_diagnostics_for_paths, optional_str, optional_u64, required_str,
14 };
15 use super::syntax_check::guard_edit;
16 use async_trait::async_trait;
17 use serde_json::{Value, json};
18 use std::borrow::Cow;
19 use std::collections::HashMap;
20 use std::fs;
21 use std::path::{Path, PathBuf};
22 use std::sync::{Arc, Mutex, OnceLock, Weak};
23 use std::time::Duration;
24 use tokio::sync::{Mutex as AsyncMutex, OwnedMutexGuard};
25 use tokio_util::sync::CancellationToken;
26 use unicode_normalization::UnicodeNormalization;
27
28 // === Content-hash edit guards (#3979) ===
29
30 /// Format a file snapshot's content hash as `sha256:<hex>`.
31 ///
32 /// The prefixed shape (rather than a bare hex digest) is deliberate: it is
33 /// self-describing in the transcript, and it makes an accidentally-truncated or
34 /// hand-invented value fail the equality check instead of matching by luck.
35 ///
36 /// A file's hash is taken over its raw bytes, always before any windowing,
37 /// truncation, or rendering, so the value `read` reports is the value `write`,
38 /// `edit`, and `patch` verify.
39 pub(super) fn content_hash(bytes: &[u8]) -> String {
40 format!("sha256:{}", crate::hashing::sha256_hex(bytes))
41 }
42
43 /// Hash a file's bytes without holding the whole file in memory.
44 ///
45 /// The read path streams a bounded window out of large files on purpose, so it
46 /// never has the full contents to hash. Digesting through a separate streaming
47 /// pass keeps that memory bound while still producing a hash over the entire
48 /// file — the only value an edit guard can verify against.
49 fn hash_file_streaming(path: &Path) -> std::io::Result<String> {
50 use sha2::{Digest, Sha256};
51 use std::io::Read as _;
52
53 let mut file = fs::File::open(path)?;
54 let mut hasher = Sha256::new();
55 let mut buf = vec![0_u8; 64 * 1024];
56 loop {
57 let read = file.read(&mut buf)?;
58 if read == 0 {
59 break;
60 }
61 hasher.update(&buf[..read]);
62 }
63 Ok(format!(
64 "sha256:{}",
65 crate::hashing::hex_bytes(hasher.finalize())
66 ))
67 }
68
69 /// The one-line header that reports a snapshot hash to the model.
70 ///
71 /// This goes in `ToolResult::content`, not in `ToolResult::metadata`, and that
72 /// placement is the whole point. `metadata` never reaches the model: the wire
73 /// `ContentBlock::ToolResult` (`crates/core/src/request.rs`) has no field for
74 /// it, and the turn loop builds the tool message from `output.content` alone
75 /// (`crates/tui/src/core/engine/turn_loop.rs`). Metadata is for the TUI,
76 /// telemetry, and the approval/mutation receipts. A hash the model cannot read
77 /// is a guard the model cannot use, so it is rendered into the content — either
78 /// as an attribute on the `<file …>` envelope, or as this header line for the
79 /// unwrapped small-file read.
80 fn content_hash_header(hash: &str) -> String {
81 format!("content_hash=\"{hash}\"\n")
82 }
83
84 /// Reject a mutation whose `expected_hash` does not describe the current file.
85 ///
86 /// Callers must run this against the exact snapshot the mutation would be
87 /// applied to, and before anything is written. `None` (parameter absent) keeps
88 /// the pre-#3979 behavior untouched — the guard is opt-in.
89 fn verify_expected_hash(
90 expected: Option<&str>,
91 current_bytes: &[u8],
92 action: &str,
93 path_str: &str,
94 ) -> Result<(), ToolError> {
95 let Some(expected) = expected else {
96 return Ok(());
97 };
98 let actual = content_hash(current_bytes);
99 if expected == actual {
100 return Ok(());
101 }
102 Err(ToolError::execution_failed(format!(
103 "File `{action}` refused: {path_str} changed since it was read. \
104 expected_hash was {expected} but the file is now {actual}, so nothing was written. \
105 Recovery: call File with action=\"read\" path=\"{path_str}\" to get the current contents \
106 and its content_hash, then retry with the new hash."
107 )))
108 }
109
110 /// Shared schema text for the optional guard parameter.
111 ///
112 /// The hidden compatibility schema still has a byte budget, so this is only the
113 /// instruction the legacy caller needs: what to pass and what happens on a
114 /// mismatch. The rationale stays in doc comments rather than schema bytes.
115 pub(super) const EXPECTED_HASH_DESCRIPTION: &str = "The `content_hash` from a prior read; the write is refused and the file left unchanged if it changed since";
116
117 // === Cross-harness parameter aliases ===
118
119 /// Rewrite well-known parameter spellings from other coding harnesses onto the
120 /// names this tool actually implements.
121 ///
122 /// Every mainstream harness names the same three file-edit arguments
123 /// differently — `old_string`/`new_string`, `old_str`/`new_str`,
124 /// `oldText`/`newText` — and models carry whichever spelling their training
125 /// saw most. CodeWhale's canonical `search`/`replace` is the odd one out, so a
126 /// model reaching for its prior used to burn a full turn on a rejection
127 /// (#5209) and then guess again. Translating an unambiguous synonym is
128 /// strictly better than refusing it: the edit the model asked for is the edit
129 /// that happens, and the schema still advertises exactly one canonical name so
130 /// there is no new ambiguity to learn.
131 ///
132 /// This is deliberately *not* a silent-acceptance path. Only exact synonyms
133 /// are mapped, a synonym that disagrees with an explicitly supplied canonical
134 /// value is an error rather than a coin flip, and any parameter that is not a
135 /// known synonym still fails validation. The #5209 guarantee — no fabricated
136 /// "Replaced 1 occurrence" for an edit that never landed — is unchanged.
137 pub(super) struct ParamAlias {
138 /// Spelling a model might emit.
139 alias: &'static str,
140 /// Parameter this tool implements.
141 canonical: &'static str,
142 }
143
144 const fn alias(alias: &'static str, canonical: &'static str) -> ParamAlias {
145 ParamAlias { alias, canonical }
146 }
147
148 /// Path spellings shared by every file action. `path` is CodeWhale's
149 /// canonical name and the most common one in the field, but `file_path` is
150 /// widespread enough in training data to be worth accepting everywhere.
151 pub(super) const PATH_ALIASES: &[ParamAlias] =
152 &[alias("file_path", "path"), alias("filePath", "path")];
153
154 /// `input` with [`PATH_ALIASES`] folded onto `path`, exactly as the file
155 /// tools' `execute` does before touching disk.
156 ///
157 /// Policy gates (typed file rules, the workspace-write carve-out, repo law,
158 /// Auto-Review) must judge the path the tool will act on. Reading only the
159 /// raw `path` key misses accepted `file_path`/`filePath` spellings. A
160 /// conflicting pair, which `execute` refuses, is returned unchanged.
161 pub(crate) fn with_canonical_path_argument(input: &Value) -> Cow<'_, Value> {
162 if !PATH_ALIASES
163 .iter()
164 .any(|ParamAlias { alias, .. }| input.get(*alias).is_some())
165 {
166 return Cow::Borrowed(input);
167 }
168 let mut folded = input.clone();
169 match apply_param_aliases(&mut folded, PATH_ALIASES, "path") {
170 Ok(()) => Cow::Owned(folded),
171 Err(_) => Cow::Borrowed(input),
172 }
173 }
174
175 /// `path` and every spelling [`PATH_ALIASES`] folds onto it, for gates that
176 /// deliberately over-collect candidate targets.
177 pub(crate) fn path_argument_keys() -> impl Iterator<Item = &'static str> {
178 std::iter::once("path").chain(PATH_ALIASES.iter().map(|ParamAlias { alias, .. }| *alias))
179 }
180
181 /// Edit-specific spellings. Ordered most- to least-common.
182 const EDIT_ALIASES: &[ParamAlias] = &[
183 alias("old_string", "search"),
184 alias("new_string", "replace"),
185 alias("old_str", "search"),
186 alias("new_str", "replace"),
187 alias("oldText", "search"),
188 alias("newText", "replace"),
189 alias("old_text", "search"),
190 alias("new_text", "replace"),
191 alias("replacement", "replace"),
192 ];
193
194 /// Read-window spellings. `offset`/`limit` and `line_offset`/`n_lines` both
195 /// name the same two numbers as CodeWhale's `start_line`/`max_lines` in widely
196 /// trained-on tool surfaces. A wrong guess here used to be ignored outright,
197 /// silently returning the head of the file instead of the window the model
198 /// asked for — a wrong answer shaped like a right one.
199 const READ_ALIASES: &[ParamAlias] = &[
200 alias("offset", "start_line"),
201 alias("line_offset", "start_line"),
202 alias("limit", "max_lines"),
203 alias("n_lines", "max_lines"),
204 alias("num_lines", "max_lines"),
205 ];
206
207 /// `search_name` spellings. The `File` wrapper advertises `max_results` for
208 /// both search actions, but only `search_content` implements that name; on
209 /// `search_name` the same number is spelled `limit`. Folding it here (rather
210 /// than copying it inside the wrapper) keeps one alias mechanism, so the
211 /// result-count cap a model asks for is the cap it gets whichever name it
212 /// reaches for, and a direct `file_search` call behaves the same way.
213 pub(super) const SEARCH_NAME_ALIASES: &[ParamAlias] = &[alias("max_results", "limit")];
214
215 /// `search_content` spellings, mirroring `SEARCH_NAME_ALIASES` in the other
216 /// direction: the wrapper advertises `query` and `limit` on the name-search
217 /// side, and a model that carries them across to a content search means
218 /// `pattern` and `max_results`.
219 pub(super) const SEARCH_CONTENT_ALIASES: &[ParamAlias] =
220 &[alias("query", "pattern"), alias("limit", "max_results")];
221
222 /// Apply `aliases` to `input`, in place.
223 ///
224 /// An alias is consumed only when the canonical key is absent. When both are
225 /// present and *equal* the alias is dropped as a harmless duplicate; when both
226 /// are present and disagree the call fails, because guessing which one the
227 /// model meant is exactly the fabrication this path exists to prevent.
228 pub(super) fn apply_param_aliases(
229 input: &mut Value,
230 aliases: &[ParamAlias],
231 tool_label: &str,
232 ) -> Result<(), ToolError> {
233 let Some(obj) = input.as_object_mut() else {
234 return Ok(());
235 };
236
237 for ParamAlias { alias, canonical } in aliases {
238 let Some(alias_value) = obj.remove(*alias) else {
239 continue;
240 };
241 match obj.get(*canonical) {
242 None => {
243 obj.insert((*canonical).to_string(), alias_value);
244 }
245 Some(existing) if existing == &alias_value => {}
246 Some(_) => {
247 return Err(ToolError::invalid_input(format!(
248 "{tool_label} received both `{canonical}` and its alias `{alias}` with different values, so the intended argument is ambiguous; nothing was changed. Pass only `{canonical}`."
249 )));
250 }
251 }
252 }
253
254 Ok(())
255 }
256
257 // === Per-action parameter contracts ===
258
259 /// The parameter contract for one `File` action.
260 ///
261 /// #5209 taught `edit` to refuse a parameter it does not implement instead of
262 /// dropping it and returning a success-shaped receipt. Only `edit` learned it.
263 /// Every other action kept silently discarding unknown keys, and for a reader
264 /// that is the same failure wearing a quieter costume: a misspelled
265 /// `start_line` on `read` is dropped, the head of the file comes back, and
266 /// nothing in the response says the requested window was never honored — a
267 /// wrong answer shaped like a right one.
268 ///
269 /// One table, one error shape, every action.
270 pub(super) struct ActionParams {
271 /// Action name as the model spells it on `File` (`read`, `write`, …).
272 action: &'static str,
273 /// Every parameter the action implements, canonical spellings only.
274 /// Aliases are folded onto these by [`apply_param_aliases`] before
275 /// validation runs, so they must not be listed here.
276 allowed: &'static [&'static str],
277 /// Parameters the action cannot run without.
278 required: &'static [&'static str],
279 /// `true` when exactly one of `required` is needed rather than all of
280 /// them — `patch` accepts `patch`, `replace`, or `changes`.
281 required_is_choice: bool,
282 }
283
284 const fn params(
285 action: &'static str,
286 allowed: &'static [&'static str],
287 required: &'static [&'static str],
288 ) -> ActionParams {
289 ActionParams {
290 action,
291 allowed,
292 required,
293 required_is_choice: false,
294 }
295 }
296
297 pub(super) const READ_PARAMS: ActionParams = params(
298 "read",
299 &["path", "start_line", "max_lines", "pages"],
300 &["path"],
301 );
302
303 pub(super) const WRITE_PARAMS: ActionParams = params(
304 "write",
305 &["path", "content", "expected_hash"],
306 &["path", "content"],
307 );
308
309 pub(super) const EDIT_PARAMS: ActionParams = params(
310 "edit",
311 &["path", "search", "replace", "expected_hash"],
312 &["path", "search", "replace"],
313 );
314
315 pub(super) const LIST_PARAMS: ActionParams = params("list", &["path"], &[]);
316
317 pub(super) const SEARCH_NAME_PARAMS: ActionParams = params(
318 "search_name",
319 &["query", "path", "limit", "extensions", "exclude"],
320 &["query"],
321 );
322
323 pub(super) const SEARCH_CONTENT_PARAMS: ActionParams = params(
324 "search_content",
325 &[
326 "pattern",
327 "path",
328 "include",
329 "exclude",
330 "context_lines",
331 "case_insensitive",
332 "max_results",
333 ],
334 &["pattern"],
335 );
336
337 pub(super) const PATCH_PARAMS: ActionParams = ActionParams {
338 action: "patch",
339 allowed: &[
340 "path",
341 "patch",
342 "replace",
343 "changes",
344 "fuzz",
345 "create_if_missing",
346 "expected_hash",
347 ],
348 required: &["patch", "replace", "changes"],
349 required_is_choice: true,
350 };
351
352 /// Render `names` as a backticked, comma-separated English list.
353 fn quoted_list(names: &[&str], conjunction: &str) -> String {
354 let quoted: Vec<String> = names.iter().map(|name| format!("`{name}`")).collect();
355 match quoted.as_slice() {
356 [] => "none".to_string(),
357 [only] => only.clone(),
358 [first, second] => format!("{first} {conjunction} {second}"),
359 [head @ .., last] => format!("{}, {conjunction} {last}", head.join(", ")),
360 }
361 }
362
363 impl ActionParams {
364 /// Reject parameter names this action does not implement.
365 ///
366 /// Must run *after* [`apply_param_aliases`], exactly as the `edit` path
367 /// does. The alias lane's reasoning stands: translating an unambiguous
368 /// synonym is better than refusing it, so by the time this runs every
369 /// spelling with a known meaning has already been folded onto its
370 /// canonical name. What is left is a name with no known meaning, where
371 /// continuing would mean guessing which argument was intended — so it
372 /// hard-errors rather than dropping the argument and reporting success.
373 pub(super) fn reject_unknown(&self, input: &Value) -> Result<(), ToolError> {
374 let action = self.action;
375 let required = if self.required_is_choice {
376 format!("one of {}", quoted_list(self.required, "or"))
377 } else {
378 quoted_list(self.required, "and")
379 };
380
381 let Some(obj) = input.as_object() else {
382 return Err(ToolError::invalid_input(format!(
383 "File {action} input must be an object. Allowed parameters are {}. Required: {required}. The {action} was not performed.",
384 quoted_list(self.allowed, "and"),
385 )));
386 };
387
388 let unexpected: Vec<&str> = obj
389 .keys()
390 .map(String::as_str)
391 .filter(|key| !self.allowed.contains(key))
392 .collect();
393 if !unexpected.is_empty() {
394 return Err(ToolError::invalid_input(format!(
395 "unexpected File {action} parameter(s): {}. Allowed parameters are {}. Required: {required}. The {action} was not performed.",
396 unexpected.join(", "),
397 quoted_list(self.allowed, "and"),
398 )));
399 }
400
401 Ok(())
402 }
403
404 /// A required parameter that is not also allowed would make the refusal
405 /// self-contradicting: it would name an argument the same check rejects.
406 #[cfg(test)]
407 pub(super) fn assert_required_is_allowed(&self) {
408 for name in self.required {
409 assert!(
410 self.allowed.contains(name),
411 "File {} requires `{name}` but does not allow it",
412 self.action
413 );
414 }
415 }
416 }
417
418 // === ReadFileTool ===
419
420 fn canonical_path_for_credential_guard(path: &Path) -> PathBuf {
421 fs::canonicalize(path).unwrap_or_else(|_| {
422 if path.is_absolute() {
423 path.to_path_buf()
424 } else {
425 std::env::current_dir()
426 .unwrap_or_else(|_| PathBuf::from("."))
427 .join(path)
428 }
429 })
430 }
431
432 fn config_backup_path_for_credential_guard(config_path: &Path) -> PathBuf {
433 let mut file_name = config_path
434 .file_name()
435 .map(std::ffi::OsString::from)
436 .unwrap_or_else(|| std::ffi::OsString::from(codewhale_config::CONFIG_FILE_NAME));
437 file_name.push(".bak");
438 config_path
439 .parent()
440 .unwrap_or_else(|| Path::new("."))
441 .join(file_name)
442 }
443
444 fn is_config_or_backup(candidate: &Path, config_path: &Path) -> bool {
445 let config_path = canonical_path_for_credential_guard(config_path);
446 let backup_path =
447 canonical_path_for_credential_guard(&config_backup_path_for_credential_guard(&config_path));
448 candidate == config_path || candidate == backup_path
449 }
450
451 /// Resolve a model-supplied path for an in-process read, applying every read
452 /// guard in the one safe order: the deny-list on the caller's raw spelling
453 /// (so a denial never names a symlink target), then `resolve_path`, then the
454 /// credential-store check and the deny-list again on the resolved path.
455 ///
456 /// Every tool that reads a file's content in-process and hands it (or a
457 /// derivative) to the model goes through this.
458 pub(crate) fn resolve_guarded_read_path(
459 context: &ToolContext,
460 raw: &str,
461 tool: &str,
462 ) -> Result<PathBuf, ToolError> {
463 enforce_read_denylist(Path::new(raw), tool)?;
464 let path = context.resolve_path(raw)?;
465 if is_codewhale_credential_path(&path) {
466 return Err(ToolError::permission_denied(format!(
467 "{tool} cannot expose Codewhale configuration or credential-store files; use `codewhale config list` or `codewhale auth status` for safe inspection"
468 )));
469 }
470 enforce_read_denylist(&path, tool)?;
471 Ok(path)
472 }
473
474 /// Refuse a read the sandbox read deny-list blocks (S1).
475 ///
476 /// `read_file`, `read`, and `read_media` all run *in-process*: they call
477 /// `std::fs` inside the harness, so `sandbox-exec` and `bwrap` never see them
478 /// and the OS-level deny rules do not apply. This is the enforcement point for
479 /// those tools, and the refusal is always an explicit error — never an empty
480 /// result, which would read as "the file is empty" and invite the model to
481 /// probe siblings.
482 pub(crate) fn enforce_read_denylist(path: &Path, tool: &str) -> Result<(), ToolError> {
483 // Expand the user's home before authorization, retaining the spelling they
484 // supplied in every denial. This shares the file tools' path resolution;
485 // expansion grants no additional access and never exposes a symlink target.
486 let home_path = path
487 .to_str()
488 .map(super::spec::resolve_home_path)
489 .transpose()?
490 .flatten();
491 if home_path
492 .as_deref()
493 .is_some_and(is_codewhale_credential_path)
494 {
495 return Err(ToolError::permission_denied(format!(
496 "{tool} cannot expose Codewhale configuration or credential-store files; use `codewhale config list` or `codewhale auth status` for safe inspection"
497 )));
498 }
499 match crate::sandbox::read_guard::active().check(home_path.as_deref().unwrap_or(path)) {
500 Ok(()) => Ok(()),
501 Err(mut denial) => {
502 denial.requested = path.to_path_buf();
503 let message = denial.message(tool);
504 tracing::warn!(
505 target: "codewhale::sandbox::read_guard",
506 requested = %denial.requested.display(),
507 via_symlink = denial.via_symlink,
508 tool = tool,
509 "sandbox read deny-list refused a read"
510 );
511 Err(ToolError::permission_denied(message))
512 }
513 }
514 }
515
516 /// Return whether `read_file` must refuse a CodeWhale-owned credential file.
517 ///
518 /// This is deliberately scoped to the active config, the two conventional
519 /// config locations (including one-time backups), and CodeWhale's file-backed
520 /// secret-store directories. Other dotfiles remain readable. Model-bound
521 /// redaction is still required because shell tools can read these files and
522 /// arbitrary commands can print credentials without reading a file at all.
523 pub(crate) fn is_codewhale_credential_path(path: &Path) -> bool {
524 let candidate = canonical_path_for_credential_guard(path);
525
526 if let Ok(active_config) = codewhale_config::resolve_config_path(None)
527 && is_config_or_backup(&candidate, &active_config)
528 {
529 return true;
530 }
531
532 // `CODEWHALE_HOME` relocates the *runtime* home; it is not a licence to read
533 // the user's real `~/.codewhale/config.toml`. `codewhale_home()` returns the
534 // override when one is set, so relying on it alone left the ambient store
535 // unguarded whenever that variable pointed elsewhere. Keep the ambient root
536 // in the set alongside the override, mirroring the deliberately
537 // unconditional `~/.codewhale/secrets` entry in `sandbox::read_guard`
538 // (read_guard.rs:481-487). `legacy_deepseek_home()` is already ambient by
539 // construction (paths/src/lib.rs:183-185), so it needs no counterpart.
540 let mut roots: Vec<PathBuf> = Vec::with_capacity(3);
541 roots.extend(codewhale_config::codewhale_home().ok());
542 roots.extend(codewhale_config::legacy_deepseek_home().ok());
543 roots.extend(
544 codewhale_paths::user_home().map(|home| home.join(codewhale_config::CODEWHALE_APP_DIR)),
545 );
546 for root in roots {
547 if is_config_or_backup(&candidate, &root.join(codewhale_config::CONFIG_FILE_NAME)) {
548 return true;
549 }
550
551 let secrets_dir = canonical_path_for_credential_guard(&root.join("secrets"));
552 if candidate.starts_with(secrets_dir) {
553 return true;
554 }
555 }
556
557 false
558 }
559
560 // === small-contract-compatible primitive implementation helpers ===
561
562 /// Default model-visible byte budget for one `read` call.
563 ///
564 /// Bytes are the *only* default bound: there is no line cap, so an ordinary
565 /// source or prose file comes back whole in one call instead of being paged
566 /// at some arbitrary line count with most of the budget unspent.
567 const READ_DEFAULT_MAX_BYTES: usize = 100_000;
568 /// Hard ceiling on a budget the *model* asks for with `max_bytes`. A larger
569 /// request clamps down to this; it is never an error.
570 const READ_REQUEST_MAX_BYTES: usize = 500_000;
571 /// Outer bound on the operator's process-wide `[workshop] read_result_max_bytes`
572 /// override, and therefore on any read result.
573 const READ_RESULT_ABSOLUTE_MAX_BYTES: usize = 2 * 1024 * 1024;
574
575 /// Whole-source processing bound, independent of the model-visible read budget.
576 /// The lowercase primitives accept only regular, single-link files. Hidden
577 /// compatibility readers and PDF extraction keep their existing contracts.
578 const CONTRACT_FILE_MAX_BYTES: usize = 16 * 1024 * 1024;
579
580 fn check_contract_file_size(size: usize) -> Result<(), ToolError> {
581 if size > CONTRACT_FILE_MAX_BYTES {
582 return Err(ToolError::execution_failed(
583 "File content exceeds the supported 16 MiB processing cap. Split or export a smaller file, or use an appropriate separately authorized tool.",
584 ));
585 }
586 Ok(())
587 }
588
589 fn check_contract_cancelled(token: Option<&CancellationToken>) -> Result<(), ToolError> {
590 if token.is_some_and(CancellationToken::is_cancelled) {
591 return Err(ToolError::cancelled("Operation aborted"));
592 }
593 Ok(())
594 }
595
596 /// Read at most cap+1 actual bytes, including files that grow after metadata.
597 /// Cancellation is polled between bounded reads; it cannot interrupt an OS
598 /// syscall already in progress. This worker never mutates the file.
599 pub(super) fn read_contract_source(
600 reader: &mut impl std::io::Read,
601 cancel: Option<&CancellationToken>,
602 ) -> Result<Vec<u8>, ToolError> {
603 let mut bytes = Vec::new();
604 let mut chunk = [0_u8; 64 * 1024];
605 loop {
606 check_contract_cancelled(cancel)?;
607 let remaining = CONTRACT_FILE_MAX_BYTES + 1 - bytes.len();
608 let read_len = remaining.min(chunk.len());
609 let count = match reader.read(&mut chunk[..read_len]) {
610 Ok(count) => count,
611 Err(error) if error.kind() == std::io::ErrorKind::Interrupted => continue,
612 Err(error) => {
613 return Err(ToolError::execution_failed(format!(
614 "Failed to read file: {error}"
615 )));
616 }
617 };
618 check_contract_cancelled(cancel)?;
619 if count == 0 {
620 return Ok(bytes);
621 }
622 bytes.extend_from_slice(&chunk[..count]);
623 check_contract_file_size(bytes.len())?;
624 }
625 }
626
627 async fn load_contract_source(
628 path: &Path,
629 writable: bool,
630 context: &ToolContext,
631 ) -> Result<Option<Vec<u8>>, ToolError> {
632 check_file_operation_cancelled(context)?;
633 let path = path.to_path_buf();
634 let cancel = context.cancel_token.clone();
635 let mut worker = tokio::task::spawn_blocking(move || {
636 check_contract_cancelled(cancel.as_ref())?;
637 let open_error = |error| {
638 if writable {
639 ToolError::execution_failed(format!(
640 "Could not edit file {}: target must be readable and writable ({error})",
641 path.display()
642 ))
643 } else {
644 ToolError::execution_failed(format!("Failed to read {}: {error}", path.display()))
645 }
646 };
647 let Some(mut file) = crate::plugins::registry::open_existing_regular_file(&path, writable)
648 .map_err(open_error)?
649 else {
650 return Ok(None);
651 };
652 let metadata = file
653 .metadata()
654 .map_err(|error| open_error(error.to_string()))?;
655 if metadata.len() > CONTRACT_FILE_MAX_BYTES as u64 {
656 check_contract_file_size(CONTRACT_FILE_MAX_BYTES + 1)?;
657 }
658 read_contract_source(&mut file, cancel.as_ref()).map(Some)
659 });
660 let result = if let Some(cancel) = context.cancel_token.as_ref() {
661 tokio::select! {
662 biased;
663 () = cancel.cancelled() => return Err(ToolError::cancelled("Operation aborted")),
664 result = &mut worker => result,
665 }
666 } else {
667 worker.await
668 };
669 result.map_err(|error| ToolError::execution_failed(format!("File read task: {error}")))?
670 }
671
672 /// Resolve the byte budget for one `read` call from the three layers that can
673 /// set it, highest wins:
674 ///
675 /// 1. **The model's own request** — `max_bytes` on this call, clamped to
676 /// [`READ_REQUEST_MAX_BYTES`] (500 000).
677 /// 2. **The operator's process-wide override** — `[workshop]
678 /// read_result_max_bytes`, clamped into
679 /// `[READ_DEFAULT_MAX_BYTES, READ_RESULT_ABSOLUTE_MAX_BYTES]` (2 MiB).
680 /// 3. **The default** — [`READ_DEFAULT_MAX_BYTES`] (100 000).
681 ///
682 /// The result is `max(1, 2-or-3)`. Both raising layers can only raise: a model
683 /// request never shrinks a budget the operator widened, and an operator who
684 /// widened it process-wide keeps that floor when the model asks for less.
685 fn effective_read_max_bytes(requested: Option<usize>) -> usize {
686 let baseline =
687 crate::tools::large_output_router::WorkshopConfig::active_read_result_max_bytes()
688 .map_or(READ_DEFAULT_MAX_BYTES, |configured| {
689 configured.clamp(READ_DEFAULT_MAX_BYTES, READ_RESULT_ABSOLUTE_MAX_BYTES)
690 });
691 match requested {
692 Some(requested) => baseline.max(requested.min(READ_REQUEST_MAX_BYTES)),
693 None => baseline,
694 }
695 }
696
697 type FileMutationMutex = AsyncMutex<()>;
698
699 /// File primitives can also be invoked outside the native engine's global
700 /// execution lock (for example by an embedded host). Keep writes to one path
701 /// ordered in those hosts without exposing any locking ceremony in the tool
702 /// schema or result.
703 fn file_mutation_lock(path: &Path) -> Result<Arc<FileMutationMutex>, ToolError> {
704 static LOCKS: OnceLock<Mutex<HashMap<PathBuf, Weak<FileMutationMutex>>>> = OnceLock::new();
705 let locks = LOCKS.get_or_init(|| Mutex::new(HashMap::new()));
706 let mut locks = locks.lock().map_err(|_| {
707 ToolError::execution_failed(
708 "file mutation queue is unavailable because its lock was poisoned",
709 )
710 })?;
711 locks.retain(|_, lock| lock.strong_count() > 0);
712 if let Some(lock) = locks.get(path).and_then(Weak::upgrade) {
713 return Ok(lock);
714 }
715 let lock = Arc::new(AsyncMutex::new(()));
716 locks.insert(path.to_path_buf(), Arc::downgrade(&lock));
717 Ok(lock)
718 }
719
720 async fn acquire_file_mutation(
721 path: &Path,
722 context: &ToolContext,
723 ) -> Result<OwnedMutexGuard<()>, ToolError> {
724 let lock = file_mutation_lock(path)?;
725 if let Some(cancel) = context.cancel_token.as_ref() {
726 tokio::select! {
727 guard = lock.lock_owned() => Ok(guard),
728 () = cancel.cancelled() => Err(ToolError::cancelled("Operation aborted")),
729 }
730 } else {
731 Ok(lock.lock_owned().await)
732 }
733 }
734
735 /// Atomic workspace write on the blocking pool: temp create plus fsync plus
736 /// rename (and a retry loop on Windows) must not park a Tokio worker
737 /// (blocking-call convention, #6149). Error shape matches the historical
738 /// inline call.
739 async fn run_blocking_write_atomic(path: &Path, contents: Vec<u8>) -> Result<(), ToolError> {
740 let path = path.to_path_buf();
741 tokio::task::spawn_blocking(move || {
742 crate::utils::write_atomic_workspace(&path, &contents).map_err(|e| {
743 ToolError::execution_failed(format!("Failed to write {}: {e}", path.display()))
744 })
745 })
746 .await
747 .map_err(|e| ToolError::execution_failed(format!("File write task: {e}")))??;
748 Ok(())
749 }
750
751 fn check_file_operation_cancelled(context: &ToolContext) -> Result<(), ToolError> {
752 check_contract_cancelled(context.cancel_token.as_ref())
753 }
754
755 /// One `mutation.files[]` receipt entry. `size` and `sha256` describe the
756 /// exact bytes this call wrote, recorded where they were written so a turn's
757 /// artifact reference never has to re-read (and race) the disk. `sha256` is
758 /// plain lowercase hex: the same value `GET /v1/workspace/files/read` reports
759 /// as `revision`.
760 pub(crate) fn mutation_file_entry(path: &str, outcome: &str, written: Option<&[u8]>) -> Value {
761 let mut entry = json!({ "path": path, "outcome": outcome });
762 if let Some(bytes) = written {
763 entry["size"] = json!(bytes.len());
764 entry["sha256"] = json!(crate::hashing::sha256_hex(bytes));
765 }
766 entry
767 }
768
769 #[allow(clippy::too_many_arguments)]
770 async fn contract_mutation_result(
771 context: &ToolContext,
772 file_path: &Path,
773 requested_path: &str,
774 before: &str,
775 after: &str,
776 written: &[u8],
777 outcome: &str,
778 summary: String,
779 ) -> ToolResult {
780 let paths = [file_path.to_path_buf()];
781 let diagnostics = lsp_diagnostics_for_paths(context, &paths).await;
782 ToolResult::success(summary).with_metadata(json!({
783 "event": "file.mutation",
784 "lsp_diagnostics": diagnostics,
785 "mutation": {
786 "diff": make_unified_diff(requested_path, before, after),
787 "files": [mutation_file_entry(requested_path, outcome, Some(written))],
788 "renames": []
789 }
790 }))
791 }
792
793 fn reject_primitive_unknown(input: &Value, tool: &str, allowed: &[&str]) -> Result<(), ToolError> {
794 let object = input
795 .as_object()
796 .ok_or_else(|| ToolError::invalid_input(format!("{tool} input must be an object")))?;
797 let unexpected = object
798 .keys()
799 .filter(|key| !allowed.contains(&key.as_str()))
800 .cloned()
801 .collect::<Vec<_>>();
802 if unexpected.is_empty() {
803 return Ok(());
804 }
805 Err(ToolError::invalid_input(format!(
806 "unexpected {tool} parameter(s): {}",
807 unexpected.join(", ")
808 )))
809 }
810
811 fn contract_nonnegative_int(input: &Value, key: &str) -> Result<Option<usize>, ToolError> {
812 let Some(value) = input.get(key) else {
813 return Ok(None);
814 };
815 let number = codewhale_tools::json_nonnegative_integer(value)
816 .ok_or_else(|| ToolError::invalid_input(format!("{key} must be a non-negative integer")))?;
817 usize::try_from(number)
818 .map(Some)
819 .map_err(|_| ToolError::invalid_input(format!("{key} exceeds platform range")))
820 }
821
822 fn primitive_image_mime(bytes: &[u8]) -> Option<&'static str> {
823 crate::image_attach::sniff_media_type(bytes)
824 .or_else(|| bytes.starts_with(b"BM").then_some("image/bmp"))
825 }
826
827 fn contract_format_size(bytes: usize) -> String {
828 if bytes < 1024 {
829 format!("{bytes}B")
830 } else if bytes < 1024 * 1024 {
831 format!("{:.1}KB", bytes as f64 / 1024.0)
832 } else {
833 format!("{:.1}MB", bytes as f64 / (1024.0 * 1024.0))
834 }
835 }
836
837 #[derive(Debug)]
838 struct ContractReadWindow {
839 content: String,
840 shown_lines: usize,
841 truncated: bool,
842 first_line_too_large: bool,
843 }
844
845 /// Retain only complete lines from the head, stopping at `max_bytes`. A
846 /// terminal newline is content but does not add a phantom line to the
847 /// truncation counter.
848 ///
849 /// The byte budget is the single bound. There is no line cap to fragment a
850 /// file that fits: every retained line costs at least its own newline, so
851 /// `max_bytes` already bounds the line count as well.
852 fn contract_read_window(content: &str, max_bytes: usize) -> ContractReadWindow {
853 if content
854 .split('\n')
855 .next()
856 .is_some_and(|line| line.len() > max_bytes)
857 {
858 return ContractReadWindow {
859 content: String::new(),
860 shown_lines: 0,
861 truncated: true,
862 first_line_too_large: true,
863 };
864 }
865 if content.len() <= max_bytes {
866 return ContractReadWindow {
867 content: content.to_string(),
868 shown_lines: content.split_terminator('\n').count(),
869 truncated: false,
870 first_line_too_large: false,
871 };
872 }
873 let mut shown_lines = 0;
874 let mut bytes = 0usize;
875 for line in content.split_terminator('\n') {
876 let next = line.len() + usize::from(shown_lines > 0);
877 if bytes.saturating_add(next) > max_bytes {
878 break;
879 }
880 shown_lines += 1;
881 bytes += next;
882 }
883 ContractReadWindow {
884 content: content[..bytes].to_string(),
885 shown_lines,
886 truncated: true,
887 first_line_too_large: false,
888 }
889 }
890
891 /// Tool for reading UTF-8 files from the workspace.
892 pub struct ReadFileTool;
893
894 impl ReadFileTool {
895 /// Execute the lowercase `read` primitive without leaking the hidden
896 /// Codewhale hash/snapshot protocol into its small-contract-shaped model contract.
897 pub(super) async fn execute_contract_read(
898 input: Value,
899 context: &ToolContext,
900 ) -> Result<RichToolResult, ToolError> {
901 reject_primitive_unknown(&input, "read", &["path", "offset", "limit", "max_bytes"])?;
902 let path_str = required_str(&input, "path")?;
903 let offset = contract_nonnegative_int(&input, "offset")?;
904 let limit = contract_nonnegative_int(&input, "limit")?;
905 let max_bytes = effective_read_max_bytes(contract_nonnegative_int(&input, "max_bytes")?);
906 // S1/F2: check the caller's own spelling BEFORE `resolve_path`
907 // canonicalizes it. A workspace symlink `notes.txt` -> a denied vault
908 // file resolves to the secret's absolute location, and a denial raised
909 // only on the resolved path would name that location in the error —
910 // answering the very question ("where is the secret?") the read was
911 // probing for. `read_guard::check` canonicalizes internally, so the
912 // raw spelling still matches by its target; the resolved check after
913 // `resolve_path` stays as defense in depth for callers whose process
914 // cwd is not the workspace.
915 let file_path = resolve_guarded_read_path(context, path_str, "read")?;
916 check_file_operation_cancelled(context)?;
917 let bytes = load_contract_source(&file_path, false, context)
918 .await?
919 .ok_or_else(|| {
920 ToolError::execution_failed(format!(
921 "Failed to read {}: file not found",
922 file_path.display()
923 ))
924 })?;
925 // #6283: every read response carries the file's byte size, line
926 // count, and truncation flag so the caller can page deliberately
927 // instead of discovering a huge file one window at a time.
928 let size_bytes = bytes.len();
929 check_file_operation_cancelled(context)?;
930 if let Some(mime_type) = primitive_image_mime(&bytes) {
931 let prepared = crate::image_attach::prepare_tool_image_bytes(&bytes, mime_type);
932 context.note_file_read(&file_path);
933 return Ok(RichToolResult::with_content_blocks(
934 ToolResult::success(prepared.note).with_metadata(json!({
935 "evidence_routing": "inline"
936 })),
937 prepared.block.into_iter().collect(),
938 ));
939 }
940
941 // The small-contract reader decodes non-image buffers as UTF-8 text with replacement
942 // characters instead of refusing the whole read on one invalid byte.
943 let text = String::from_utf8_lossy(&bytes);
944 // Count and select with byte boundaries, without a pointer per newline.
945 let line_count = text.bytes().filter(|byte| *byte == b'\n').count() + 1;
946 let requested_offset = offset.unwrap_or(1);
947 let start = requested_offset.saturating_sub(1);
948 if start >= line_count {
949 return Err(ToolError::execution_failed(format!(
950 "Offset {requested_offset} is beyond end of file ({line_count} lines total)"
951 )));
952 }
953 let start_byte = if start == 0 {
954 0
955 } else {
956 text.match_indices('\n')
957 .nth(start - 1)
958 .expect("line count validated offset")
959 .0
960 + 1
961 };
962 let selected_lines = limit.unwrap_or(usize::MAX).min(line_count - start);
963 let end_byte = if selected_lines == 0 {
964 start_byte
965 } else {
966 text[start_byte..]
967 .match_indices('\n')
968 .nth(selected_lines - 1)
969 .map_or(text.len(), |(index, _)| start_byte + index)
970 };
971 let selected_content = &text[start_byte..end_byte];
972 let window = contract_read_window(selected_content, max_bytes);
973 // Truncated means the file holds more than this response shows:
974 // either the byte budget cut the window, or a bounded range stopped
975 // before EOF. A whole file that fits is never truncated.
976 let truncated = window.truncated || limit.is_some() && start + selected_lines < line_count;
977 let first_display = start + 1;
978 let mut output = if window.first_line_too_large {
979 let size = selected_content.split('\n').next().map_or(0, str::len);
980 format!(
981 "[Line {first_display} is {}, exceeds the {max_bytes}-byte output budget for this call. Use bash: sed -n '{first_display}p' {path_str} | head -c {max_bytes}]",
982 contract_format_size(size)
983 )
984 } else {
985 window.content
986 };
987
988 if !window.first_line_too_large && window.truncated {
989 let last_display = first_display + window.shown_lines.saturating_sub(1);
990 let next_offset = last_display + 1;
991 // Continuation must be exact: name the next offset, and when the
992 // caller asked for a bounded range, the part of that range still
993 // unread. `max_bytes` is only offered while it can still go up.
994 let mut hint = format!("offset={next_offset}");
995 if let Some(limit) = limit {
996 let remaining = limit.saturating_sub(window.shown_lines);
997 if remaining > 0 {
998 hint.push_str(&format!(" limit={remaining}"));
999 }
1000 }
1001 let raise = if max_bytes < READ_REQUEST_MAX_BYTES {
1002 format!(", or max_bytes up to {READ_REQUEST_MAX_BYTES} to read more per call")
1003 } else {
1004 String::new()
1005 };
1006 output.push_str(&format!(
1007 "\n\n[Showing lines {first_display}-{last_display} of {} ({} total, {max_bytes}-byte output budget). Use {hint} to continue{raise}.]",
1008 line_count,
1009 contract_format_size(size_bytes)
1010 ));
1011 } else if limit.is_some() {
1012 let consumed = selected_lines;
1013 if start + consumed < line_count {
1014 let remaining = line_count - (start + consumed);
1015 let next_offset = start + consumed + 1;
1016 output.push_str(&format!(
1017 "\n\n[{remaining} more lines in file ({} total). Use offset={next_offset} to continue.]",
1018 contract_format_size(size_bytes)
1019 ));
1020 }
1021 }
1022
1023 // This internal observation keeps hidden legacy edit replay working,
1024 // but no hash or read-before-edit ceremony reaches the lowercase
1025 // schema or result.
1026 context.note_file_read(&file_path);
1027 Ok(RichToolResult::plain(
1028 ToolResult::success(output).with_metadata(json!({
1029 "evidence_routing": "inline",
1030 // The budget this call actually enforced. The context
1031 // compactor honors it so an already-bounded read is never
1032 // truncated a second time on its way into the conversation.
1033 "read_budget_bytes": max_bytes,
1034 // #6283: paging contract. `size` is the whole file in bytes,
1035 // `line_count` its total lines, `truncated` whether the file
1036 // holds more than this response shows.
1037 "size": size_bytes,
1038 "truncated": truncated,
1039 "line_count": line_count
1040 })),
1041 ))
1042 }
1043 }
1044
1045 #[async_trait]
1046 impl ToolSpec for ReadFileTool {
1047 fn name(&self) -> &'static str {
1048 "read_file"
1049 }
1050
1051 fn model_visible(&self) -> bool {
1052 false
1053 }
1054
1055 fn description(&self) -> &'static str {
1056 "Read a UTF-8 file from the workspace. Use this instead of `cat`, `head`, `tail`, or `sed -n '..p'` in `Bash` — it's faster, sandbox-aware, and skips the approval prompt. Plain text is returned as-is and records the file snapshot required before `edit` will make a narrow in-place edit. Text reads report the whole file's `content_hash=\"sha256:…\"`; pass that value back as `expected_hash` on a later `write`, `edit`, or `patch` to have the write refused if the file changed in between. Codewhale config files and file-backed credential stores cannot be read with this tool; use `codewhale config list` or `codewhale auth status` for safe inspection. PDFs are text-extracted when the optional `pdftotext` executable (Poppler) is installed. Image screenshots are OCR-extracted when local OCR is available. Cannot read other non-PDF binaries.\n\nFor large files, use `start_line` and `max_lines` to read in chunks. By default, returns up to 500 lines or 16KB, whichever comes first. If `truncated=\"true\"` and `next_start_line` is present, continue reading from there; a byte-limited window instead shows head + tail with a `[CONTENT TRUNCATED]` marker and its note says how to narrow the range. For PDFs, use `pages` instead — `start_line`/`max_lines` only apply to text files."
1057 }
1058
1059 fn input_schema(&self) -> Value {
1060 json!({
1061 "type": "object",
1062 "properties": {
1063 "path": {
1064 "type": "string",
1065 "description": "Path to the file (relative to workspace, absolute, or ~/ home-relative). Alias: `file_path`"
1066 },
1067 "start_line": {
1068 "type": "integer",
1069 "description": "Starting line (1-based, default 1). Aliases: `offset`, `line_offset`"
1070 },
1071 "max_lines": {
1072 "type": "integer",
1073 "description": "Maximum lines to return (default 500, max 500; a 16KB byte budget applies regardless). Aliases: `limit`, `n_lines`"
1074 },
1075 "pages": {
1076 "type": "string",
1077 "description": "PDF only: page range to extract, e.g. \"1-5\" or \"10\". Ignored for non-PDF files."
1078 }
1079 },
1080 "required": ["path"]
1081 })
1082 }
1083
1084 fn capabilities(&self) -> Vec<ToolCapability> {
1085 vec![ToolCapability::ReadOnly, ToolCapability::Sandboxable]
1086 }
1087
1088 fn supports_parallel(&self) -> bool {
1089 true
1090 }
1091
1092 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
1093 let mut input = input;
1094 apply_param_aliases(&mut input, PATH_ALIASES, "File read")?;
1095 apply_param_aliases(&mut input, READ_ALIASES, "File read")?;
1096 READ_PARAMS.reject_unknown(&input)?;
1097
1098 let path_str = required_str(&input, "path")?;
1099 // S1/F2: raw spelling first, resolved path after — see the matching
1100 // comment in `execute_contract_read`. Only the raw-spelling denial can
1101 // promise an error that never names the symlink target's location.
1102 let file_path = resolve_guarded_read_path(context, path_str, "read_file")?;
1103 let pages = optional_str(&input, "pages")?;
1104
1105 if let Some(result) = read_pdf_if_detected(
1106 &file_path,
1107 pages,
1108 super::pdf::PdfTextCommand::system(Some(context)),
1109 )
1110 .await?
1111 {
1112 return Ok(result);
1113 }
1114 if is_image_for_ocr(&file_path) {
1115 return read_image_via_ocr(&file_path, path_str, context).await;
1116 }
1117
1118 // Open before parameter parsing so a missing file keeps the
1119 // historical "Failed to read …" error shape regardless of the other
1120 // arguments. The open and size probe run on the blocking pool —
1121 // tool handlers execute on the Tokio runtime (blocking-call
1122 // convention, #6149).
1123 let file_bytes = tokio::task::spawn_blocking({
1124 let file_path = file_path.clone();
1125 move || {
1126 let file = fs::File::open(&file_path).map_err(|e| {
1127 ToolError::execution_failed(format!(
1128 "Failed to read {}: {}",
1129 file_path.display(),
1130 e
1131 ))
1132 })?;
1133 Ok::<_, ToolError>(file.metadata().map(|meta| meta.len()).unwrap_or(u64::MAX))
1134 }
1135 })
1136 .await
1137 .map_err(|e| ToolError::execution_failed(format!("File open task: {e}")))??;
1138
1139 let explicit_range = input
1140 .get("start_line")
1141 .or_else(|| input.get("max_lines"))
1142 .is_some();
1143
1144 // Small-file fast path. Only applies when the caller didn't pass an
1145 // explicit range — otherwise an explicit `start_line = 5` on a
1146 // tiny file would silently ignore the request.
1147 if !explicit_range && file_bytes <= SMALL_FILE_BYTES as u64 {
1148 let contents = tokio::fs::read_to_string(&file_path).await.map_err(|e| {
1149 ToolError::execution_failed(format!(
1150 "Failed to read {}: {}",
1151 file_path.display(),
1152 e
1153 ))
1154 })?;
1155 context.note_file_read(&file_path);
1156
1157 let total_lines = contents.lines().count();
1158 if total_lines <= SMALL_FILE_LINES {
1159 // The whole file is in hand, so hash it directly rather than
1160 // re-reading it. Prefixed as a header line because this branch
1161 // returns the contents unwrapped — there is no `<file …>` tag
1162 // to hang the attribute on.
1163 let hash = content_hash(contents.as_bytes());
1164 let body = format!("{}{contents}", content_hash_header(&hash));
1165 return Ok(ToolResult::success(body).with_metadata(json!({
1166 "evidence_routing": "inline",
1167 "content_hash": hash
1168 })));
1169 }
1170
1171 // Small in bytes but too many lines: render the default window
1172 // straight from the in-memory contents.
1173 let hash = content_hash(contents.as_bytes());
1174 let window: Vec<String> = contents
1175 .lines()
1176 .take(DEFAULT_READ_LINES)
1177 .map(str::to_string)
1178 .collect();
1179 return Ok(render_line_window(
1180 path_str,
1181 &window,
1182 total_lines,
1183 1,
1184 DEFAULT_READ_LINES,
1185 Some(hash.as_str()),
1186 ));
1187 }
1188
1189 // Strict types (2026-08-04 review): a `start_line:"1200"` string or a
1190 // negative/float value used to silently fall back to the defaults —
1191 // returning the head of the file instead of the window the model
1192 // asked for, the exact wrong-answer-shaped-like-a-right-one this
1193 // action's alias/unknown-parameter hardening exists to prevent.
1194 let start_line = match optional_u64(&input, "start_line", 1)? {
1195 0 => {
1196 return Err(ToolError::invalid_input(
1197 "start_line must be 1-based and greater than 0".to_string(),
1198 ));
1199 }
1200 v => usize::try_from(v).map_err(|_| {
1201 ToolError::invalid_input(
1202 "start_line exceeds platform addressable range".to_string(),
1203 )
1204 })?,
1205 };
1206
1207 let max_lines = match optional_u64(&input, "max_lines", DEFAULT_READ_LINES as u64)? {
1208 0 => {
1209 return Err(ToolError::invalid_input(
1210 "max_lines must be greater than 0".to_string(),
1211 ));
1212 }
1213 v => {
1214 let converted = usize::try_from(v).map_err(|_| {
1215 ToolError::invalid_input(
1216 "max_lines exceeds platform addressable range".to_string(),
1217 )
1218 })?;
1219 std::cmp::min(converted, HARD_MAX_READ_LINES)
1220 }
1221 };
1222
1223 // Bounded read for ranged/large files: skip and take lines through a
1224 // BufReader instead of materializing the whole file. The stream still
1225 // runs to EOF so the total line count and whole-file UTF-8 validation
1226 // match the historical read_to_string behavior. Open, stream, and hash
1227 // all run on the blocking pool (blocking-call convention, #6149).
1228 let (window, total_lines, hash) = tokio::task::spawn_blocking({
1229 let file_path = file_path.clone();
1230 move || {
1231 let file = fs::File::open(&file_path).map_err(|e| {
1232 ToolError::execution_failed(format!(
1233 "Failed to read {}: {}",
1234 file_path.display(),
1235 e
1236 ))
1237 })?;
1238 let (window, total_lines) = read_window_streaming(file, start_line, max_lines)
1239 .map_err(|e| {
1240 ToolError::execution_failed(format!(
1241 "Failed to read {}: {}",
1242 file_path.display(),
1243 e
1244 ))
1245 })?;
1246 // The window is a slice; the guard needs the whole file. A
1247 // second streaming pass digests the rest without ever
1248 // materializing it. A failure here only costs the guard — the
1249 // read itself already succeeded, so the window is still
1250 // returned, just without a hash to pass back to `edit`.
1251 // Special files are skipped: reopening a FIFO or device can
1252 // block indefinitely (or re-consume a one-shot stream), and a
1253 // stream has no stable content an edit guard could pin.
1254 let hash = match fs::metadata(&file_path) {
1255 Ok(meta) if meta.is_file() => hash_file_streaming(&file_path).ok(),
1256 _ => None,
1257 };
1258 Ok::<_, ToolError>((window, total_lines, hash))
1259 }
1260 })
1261 .await
1262 .map_err(|e| ToolError::execution_failed(format!("File read task: {e}")))??;
1263 context.note_file_read(&file_path);
1264
1265 // `start_line > total_lines` is not an error — it lets the model
1266 // page past the end without raising. Returns an empty-content
1267 // sentinel so subsequent reads can stop.
1268 if start_line > total_lines {
1269 let hash_attr = hash
1270 .as_deref()
1271 .map(|hash| format!(" content_hash=\"{hash}\""))
1272 .unwrap_or_default();
1273 let output = format!(
1274 "<file path=\"{path_str}\" total_lines=\"{total_lines}\" shown_lines=\"none\" truncated=\"false\"{hash_attr}>\n\
1275 \n\
1276 [NO CONTENT] start_line {start_line} is beyond total_lines {total_lines}.\n\
1277 </file>"
1278 );
1279 return Ok(ToolResult::success(output).with_metadata(json!({
1280 "evidence_routing": "inline",
1281 "content_hash": hash
1282 })));
1283 }
1284
1285 Ok(render_line_window(
1286 path_str,
1287 &window,
1288 total_lines,
1289 start_line,
1290 max_lines,
1291 hash.as_deref(),
1292 ))
1293 }
1294 }
1295
1296 // Bounded output for large files. The small-file fast path keeps the
1297 // historical "return contents unchanged" behavior so existing flows
1298 // (small configs, single source files, etc.) don't suddenly start
1299 // seeing wrapped output. Once a file is large or the caller asks
1300 // for an explicit range, we switch to a numbered, line-tagged
1301 // window with continuation hints so the model can page through
1302 // without re-loading the entire file on every turn. Harvested
1303 // from PR #1451 by @Oliver-ZPLiu, closes part of #1450.
1304 // One bound, not two competing ones. The real cost of a read is BYTES of
1305 // context, and `MAX_VISIBLE_BYTES` already enforces that. A separate 200-line
1306 // default fired long before the byte budget on any prose file — a 229-line,
1307 // 12 KB document truncated at line 200 with a third of the budget unspent,
1308 // costing a second round trip to fetch 29 lines. The line cap now only guards
1309 // pathologically short lines, where 500 lines is still a small read.
1310 const DEFAULT_READ_LINES: usize = HARD_MAX_READ_LINES;
1311 const HARD_MAX_READ_LINES: usize = 500;
1312 const MAX_VISIBLE_BYTES: usize = 16 * 1024;
1313 const SMALL_FILE_LINES: usize = HARD_MAX_READ_LINES;
1314 const SMALL_FILE_BYTES: usize = 16 * 1024;
1315
1316 /// Stream a line window out of `file`: skip `start_line - 1` lines, collect
1317 /// up to `max_lines`, then keep counting (and validating UTF-8) to EOF.
1318 /// Returns the collected window plus the total line count. Only the window
1319 /// is ever held in memory.
1320 fn read_window_streaming(
1321 file: fs::File,
1322 start_line: usize,
1323 max_lines: usize,
1324 ) -> std::io::Result<(Vec<String>, usize)> {
1325 use std::io::BufRead;
1326
1327 let mut reader = std::io::BufReader::new(file);
1328 let mut raw: Vec<u8> = Vec::new();
1329 let mut window: Vec<String> = Vec::new();
1330 let mut total_lines = 0usize;
1331 let start_idx = start_line - 1;
1332
1333 loop {
1334 raw.clear();
1335 let n = reader.read_until(b'\n', &mut raw)?;
1336 if n == 0 {
1337 break;
1338 }
1339 // Mirror `str::lines`: strip the trailing '\n', and a '\r' only when
1340 // it directly precedes that '\n'.
1341 let mut end = raw.len();
1342 if raw[..end].ends_with(b"\n") {
1343 end -= 1;
1344 if raw[..end].ends_with(b"\r") {
1345 end -= 1;
1346 }
1347 }
1348 // Validate every line so invalid UTF-8 anywhere in the file fails
1349 // exactly like the previous whole-file read_to_string did.
1350 let line = std::str::from_utf8(&raw[..end]).map_err(|_| {
1351 std::io::Error::new(
1352 std::io::ErrorKind::InvalidData,
1353 "stream did not contain valid UTF-8",
1354 )
1355 })?;
1356 if total_lines >= start_idx && window.len() < max_lines {
1357 window.push(line.to_string());
1358 }
1359 total_lines += 1;
1360 }
1361
1362 Ok((window, total_lines))
1363 }
1364
1365 /// Marker placed between the retained head and tail when a read window is
1366 /// truncated by the byte budget. Mirrors qwen-code's truncation style so the
1367 /// model sees both ends of the range.
1368 const BYTE_TRUNCATION_SEPARATOR: &str = "\n\n---\n... [CONTENT TRUNCATED] ...\n---\n\n";
1369
1370 /// Split `content` into a head of at most `head_budget` bytes and a tail that
1371 /// fills the remainder of `total_budget` (separator accounted for). Never
1372 /// overlaps and never splits mid-codepoint. Style matches qwen-code:
1373 /// `head_budget = total_budget / 5`.
1374 fn head_tail_for_budget(content: &str, total_budget: usize) -> (String, String) {
1375 let head_budget = (total_budget / 5).max(1);
1376 let head_end = (0..=head_budget.min(content.len()))
1377 .rev()
1378 .find(|&i| content.is_char_boundary(i))
1379 .unwrap_or(0);
1380 let sep_len = BYTE_TRUNCATION_SEPARATOR.len();
1381 let tail_budget = total_budget
1382 .saturating_sub(head_end)
1383 .saturating_sub(sep_len)
1384 .max(1);
1385 let tail_floor = content.len().saturating_sub(tail_budget).max(head_end);
1386 let tail_start = (tail_floor..=content.len())
1387 .find(|&i| content.is_char_boundary(i))
1388 .unwrap_or(content.len());
1389 (
1390 content[..head_end].to_string(),
1391 content[tail_start..].to_string(),
1392 )
1393 }
1394
1395 /// Render a collected line window into the `<file …>` wrapper used for
1396 /// ranged/large reads. `window` must hold the lines for
1397 /// `start_line..start_line + max_lines` (clamped to EOF).
1398 fn render_line_window(
1399 path_str: &str,
1400 window: &[String],
1401 total_lines: usize,
1402 start_line: usize,
1403 max_lines: usize,
1404 content_hash: Option<&str>,
1405 ) -> ToolResult {
1406 let zero_based_start = start_line - 1;
1407 let zero_based_end = std::cmp::min(zero_based_start + max_lines, total_lines);
1408 let shown_first = start_line;
1409 let shown_last = zero_based_end; // 1-based inclusive line number of the last shown line
1410
1411 let mut numbered = String::new();
1412 for (offset, line) in window.iter().enumerate() {
1413 let line_no = start_line + offset;
1414 numbered.push_str(&format!("{line_no:>6}│ {line}\n"));
1415 }
1416
1417 // UTF-8-safe byte truncation of the rendered range. Qwen-style: keep a
1418 // short head (budget/5) plus the matching tail so the model sees both
1419 // ends of a long range. The full file already lives at `path_str` — the
1420 // recovery note names that absolute/workspace path for a re-read.
1421 let visible_bytes =
1422 crate::tools::large_output_router::WorkshopConfig::active_read_result_max_bytes()
1423 .map(|n| n.clamp(MAX_VISIBLE_BYTES, READ_RESULT_ABSOLUTE_MAX_BYTES))
1424 .unwrap_or(MAX_VISIBLE_BYTES);
1425 let truncated_by_bytes = numbered.len() > visible_bytes;
1426 let shown_content = if truncated_by_bytes {
1427 let (head, tail) = head_tail_for_budget(&numbered, visible_bytes);
1428 format!("{head}{BYTE_TRUNCATION_SEPARATOR}{tail}")
1429 } else {
1430 numbered
1431 };
1432
1433 let truncated_by_lines = zero_based_end < total_lines;
1434 let truncated = truncated_by_lines || truncated_by_bytes;
1435 let next_start = zero_based_end + 1;
1436
1437 let mut attrs = format!(
1438 "path=\"{path_str}\" total_lines=\"{total_lines}\" shown_lines=\"{shown_first}-{shown_last}\" truncated=\"{truncated}\""
1439 );
1440 if truncated_by_lines {
1441 attrs.push_str(&format!(" next_start_line=\"{next_start}\""));
1442 }
1443 // Hashes the whole file, not the shown window — a partial read still
1444 // yields a guard the model can pass to `edit`/`patch`.
1445 if let Some(hash) = content_hash {
1446 attrs.push_str(&format!(" content_hash=\"{hash}\""));
1447 }
1448
1449 let mut output = format!("<file {attrs}>\n{shown_content}");
1450 if truncated_by_lines {
1451 output.push_str(&format!(
1452 "\n[TRUNCATED] Showing lines {shown_first}-{shown_last} of {total_lines}. To continue, call read with path=\"{path_str}\" offset={next_start} limit={max_lines}\n"
1453 ));
1454 }
1455 if truncated_by_bytes {
1456 if shown_first == shown_last {
1457 // One line alone exceeds the byte budget: no start_line/max_lines
1458 // combination can ever reveal the elided middle, so the note must
1459 // not pretend otherwise — name the escape hatch that works.
1460 output.push_str(&format!(
1461 "\n[TRUNCATED] Line {shown_first} alone exceeds the {visible_bytes}-byte output budget; showing its head + tail. No line window can reveal the middle of one line — use a searched shell slice when needed.\n"
1462 ));
1463 } else {
1464 let narrower = (shown_last - shown_first).div_ceil(2).max(1);
1465 output.push_str(&format!(
1466 "\n[TRUNCATED] The selected range exceeded the {visible_bytes}-byte output budget; showing head + tail of lines {shown_first}-{shown_last}. Re-read narrower windows to see the middle, e.g. offset={shown_first} limit={narrower}, then advance offset.\n"
1467 ));
1468 }
1469 }
1470 output.push_str("</file>");
1471
1472 // The file tool self-bounds at its own byte budget and carries its own continuation
1473 // contract (`next_start_line`), so the large-output spillover envelope
1474 // must never re-wrap a read result with a second, weaker truncation.
1475 ToolResult::success(output).with_metadata(json!({
1476 "evidence_routing": "inline",
1477 "content_hash": content_hash
1478 }))
1479 }
1480
1481 async fn read_image_via_ocr(
1482 path: &Path,
1483 requested_path: &str,
1484 context: &ToolContext,
1485 ) -> Result<ToolResult, ToolError> {
1486 let text = crate::tools::image_ocr::ocr_image_path(path, context).await?;
1487 Ok(ToolResult::success(format!(
1488 "<image_ocr path=\"{requested_path}\">\n{text}\n</image_ocr>"
1489 )))
1490 }
1491
1492 /// Detect an existing PDF by extension or by sniffing `%PDF` magic bytes.
1493 async fn is_pdf(path: &Path) -> Result<bool, ToolError> {
1494 let extension_matches = path
1495 .extension()
1496 .and_then(|e| e.to_str())
1497 .is_some_and(|ext| ext.eq_ignore_ascii_case("pdf"));
1498 let mut file = tokio::fs::File::open(path).await.map_err(|error| {
1499 ToolError::execution_failed(format!("Failed to read {}: {error}", path.display()))
1500 })?;
1501 if extension_matches {
1502 return Ok(true);
1503 }
1504 let mut buf = [0u8; 4];
1505 use tokio::io::AsyncReadExt;
1506 Ok(file.read_exact(&mut buf).await.is_ok() && &buf == b"%PDF")
1507 }
1508
1509 fn is_image_for_ocr(path: &Path) -> bool {
1510 path.extension()
1511 .and_then(|e| e.to_str())
1512 .is_some_and(|ext| {
1513 matches!(
1514 ext.to_ascii_lowercase().as_str(),
1515 "png" | "jpg" | "jpeg" | "tif" | "tiff" | "bmp"
1516 )
1517 })
1518 }
1519
1520 fn parse_pages_arg(spec: &str) -> Option<(u32, u32)> {
1521 let trimmed = spec.trim();
1522 if trimmed.is_empty() {
1523 return None;
1524 }
1525 if let Some((a, b)) = trimmed.split_once('-') {
1526 let start: u32 = a.trim().parse().ok()?;
1527 let end: u32 = b.trim().parse().ok()?;
1528 if start == 0 || end < start {
1529 return None;
1530 }
1531 Some((start, end))
1532 } else {
1533 let n: u32 = trimmed.parse().ok()?;
1534 if n == 0 {
1535 return None;
1536 }
1537 Some((n, n))
1538 }
1539 }
1540
1541 /// Clean PDF-extracted text for TUI display: collapse consecutive blank
1542 /// lines (more than 1 becomes 1), replace NUL bytes with U+FFFD, replace
1543 /// non-breaking spaces with regular spaces, and trim trailing whitespace
1544 /// on each line. Produces output that won't clutter the transcript with
1545 /// vertical gaps or invisible control characters.
1546 fn clean_pdf_text(raw: &str) -> String {
1547 let mut out = String::with_capacity(raw.len());
1548 let mut blank_run = 0usize;
1549 let mut any_content = false;
1550 for line in raw.lines() {
1551 let trimmed = line.trim_end();
1552 if trimmed.is_empty() {
1553 blank_run = blank_run.saturating_add(1);
1554 if blank_run <= 1 {
1555 out.push('\n');
1556 }
1557 } else {
1558 blank_run = 0;
1559 any_content = true;
1560 // Push cleaned characters directly — avoids a per-line
1561 // temporary String allocation.
1562 for c in trimmed.chars() {
1563 match c {
1564 '\0' => out.push('\u{FFFD}'),
1565 '\u{A0}' => out.push(' '),
1566 other => out.push(other),
1567 }
1568 }
1569 out.push('\n');
1570 }
1571 }
1572 // Trim leading blank lines only — don't use str::trim() which
1573 // would also strip intentional indentation (e.g. centred titles).
1574 if any_content {
1575 let start = out.find(|c: char| c != '\n').unwrap_or(0);
1576 // Walk back from end to find the last non-newline character.
1577 let end = out.rfind(|c: char| c != '\n').map_or(out.len(), |i| {
1578 i + out[i..].chars().next().map_or(1, |c| c.len_utf8())
1579 });
1580 out[start..end].to_string()
1581 } else {
1582 String::new()
1583 }
1584 }
1585
1586 async fn read_pdf_if_detected(
1587 path: &Path,
1588 pages: Option<&str>,
1589 command: super::pdf::PdfTextCommand<'_>,
1590 ) -> Result<Option<ToolResult>, ToolError> {
1591 if !is_pdf(path).await? {
1592 return Ok(None);
1593 }
1594 // Validate the `pages` spec once, up front, so both extractor paths
1595 // surface the same error shape on bad input.
1596 let page_range = match pages {
1597 Some(spec) => match parse_pages_arg(spec) {
1598 Some((start, end)) => Some((start, end)),
1599 None => {
1600 return Err(ToolError::invalid_input(format!(
1601 "invalid `pages` value `{spec}` (expected `N` or `N-M`, e.g. `1-5`)"
1602 )));
1603 }
1604 },
1605 None => None,
1606 };
1607
1608 read_pdf_with_command(path, page_range, command)
1609 .await
1610 .map(Some)
1611 }
1612
1613 async fn read_pdf_with_command(
1614 path: &Path,
1615 page_range: Option<(u32, u32)>,
1616 command: super::pdf::PdfTextCommand<'_>,
1617 ) -> Result<ToolResult, ToolError> {
1618 let text = super::pdf::extract_path(path, page_range, command)
1619 .await
1620 .map_err(super::pdf::into_tool_error)?;
1621 Ok(ToolResult::success(clean_pdf_text(&text)))
1622 }
1623
1624 // === WriteFileTool ===
1625
1626 /// Tool for writing UTF-8 files to the workspace.
1627 pub struct WriteFileTool;
1628
1629 impl WriteFileTool {
1630 /// Execute the small-contract-shaped lowercase writer. Compatibility-only hash
1631 /// arguments remain on the hidden `write_file`/`File` paths.
1632 pub(super) async fn execute_contract_write(
1633 input: Value,
1634 context: &ToolContext,
1635 ) -> Result<ToolResult, ToolError> {
1636 reject_primitive_unknown(&input, "write", &["path", "content"])?;
1637 let path_str = required_str(&input, "path")?;
1638 let file_content = required_str(&input, "content")?;
1639 check_contract_file_size(file_content.len())?;
1640 let file_path = context.resolve_path(path_str)?;
1641 let mutation_guard = acquire_file_mutation(&file_path, context).await?;
1642 check_file_operation_cancelled(context)?;
1643
1644 let prior = load_contract_source(&file_path, false, context).await?;
1645 let existed_before = prior.is_some();
1646 let prior_bytes = prior.unwrap_or_default();
1647 let prior_contents = String::from_utf8_lossy(&prior_bytes);
1648
1649 if let Some(parent) = file_path.parent() {
1650 tokio::fs::create_dir_all(parent).await.map_err(|error| {
1651 ToolError::execution_failed(format!(
1652 "Failed to create directory {}: {error}",
1653 parent.display()
1654 ))
1655 })?;
1656 }
1657 check_file_operation_cancelled(context)?;
1658 // Preserve the existing file's line-ending style on overwrite (see
1659 // `preserve_prior_line_endings`); otherwise a CRLF (Windows) file is
1660 // silently rewritten with LF line endings.
1661 let mut written = if prior_contents.is_empty() {
1662 file_content.to_string()
1663 } else {
1664 let normalized = normalize_contract_line_endings(file_content);
1665 let ending = contract_line_ending(&prior_contents);
1666 let extra = if ending == "\r\n" {
1667 normalized.bytes().filter(|byte| *byte == b'\n').count()
1668 } else {
1669 0
1670 };
1671 check_contract_file_size(normalized.len().saturating_add(extra))?;
1672 restore_contract_line_endings(&normalized, ending)
1673 };
1674 guard_edit(
1675 &file_path,
1676 path_str,
1677 existed_before.then(|| prior_contents.as_ref()),
1678 &written,
1679 )?;
1680 if existed_before
1681 && let Some(normalized) = normalize_edit(&file_path, &prior_contents, &written).await
1682 {
1683 written = normalized;
1684 }
1685 check_contract_file_size(written.len())?;
1686 check_file_operation_cancelled(context)?;
1687 // Once replacement starts, report its actual completion even if cancelled.
1688 run_blocking_write_atomic(&file_path, written.clone().into_bytes()).await?;
1689 context.note_file_read(&file_path);
1690 drop(mutation_guard);
1691
1692 let outcome = if existed_before { "updated" } else { "created" };
1693 let utf16_units = written.encode_utf16().count();
1694 Ok(contract_mutation_result(
1695 context,
1696 &file_path,
1697 path_str,
1698 prior_contents.as_ref(),
1699 &written,
1700 written.as_bytes(),
1701 outcome,
1702 format!("Successfully wrote {utf16_units} bytes to {path_str}"),
1703 )
1704 .await)
1705 }
1706 }
1707
1708 #[async_trait]
1709 impl ToolSpec for WriteFileTool {
1710 fn name(&self) -> &'static str {
1711 "write_file"
1712 }
1713
1714 fn model_visible(&self) -> bool {
1715 false
1716 }
1717
1718 fn description(&self) -> &'static str {
1719 "Write content to a UTF-8 file in the workspace. Use this instead of heredocs (`cat <<EOF > file`) or `echo > file` in `Bash` — diffs render inline and approval is handled cleanly. Creates or overwrites; parent directories are auto-created. Pass `expected_hash` (the `content_hash` from a prior `read`) to have the overwrite refused if the file changed since that read."
1720 }
1721
1722 fn input_schema(&self) -> Value {
1723 json!({
1724 "type": "object",
1725 "properties": {
1726 "path": {
1727 "type": "string",
1728 "description": "Path to the file. Alias: `file_path`"
1729 },
1730 "content": {
1731 "type": "string",
1732 "description": "Content to write"
1733 },
1734 "expected_hash": {
1735 "type": "string",
1736 "description": EXPECTED_HASH_DESCRIPTION
1737 }
1738 },
1739 "required": ["path", "content"]
1740 })
1741 }
1742
1743 fn capabilities(&self) -> Vec<ToolCapability> {
1744 vec![
1745 ToolCapability::WritesFiles,
1746 ToolCapability::Sandboxable,
1747 ToolCapability::RequiresApproval,
1748 ]
1749 }
1750
1751 fn approval_requirement(&self) -> ApprovalRequirement {
1752 ApprovalRequirement::Suggest
1753 }
1754
1755 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
1756 let mut input = input;
1757 apply_param_aliases(&mut input, PATH_ALIASES, "File write")?;
1758 WRITE_PARAMS.reject_unknown(&input)?;
1759
1760 let path_str = required_str(&input, "path")?;
1761 let file_content = required_str(&input, "content")?;
1762 let expected_hash = optional_str(&input, "expected_hash")?;
1763
1764 let file_path = context.resolve_path(path_str)?;
1765
1766 // Snapshot the existing contents (if any) before we overwrite — used
1767 // to render an inline diff in the tool result. Only a genuinely
1768 // absent path is "new": a stat that fails for any other reason must
1769 // not let an existing file be overwritten as if it were empty.
1770 let existed_before = tokio::fs::try_exists(&file_path).await.map_err(|error| {
1771 ToolError::execution_failed(format!(
1772 "Failed to inspect {}: {error}",
1773 file_path.display()
1774 ))
1775 })?;
1776 let prior_contents = if existed_before {
1777 tokio::fs::read_to_string(&file_path)
1778 .await
1779 .map_err(|error| {
1780 ToolError::execution_failed(format!(
1781 "Failed to read {}: {error}",
1782 file_path.display()
1783 ))
1784 })?
1785 } else {
1786 String::new()
1787 };
1788
1789 // Content-hash guard (#3979), checked against the same snapshot the
1790 // diff is rendered from and before any directory or file is touched.
1791 if let Some(expected) = expected_hash {
1792 if !existed_before {
1793 // A hash describes a file that was read. Guarding a create is
1794 // a contradiction, and silently creating the file anyway would
1795 // defeat the guard the caller asked for — fail closed.
1796 return Err(ToolError::execution_failed(format!(
1797 "File `write` refused: expected_hash was supplied but {path_str} does not exist, so there is no snapshot to verify and nothing was written. Recovery: drop `expected_hash` to create the file, or read the intended path first."
1798 )));
1799 }
1800 verify_expected_hash(Some(expected), prior_contents.as_bytes(), "write", path_str)?;
1801 }
1802
1803 // Create parent directories if needed
1804 if let Some(parent) = file_path.parent() {
1805 tokio::fs::create_dir_all(parent).await.map_err(|e| {
1806 ToolError::execution_failed(format!(
1807 "Failed to create directory {}: {}",
1808 parent.display(),
1809 e
1810 ))
1811 })?;
1812 }
1813
1814 // Preserve the existing file's line-ending style on overwrite (see
1815 // `preserve_prior_line_endings`); a full `write_file` over a CRLF
1816 // (Windows) file otherwise silently rewrites every line ending to LF.
1817 let mut written = preserve_prior_line_endings(file_content, &prior_contents);
1818
1819 guard_edit(
1820 &file_path,
1821 path_str,
1822 existed_before.then(|| prior_contents.as_ref()),
1823 &written,
1824 )?;
1825 if existed_before
1826 && let Some(normalized) = normalize_edit(&file_path, &prior_contents, &written).await
1827 {
1828 written = normalized;
1829 }
1830
1831 run_blocking_write_atomic(&file_path, written.clone().into_bytes()).await?;
1832 context.note_file_read(&file_path);
1833
1834 let display = file_path.display().to_string();
1835 let diff = make_unified_diff(&display, &prior_contents, &written);
1836 let summary = if existed_before {
1837 format!("Wrote {} bytes to {}", written.len(), display)
1838 } else {
1839 format!("Created {} ({} bytes)", display, written.len())
1840 };
1841 let body = if diff.is_empty() {
1842 format!("{summary}\n(no changes)")
1843 } else {
1844 format!("{diff}\n{summary}")
1845 };
1846
1847 // Append LSP diagnostics for the written file when enabled (#428).
1848 let diag_block = lsp_diagnostics_for_paths(context, &[file_path]).await;
1849 let full_body = if diag_block.is_empty() {
1850 body
1851 } else {
1852 format!("{body}\n{diag_block}")
1853 };
1854
1855 let outcome = if existed_before { "updated" } else { "created" };
1856 // Keep the execution-owned receipt workspace-relative even though the
1857 // legacy model-facing output above retains its resolved-path wording.
1858 let receipt_diff = make_unified_diff(path_str, &prior_contents, &written);
1859 Ok(ToolResult::success(full_body).with_metadata(json!({
1860 "event": "file.mutation",
1861 "mutation": {
1862 "diff": receipt_diff,
1863 "files": [mutation_file_entry(path_str, outcome, Some(written.as_bytes()))],
1864 "renames": []
1865 }
1866 })))
1867 }
1868 }
1869
1870 // === EditFileTool ===
1871
1872 /// Tool for search/replace editing of files.
1873 pub struct EditFileTool;
1874
1875 #[derive(Clone, Debug)]
1876 struct ContractEdit {
1877 index: usize,
1878 old_text: String,
1879 new_text: String,
1880 }
1881
1882 #[derive(Clone, Debug)]
1883 struct ResolvedContractEdit {
1884 index: usize,
1885 start: usize,
1886 end: usize,
1887 replacement: String,
1888 }
1889
1890 fn normalize_contract_line_endings(text: &str) -> String {
1891 text.replace("\r\n", "\n").replace('\r', "\n")
1892 }
1893
1894 fn contract_line_ending(text: &str) -> &'static str {
1895 match text.find('\n') {
1896 Some(index) if index > 0 && text.as_bytes()[index - 1] == b'\r' => "\r\n",
1897 _ => "\n",
1898 }
1899 }
1900
1901 fn restore_contract_line_endings(text: &str, ending: &str) -> String {
1902 if ending == "\r\n" {
1903 text.replace('\n', "\r\n")
1904 } else {
1905 text.to_string()
1906 }
1907 }
1908
1909 /// First code point of the placeholder range that carries one non-UTF-8 byte
1910 /// through a text edit (Supplementary Private Use Area-A, U+F0000..=U+F00FF).
1911 const RAW_BYTE_PLACEHOLDER_BASE: u32 = 0xF_0000;
1912
1913 fn is_raw_byte_placeholder(ch: char) -> bool {
1914 (RAW_BYTE_PLACEHOLDER_BASE..=RAW_BYTE_PLACEHOLDER_BASE + 0xFF).contains(&u32::from(ch))
1915 }
1916
1917 /// Decode `bytes` for a text edit without losing any of them: valid UTF-8 is
1918 /// kept as text and each invalid byte becomes a placeholder code point that
1919 /// [`encode_lossless_text`] turns back into the same byte. A file that is not
1920 /// UTF-8 and already uses the placeholder range (or edits that do) cannot be
1921 /// round-tripped, so the edit is refused rather than risk a silent rewrite.
1922 ///
1923 /// The flag is `true` only when placeholders were introduced. A valid UTF-8
1924 /// file keeps its characters as they are, including any in the placeholder
1925 /// range (Nerd Font icons live there), and is written back as plain UTF-8.
1926 fn decode_bytes_losslessly(
1927 bytes: &[u8],
1928 edits: &[ContractEdit],
1929 path: &str,
1930 ) -> Result<(String, bool), ToolError> {
1931 if let Ok(text) = std::str::from_utf8(bytes) {
1932 return Ok((text.to_string(), false));
1933 }
1934 let refuse = || {
1935 ToolError::execution_failed(format!(
1936 "Could not edit file {path}: it is not valid UTF-8 and uses characters Codewhale needs to keep its raw bytes intact. The file was not changed; use File `patch` or a shell tool for this file."
1937 ))
1938 };
1939 if edits.iter().any(|edit| {
1940 edit.old_text.chars().any(is_raw_byte_placeholder)
1941 || edit.new_text.chars().any(is_raw_byte_placeholder)
1942 }) {
1943 return Err(refuse());
1944 }
1945 let mut out = String::with_capacity(bytes.len());
1946 for chunk in bytes.utf8_chunks() {
1947 if chunk.valid().chars().any(is_raw_byte_placeholder) {
1948 return Err(refuse());
1949 }
1950 out.push_str(chunk.valid());
1951 for &byte in chunk.invalid() {
1952 out.push(
1953 char::from_u32(RAW_BYTE_PLACEHOLDER_BASE + u32::from(byte))
1954 .expect("placeholder range holds valid code points"),
1955 );
1956 }
1957 }
1958 Ok((out, true))
1959 }
1960
1961 /// Inverse of [`decode_bytes_losslessly`] for a file it decoded with
1962 /// placeholders. Never call it on text from a valid UTF-8 file.
1963 fn encode_lossless_text(text: &str) -> Vec<u8> {
1964 if !text.chars().any(is_raw_byte_placeholder) {
1965 return text.as_bytes().to_vec();
1966 }
1967 let mut out = Vec::with_capacity(text.len());
1968 let mut buf = [0_u8; 4];
1969 for ch in text.chars() {
1970 if is_raw_byte_placeholder(ch) {
1971 out.push((u32::from(ch) - RAW_BYTE_PLACEHOLDER_BASE) as u8);
1972 } else {
1973 out.extend_from_slice(ch.encode_utf8(&mut buf).as_bytes());
1974 }
1975 }
1976 out
1977 }
1978
1979 // Raw-byte placeholders use four UTF-8 bytes in memory but encode as one.
1980 fn contract_encoded_size(text: &str, has_raw_bytes: bool) -> usize {
1981 if has_raw_bytes {
1982 text.chars()
1983 .map(|ch| {
1984 if is_raw_byte_placeholder(ch) {
1985 1
1986 } else {
1987 ch.len_utf8()
1988 }
1989 })
1990 .sum()
1991 } else {
1992 text.len()
1993 }
1994 }
1995
1996 fn append_contract_text(
1997 result: &mut String,
1998 encoded_size: &mut usize,
1999 text: &str,
2000 has_raw_bytes: bool,
2001 ) -> Result<(), ToolError> {
2002 let next = encoded_size.saturating_add(contract_encoded_size(text, has_raw_bytes));
2003 check_contract_file_size(next)?;
2004 result.push_str(text);
2005 *encoded_size = next;
2006 Ok(())
2007 }
2008
2009 /// The line terminator of each line in `text`, in order (`\r\n`, `\n` or a
2010 /// lone `\r`). The k-th entry ends the k-th line of the LF-normalized text.
2011 fn line_terminators(text: &str) -> Vec<&'static str> {
2012 let bytes = text.as_bytes();
2013 let mut out = Vec::new();
2014 let mut index = 0;
2015 while index < bytes.len() {
2016 match bytes[index] {
2017 b'\r' if bytes.get(index + 1) == Some(&b'\n') => {
2018 out.push("\r\n");
2019 index += 2;
2020 }
2021 b'\r' => {
2022 out.push("\r");
2023 index += 1;
2024 }
2025 b'\n' => {
2026 out.push("\n");
2027 index += 1;
2028 }
2029 _ => index += 1,
2030 }
2031 }
2032 out
2033 }
2034
2035 /// Give every line an edit left alone its original terminator back, and the
2036 /// file's dominant `fallback` terminator to lines the edit wrote (B6). The
2037 /// edit ran on LF-normalized text; restoring one style for the whole file
2038 /// rewrote mixed-ending files on lines nobody touched.
2039 fn restore_line_endings_per_line(
2040 original: &str,
2041 normalized_original: &str,
2042 updated: &str,
2043 fallback: &str,
2044 has_raw_bytes: bool,
2045 ) -> Result<String, ToolError> {
2046 let terminators = line_terminators(original);
2047 if terminators.iter().all(|ending| *ending == fallback) {
2048 let extra = if fallback == "\r\n" {
2049 updated.bytes().filter(|byte| *byte == b'\n').count()
2050 } else {
2051 0
2052 };
2053 check_contract_file_size(
2054 contract_encoded_size(updated, has_raw_bytes).saturating_add(extra),
2055 )?;
2056 return Ok(restore_contract_line_endings(updated, fallback));
2057 }
2058 let diff = similar::TextDiff::configure()
2059 .timeout(std::time::Duration::from_secs(1))
2060 .diff_lines(normalized_original, updated);
2061 let mut out = String::new();
2062 let mut encoded_size = 0;
2063 for change in diff.iter_all_changes() {
2064 let ending = match change.tag() {
2065 similar::ChangeTag::Delete => continue,
2066 similar::ChangeTag::Equal => change
2067 .old_index()
2068 .and_then(|index| terminators.get(index).copied())
2069 .unwrap_or(fallback),
2070 similar::ChangeTag::Insert => fallback,
2071 };
2072 let line = change.value();
2073 match line.strip_suffix('\n') {
2074 Some(body) => {
2075 append_contract_text(&mut out, &mut encoded_size, body, has_raw_bytes)?;
2076 append_contract_text(&mut out, &mut encoded_size, ending, has_raw_bytes)?;
2077 }
2078 None => append_contract_text(&mut out, &mut encoded_size, line, has_raw_bytes)?,
2079 }
2080 }
2081 Ok(out)
2082 }
2083
2084 /// Rewrite `content` to match the line-ending style of an existing file's
2085 /// `prior` content, so a full-file overwrite (`write_file` / contract `write`)
2086 /// does not silently flip a CRLF (Windows) file to LF — the same policy
2087 /// `edit_file` applies. A brand-new file (no prior content) is returned
2088 /// verbatim: there is no style to preserve.
2089 fn preserve_prior_line_endings(content: &str, prior: &str) -> String {
2090 if prior.is_empty() {
2091 return content.to_string();
2092 }
2093 restore_contract_line_endings(
2094 &normalize_contract_line_endings(content),
2095 contract_line_ending(prior),
2096 )
2097 }
2098
2099 /// Fallback matching view used only after a literal match fails. It follows
2100 /// The small-contract normalization categories while leaving the public schema as
2101 /// exact-text replacement rather than teaching a second edit mode.
2102 fn normalize_contract_fuzzy(text: &str) -> String {
2103 let compatible = text.nfkc().collect::<String>();
2104 let mut trimmed = String::with_capacity(compatible.len());
2105 for (index, line) in compatible.split('\n').enumerate() {
2106 if index > 0 {
2107 trimmed.push('\n');
2108 }
2109 trimmed.push_str(line.trim_end());
2110 }
2111 trimmed
2112 .chars()
2113 .map(|ch| match ch {
2114 '\u{2018}' | '\u{2019}' | '\u{201A}' | '\u{201B}' => '\'',
2115 '\u{201C}' | '\u{201D}' | '\u{201E}' | '\u{201F}' => '"',
2116 '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2014}' | '\u{2015}'
2117 | '\u{2212}' => '-',
2118 '\u{00A0}' | '\u{2002}'..='\u{200A}' | '\u{202F}' | '\u{205F}' | '\u{3000}' => ' ',
2119 other => other,
2120 })
2121 .collect()
2122 }
2123
2124 fn text_matches(haystack: &str, needle: &str) -> Vec<(usize, usize)> {
2125 if needle.is_empty() {
2126 return Vec::new();
2127 }
2128 haystack
2129 .match_indices(needle)
2130 .map(|(start, matched)| (start, start + matched.len()))
2131 .collect()
2132 }
2133
2134 fn contract_edit_not_found(path: &str, index: usize, total: usize) -> ToolError {
2135 if total == 1 {
2136 ToolError::execution_failed(format!(
2137 "Could not find the exact text in {path}. The old text must match exactly including all whitespace and newlines."
2138 ))
2139 } else {
2140 ToolError::execution_failed(format!(
2141 "Could not find edits[{index}] in {path}. The oldText must match exactly including all whitespace and newlines."
2142 ))
2143 }
2144 }
2145
2146 fn contract_edit_duplicate(path: &str, index: usize, total: usize, matches: usize) -> ToolError {
2147 if total == 1 {
2148 ToolError::execution_failed(format!(
2149 "Found {matches} occurrences of the text in {path}. The text must be unique. Please provide more context to make it unique."
2150 ))
2151 } else {
2152 ToolError::execution_failed(format!(
2153 "Found {matches} occurrences of edits[{index}] in {path}. Each oldText must be unique. Please provide more context to make it unique."
2154 ))
2155 }
2156 }
2157
2158 fn prepare_contract_edit_input(mut input: Value) -> Result<Value, ToolError> {
2159 for key in ["oldText", "newText"] {
2160 if let Some(text) = input.get(key).and_then(Value::as_str) {
2161 check_contract_file_size(text.len())?;
2162 }
2163 }
2164 if let Some(encoded) = input.get("edits").and_then(Value::as_str) {
2165 check_contract_file_size(encoded.len())?;
2166 }
2167 let object = input
2168 .as_object_mut()
2169 .ok_or_else(|| ToolError::invalid_input("edit input must be an object"))?;
2170 if let Some(Value::String(encoded)) = object.get("edits")
2171 && let Ok(decoded) = serde_json::from_str::<Value>(encoded)
2172 && decoded.is_array()
2173 {
2174 object.insert("edits".to_string(), decoded);
2175 }
2176
2177 let legacy_old = object
2178 .get("oldText")
2179 .and_then(Value::as_str)
2180 .map(str::to_string);
2181 let legacy_new = object
2182 .get("newText")
2183 .and_then(Value::as_str)
2184 .map(str::to_string);
2185 if let (Some(old_text), Some(new_text)) = (legacy_old, legacy_new) {
2186 let legacy = json!({"oldText": old_text, "newText": new_text});
2187 let mut edits = match object.remove("edits") {
2188 Some(Value::Array(edits)) => edits,
2189 _ => Vec::new(),
2190 };
2191 edits.push(legacy);
2192 object.insert("edits".to_string(), Value::Array(edits));
2193 object.remove("oldText");
2194 object.remove("newText");
2195 }
2196 Ok(input)
2197 }
2198
2199 fn parse_contract_edits(input: &Value) -> Result<Vec<ContractEdit>, ToolError> {
2200 let raw = input
2201 .get("edits")
2202 .and_then(Value::as_array)
2203 .ok_or_else(|| ToolError::invalid_input("edits must be an array"))?;
2204 if raw.is_empty() {
2205 return Err(ToolError::invalid_input(
2206 "edit requires at least one replacement in edits",
2207 ));
2208 }
2209 let mut replacement_bytes = 0usize;
2210 raw.iter()
2211 .enumerate()
2212 .map(|(index, edit)| {
2213 reject_primitive_unknown(edit, &format!("edits[{index}]"), &["oldText", "newText"])?;
2214 let old_text = required_str(edit, "oldText")?;
2215 let new_text = required_str(edit, "newText")?;
2216 check_contract_file_size(old_text.len())?;
2217 check_contract_file_size(new_text.len())?;
2218 // All non-overlapping replacements survive into the output. Count
2219 // normalized bytes before copying another large replacement.
2220 replacement_bytes = replacement_bytes
2221 .saturating_add(new_text.len() - new_text.match_indices("\r\n").count());
2222 check_contract_file_size(replacement_bytes)?;
2223 if old_text.is_empty() {
2224 return Err(ToolError::invalid_input(format!(
2225 "edits[{index}].oldText must not be empty"
2226 )));
2227 }
2228 Ok(ContractEdit {
2229 index,
2230 old_text: normalize_contract_line_endings(old_text),
2231 new_text: normalize_contract_line_endings(new_text),
2232 })
2233 })
2234 .collect()
2235 }
2236
2237 fn apply_resolved_edits(
2238 base: &str,
2239 edits: &[ResolvedContractEdit],
2240 offset: usize,
2241 has_raw_bytes: bool,
2242 ) -> Result<String, ToolError> {
2243 let mut encoded_size = contract_encoded_size(base, has_raw_bytes);
2244 check_contract_file_size(encoded_size)?;
2245 let mut updated = base.to_string();
2246 for edit in edits.iter().rev() {
2247 let range = edit.start.saturating_sub(offset)..edit.end.saturating_sub(offset);
2248 let removed = &updated[range.clone()];
2249 let projected_string_size = updated
2250 .len()
2251 .saturating_sub(removed.len())
2252 .saturating_add(edit.replacement.len());
2253 encoded_size = encoded_size
2254 .saturating_sub(contract_encoded_size(removed, has_raw_bytes))
2255 .saturating_add(contract_encoded_size(&edit.replacement, has_raw_bytes));
2256 // Reject each intermediate replacement before String::replace_range
2257 // allocates; raw-byte placeholders are at most four bytes in memory.
2258 check_contract_file_size(encoded_size)?;
2259 if projected_string_size > CONTRACT_FILE_MAX_BYTES * if has_raw_bytes { 4 } else { 1 } {
2260 check_contract_file_size(CONTRACT_FILE_MAX_BYTES + 1)?;
2261 }
2262 updated.replace_range(range, &edit.replacement);
2263 }
2264 Ok(updated)
2265 }
2266
2267 fn lines_with_endings(text: &str) -> Vec<&str> {
2268 if text.is_empty() {
2269 Vec::new()
2270 } else {
2271 text.split_inclusive('\n').collect()
2272 }
2273 }
2274
2275 fn line_spans(text: &str) -> Vec<(usize, usize)> {
2276 let mut offset = 0usize;
2277 lines_with_endings(text)
2278 .into_iter()
2279 .map(|line| {
2280 let span = (offset, offset + line.len());
2281 offset = span.1;
2282 span
2283 })
2284 .collect()
2285 }
2286
2287 fn touched_line_range(
2288 spans: &[(usize, usize)],
2289 edit: &ResolvedContractEdit,
2290 ) -> Result<(usize, usize), ToolError> {
2291 let start = spans
2292 .iter()
2293 .position(|(line_start, line_end)| edit.start >= *line_start && edit.start < *line_end)
2294 .ok_or_else(|| ToolError::execution_failed("edit match fell outside the file"))?;
2295 let mut end = start;
2296 while end < spans.len() && spans[end].1 < edit.end {
2297 end += 1;
2298 }
2299 if end >= spans.len() {
2300 return Err(ToolError::execution_failed(
2301 "edit match fell outside the file",
2302 ));
2303 }
2304 Ok((start, end + 1))
2305 }
2306
2307 fn apply_fuzzy_edits_preserving_other_lines(
2308 original: &str,
2309 normalized: &str,
2310 edits: &[ResolvedContractEdit],
2311 has_raw_bytes: bool,
2312 ) -> Result<String, ToolError> {
2313 let original_lines = lines_with_endings(original);
2314 let spans = line_spans(normalized);
2315 if original_lines.len() != spans.len() {
2316 return Err(ToolError::execution_failed(
2317 "fuzzy edit could not preserve the file's untouched lines",
2318 ));
2319 }
2320
2321 #[derive(Debug)]
2322 struct Group {
2323 start_line: usize,
2324 end_line: usize,
2325 edits: Vec<ResolvedContractEdit>,
2326 }
2327
2328 let mut groups: Vec<Group> = Vec::new();
2329 for edit in edits {
2330 let (start_line, end_line) = touched_line_range(&spans, edit)?;
2331 if let Some(group) = groups.last_mut()
2332 && start_line < group.end_line
2333 {
2334 group.end_line = group.end_line.max(end_line);
2335 group.edits.push(edit.clone());
2336 } else {
2337 groups.push(Group {
2338 start_line,
2339 end_line,
2340 edits: vec![edit.clone()],
2341 });
2342 }
2343 }
2344
2345 let mut result = String::new();
2346 let mut encoded_size = 0;
2347 let mut original_line = 0usize;
2348 for group in groups {
2349 for line in &original_lines[original_line..group.start_line] {
2350 append_contract_text(&mut result, &mut encoded_size, line, has_raw_bytes)?;
2351 }
2352 let group_start = spans[group.start_line].0;
2353 let group_end = spans[group.end_line - 1].1;
2354 let replacement = apply_resolved_edits(
2355 &normalized[group_start..group_end],
2356 &group.edits,
2357 group_start,
2358 has_raw_bytes,
2359 )?;
2360 append_contract_text(&mut result, &mut encoded_size, &replacement, has_raw_bytes)?;
2361 original_line = group.end_line;
2362 }
2363 for line in &original_lines[original_line..] {
2364 append_contract_text(&mut result, &mut encoded_size, line, has_raw_bytes)?;
2365 }
2366 Ok(result)
2367 }
2368
2369 fn apply_contract_edits(
2370 base: &str,
2371 edits: &[ContractEdit],
2372 path: &str,
2373 has_raw_bytes: bool,
2374 ) -> Result<String, ToolError> {
2375 let fuzzy_base = normalize_contract_fuzzy(base);
2376 let initial = edits
2377 .iter()
2378 .map(|edit| {
2379 if base.contains(&edit.old_text) {
2380 Ok(false)
2381 } else if fuzzy_base.contains(&normalize_contract_fuzzy(&edit.old_text)) {
2382 Ok(true)
2383 } else {
2384 Err(contract_edit_not_found(path, edit.index, edits.len()))
2385 }
2386 })
2387 .collect::<Result<Vec<_>, _>>()?;
2388 let use_fuzzy = initial.into_iter().any(|used| used);
2389 let replacement_base = if use_fuzzy { fuzzy_base.as_str() } else { base };
2390
2391 let mut resolved = Vec::with_capacity(edits.len());
2392 for edit in edits {
2393 let exact = text_matches(replacement_base, &edit.old_text);
2394 let fuzzy_old = normalize_contract_fuzzy(&edit.old_text);
2395 let fuzzy_occurrences = text_matches(&fuzzy_base, &fuzzy_old).len();
2396 if fuzzy_occurrences > 1 {
2397 return Err(contract_edit_duplicate(
2398 path,
2399 edit.index,
2400 edits.len(),
2401 fuzzy_occurrences,
2402 ));
2403 }
2404 let matches = if exact.is_empty() {
2405 text_matches(replacement_base, &fuzzy_old)
2406 } else {
2407 exact
2408 };
2409 let Some(&(start, end)) = matches.first() else {
2410 return Err(contract_edit_not_found(path, edit.index, edits.len()));
2411 };
2412 if matches.len() > 1 {
2413 return Err(contract_edit_duplicate(
2414 path,
2415 edit.index,
2416 edits.len(),
2417 matches.len(),
2418 ));
2419 }
2420 resolved.push(ResolvedContractEdit {
2421 index: edit.index,
2422 start,
2423 end,
2424 replacement: edit.new_text.clone(),
2425 });
2426 }
2427
2428 resolved.sort_by_key(|edit| (edit.start, edit.end));
2429 for pair in resolved.windows(2) {
2430 if pair[0].end > pair[1].start {
2431 return Err(ToolError::execution_failed(format!(
2432 "edits[{}] and edits[{}] overlap in {path}; merge them or target separate regions",
2433 pair[0].index, pair[1].index
2434 )));
2435 }
2436 }
2437
2438 let updated = if use_fuzzy {
2439 apply_fuzzy_edits_preserving_other_lines(base, replacement_base, &resolved, has_raw_bytes)?
2440 } else {
2441 apply_resolved_edits(replacement_base, &resolved, 0, has_raw_bytes)?
2442 };
2443 if updated == base {
2444 return Err(ToolError::execution_failed(format!(
2445 "No changes made to {path}; the replacement produced identical content."
2446 )));
2447 }
2448 Ok(updated)
2449 }
2450
2451 impl EditFileTool {
2452 pub(super) async fn execute_contract_edits(
2453 input: Value,
2454 context: &ToolContext,
2455 ) -> Result<ToolResult, ToolError> {
2456 let input = prepare_contract_edit_input(input)?;
2457 reject_primitive_unknown(&input, "edit", &["path", "edits"])?;
2458 let path_str = required_str(&input, "path")?;
2459 let edits = parse_contract_edits(&input)?;
2460 let file_path = context.resolve_path(path_str)?;
2461 let mutation_guard = acquire_file_mutation(&file_path, context).await?;
2462 check_file_operation_cancelled(context)?;
2463
2464 let raw_bytes = load_contract_source(&file_path, true, context)
2465 .await?
2466 .ok_or_else(|| {
2467 ToolError::execution_failed(format!(
2468 "Could not edit file {path_str}: file not found"
2469 ))
2470 })?;
2471 check_file_operation_cancelled(context)?;
2472 // Bytes that are not UTF-8 ride through the edit as placeholders and
2473 // are written back unchanged (B6), instead of becoming U+FFFD.
2474 let (raw, has_raw_bytes) = decode_bytes_losslessly(&raw_bytes, &edits, path_str)?;
2475 let (bom, without_bom) = raw
2476 .strip_prefix('\u{FEFF}')
2477 .map_or(("", raw.as_str()), |text| ("\u{FEFF}", text));
2478 let ending = contract_line_ending(without_bom);
2479 let normalized = normalize_contract_line_endings(without_bom);
2480 let updated = apply_contract_edits(&normalized, &edits, path_str, has_raw_bytes)?;
2481 check_file_operation_cancelled(context)?;
2482 let restored = restore_line_endings_per_line(
2483 without_bom,
2484 &normalized,
2485 &updated,
2486 ending,
2487 has_raw_bytes,
2488 )?;
2489 check_contract_file_size(
2490 bom.len()
2491 .saturating_add(contract_encoded_size(&restored, has_raw_bytes)),
2492 )?;
2493 let mut final_content = format!("{bom}{restored}");
2494 guard_edit(&file_path, path_str, Some(&raw), &final_content)?;
2495 if let Some(normalized) = normalize_edit(&file_path, &raw, &final_content).await {
2496 final_content = normalized;
2497 }
2498
2499 check_contract_file_size(contract_encoded_size(&final_content, has_raw_bytes))?;
2500 check_file_operation_cancelled(context)?;
2501 let bytes = if has_raw_bytes {
2502 encode_lossless_text(&final_content)
2503 } else {
2504 final_content.clone().into_bytes()
2505 };
2506 check_contract_file_size(bytes.len())?;
2507 check_file_operation_cancelled(context)?;
2508 // The mutation worker owns completion after atomic replacement starts.
2509 run_blocking_write_atomic(&file_path, bytes.clone()).await?;
2510 context.note_file_read(&file_path);
2511 drop(mutation_guard);
2512
2513 Ok(contract_mutation_result(
2514 context,
2515 &file_path,
2516 path_str,
2517 &raw,
2518 &final_content,
2519 &bytes,
2520 "updated",
2521 format!(
2522 "Successfully replaced {} block(s) in {path_str}.",
2523 edits.len()
2524 ),
2525 )
2526 .await)
2527 }
2528 }
2529
2530 #[async_trait]
2531 impl ToolSpec for EditFileTool {
2532 fn name(&self) -> &'static str {
2533 "edit_file"
2534 }
2535
2536 fn model_visible(&self) -> bool {
2537 false
2538 }
2539
2540 fn description(&self) -> &'static str {
2541 "Replace text in a single file via exact search/replace after the file has been read with File `read` in this session. Use this instead of `sed -i` in `Bash` for one unambiguous in-place edit. `search` must match exactly one location by default; when no exact match is found the tool retries with leading-whitespace-tolerant fuzzy matching automatically. Returns a compact unified diff, not the full file. Pass `expected_hash` (the `content_hash` from that `read`) to have the edit refused, with the file untouched, if it changed in between. For structural, multi-block, or cross-file changes, use File `patch` or `write` instead."
2542 }
2543
2544 fn input_schema(&self) -> Value {
2545 json!({
2546 "type": "object",
2547 "properties": {
2548 "path": {
2549 "type": "string",
2550 "description": "Path to the file. Alias: `file_path`"
2551 },
2552 "search": {
2553 "type": "string",
2554 "description": "Exact text to search for, including whitespace, indentation, and newlines. Aliases: `old_string`, `old_str`, `oldText`"
2555 },
2556 "replace": {
2557 "type": "string",
2558 "description": "Text to replace with. Aliases: `new_string`, `new_str`, `newText`"
2559 },
2560 "expected_hash": {
2561 "type": "string",
2562 "description": EXPECTED_HASH_DESCRIPTION
2563 }
2564 },
2565 "required": ["path", "search", "replace"]
2566 })
2567 }
2568
2569 fn capabilities(&self) -> Vec<ToolCapability> {
2570 vec![
2571 ToolCapability::WritesFiles,
2572 ToolCapability::Sandboxable,
2573 ToolCapability::RequiresApproval,
2574 ]
2575 }
2576
2577 fn approval_requirement(&self) -> ApprovalRequirement {
2578 ApprovalRequirement::Suggest
2579 }
2580
2581 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
2582 // Translate known cross-harness spellings (`old_string`/`new_string`,
2583 // `old_str`/`new_str`, …) onto `search`/`replace` first, then reject
2584 // whatever is left that we do not implement. #5209 required that a
2585 // mis-named edit never produce a success-shaped receipt for a file
2586 // that did not change; performing the edit the model unambiguously
2587 // asked for satisfies that more directly than refusing it did.
2588 let mut input = input;
2589 apply_param_aliases(&mut input, PATH_ALIASES, "File edit")?;
2590 apply_param_aliases(&mut input, EDIT_ALIASES, "File edit")?;
2591 EDIT_PARAMS.reject_unknown(&input)?;
2592
2593 let path_str = required_str(&input, "path")?;
2594 let search = required_str(&input, "search")?;
2595 let replace = required_str(&input, "replace")?;
2596 let expected_hash = optional_str(&input, "expected_hash")?;
2597
2598 if search == replace {
2599 // #5003 — long-text edits repeatedly failed here because the model
2600 // generated a `replace` identical to `search`. A bare "no change"
2601 // message gave no hint of the root cause, so the model retried the
2602 // same broken call. Spell out the failure and the recovery path.
2603 let char_count = search.chars().count();
2604 let line_count = search.lines().count();
2605 return Err(ToolError::invalid_input(format!(
2606 "search and replace are identical ({char_count} chars, {line_count} lines), so no change is possible. This usually means `replace` was copied verbatim from `search` instead of carrying the intended edits. Recovery: re-read the file with File action=\"read\", then retry with a `replace` that is genuinely different from `search`; for large multi-line rewrites prefer apply_patch with a unified diff."
2607 )));
2608 }
2609 if search.is_empty() {
2610 return Err(ToolError::invalid_input("search must not be empty"));
2611 }
2612 if let Some(reason) = edit_payload_looks_corrupted(search, replace) {
2613 return Err(ToolError::invalid_input(format!(
2614 "edit_file refused corrupted payload: {reason}. Recovery: re-read the file and retry with a complete replace (or use apply_patch for brace-heavy multi-line edits)."
2615 )));
2616 }
2617
2618 let file_path = context.resolve_path(path_str)?;
2619 context.require_fresh_file_read(&file_path, path_str)?;
2620
2621 let contents = tokio::fs::read_to_string(&file_path).await.map_err(|e| {
2622 ToolError::execution_failed(format!("Failed to read {}: {}", file_path.display(), e))
2623 })?;
2624
2625 // Content-hash guard (#3979). Verified against `contents` — the exact
2626 // snapshot every match below is computed from and that the write is
2627 // derived from — and before any search/replace work, so a stale hash
2628 // can never reach the filesystem regardless of what the search would
2629 // have matched.
2630 verify_expected_hash(expected_hash, contents.as_bytes(), "edit", path_str)?;
2631
2632 // Models provide LF newlines even when the file on disk uses CRLF.
2633 // Match in a newline-normalized view, while retaining the sparse
2634 // positions where CR bytes were removed so only the original span is
2635 // replaced and the rest of the file stays byte-for-byte untouched.
2636 let (normalized_contents, crlf_positions) = normalize_crlf_with_positions(&contents);
2637 let normalized_search = normalize_crlf(search);
2638 let mut exact_ranges = normalized_contents
2639 .match_indices(normalized_search.as_ref())
2640 .map(|(start, matched)| (start, start + matched.len()));
2641 let first_exact_match = exact_ranges
2642 .next()
2643 .map(|range| map_normalized_range(range, crlf_positions.as_deref()));
2644 let exact_count = usize::from(first_exact_match.is_some()) + exact_ranges.count();
2645
2646 let ((match_start, match_end), fuzz_kind) = if exact_count == 0 {
2647 // First fallback: tolerate indentation differences.
2648 let indent_matches = map_normalized_ranges(
2649 leading_whitespace_fuzzy_matches(
2650 normalized_contents.as_ref(),
2651 normalized_search.as_ref(),
2652 ),
2653 crlf_positions.as_deref(),
2654 );
2655 match indent_matches.as_slice() {
2656 [(start, end)] => ((*start, *end), Some("indentation")),
2657 [] => {
2658 // Second fallback: tolerate typographic-punctuation
2659 // drift (smart quotes, em-dashes, NBSP). Picks up the
2660 // copy-paste failure mode where a browser/chat client
2661 // silently substituted Unicode punctuation in for the
2662 // ASCII the file actually contains.
2663 let punct_matches = map_normalized_ranges(
2664 punctuation_normalized_matches(
2665 normalized_contents.as_ref(),
2666 normalized_search.as_ref(),
2667 ),
2668 crlf_positions.as_deref(),
2669 );
2670 match punct_matches.as_slice() {
2671 [] => {
2672 // #5003 — the model could not tell why its search
2673 // missed; show the first lines of the search text
2674 // so it can compare against the file's contents.
2675 return Err(ToolError::execution_failed(format!(
2676 "Search string not found in {}. The search text starts with:\n{}\n{}Recovery: retry with the search copied from the lines above, or call File with action=\"read\" path=\"{path_str}\" to inspect the current contents.",
2677 file_path.display(),
2678 preview_search_for_error(search),
2679 nearest_match_hint(
2680 normalized_contents.as_ref(),
2681 normalized_search.as_ref(),
2682 crlf_positions.is_some(),
2683 search.contains('\r'),
2684 ),
2685 )));
2686 }
2687 [(start, end)] => ((*start, *end), Some("punctuation")),
2688 _ => {
2689 return Err(ToolError::execution_failed(format!(
2690 "File `edit` search is non-unique after punctuation normalization: matched {} locations in {}. Recovery: call File with action=\"read\" path=\"{path_str}\" and retry with surrounding lines that make the search unique.",
2691 punct_matches.len(),
2692 file_path.display()
2693 )));
2694 }
2695 }
2696 }
2697 _ => {
2698 return Err(ToolError::execution_failed(format!(
2699 "File `edit` search is non-unique after indentation normalization: matched {} locations in {}. Recovery: call File with action=\"read\" path=\"{path_str}\" and retry with surrounding lines that make the search unique.",
2700 indent_matches.len(),
2701 file_path.display()
2702 )));
2703 }
2704 }
2705 } else if exact_count > 1 {
2706 return Err(ToolError::execution_failed(format!(
2707 "File `edit` search is non-unique: matched {} locations in {}. \
2708 Recovery: call File with action=\"read\" path=\"{path_str}\" and retry with surrounding lines that make the search unique.",
2709 exact_count,
2710 file_path.display()
2711 )));
2712 } else {
2713 let Some((start, end)) = first_exact_match else {
2714 return Err(ToolError::execution_failed(
2715 "edit_file internal range accounting failed — refusing write",
2716 ));
2717 };
2718 let fuzz_kind = (&contents[start..end] != search).then_some("line endings");
2719 ((start, end), fuzz_kind)
2720 };
2721
2722 let effective_replace =
2723 normalize_replacement_line_endings(replace, crlf_positions.is_some());
2724 let mut updated = contents.clone();
2725 updated.replace_range(match_start..match_end, &effective_replace);
2726 if updated == contents {
2727 return Err(ToolError::invalid_input(
2728 "search and replace resolve to identical file contents after line-ending normalization, no change intended",
2729 ));
2730 }
2731
2732 if let Some(reason) = invalid_preprocessor_edit(&file_path, &contents, &updated) {
2733 return Err(ToolError::invalid_input(format!(
2734 "edit_file refused corrupted payload: {reason}. Recovery: re-read the file and retry with a complete replace (or use apply_patch for brace-heavy multi-line edits)."
2735 )));
2736 }
2737
2738 // Fidelity: the intended replace text must appear in the updated buffer
2739 // (empty replace is a valid deletion). Catches host/tool bridges that
2740 // claim success after mangling the payload.
2741 if !effective_replace.is_empty() && !updated.contains(&effective_replace) {
2742 return Err(ToolError::execution_failed(
2743 "edit_file internal fidelity check failed: replace text missing from updated buffer — refusing write",
2744 ));
2745 }
2746
2747 guard_edit(&file_path, path_str, Some(&contents), &updated)?;
2748
2749 // #6205 — normalize after the syntax gate so the next turn's anchors
2750 // match the bytes on disk rather than the text the model emitted.
2751 let normalized_formatting = match normalize_edit(&file_path, &contents, &updated).await {
2752 Some(normalized) => {
2753 updated = normalized;
2754 true
2755 }
2756 None => false,
2757 };
2758
2759 run_blocking_write_atomic(&file_path, updated.clone().into_bytes()).await?;
2760
2761 // #5209 — never emit a success receipt unless the on-disk write
2762 // actually applied. A fabricated "Replaced 1 occurrence" + diff is
2763 // worse than a hard error: models trust it and re-edit the same
2764 // span 3–5× before noticing nothing changed.
2765 let on_disk = tokio::fs::read_to_string(&file_path).await.map_err(|e| {
2766 ToolError::execution_failed(format!(
2767 "Failed to verify write to {}: {}",
2768 file_path.display(),
2769 e
2770 ))
2771 })?;
2772 if on_disk != updated {
2773 return Err(ToolError::execution_failed(format!(
2774 "edit_file write verification failed for {}: on-disk contents do not match the applied edit — refusing success receipt",
2775 file_path.display()
2776 )));
2777 }
2778
2779 context.note_file_read(&file_path);
2780
2781 let display = file_path.display().to_string();
2782 let diff = make_unified_diff(&display, &contents, &updated);
2783 let fuzz_note = match fuzz_kind {
2784 Some("indentation") => " (fuzzy indentation match)",
2785 Some("punctuation") => {
2786 " (fuzzy punctuation match — typographic quotes/dashes normalized)"
2787 }
2788 Some("line endings") => " (CRLF/LF-normalized match)",
2789 Some(other) => other,
2790 None => "",
2791 };
2792 let format_note = if normalized_formatting {
2793 NORMALIZED_NOTE
2794 } else {
2795 ""
2796 };
2797 let summary = format!("Replaced 1 occurrence in {display}{fuzz_note}{format_note}");
2798 let body = if diff.is_empty() {
2799 format!("{summary}\n(no textual changes)")
2800 } else {
2801 format!("{diff}\n{summary}")
2802 };
2803
2804 // Append LSP diagnostics for the edited file when enabled (#428).
2805 let diag_block = lsp_diagnostics_for_paths(context, &[file_path]).await;
2806 let full_body = if diag_block.is_empty() {
2807 body
2808 } else {
2809 format!("{body}\n{diag_block}")
2810 };
2811
2812 // The structured receipt uses the requested workspace path instead of
2813 // the resolved host path retained by the legacy model-facing body.
2814 let receipt_diff = make_unified_diff(path_str, &contents, &updated);
2815 Ok(ToolResult::success(full_body).with_metadata(json!({
2816 "event": "file.mutation",
2817 "mutation": {
2818 "diff": receipt_diff,
2819 "files": [mutation_file_entry(path_str, "updated", Some(updated.as_bytes()))],
2820 "renames": []
2821 }
2822 })))
2823 }
2824 }
2825
2826 /// Detect catastrophic argument corruption of brace-structured edits.
2827 ///
2828 /// Models (and some host XML/JSON bridges) occasionally deliver a `replace`
2829 /// payload where a multi-line `{ ... }` block collapsed to empty `[]` or `{}`
2830 /// while `search` still contains the full structured original. Writing that
2831 /// would brick Rust match arms / JSON objects. Fail closed with recovery text
2832 /// instead of applying the mangled payload (dogfood 2026-07-24).
2833 ///
2834 /// Unbalanced-to-unbalanced edits with the **same** brace/bracket delta are
2835 /// legitimate (e.g. adding `});` inside a nested fragment). Only a *change*
2836 /// in balance is treated as truncation/mangling. Empty-bracket collapse and
2837 /// extreme-shrinkage guards remain.
2838 fn edit_payload_looks_corrupted(search: &str, replace: &str) -> Option<&'static str> {
2839 let search_curly_open = search.matches('{').count();
2840 let search_curly_close = search.matches('}').count();
2841 let replace_curly_open = replace.matches('{').count();
2842 let replace_curly_close = replace.matches('}').count();
2843 let search_square_open = search.matches('[').count();
2844 let search_square_close = search.matches(']').count();
2845 let replace_square_open = replace.matches('[').count();
2846 let replace_square_close = replace.matches(']').count();
2847
2848 let search_curly_delta = search_curly_open as i32 - search_curly_close as i32;
2849 let replace_curly_delta = replace_curly_open as i32 - replace_curly_close as i32;
2850 let search_square_delta = search_square_open as i32 - search_square_close as i32;
2851 let replace_square_delta = replace_square_open as i32 - replace_square_close as i32;
2852
2853 // Same delta on both sides (including both unbalanced the same way) is
2854 // normal for fragment edits. Divergent deltas usually mean truncation.
2855 if search_curly_delta != replace_curly_delta {
2856 return Some(
2857 "search/replace change `{`/`}` brace balance — the tool-call arguments were likely truncated or mangled before apply",
2858 );
2859 }
2860 if search_square_delta != replace_square_delta {
2861 return Some(
2862 "search/replace change `[`/`]` bracket balance — the tool-call arguments were likely truncated or mangled before apply",
2863 );
2864 }
2865
2866 // Dogfood 2026-07-24: multi-line Rust `{ ... }` search collapsed into an
2867 // empty `[ ... ]` placeholder (host/XML arg bridge ate the brace body).
2868 // Count non-whitespace, non-bracket payload chars; a near-empty bracket
2869 // husk with a tiny tail like `=> {},` is the signature of that failure.
2870 if search_curly_open >= 1 && replace_square_open >= 1 {
2871 let significant = replace
2872 .chars()
2873 .filter(|c| !c.is_whitespace() && *c != '[' && *c != ']')
2874 .count();
2875 if significant <= 12 {
2876 return Some(
2877 "replace collapsed a brace-structured search block into an empty/placeholder bracket span — refusing to brick the file; re-send the full replace text (prefer apply_patch for multi-line match arms)",
2878 );
2879 }
2880 }
2881
2882 // Extreme shrinkage with lost braces (e.g. 200-char match arm -> tiny stub).
2883 // Balanced-to-balanced nesting changes that shrink hard still look like
2884 // mangling; keep this guard even when deltas match.
2885 if search.len() >= 80
2886 && replace.len() * 8 < search.len()
2887 && search_curly_open >= 1
2888 && replace_curly_open < search_curly_open
2889 {
2890 return Some(
2891 "replace is drastically shorter than search and lost brace structure — likely argument mangling; refuse apply",
2892 );
2893 }
2894
2895 None
2896 }
2897
2898 const PREPROCESSOR_CONDITIONAL_ERROR: &str = "replace would change the C/C++ preprocessor conditional balance (#if/#ifdef/#ifndef vs #endif) — the search or replace text is missing a matching directive; copy the complete block including both its opening and closing directives";
2899
2900 #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
2901 struct PreprocessorConditionalDebt {
2902 orphaned_closes: usize,
2903 unclosed_opens: usize,
2904 }
2905
2906 impl PreprocessorConditionalDebt {
2907 fn total(self) -> usize {
2908 self.orphaned_closes + self.unclosed_opens
2909 }
2910 }
2911
2912 /// Reject an edit only when it introduces new conditional-structure damage in
2913 /// a file whose extension identifies it as C-family source. The whole file is
2914 /// checked before and after the edit: complete block insertion/removal is safe,
2915 /// while an orphaned opener or closer increases the structural debt. Existing
2916 /// debt may be preserved or reduced so this guard never prevents a repair.
2917 fn invalid_preprocessor_edit(path: &Path, before: &str, after: &str) -> Option<&'static str> {
2918 if !is_c_family_source(path) {
2919 return None;
2920 }
2921
2922 let before_debt = preprocessor_conditional_debt(before);
2923 let after_debt = preprocessor_conditional_debt(after);
2924 let safe = after_debt == before_debt
2925 || after_debt.total() == 0
2926 || after_debt.total() < before_debt.total();
2927
2928 (!safe).then_some(PREPROCESSOR_CONDITIONAL_ERROR)
2929 }
2930
2931 fn is_c_family_source(path: &Path) -> bool {
2932 const EXTENSIONS: &[&str] = &[
2933 "c", "cc", "cp", "cpp", "cxx", "h", "h++", "hh", "hpp", "hxx", "inl", "ipp", "ixx", "m",
2934 "mm", "tpp", "cu", "cuh", "cppm",
2935 ];
2936
2937 path.extension()
2938 .and_then(|extension| extension.to_str())
2939 .is_some_and(|extension| {
2940 EXTENSIONS
2941 .iter()
2942 .any(|candidate| extension.eq_ignore_ascii_case(candidate))
2943 })
2944 }
2945
2946 /// Measure unmatched preprocessor conditionals across an entire source file.
2947 /// Tracking nesting (instead of comparing span-level tuple counts) also catches
2948 /// an `#endif` moved before its opener. Whitespace between `#` and the directive
2949 /// name is accepted, as it is by C preprocessors.
2950 fn preprocessor_conditional_debt(text: &str) -> PreprocessorConditionalDebt {
2951 let mut depth = 0usize;
2952 let mut orphaned_closes = 0usize;
2953
2954 for line in text.lines() {
2955 match preprocessor_directive(line) {
2956 Some("if" | "ifdef" | "ifndef") => depth += 1,
2957 Some("endif") if depth == 0 => orphaned_closes += 1,
2958 Some("endif") => depth -= 1,
2959 _ => {}
2960 }
2961 }
2962
2963 PreprocessorConditionalDebt {
2964 orphaned_closes,
2965 unclosed_opens: depth,
2966 }
2967 }
2968
2969 fn preprocessor_directive(line: &str) -> Option<&str> {
2970 let rest = line.trim_start().strip_prefix('#')?.trim_start();
2971 let name_end = rest
2972 .find(|character: char| !character.is_ascii_alphabetic())
2973 .unwrap_or(rest.len());
2974 (name_end > 0).then_some(&rest[..name_end])
2975 }
2976
2977 /// Build a short, line-truncated preview of a (possibly very long) search
2978 /// payload for error messages, so the model can compare what it searched for
2979 /// against the file's actual contents without the error message ballooning.
2980 /// The file region most like a search that did not match (#6542), with
2981 /// 1-based line numbers and a note on whitespace / line-ending differences,
2982 /// so the next edit can copy the real text instead of re-reading the file.
2983 ///
2984 /// Known limitation: candidates are anchored on the search's first
2985 /// non-blank line, so a search whose first line is also wrong may report
2986 /// no similar region even when later lines exist in the file.
2987 fn nearest_match_hint(
2988 contents: &str,
2989 search: &str,
2990 file_has_crlf: bool,
2991 search_has_cr: bool,
2992 ) -> String {
2993 const MAX_SCANNED_LINES: usize = 50_000;
2994 const MAX_EXCERPT_LINES: usize = 12;
2995 const MAX_EXCERPT_LINE_LEN: usize = 200;
2996 const MIN_SCORE: f32 = 0.5;
2997
2998 let line_ratio = |a: &str, b: &str| -> f32 {
2999 let (a, b) = (a.trim(), b.trim());
3000 if a == b {
3001 1.0
3002 } else {
3003 similar::TextDiff::from_chars(a, b).ratio()
3004 }
3005 };
3006 let file_lines: Vec<&str> = contents.lines().take(MAX_SCANNED_LINES).collect();
3007 let search_lines: Vec<&str> = search.lines().collect();
3008 let Some(anchor) = search_lines.iter().position(|line| !line.trim().is_empty()) else {
3009 return String::new();
3010 };
3011 let window = search_lines.len().min(file_lines.len()).max(1);
3012
3013 let mut anchors: Vec<(f32, usize)> = file_lines
3014 .iter()
3015 .enumerate()
3016 .filter(|(index, line)| *index >= anchor && !line.trim().is_empty())
3017 .map(|(index, line)| (line_ratio(search_lines[anchor], line), index - anchor))
3018 .collect();
3019 anchors.sort_by(|a, b| b.0.total_cmp(&a.0).then(a.1.cmp(&b.1)));
3020 let best = anchors
3021 .into_iter()
3022 .take(8)
3023 .map(|(_, start)| {
3024 let end = (start + window).min(file_lines.len());
3025 let score = search_lines
3026 .iter()
3027 .zip(&file_lines[start..end])
3028 .map(|(want, have)| line_ratio(want, have))
3029 .sum::<f32>()
3030 / window as f32;
3031 (score, start, end)
3032 })
3033 .max_by(|a, b| a.0.total_cmp(&b.0).then(b.1.cmp(&a.1)));
3034
3035 let mut notes = Vec::new();
3036 if search_has_cr && !file_has_crlf {
3037 notes.push(
3038 "the search contains carriage returns (CRLF) but the file uses LF line endings"
3039 .to_string(),
3040 );
3041 }
3042 let Some((score, start, end)) = best.filter(|(score, ..)| *score >= MIN_SCORE) else {
3043 let mut hint = String::from("No similar region found in the file.\n");
3044 for note in notes {
3045 hint.push_str(&format!("Note: {note}.\n"));
3046 }
3047 return hint;
3048 };
3049 let region = &file_lines[start..end];
3050 let strip_trailing = |lines: &[&str]| -> Vec<String> {
3051 lines
3052 .iter()
3053 .map(|line| line.trim_end().to_string())
3054 .collect()
3055 };
3056 let collapse = |lines: &[&str]| -> String {
3057 lines
3058 .iter()
3059 .flat_map(|line| line.split_whitespace())
3060 .collect::<Vec<_>>()
3061 .join(" ")
3062 };
3063 if strip_trailing(region) == strip_trailing(&search_lines) {
3064 notes.push("the closest region differs only in trailing whitespace".to_string());
3065 } else if collapse(region) == collapse(&search_lines) {
3066 let tabs = |lines: &[&str]| lines.iter().any(|line| line.starts_with('\t'));
3067 if tabs(region) != tabs(&search_lines) {
3068 notes
3069 .push("the closest region differs only in whitespace (tabs vs spaces)".to_string());
3070 } else {
3071 notes.push("the closest region differs only in whitespace".to_string());
3072 }
3073 }
3074 if file_has_crlf {
3075 notes.push("the file uses CRLF line endings; LF in the search is fine".to_string());
3076 }
3077
3078 let width = end.to_string().len();
3079 let mut hint = format!(
3080 "Closest match (lines {}-{}, {:.0}% similar):\n",
3081 start + 1,
3082 end,
3083 score * 100.0
3084 );
3085 for (offset, line) in region.iter().take(MAX_EXCERPT_LINES).enumerate() {
3086 let mut shown: String = line.chars().take(MAX_EXCERPT_LINE_LEN).collect();
3087 if line.chars().count() > MAX_EXCERPT_LINE_LEN {
3088 shown.push_str("...");
3089 }
3090 hint.push_str(&format!("{:>width$}\t{shown}\n", start + offset + 1));
3091 }
3092 if region.len() > MAX_EXCERPT_LINES {
3093 hint.push_str(&format!(
3094 "... ({} more lines)\n",
3095 region.len() - MAX_EXCERPT_LINES
3096 ));
3097 }
3098 for note in notes {
3099 hint.push_str(&format!("Note: {note}.\n"));
3100 }
3101 hint
3102 }
3103
3104 fn preview_search_for_error(search: &str) -> String {
3105 const MAX_PREVIEW_LINES: usize = 3;
3106 const MAX_PREVIEW_LINE_LEN: usize = 80;
3107 search
3108 .lines()
3109 .take(MAX_PREVIEW_LINES)
3110 .map(|line| {
3111 if line.chars().count() > MAX_PREVIEW_LINE_LEN {
3112 let mut truncated: String = line.chars().take(MAX_PREVIEW_LINE_LEN).collect();
3113 truncated.push_str("...");
3114 truncated
3115 } else {
3116 line.to_string()
3117 }
3118 })
3119 .collect::<Vec<_>>()
3120 .join("\n")
3121 }
3122
3123 /// Normalize Windows CRLF pairs to LF while retaining the normalized byte
3124 /// positions where a `\r` was removed. Lone carriage returns are preserved.
3125 /// Inputs without CRLF are borrowed and use identity offsets.
3126 ///
3127 /// A normalized boundary maps back to the original by adding the number of
3128 /// removed CR bytes strictly before it. At the normalized newline itself that
3129 /// excludes the current CR, so the start maps to `\r`; after the newline (or
3130 /// at EOF) it includes that CR and spans the full pair.
3131 fn normalize_crlf(input: &str) -> Cow<'_, str> {
3132 if input.contains("\r\n") {
3133 Cow::Owned(input.replace("\r\n", "\n"))
3134 } else {
3135 Cow::Borrowed(input)
3136 }
3137 }
3138
3139 fn normalize_crlf_with_positions(input: &str) -> (Cow<'_, str>, Option<Vec<usize>>) {
3140 if !input.contains("\r\n") {
3141 return (Cow::Borrowed(input), None);
3142 }
3143
3144 let mut normalized = String::with_capacity(input.len());
3145 let mut crlf_positions = Vec::new();
3146 let mut chars = input.char_indices().peekable();
3147
3148 while let Some((_, ch)) = chars.next() {
3149 if ch == '\r' && matches!(chars.peek(), Some((_, '\n'))) {
3150 let _ = chars.next();
3151 crlf_positions.push(normalized.len());
3152 normalized.push('\n');
3153 continue;
3154 }
3155
3156 normalized.push(ch);
3157 }
3158
3159 (Cow::Owned(normalized), Some(crlf_positions))
3160 }
3161
3162 fn map_normalized_range(
3163 (start, end): (usize, usize),
3164 crlf_positions: Option<&[usize]>,
3165 ) -> (usize, usize) {
3166 let Some(crlf_positions) = crlf_positions else {
3167 return (start, end);
3168 };
3169 let map_boundary =
3170 |offset| offset + crlf_positions.partition_point(|position| *position < offset);
3171 (map_boundary(start), map_boundary(end))
3172 }
3173
3174 fn map_normalized_ranges(
3175 ranges: impl IntoIterator<Item = (usize, usize)>,
3176 crlf_positions: Option<&[usize]>,
3177 ) -> Vec<(usize, usize)> {
3178 ranges
3179 .into_iter()
3180 .map(|range| map_normalized_range(range, crlf_positions))
3181 .collect()
3182 }
3183
3184 /// Convert model-provided replacement newlines to the base file's convention.
3185 /// Fold CRLF first so an already-CRLF payload never becomes `\r\r\n`.
3186 fn normalize_replacement_line_endings(replace: &str, use_crlf: bool) -> String {
3187 let lf = replace.replace("\r\n", "\n");
3188 if use_crlf {
3189 lf.replace('\n', "\r\n")
3190 } else {
3191 lf
3192 }
3193 }
3194
3195 fn strip_line_leading_whitespace_with_map(input: &str) -> (String, Vec<usize>) {
3196 let mut normalized = String::with_capacity(input.len());
3197 let mut byte_map = Vec::with_capacity(input.len());
3198 let mut at_line_start = true;
3199 for (idx, ch) in input.char_indices() {
3200 if at_line_start && matches!(ch, ' ' | '\t') {
3201 continue;
3202 }
3203 normalized.push(ch);
3204 for _ in 0..ch.len_utf8() {
3205 byte_map.push(idx);
3206 }
3207 at_line_start = ch == '\n';
3208 }
3209 (normalized, byte_map)
3210 }
3211
3212 fn line_start_before(input: &str, idx: usize) -> usize {
3213 input[..idx]
3214 .rfind('\n')
3215 .map_or(0, |newline| newline.saturating_add(1))
3216 }
3217
3218 fn next_char_boundary(input: &str, idx: usize) -> usize {
3219 if idx >= input.len() {
3220 return input.len();
3221 }
3222
3223 let mut next = idx.saturating_add(1);
3224 while next < input.len() && !input.is_char_boundary(next) {
3225 next = next.saturating_add(1);
3226 }
3227 next
3228 }
3229
3230 fn leading_whitespace_fuzzy_matches(contents: &str, search: &str) -> Vec<(usize, usize)> {
3231 let (normalized_contents, byte_map) = strip_line_leading_whitespace_with_map(contents);
3232 let (normalized_search, _) = strip_line_leading_whitespace_with_map(search);
3233 if normalized_search.is_empty() {
3234 return Vec::new();
3235 }
3236
3237 let mut matches = Vec::new();
3238 let mut cursor = 0;
3239 while let Some(rel_idx) = normalized_contents[cursor..].find(&normalized_search) {
3240 let norm_start = cursor + rel_idx;
3241 let norm_end = norm_start + normalized_search.len();
3242 let Some(&mapped_start) = byte_map.get(norm_start) else {
3243 break;
3244 };
3245 // Use the actual match start position, expanding to line start only
3246 // when the match begins at a line boundary in the normalized text.
3247 // This prevents destroying preceding text on the same line when
3248 // the match starts mid-line after whitespace stripping.
3249 let original_start =
3250 if norm_start == 0 || normalized_contents.as_bytes()[norm_start - 1] == b'\n' {
3251 // Match starts at a line boundary — use line start for full-line replacement.
3252 line_start_before(contents, mapped_start)
3253 } else {
3254 // Match starts mid-line — use the exact mapped position.
3255 mapped_start
3256 };
3257 let original_end = byte_map.get(norm_end).copied().unwrap_or(contents.len());
3258 matches.push((original_start, original_end));
3259 cursor = next_char_boundary(&normalized_contents, norm_start);
3260 }
3261 matches
3262 }
3263
3264 /// Normalize typographic punctuation to its ASCII counterpart:
3265 ///
3266 /// * `"` `"` / U+201C U+201D → `"`
3267 /// * `'` `'` / U+2018 U+2019 → `'`
3268 /// * `–` `—` / U+2013 U+2014 → `-`
3269 /// * U+00A0 (non-breaking space) → ASCII space
3270 ///
3271 /// Returns the normalized string plus a byte-map sized to
3272 /// `normalized.len()` whose i-th entry is the original byte offset of
3273 /// the character that produced normalized byte i. Used to recover the
3274 /// original-byte range after finding a match in normalized space.
3275 fn punctuation_normalized_with_map(input: &str) -> (String, Vec<usize>) {
3276 let mut normalized = String::with_capacity(input.len());
3277 let mut byte_map = Vec::with_capacity(input.len());
3278 for (idx, ch) in input.char_indices() {
3279 let replacement: Option<char> = match ch {
3280 '\u{201C}' | '\u{201D}' => Some('"'),
3281 '\u{2018}' | '\u{2019}' => Some('\''),
3282 '\u{2013}' | '\u{2014}' => Some('-'),
3283 '\u{00A0}' => Some(' '),
3284 _ => None,
3285 };
3286 let written = replacement.unwrap_or(ch);
3287 normalized.push(written);
3288 for _ in 0..written.len_utf8() {
3289 byte_map.push(idx);
3290 }
3291 }
3292 (normalized, byte_map)
3293 }
3294
3295 /// Try to find `search` inside `contents` after normalizing typographic
3296 /// punctuation in both. Catches the copy-paste failure mode where a
3297 /// browser, word processor, or chat client silently converted ASCII
3298 /// quotes/dashes to their Unicode "pretty" forms.
3299 fn punctuation_normalized_matches(contents: &str, search: &str) -> Vec<(usize, usize)> {
3300 let (norm_contents, byte_map) = punctuation_normalized_with_map(contents);
3301 let (norm_search, _) = punctuation_normalized_with_map(search);
3302 if norm_search.is_empty() {
3303 return Vec::new();
3304 }
3305 // If normalization didn't change anything, the exact-match pass
3306 // already considered this case — skip to avoid double-reporting.
3307 if norm_contents == contents && norm_search == search {
3308 return Vec::new();
3309 }
3310
3311 let mut matches = Vec::new();
3312 let mut cursor = 0;
3313 while let Some(rel_idx) = norm_contents[cursor..].find(&norm_search) {
3314 let norm_start = cursor + rel_idx;
3315 let norm_end = norm_start + norm_search.len();
3316 let Some(&original_start) = byte_map.get(norm_start) else {
3317 break;
3318 };
3319 let original_end = byte_map.get(norm_end).copied().unwrap_or(contents.len());
3320 matches.push((original_start, original_end));
3321 cursor = next_char_boundary(&norm_contents, norm_start);
3322 }
3323 matches
3324 }
3325
3326 // === ListDirTool ===
3327
3328 /// Tool for listing directory contents.
3329 pub struct ListDirTool;
3330
3331 const LIST_DIR_TIMEOUT: Duration = Duration::from_secs(30);
3332
3333 /// Cap on entries returned by a single `list_dir` call so a huge directory
3334 /// (node_modules, build output, photo dumps) can't balloon the tool result.
3335 /// Mirrors the bounded-output idiom of `read_file`'s `HARD_MAX_READ_LINES`.
3336 /// Directories at or under the cap keep the historical plain-array response;
3337 /// larger ones return an object with truncation metadata.
3338 const LIST_DIR_MAX_ENTRIES: usize = 500;
3339
3340 #[async_trait]
3341 impl ToolSpec for ListDirTool {
3342 fn name(&self) -> &'static str {
3343 "list_dir"
3344 }
3345
3346 fn model_visible(&self) -> bool {
3347 true
3348 }
3349
3350 fn description(&self) -> &'static str {
3351 "List entries in a workspace directory. This bounded, sandbox-aware tool is searchable when the core read/write/edit/bash toolbox is not enough."
3352 }
3353
3354 fn input_schema(&self) -> Value {
3355 json!({
3356 "type": "object",
3357 "properties": {
3358 "path": {
3359 "type": "string",
3360 "description": "Path to inspect (relative to workspace, absolute, or ~/ home-relative; default: .)"
3361 }
3362 },
3363 "required": []
3364 })
3365 }
3366
3367 fn capabilities(&self) -> Vec<ToolCapability> {
3368 vec![ToolCapability::ReadOnly, ToolCapability::Sandboxable]
3369 }
3370
3371 fn supports_parallel(&self) -> bool {
3372 true
3373 }
3374
3375 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
3376 let mut input = input;
3377 apply_param_aliases(&mut input, PATH_ALIASES, "File list")?;
3378 LIST_PARAMS.reject_unknown(&input)?;
3379
3380 let path_str = optional_str(&input, "path")?.unwrap_or(".");
3381 // S1: enumerating a denied directory is a read of it — `list_dir ~/.ssh`
3382 // hands back the key file names. Seatbelt's `deny file-read*` blocks
3383 // readdir of denied dirs, so refusing here matches the OS layer. The
3384 // raw spelling is checked first (F2) so the refusal names the caller's
3385 // path, never a symlink target it might resolve to.
3386 enforce_read_denylist(Path::new(path_str), "list_dir")?;
3387 let dir_path = context.resolve_path(path_str)?;
3388 enforce_read_denylist(&dir_path, "list_dir")?;
3389
3390 let entries =
3391 list_dir_entries_async(dir_path, context.cancel_token.clone(), LIST_DIR_TIMEOUT)
3392 .await?;
3393
3394 ToolResult::json(&entries).map_err(|e| ToolError::execution_failed(e.to_string()))
3395 }
3396 }
3397
3398 async fn list_dir_entries_async(
3399 dir_path: PathBuf,
3400 cancel_token: Option<CancellationToken>,
3401 timeout: Duration,
3402 ) -> Result<Value, ToolError> {
3403 let worker_cancel_token = cancel_token.clone();
3404 run_blocking_list_dir(timeout, cancel_token, move || {
3405 list_dir_entries(&dir_path, worker_cancel_token.as_ref())
3406 })
3407 .await
3408 }
3409
3410 async fn run_blocking_list_dir<F>(
3411 timeout: Duration,
3412 cancel_token: Option<CancellationToken>,
3413 list_dir: F,
3414 ) -> Result<Value, ToolError>
3415 where
3416 F: FnOnce() -> Result<Value, ToolError> + Send + 'static,
3417 {
3418 if cancel_token
3419 .as_ref()
3420 .is_some_and(CancellationToken::is_cancelled)
3421 {
3422 return Err(list_dir_cancelled());
3423 }
3424
3425 let task = tokio::task::spawn_blocking(list_dir);
3426 let result = match cancel_token {
3427 Some(token) => {
3428 tokio::select! {
3429 biased;
3430 () = token.cancelled() => return Err(list_dir_cancelled()),
3431 result = tokio::time::timeout(timeout, task) => result,
3432 }
3433 }
3434 None => tokio::time::timeout(timeout, task).await,
3435 };
3436
3437 let joined = result.map_err(|_| list_dir_timeout(timeout))?;
3438 joined.map_err(|err| {
3439 ToolError::execution_failed(format!("list_dir worker failed before completion: {err}"))
3440 })?
3441 }
3442
3443 fn list_dir_entries(
3444 dir_path: &Path,
3445 cancel_token: Option<&CancellationToken>,
3446 ) -> Result<Value, ToolError> {
3447 check_list_dir_cancelled(cancel_token)?;
3448
3449 let mut entries = Vec::new();
3450 let mut total_entries = 0usize;
3451
3452 for entry in fs::read_dir(dir_path).map_err(|e| {
3453 ToolError::execution_failed(format!(
3454 "Failed to read directory {}: {}",
3455 dir_path.display(),
3456 e
3457 ))
3458 })? {
3459 check_list_dir_cancelled(cancel_token)?;
3460
3461 let entry = entry.map_err(|e| ToolError::execution_failed(e.to_string()))?;
3462 total_entries += 1;
3463 // Past the cap, keep counting for the truncation metadata but stop
3464 // materializing entries.
3465 if entries.len() >= LIST_DIR_MAX_ENTRIES {
3466 continue;
3467 }
3468 let file_type = entry
3469 .file_type()
3470 .map_err(|e| ToolError::execution_failed(e.to_string()))?;
3471
3472 entries.push(json!({
3473 "name": entry.file_name().to_string_lossy().to_string(),
3474 "is_dir": file_type.is_dir(),
3475 }));
3476 }
3477
3478 if total_entries > entries.len() {
3479 Ok(json!({
3480 "entries": entries,
3481 "listed_entries": LIST_DIR_MAX_ENTRIES,
3482 "total_entries": total_entries,
3483 "truncated": true,
3484 }))
3485 } else {
3486 Ok(Value::Array(entries))
3487 }
3488 }
3489
3490 fn check_list_dir_cancelled(cancel_token: Option<&CancellationToken>) -> Result<(), ToolError> {
3491 if cancel_token.is_some_and(CancellationToken::is_cancelled) {
3492 return Err(list_dir_cancelled());
3493 }
3494 Ok(())
3495 }
3496
3497 fn list_dir_cancelled() -> ToolError {
3498 ToolError::cancelled("list_dir cancelled before completion")
3499 }
3500
3501 fn list_dir_timeout(timeout: Duration) -> ToolError {
3502 ToolError::Timeout {
3503 seconds: timeout.as_secs().max(1),
3504 }
3505 }
3506
3507 // === Unit Tests ===
3508
3509 #[cfg(test)]
3510 #[path = "file/tests.rs"]
3511 mod pdf_tests;
3512
3513 #[cfg(test)]
3514 #[path = "file/tests/tools.rs"]
3515 mod tests;
3516
3516 lines RUST