返回 CodeWhale
tool_catalog.rs
根目录 / crates / tui / src / core / engine / tool_catalog.rs
1 //! Deferred tool catalog and built-in advanced tool helpers.
2 //!
3 //! The streaming turn loop owns when tools are offered or executed. This module
4 //! owns the catalog-level policy around deferred loading, tool search, missing
5 //! tool suggestions, and the small set of built-in advanced tools that are not
6 //! registered by the normal runtime tool registry.
7
8 use std::collections::HashSet;
9 use std::path::Path;
10 use std::time::Duration;
11
12 use serde_json::{Value, json};
13
14 #[cfg(test)]
15 use crate::mcp::McpPool;
16 use crate::model_profile::ToolSurfaceBudget;
17 use crate::tools::spec::{
18 ToolContext, ToolError, ToolResult, optional_str, optional_u64, required_str,
19 };
20 use codewhale_config::AppMode;
21 use codewhale_models::Tool;
22
23 use crate::core::session::ToolActivationCache;
24 use crate::dependencies::ExternalTool;
25 use crate::features::{Feature, Features};
26 use crate::regex_cache::compile_user_regex;
27
28 pub(crate) const MULTI_TOOL_PARALLEL_NAME: &str = "multi_tool_use.parallel";
29 pub(crate) const REQUEST_USER_INPUT_NAME: &str = "request_user_input";
30 pub(crate) const CODE_EXECUTION_TOOL_NAME: &str = "code_execution";
31 const CODE_EXECUTION_TOOL_TYPE: &str = "code_execution_20250825";
32 const CODE_EXECUTION_DESCRIPTION: &str = "Execute Python code with the local Python interpreter using this call's execution policy and return stdout/stderr/return_code as JSON.";
33 pub(super) use crate::tools::codemode::EXECUTE_TOOLS_TOOL_NAME;
34 pub(super) use crate::tools::js_execution::JS_EXECUTION_TOOL_NAME;
35 pub(crate) const TOOL_SEARCH_NAME: &str = "tool_search";
36 const TOOL_RESULT_RETRIEVAL_NAME: &str = "retrieve_tool_result";
37 const TOOL_SEARCH_TYPE: &str = "tool_search_20251119";
38 pub(super) const LEGACY_TOOL_SEARCH_REGEX_NAME: &str = "tool_search_tool_regex";
39 pub(super) const LEGACY_TOOL_SEARCH_BM25_NAME: &str = "tool_search_tool_bm25";
40 const TOOL_SEARCH_DEFAULT_MAX_RESULTS: usize = 8;
41 const TOOL_SEARCH_MAX_RESULTS_LIMIT: usize = 8;
42
43 pub(crate) fn is_tool_search_tool(name: &str) -> bool {
44 matches!(
45 name,
46 TOOL_SEARCH_NAME | LEGACY_TOOL_SEARCH_REGEX_NAME | LEGACY_TOOL_SEARCH_BM25_NAME
47 )
48 }
49
50 // Crate-visible so the hook gate tests the real eager names instead of a copy.
51 #[rustfmt::skip]
52 pub(crate) const DEFAULT_ACTIVE_NATIVE_TOOLS: &[&str] = &[
53 // Core work controls are eager; specialized tools stay searchable.
54 "read", "write", "edit", "bash", "agent", "workflow", "todo_write",
55 // Continuation instructions require these controls. Hiding them behind
56 // discovery leaves a model unable to stop the work it was asked to run.
57 "create_goal", "get_goal", "update_goal",
58 // The pinned `## Skills` index tells the model to call `load_skill`, so
59 // the tool has to be on the wire for that instruction to be true. Behind
60 // `tool_search` it cost a discovery hop plus a `change:tool_surface`
61 // re-pin every time a skill was used, against ~134 pinned bytes to have
62 // it eager beside the index the prefix already carries.
63 "load_skill",
64 ];
65
66 const CORE_ACTION_TOOL_FALLBACKS: &[CoreActionToolFallback] = &[
67 CoreActionToolFallback {
68 name: "bash",
69 description: "Run shell commands in the workspace.",
70 unavailable_reason: "Not present in the current model-visible catalog. The session profile, feature availability, or a command tool allow/deny gate can remove shell access. Plan keeps the same primitive identity but centrally refuses execution.",
71 },
72 CoreActionToolFallback {
73 name: "read",
74 description: "Read workspace files.",
75 unavailable_reason: "Not present in the current model-visible catalog. File reads are available in Plan and executable modes unless a command allow/deny gate removes them.",
76 },
77 CoreActionToolFallback {
78 name: "write",
79 description: "Create or replace workspace files.",
80 unavailable_reason: "Not present in the current model-visible catalog. Plan mode has no file-mutation authority; switch to Work mode before writing.",
81 },
82 CoreActionToolFallback {
83 name: "edit",
84 description: "Apply exact replacements to workspace files.",
85 unavailable_reason: "Not present in the current model-visible catalog. Plan mode has no file-mutation authority; switch to Work mode before editing.",
86 },
87 ];
88
89 #[derive(Debug, Clone, Copy)]
90 struct CoreActionToolFallback {
91 name: &'static str,
92 description: &'static str,
93 unavailable_reason: &'static str,
94 }
95
96 /// Pre-computed lowercased haystack + name for each fallback; built once.
97 struct CachedFallback {
98 fallback: CoreActionToolFallback,
99 haystack: String,
100 name_lower: String,
101 }
102
103 static CACHED_FALLBACKS: std::sync::OnceLock<Vec<CachedFallback>> = std::sync::OnceLock::new();
104
105 fn cached_fallbacks() -> &'static [CachedFallback] {
106 CACHED_FALLBACKS.get_or_init(|| {
107 CORE_ACTION_TOOL_FALLBACKS
108 .iter()
109 .map(|f| CachedFallback {
110 fallback: *f,
111 haystack: format!(
112 "{}\n{}\n{}",
113 f.name.to_lowercase(),
114 f.description.to_lowercase(),
115 f.unavailable_reason.to_lowercase(),
116 ),
117 name_lower: f.name.to_lowercase(),
118 })
119 .collect()
120 })
121 }
122
123 /// Membership index over [`DEFAULT_ACTIVE_NATIVE_TOOLS`], built once for the
124 /// process lifetime. The array stays the source of truth for ordered
125 /// inspection; this set only accelerates the
126 /// hot membership check in [`should_default_defer_tool`], which runs once per
127 /// catalog tool on every catalog rebuild (i.e. per turn) — an O(n·m) linear
128 /// scan over the array collapses to O(1) hashed lookups.
129 static DEFAULT_ACTIVE_NATIVE_TOOLS_SET: std::sync::OnceLock<HashSet<&'static str>> =
130 std::sync::OnceLock::new();
131
132 fn default_active_native_tools_set() -> &'static HashSet<&'static str> {
133 DEFAULT_ACTIVE_NATIVE_TOOLS_SET
134 .get_or_init(|| DEFAULT_ACTIVE_NATIVE_TOOLS.iter().copied().collect())
135 }
136
137 pub(super) fn should_default_defer_tool(name: &str, always_load: &HashSet<String>) -> bool {
138 if always_load.contains(name) {
139 return false;
140 }
141
142 if is_tool_search_tool(name) {
143 return false;
144 }
145
146 // Membership-only test (no ordering dependency): the side set built from
147 // DEFAULT_ACTIVE_NATIVE_TOOLS returns identical hit/miss results as the
148 // former `.iter().any(...)` linear scan.
149 !default_active_native_tools_set().contains(name)
150 }
151
152 pub(crate) fn apply_native_tool_deferral(catalog: &mut [Tool], always_load: &HashSet<String>) {
153 for tool in catalog {
154 tool.defer_loading = Some(should_default_defer_tool(&tool.name, always_load));
155 }
156 }
157
158 pub(super) fn apply_mcp_tool_deferral(
159 catalog: &mut [Tool],
160 _mode: AppMode,
161 always_load: &HashSet<String>,
162 ) {
163 for tool in catalog {
164 if always_load.contains(&tool.name) {
165 tool.defer_loading = Some(false);
166 continue;
167 }
168 tool.defer_loading = Some(true);
169 }
170 }
171
172 /// Build the model tool catalog from native and MCP tool lists.
173 ///
174 /// **Catalog-head stability invariant.** The head of the catalog (all
175 /// non-deferred tools) must remain byte-identical across mode toggles
176 /// (Plan ↔ Agent ↔ YOLO) for tools that are common to both modes.
177 /// Deferred tool activations append to the tail and never reorder the
178 /// head. This invariant is critical for DeepSeek's KV prefix cache:
179 /// the tools array is part of the immutable prefix, and any byte-level
180 /// change in the head forces a full re-prefill on the next turn.
181 #[cfg(test)]
182 pub(super) fn build_model_tool_catalog(
183 native_tools: Vec<Tool>,
184 mcp_tools: Vec<Tool>,
185 mode: AppMode,
186 always_load: &HashSet<String>,
187 ) -> Vec<Tool> {
188 build_model_tool_catalog_with_surface(
189 native_tools,
190 mcp_tools,
191 mode,
192 always_load,
193 ToolSurfaceBudget::Standard,
194 )
195 }
196
197 pub(super) fn build_model_tool_catalog_with_surface(
198 mut native_tools: Vec<Tool>,
199 mut mcp_tools: Vec<Tool>,
200 mode: AppMode,
201 always_load: &HashSet<String>,
202 surface_budget: ToolSurfaceBudget,
203 ) -> Vec<Tool> {
204 apply_native_tool_deferral(&mut native_tools, always_load);
205 apply_mcp_tool_deferral(&mut mcp_tools, mode, always_load);
206 apply_tool_surface_budget(&mut native_tools, surface_budget, always_load);
207 apply_tool_surface_budget(&mut mcp_tools, surface_budget, always_load);
208 // Sort each partition by name for prefix-cache stability (#263). The
209 // upstream `to_api_tools()` already sorts the registry's HashMap output;
210 // this catalog is built from caller-supplied Vecs which the test harness
211 // and (future) caller refactors may not pre-sort. Built-ins stay as a
212 // contiguous prefix ahead of MCP tools so adding/removing an MCP tool
213 // never shifts a built-in's position.
214 native_tools.sort_by(|a, b| a.name.cmp(&b.name));
215 mcp_tools.sort_by(|a, b| a.name.cmp(&b.name));
216 native_tools.extend(mcp_tools);
217 native_tools
218 }
219
220 // A second Registry authority used to live here: it appended a "call
221 // registry_sync before this tool" paragraph to a model-visible `exec_shell`
222 // description. The model-visible shell tool is `bash` — `exec_shell` is only a
223 // canonical *action* name (see `tools::canonical_action`) — so the hook never
224 // fired on a live catalog, and its own test pinned that it must not touch
225 // `bash`. The Registry instruction in `Engine::new` is the single prompt
226 // authority for this decision; a per-tool copy of it is not revived here.
227
228 pub(super) fn apply_tool_surface_budget(
229 catalog: &mut [Tool],
230 surface_budget: ToolSurfaceBudget,
231 always_load: &HashSet<String>,
232 ) {
233 if !matches!(surface_budget, ToolSurfaceBudget::Compact) {
234 return;
235 }
236 for tool in catalog {
237 if always_load.contains(&tool.name) {
238 continue;
239 }
240 if matches!(tool.name.as_str(), "Run" | "tasks" | "Web") {
241 tool.defer_loading = Some(true);
242 }
243 }
244 }
245
246 /// Whether two tool-surface budgets currently produce the same catalog.
247 ///
248 /// Runs [`apply_tool_surface_budget`] over `catalog` under both budgets and
249 /// compares the results. `/preview-request` publishes the Standard-vs-Full
250 /// answer as a derived field so the truthful "these are currently collapsed"
251 /// disclosure cannot drift from the code: the day the shaper narrows Standard
252 /// differently from Full, this starts returning `false` on its own.
253 pub(super) fn surface_budgets_produce_same_catalog(
254 catalog: &[Tool],
255 always_load: &HashSet<String>,
256 left_budget: ToolSurfaceBudget,
257 right_budget: ToolSurfaceBudget,
258 ) -> bool {
259 let mut left = catalog.to_vec();
260 let mut right = catalog.to_vec();
261 apply_tool_surface_budget(&mut left, left_budget, always_load);
262 apply_tool_surface_budget(&mut right, right_budget, always_load);
263 serde_json::to_string(&left).ok() == serde_json::to_string(&right).ok()
264 }
265
266 /// How the harness exposes tool-calling to the model, resolved per turn.
267 ///
268 /// Mirrors Codex's `ToolMode`: the model's own metadata wins, `[features]`
269 /// flags override the default, and anything else is [`ToolMode::Direct`].
270 /// There is no user-facing mode to enter — the catalog shape is the whole
271 /// mechanism, so `CodeMode` only promotes `execute_tools` from deferred to
272 /// eager. (A `CodeModeOnly` restriction needs dispatch enforcement and is a
273 /// later slice, not a third variant here.)
274 ///
275 /// KV-cache effect: the inputs are session config (plus future per-model
276 /// metadata), so the resolved mode is prefix-stable within a session; a flag
277 /// flip refreshes the prefix under an explicit config-change reason like any
278 /// other catalog reshape.
279 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
280 pub(crate) enum ToolMode {
281 /// Composition by choice: `execute_tools` stays deferred until `tool_search`.
282 Direct,
283 /// Composition by default: `execute_tools` is eager alongside direct tools.
284 CodeMode,
285 }
286
287 /// Resolve the turn's tool mode: model hint first, `[features] code_mode`
288 /// second, [`ToolMode::Direct`] otherwise. The engine passes `None` for the
289 /// hint until per-model metadata is wired (model_registry follow-up).
290 pub(crate) fn requested_tool_mode(model_hint: Option<ToolMode>, features: &Features) -> ToolMode {
291 model_hint.unwrap_or_else(|| {
292 if features.enabled(Feature::CodeMode) {
293 ToolMode::CodeMode
294 } else {
295 ToolMode::Direct
296 }
297 })
298 }
299
300 pub(crate) fn ensure_advanced_tooling(
301 catalog: &mut Vec<Tool>,
302 mode: AppMode,
303 always_load: &HashSet<String>,
304 tool_mode: ToolMode,
305 ) {
306 // code_execution depends on a locally-installed Python interpreter
307 // (python3 / python / py -3). Before v0.8.31, the tool was always
308 // advertised and would fail at execution time on Windows where
309 // `python3` isn't on PATH — the model treated the tool as reliable
310 // once it appeared in the catalog. We now probe at catalog-build
311 // time and only advertise when an interpreter resolves. See
312 // `crate::dependencies::resolve_python_interpreter` for the probe.
313 if mode != AppMode::Plan
314 && !catalog.iter().any(|t| t.name == CODE_EXECUTION_TOOL_NAME)
315 && crate::dependencies::host_tool_available(CODE_EXECUTION_TOOL_NAME, || {
316 crate::dependencies::resolve_python_interpreter().is_some()
317 })
318 {
319 catalog.push(Tool {
320 tool_type: Some(CODE_EXECUTION_TOOL_TYPE.to_string()),
321 name: CODE_EXECUTION_TOOL_NAME.to_string(),
322 description: CODE_EXECUTION_DESCRIPTION.to_string(),
323 input_schema: json!({
324 "type": "object",
325 "properties": {
326 "code": { "type": "string", "description": "Python source code to execute." },
327 "sandbox_permissions": {
328 "type": "string",
329 "enum": ["workspace-write", "danger-full-access"],
330 "description": "Request a wider policy for this exact execution after a sandbox denial; requires justification and explicit user approval."
331 },
332 "justification": {
333 "type": "string",
334 "description": "Required with sandbox_permissions: why this exact code needs wider access."
335 }
336 },
337 "required": ["code"]
338 }),
339 allowed_callers: Some(vec!["direct".to_string()]),
340 defer_loading: Some(should_default_defer_tool(
341 CODE_EXECUTION_TOOL_NAME,
342 always_load,
343 )),
344 input_examples: None,
345 strict: None,
346 cache_control: None,
347 });
348 }
349
350 // js_execution mirrors code_execution: gate on Node.js being
351 // present locally so the model never sees a runtime it can't
352 // actually use. Plan mode hides shell/exec surfaces (including
353 // both interpreter tools) by construction; Agent / YOLO advertise
354 // the tool only when `resolve_node()` succeeds.
355 if mode != AppMode::Plan
356 && !catalog.iter().any(|t| t.name == JS_EXECUTION_TOOL_NAME)
357 && crate::dependencies::host_tool_available(JS_EXECUTION_TOOL_NAME, || {
358 crate::dependencies::resolve_node().is_some()
359 })
360 {
361 let mut tool = crate::tools::js_execution::js_execution_tool_definition();
362 tool.defer_loading = Some(should_default_defer_tool(&tool.name, always_load));
363 catalog.push(tool);
364 }
365
366 // execute_tools needs no dependency probe: QuickJS is compiled in.
367 // Otherwise it follows the interpreter tools exactly — hidden from Plan,
368 // deferred everywhere else — except under CodeMode, where the harness
369 // promotes composition to eager instead of waiting for tool_search.
370 if mode != AppMode::Plan && !catalog.iter().any(|t| t.name == EXECUTE_TOOLS_TOOL_NAME) {
371 let mut tool = crate::tools::codemode::execute_tools_tool_definition();
372 tool.defer_loading = Some(
373 tool_mode == ToolMode::Direct && should_default_defer_tool(&tool.name, always_load),
374 );
375 catalog.push(tool);
376 }
377
378 if !catalog.iter().any(|t| t.name == TOOL_SEARCH_NAME) {
379 catalog.push(Tool {
380 tool_type: Some(TOOL_SEARCH_TYPE.to_string()),
381 name: TOOL_SEARCH_NAME.to_string(),
382 description: "Search deferred tool definitions and return matching tool references.".to_string(),
383 input_schema: json!({
384 "type": "object",
385 "properties": {
386 "query": { "type": "string", "description": "Search query for tool discovery." },
387 "match": {
388 "type": "string",
389 "enum": ["bm25", "regex"],
390 "default": "bm25",
391 "description": "Matching algorithm: bm25 for natural-language matching, regex for a regular expression over tool names/descriptions/schema."
392 },
393 "max_results": {
394 "type": "integer",
395 "minimum": 1,
396 "maximum": TOOL_SEARCH_MAX_RESULTS_LIMIT,
397 "default": TOOL_SEARCH_DEFAULT_MAX_RESULTS,
398 "description": "Maximum number of matching tool references to return."
399 }
400 },
401 "required": ["query"]
402 }),
403 allowed_callers: Some(vec!["direct".to_string()]),
404 defer_loading: Some(false),
405 input_examples: None,
406 strict: None,
407 cache_control: None,
408 });
409 }
410 }
411
412 pub(crate) fn initial_active_tools(catalog: &[Tool]) -> HashSet<String> {
413 let mut active = HashSet::new();
414 for tool in catalog {
415 if !tool.defer_loading.unwrap_or(false) || is_tool_search_tool(&tool.name) {
416 active.insert(tool.name.clone());
417 }
418 }
419 if active.is_empty()
420 && !catalog.is_empty()
421 && let Some(first) = catalog.first()
422 {
423 active.insert(first.name.clone());
424 }
425 active
426 }
427
428 /// Remove schemas evicted from the conversation cache without hiding tools
429 /// that the current catalog now exposes eagerly.
430 ///
431 /// A cached tool can become eager after an explicit `tools_always_load`
432 /// change or another policy update. Cache revalidation correctly forgets the
433 /// old deferred entry, but the eager catalog entry must remain active.
434 pub(crate) fn remove_evicted_cache_activations(
435 catalog: &[Tool],
436 active: &mut HashSet<String>,
437 evicted: impl IntoIterator<Item = String>,
438 ) {
439 for name in evicted {
440 let is_eager_now = catalog
441 .iter()
442 .any(|tool| tool.name == name && !tool.defer_loading.unwrap_or(false));
443 if !is_eager_now {
444 active.remove(&name);
445 }
446 }
447 }
448
449 /// Promote a successfully executed deferred tool only when this conversation
450 /// had already activated it. Execution can update recency, never grant a name.
451 pub(crate) fn touch_cached_tool_after_execution(
452 catalog: &[Tool],
453 active: &mut HashSet<String>,
454 cache: &mut ToolActivationCache,
455 name: &str,
456 ) -> bool {
457 if !cache.names().any(|cached| cached == name) {
458 return false;
459 }
460 let delta = cache.activate(catalog, &[name.to_string()]);
461 remove_evicted_cache_activations(catalog, active, delta.evicted);
462 active.extend(delta.admitted);
463 true
464 }
465
466 /// Make the recovery schema visible on the next provider step when a tool
467 /// result actually publishes retrievable evidence. The initial toolbox stays
468 /// small, while a receipt never advertises a deferred route that merely asks
469 /// the model to repeat the same call.
470 pub(crate) fn activate_result_dependencies(
471 catalog: &[Tool],
472 active: &mut HashSet<String>,
473 cache: &mut ToolActivationCache,
474 result: &ToolResult,
475 ) -> bool {
476 let needs_retrieval = result
477 .metadata
478 .as_ref()
479 .and_then(|metadata| metadata.get("evidence_available"))
480 .and_then(Value::as_bool)
481 == Some(true);
482 if !needs_retrieval {
483 return false;
484 }
485 let delta = cache.activate(catalog, &[TOOL_RESULT_RETRIEVAL_NAME.to_string()]);
486 remove_evicted_cache_activations(catalog, active, delta.evicted.iter().cloned());
487 active.extend(delta.admitted.iter().cloned());
488 !delta.admitted.is_empty() || !delta.evicted.is_empty()
489 }
490
491 fn active_tool_list_from_catalog(catalog: &[Tool], active: &HashSet<String>) -> Vec<Tool> {
492 // Two-pass for prefix-cache stability (#263). Always-loaded tools come
493 // first in their stable catalog order; tools that started life deferred
494 // and were activated mid-conversation by ToolSearch get appended at the
495 // tail. Otherwise activating a deferred tool shifts every later tool's
496 // byte offset and busts the cached prefix from that point onwards.
497 let catalog_len = catalog.len();
498 let mut head: Vec<Tool> = Vec::with_capacity(catalog_len);
499 let mut tail: Vec<Tool> = Vec::with_capacity(catalog_len);
500 for tool in catalog {
501 if !active.contains(&tool.name) {
502 continue;
503 }
504 if tool.defer_loading.unwrap_or(false) {
505 tail.push(tool.clone());
506 } else {
507 head.push(tool.clone());
508 }
509 }
510 head.extend(tail);
511 head
512 }
513
514 pub(super) fn active_tools_for_step(catalog: &[Tool], active: &HashSet<String>) -> Vec<Tool> {
515 active_tool_list_from_catalog(catalog, active)
516 }
517
518 /// One turn's executable and model-visible tool contract.
519 ///
520 /// The catalog is also the prompt's only availability taxonomy: stable prompt
521 /// prose deliberately does not enumerate tool names. Keeping the concrete
522 /// registry, searchable catalog, initial request subset, and command gates in
523 /// one value prevents preview, dispatch, and execution from reconstructing
524 /// different surfaces from mutable engine configuration.
525 pub(super) struct ToolSurfacePolicy {
526 /// Runtime registry that executes native and plugin tools.
527 pub(super) registry: crate::tools::ToolRegistry,
528 /// Full model-facing catalog, including deferred entries.
529 pub(super) catalog: Vec<Tool>,
530 /// Names active at the start of the turn.
531 pub(super) active_names: HashSet<String>,
532 /// Exact initial `tools` field. `None` means no field is sent.
533 pub(super) active: Option<Vec<Tool>>,
534 pub(super) mode: AppMode,
535 pub(super) strict_tool_mode: bool,
536 allowed_tools: Option<Vec<String>>,
537 disallowed_tools: Option<Vec<String>>,
538 /// Hard per-turn cap on admitted tool calls (#4415). The turn loop copies
539 /// this limit into its own admission counter at turn start; `None` means
540 /// unlimited (the default), which keeps the admission gate inert.
541 pub(super) max_tool_calls: Option<u32>,
542 }
543
544 impl ToolSurfacePolicy {
545 #[allow(clippy::too_many_arguments)]
546 pub(super) fn new(
547 registry: crate::tools::ToolRegistry,
548 tools: Option<Vec<Tool>>,
549 mode: AppMode,
550 always_load: &HashSet<String>,
551 dynamic_active_tools: &[&'static str],
552 strict_tool_mode: bool,
553 allowed_tools: Option<Vec<String>>,
554 disallowed_tools: Option<Vec<String>>,
555 max_tool_calls: Option<u32>,
556 tool_mode: ToolMode,
557 ) -> Self {
558 let mut catalog = tools.unwrap_or_default();
559 if !catalog.is_empty() {
560 ensure_advanced_tooling(&mut catalog, mode, always_load, tool_mode);
561 }
562
563 // Synthetic tools are injected before narrowing. Doing this after the
564 // retain would re-advertise tool_search/code execution despite an
565 // explicit command gate.
566 catalog.retain(|tool| {
567 !tool_denied(disallowed_tools.as_deref(), &tool.name)
568 && tool_allowed(allowed_tools.as_deref(), &tool.name)
569 });
570 for tool in &mut catalog {
571 if let Some(actions) = tool
572 .input_schema
573 .pointer_mut("/properties/action/enum")
574 .and_then(Value::as_array_mut)
575 {
576 actions.retain(|action| {
577 !tool_call_denied(
578 disallowed_tools.as_deref(),
579 &tool.name,
580 &json!({"action": action}),
581 )
582 });
583 }
584 }
585 catalog.retain(|tool| {
586 tool.input_schema
587 .pointer("/properties/action/enum")
588 .and_then(Value::as_array)
589 .is_none_or(|actions| !actions.is_empty())
590 });
591 let mut active_names = initial_active_tools(&catalog);
592 active_names.extend(dynamic_active_tools.iter().map(|name| (*name).to_string()));
593 active_names.retain(|name| catalog.iter().any(|tool| tool.name == *name));
594 let active = active_tools_for_request(&catalog, &active_names, strict_tool_mode);
595
596 Self {
597 registry,
598 catalog,
599 active_names,
600 active,
601 mode,
602 strict_tool_mode,
603 allowed_tools,
604 disallowed_tools,
605 max_tool_calls,
606 }
607 }
608
609 #[cfg(test)]
610 pub(super) fn allows_tool(&self, name: &str) -> bool {
611 !self.denies_tool(name) && self.passes_allow_list(name)
612 }
613
614 pub(super) fn passes_allow_list(&self, name: &str) -> bool {
615 tool_allowed(self.allowed_tools.as_deref(), name)
616 }
617
618 /// A configured-server handshake may only serve this captured turn's
619 /// MCP namespace. Do not replace the command-scoped ceiling with config.
620 pub(super) fn permits_mcp_discovery(&self, server: &str) -> bool {
621 !crate::mcp::McpPool::server_denied_by(
622 self.disallowed_tools.as_deref().unwrap_or_default(),
623 server,
624 ) && self.allowed_tools.as_ref().is_none_or(|rules| {
625 let normalized = rules
626 .iter()
627 .map(|rule| rule.to_ascii_lowercase())
628 .collect::<Vec<_>>();
629 crate::mcp::tool_selection_covers_server(&normalized, server)
630 })
631 }
632
633 pub(super) fn denies_tool(&self, name: &str) -> bool {
634 tool_denied(self.disallowed_tools.as_deref(), name)
635 }
636
637 pub(super) fn denies_call(&self, name: &str, input: &Value) -> bool {
638 tool_call_denied(self.disallowed_tools.as_deref(), name, input)
639 }
640 }
641
642 pub(super) fn tool_allowed(allowed_tools: Option<&[String]>, tool_name: &str) -> bool {
643 let Some(allowed_tools) = allowed_tools else {
644 return true;
645 };
646 tool_matches_any_rule(allowed_tools, tool_name)
647 }
648
649 pub(crate) fn tool_denied(disallowed_tools: Option<&[String]>, tool_name: &str) -> bool {
650 disallowed_tools.is_some_and(|rules| {
651 tool_matches_any_rule(rules, tool_name)
652 || (requires_raw_shell(tool_name) && tool_matches_any_rule(rules, "Bash"))
653 })
654 }
655
656 /// Execution dependencies narrow denials only. Treating these as symmetric
657 /// aliases would also grant task/terminal execution to an allowlist of Bash.
658 fn requires_raw_shell(name: &str) -> bool {
659 matches!(
660 name.to_ascii_lowercase().as_str(),
661 "bash"
662 | "exec_shell"
663 | "exec_shell_interact"
664 | "exec_interact"
665 | "task_shell_start"
666 | "task_gate_run"
667 // These owners start a fresh execution and currently cannot
668 // transport this command's deny ceiling. Fail closed until they
669 // can preserve it; inspection and cancellation stay available.
670 | "task_create"
671 | "automation_create"
672 | "automation_update"
673 | "automation_resume"
674 | "automation_run"
675 | "terminal/run"
676 | "terminal/send"
677 | "terminal/reset"
678 | "code_execution"
679 | "js_execution"
680 | "rlm_eval"
681 // Launching an MCP server spawns an arbitrary local process.
682 | "start_mcp_server"
683 | "start_registry_mcp_server"
684 )
685 }
686
687 pub(crate) fn tool_call_denied(rules: Option<&[String]>, name: &str, input: &Value) -> bool {
688 use crate::tools::canonical_action::canonical_action_alias;
689 use crate::tools::execution_envelope::{VerificationBound, classify_verification};
690
691 let action = canonical_action_alias(name, input);
692 tool_denied(rules, name)
693 || tool_denied(rules, action)
694 || (action == "rlm_open"
695 && input
696 .get("url")
697 .and_then(Value::as_str)
698 .is_some_and(|url| !url.trim().is_empty())
699 && tool_denied(rules, "fetch_url"))
700 || (matches!(
701 classify_verification(action, input),
702 Some(VerificationBound::Unbounded)
703 ) && rules.is_some_and(|rules| tool_matches_any_rule(rules, "Bash")))
704 }
705
706 /// Repeat the command ceiling at native dispatch and direct delegation sinks.
707 /// The existing child evidence exception is limited to the canonical lowercase
708 /// tool, a child-owned context, and the same strict read-only grammar enforced
709 /// by Bash itself. It never admits a session, stdin, or background command.
710 pub(crate) fn enforce_tool_denial(
711 context: &crate::tools::spec::ToolContext,
712 name: &str,
713 input: &Value,
714 ) -> Result<(), ToolError> {
715 let bounded_child_read = name == "bash"
716 && context.owner_agent_id.is_some()
717 && context.shell_policy == crate::worker_profile::ShellPolicy::ReadOnly
718 && crate::tools::shell::agent_readonly_bash_input(input);
719 if !bounded_child_read && tool_call_denied(Some(&context.disallowed_tools), name, input) {
720 return Err(ToolError::permission_denied(format!(
721 "Tool '{name}' or its execution dependency is in the disallowed-tools list"
722 )));
723 }
724 Ok(())
725 }
726
727 pub(crate) fn tool_matches_any_rule(rules: &[String], tool_name: &str) -> bool {
728 let tool_name = tool_name.to_ascii_lowercase();
729 rules.iter().any(|rule| {
730 let rule = rule.to_ascii_lowercase();
731 let (rule_body, is_prefix) = rule
732 .strip_suffix('*')
733 .map_or((rule.as_str(), false), |prefix| (prefix, true));
734 std::iter::once(tool_name.as_str())
735 .chain(policy_tool_aliases(&tool_name).iter().copied())
736 .any(|candidate| {
737 if is_prefix {
738 candidate.starts_with(rule_body)
739 } else {
740 candidate == rule_body
741 }
742 })
743 })
744 }
745
746 /// Whether an explicit tool allowlist is provably limited to the native file
747 /// and shell primitives. Unknown names and wildcards remain conservative
748 /// because a configured MCP server may own them.
749 pub(crate) fn allowlist_is_native_file_and_shell_only(allowed_tools: Option<&[String]>) -> bool {
750 let Some(rules) = allowed_tools else {
751 return false;
752 };
753 const NATIVE_NAMES: &[&str] = &[
754 "bash",
755 "exec_shell",
756 "read",
757 "read_file",
758 "write",
759 "write_file",
760 "edit",
761 "edit_file",
762 "file",
763 ];
764 rules.iter().all(|rule| {
765 let rule = rule.trim();
766 !rule.is_empty()
767 && !rule.ends_with('*')
768 && NATIVE_NAMES.contains(&rule.to_ascii_lowercase().as_str())
769 })
770 }
771
772 fn policy_tool_aliases(name: &str) -> &'static [&'static str] {
773 match name {
774 "read" | "read_file" => &["read", "read_file", "file"],
775 "write" | "write_file" => &["write", "write_file", "file"],
776 "edit" | "edit_file" => &["edit", "edit_file", "file"],
777 "file" => &[
778 "file",
779 "read",
780 "read_file",
781 "write",
782 "write_file",
783 "edit",
784 "edit_file",
785 ],
786 "bash" | "exec_shell" => &["bash", "exec_shell"],
787 "mcp_read_resource" | "read_mcp_resource" => &["mcp_read_resource", "read_mcp_resource"],
788 _ => &[],
789 }
790 }
791
792 /// The `tools` field of one outbound request, from a catalog and the set of
793 /// currently-active tool names.
794 ///
795 /// Shared by [`ToolSurfacePolicy`] and the per-step rebuild inside the turn
796 /// loop, so activating a deferred tool mid-turn goes through one code path.
797 pub(crate) fn active_tools_for_request(
798 catalog: &[Tool],
799 active: &HashSet<String>,
800 strict_tool_mode: bool,
801 ) -> Option<Vec<Tool>> {
802 if catalog.is_empty() {
803 return None;
804 }
805 let mut tools = active_tools_for_step(catalog, active);
806 if strict_tool_mode {
807 crate::tools::schema_sanitize::prepare_tools_for_strict_mode(&mut tools);
808 }
809 Some(tools)
810 }
811
812 /// Reusable scratch for one `tool_search` catalog scan.
813 ///
814 /// Each deferred tool needs a lowercased `name\ndescription\ninput_schema` blob
815 /// that is compared once and dropped. Building it with `format!` also copied all
816 /// three pieces a second time into the concatenation, and the bm25 scorer then
817 /// re-lowered `tool.name` once per query term for a value that does not vary
818 /// across terms. Reusing one set of buffers across the scan removes the
819 /// concatenation copy and the per-term lowering, and keeps the buffers' capacity
820 /// instead of reallocating per tool (#6213 T5).
821 ///
822 /// This is the same precomputed-index idiom `CachedFallback` already uses for
823 /// the static core-action fallbacks in this file; it is not a new pattern.
824 ///
825 /// Lowercasing deliberately stays `str::to_lowercase`, matching the original
826 /// exactly. A per-`char` fold would allocate less but is not the same function —
827 /// it differs on Greek final sigma — and this path runs a handful of times per
828 /// turn beside a multi-second provider call, so it is not worth a semantic
829 /// change.
830 #[derive(Default)]
831 struct ToolSearchScratch {
832 /// `tool.name`, lowercased. Loop-invariant across query terms, so the bm25
833 /// scorer reads this instead of re-lowering the name once per term.
834 name_lower: String,
835 /// Compact JSON of `tool.input_schema`, before lowercasing.
836 schema_json: String,
837 /// The match target: `name\ndescription\nschema`, all lowercased.
838 hay: String,
839 }
840
841 impl ToolSearchScratch {
842 fn load(&mut self, tool: &Tool) {
843 use std::fmt::Write as _;
844
845 self.name_lower.clear();
846 self.name_lower.push_str(&tool.name.to_lowercase());
847
848 self.schema_json.clear();
849 // `Value`'s `Display` is what `to_string()` calls, so this is the same
850 // text without materializing an owned copy first. Infallible for a
851 // `String` sink; a formatting error could only shorten the schema,
852 // which weakens matching and never breaks correctness.
853 let _ = write!(self.schema_json, "{}", tool.input_schema);
854
855 self.hay.clear();
856 self.hay.push_str(&self.name_lower);
857 self.hay.push('\n');
858 self.hay.push_str(&tool.description.to_lowercase());
859 self.hay.push('\n');
860 self.hay.push_str(&self.schema_json.to_lowercase());
861 }
862 }
863
864 fn catalog_contains_tool(catalog: &[Tool], name: &str) -> bool {
865 catalog.iter().any(|tool| tool.name == name)
866 }
867
868 fn unavailable_core_action_tools_with_regex(
869 catalog: &[Tool],
870 query: &str,
871 max_results: usize,
872 ) -> Result<Vec<CoreActionToolFallback>, ToolError> {
873 if max_results == 0 {
874 return Ok(Vec::new());
875 }
876 let regex = compile_user_regex(query)
877 .map_err(|err| ToolError::invalid_input(format!("Invalid regex query: {err}")))?;
878 Ok(cached_fallbacks()
879 .iter()
880 .filter(|cf| !catalog_contains_tool(catalog, cf.fallback.name))
881 .filter(|cf| regex.is_match(&cf.haystack))
882 .take(max_results)
883 .map(|cf| cf.fallback)
884 .collect())
885 }
886
887 fn unavailable_core_action_tools_with_bm25_like(
888 catalog: &[Tool],
889 query: &str,
890 max_results: usize,
891 ) -> Vec<CoreActionToolFallback> {
892 if max_results == 0 {
893 return Vec::new();
894 }
895 let terms: Vec<String> = query
896 .split_whitespace()
897 .map(|term| term.trim().to_lowercase())
898 .filter(|term| !term.is_empty())
899 .collect();
900 if terms.is_empty() {
901 return Vec::new();
902 }
903
904 let mut scored: Vec<(i64, CoreActionToolFallback)> = Vec::new();
905 for cf in cached_fallbacks() {
906 if catalog_contains_tool(catalog, cf.fallback.name) {
907 continue;
908 }
909 let hay = &cf.haystack;
910 let name = &cf.name_lower;
911 let mut score = 0i64;
912 for term in &terms {
913 if hay.contains(term) {
914 score += 1;
915 }
916 if name.contains(term) {
917 score += 2;
918 }
919 }
920 if score > 0 {
921 scored.push((score, cf.fallback));
922 }
923 }
924 scored.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.name.cmp(b.1.name)));
925 scored
926 .into_iter()
927 .take(max_results)
928 .map(|(_, fallback)| fallback)
929 .collect()
930 }
931
932 fn discover_tools_with_regex(
933 catalog: &[Tool],
934 query: &str,
935 max_results: usize,
936 ) -> Result<Vec<String>, ToolError> {
937 let regex = compile_user_regex(query)
938 .map_err(|err| ToolError::invalid_input(format!("Invalid regex query: {err}")))?;
939
940 let mut matches = Vec::new();
941 let mut scratch = ToolSearchScratch::default();
942 for tool in catalog {
943 // tool_search loads definitions omitted from the current request. An
944 // eager tool is already present, so returning it as a cache candidate
945 // would misclassify it as rejected (the cache intentionally accepts
946 // deferred definitions only).
947 if !tool.defer_loading.unwrap_or(false) || is_tool_search_tool(&tool.name) {
948 continue;
949 }
950 scratch.load(tool);
951 if regex.is_match(&scratch.hay) {
952 matches.push(tool.name.clone());
953 }
954 if matches.len() >= max_results {
955 break;
956 }
957 }
958 Ok(matches)
959 }
960
961 fn discover_tools_with_bm25_like(catalog: &[Tool], query: &str, max_results: usize) -> Vec<String> {
962 let terms: Vec<String> = query
963 .split_whitespace()
964 .map(|term| term.trim().to_lowercase())
965 .filter(|term| !term.is_empty())
966 .collect();
967 if terms.is_empty() {
968 return Vec::new();
969 }
970
971 let mut scored: Vec<(i64, String)> = Vec::new();
972 let mut scratch = ToolSearchScratch::default();
973 for tool in catalog {
974 if !tool.defer_loading.unwrap_or(false) || is_tool_search_tool(&tool.name) {
975 continue;
976 }
977 scratch.load(tool);
978 let mut score = 0i64;
979 for term in &terms {
980 if scratch.hay.contains(term) {
981 score += 1;
982 }
983 // Loop-invariant: lowered once by `load`, not once per term.
984 if scratch.name_lower.contains(term) {
985 score += 2;
986 }
987 }
988 if score > 0 {
989 scored.push((score, tool.name.clone()));
990 }
991 }
992 scored.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.cmp(&b.1)));
993 scored
994 .into_iter()
995 .take(max_results)
996 .map(|(_, name)| name)
997 .collect()
998 }
999
1000 fn edit_distance(a: &str, b: &str) -> usize {
1001 if a == b {
1002 return 0;
1003 }
1004 if a.is_empty() {
1005 return b.chars().count();
1006 }
1007 if b.is_empty() {
1008 return a.chars().count();
1009 }
1010
1011 let b_chars: Vec<char> = b.chars().collect();
1012 let mut prev: Vec<usize> = (0..=b_chars.len()).collect();
1013 let mut curr = vec![0usize; b_chars.len() + 1];
1014
1015 for (i, a_ch) in a.chars().enumerate() {
1016 curr[0] = i + 1;
1017 for (j, b_ch) in b_chars.iter().enumerate() {
1018 let cost = if a_ch == *b_ch { 0 } else { 1 };
1019 let delete = prev[j + 1] + 1;
1020 let insert = curr[j] + 1;
1021 let substitute = prev[j] + cost;
1022 curr[j + 1] = delete.min(insert).min(substitute);
1023 }
1024 std::mem::swap(&mut prev, &mut curr);
1025 }
1026
1027 prev[b_chars.len()]
1028 }
1029
1030 fn suggest_tool_names(catalog: &[Tool], requested: &str, limit: usize) -> Vec<String> {
1031 let requested = requested.trim().to_ascii_lowercase();
1032 if requested.is_empty() || limit == 0 {
1033 return Vec::new();
1034 }
1035
1036 let mut candidates: Vec<(u8, usize, String)> = Vec::new();
1037 for tool in catalog {
1038 let candidate = tool.name.to_ascii_lowercase();
1039 let prefix_match = candidate.starts_with(&requested) || requested.starts_with(&candidate);
1040 let contains_match = candidate.contains(&requested) || requested.contains(&candidate);
1041 let distance = edit_distance(&candidate, &requested);
1042 let close_typo = distance <= 3;
1043
1044 if !(prefix_match || contains_match || close_typo) {
1045 continue;
1046 }
1047
1048 let rank = if prefix_match {
1049 0
1050 } else if contains_match {
1051 1
1052 } else {
1053 2
1054 };
1055 candidates.push((rank, distance, tool.name.clone()));
1056 }
1057
1058 candidates.sort_by(|a, b| {
1059 a.0.cmp(&b.0)
1060 .then_with(|| a.1.cmp(&b.1))
1061 .then_with(|| a.2.cmp(&b.2))
1062 });
1063 candidates.dedup_by(|a, b| a.2 == b.2);
1064 candidates
1065 .into_iter()
1066 .take(limit)
1067 .map(|(_, _, name)| name)
1068 .collect()
1069 }
1070
1071 /// Catalog tools the engine injects itself rather than registering, plus the
1072 /// legacy tool-search spellings. Exposed so the read-only request projection
1073 /// can label their provenance as `synthetic` from the same source of truth as
1074 /// `is_synthetic_catalog_tool` test coverage instead of guessing.
1075 ///
1076 /// MCP-contributed names are deliberately *not* here: those resolve through the
1077 /// real pool, and stay unknown when the pool did not resolve them.
1078 /// `multi_tool_use.parallel` is not here either — it is a legacy name the model
1079 /// may emit, never a catalog entry, so it can never appear in a transmitted
1080 /// tool array and has no catalog provenance to report.
1081 pub(super) fn default_synthetic_catalog_tool_names() -> Vec<String> {
1082 let mut names: Vec<String> = vec![
1083 TOOL_SEARCH_NAME.to_string(),
1084 LEGACY_TOOL_SEARCH_REGEX_NAME.to_string(),
1085 LEGACY_TOOL_SEARCH_BM25_NAME.to_string(),
1086 CODE_EXECUTION_TOOL_NAME.to_string(),
1087 JS_EXECUTION_TOOL_NAME.to_string(),
1088 EXECUTE_TOOLS_TOOL_NAME.to_string(),
1089 ];
1090 names.sort();
1091 names.dedup();
1092 names
1093 }
1094
1095 #[cfg(test)]
1096 fn is_synthetic_catalog_tool(name: &str) -> bool {
1097 is_tool_search_tool(name)
1098 || matches!(
1099 name,
1100 CODE_EXECUTION_TOOL_NAME | JS_EXECUTION_TOOL_NAME | EXECUTE_TOOLS_TOOL_NAME
1101 )
1102 || McpPool::is_mcp_tool(name)
1103 }
1104
1105 #[cfg(test)]
1106 pub(super) fn tool_catalog_consistency_issues(
1107 catalog: &[Tool],
1108 registry: &crate::tools::ToolRegistry,
1109 ) -> Vec<String> {
1110 let catalog_names = catalog
1111 .iter()
1112 .map(|tool| tool.name.as_str())
1113 .collect::<HashSet<_>>();
1114 let registry_api_tools = registry.to_api_tools();
1115 let registry_model_visible_names = registry_api_tools
1116 .iter()
1117 .map(|tool| tool.name.as_str())
1118 .collect::<HashSet<_>>();
1119 let mut issues = Vec::new();
1120
1121 for tool in catalog {
1122 if is_synthetic_catalog_tool(&tool.name) {
1123 continue;
1124 }
1125 if !registry.contains(&tool.name) {
1126 issues.push(format!(
1127 "catalog advertises '{}' but no registered handler exists",
1128 tool.name
1129 ));
1130 }
1131 }
1132
1133 for name in DEFAULT_ACTIVE_NATIVE_TOOLS {
1134 if registry_model_visible_names.contains(name) && !catalog_names.contains(name) {
1135 issues.push(format!(
1136 "registered core tool '{name}' is missing from the model/search catalog"
1137 ));
1138 }
1139 }
1140
1141 issues.sort();
1142 issues
1143 }
1144
1145 pub(super) fn missing_tool_error_message(tool_name: &str, catalog: &[Tool]) -> String {
1146 // Dogfood A5 (#4092): models mid-checklist sometimes emit each list entry
1147 // as its own tool call named `item`/`todo`/... . Fuzzy suggestions are
1148 // actively misleading there ("Did you mean: note, tts?"); name the actual
1149 // fix instead.
1150 if matches!(
1151 tool_name,
1152 "item" | "items" | "todo" | "todos" | "checklist" | "checklist_item" | "plan_item"
1153 ) {
1154 return format!(
1155 "Tool '{tool_name}' is not available in the current tool catalog. \
1156 Checklist entries are not separate tool calls — write the whole list \
1157 in one `todo_write` call with a `todos` array of \
1158 {{content, status}} objects."
1159 );
1160 }
1161 let suggestions = suggest_tool_names(catalog, tool_name, 3);
1162 let shell_hint = if is_shell_tool_name(tool_name) {
1163 Some(shell_tool_allow_shell_hint())
1164 } else {
1165 None
1166 };
1167 // #5123-class: `exec_shell` was replaced by lowercase `bash`. Name it first —
1168 // otherwise the error misdiagnoses a retired-name call as an allow_shell
1169 // permission problem and sends the model fixing the wrong thing.
1170 if tool_name == "exec_shell" {
1171 return format!(
1172 "Tool '{tool_name}' is not available in the current tool catalog. \
1173 `exec_shell` was replaced by `bash` — call `bash` with a `command` instead. \
1174 If `bash` is also absent: {shell_hint}.",
1175 shell_hint = shell_tool_allow_shell_hint()
1176 );
1177 }
1178 if matches!(
1179 tool_name,
1180 "exec_shell_wait" | "exec_shell_interact" | "exec_shell_cancel"
1181 ) {
1182 return format!(
1183 "Tool '{tool_name}' is not available in the current tool catalog. \
1184 Lowercase `bash` is foreground-only; use {TOOL_SEARCH_NAME} to discover shell session controls."
1185 );
1186 }
1187 if suggestions.is_empty() {
1188 if let Some(shell_hint) = shell_hint {
1189 return format!(
1190 "Tool '{tool_name}' is not available in the current tool catalog. \
1191 {shell_hint}, or use {TOOL_SEARCH_NAME} with a short query."
1192 );
1193 }
1194 return format!(
1195 "Tool '{tool_name}' is not available in the current tool catalog. \
1196 Verify mode/feature flags, or use {TOOL_SEARCH_NAME} with a short query."
1197 );
1198 }
1199
1200 let suggestion_text = format!("Did you mean: {}?", suggestions.join(", "));
1201 if let Some(shell_hint) = shell_hint {
1202 return format!(
1203 "Tool '{tool_name}' is not available in the current tool catalog. \
1204 {suggestion_text} {shell_hint}. \
1205 You can also use {TOOL_SEARCH_NAME} to discover tools."
1206 );
1207 }
1208
1209 format!(
1210 "Tool '{tool_name}' is not available in the current tool catalog. \
1211 {suggestion_text} You can also use {TOOL_SEARCH_NAME} to discover tools."
1212 )
1213 }
1214
1215 fn shell_tool_allow_shell_hint() -> &'static str {
1216 "Shell tools are absent because this session or profile disabled shell access, \
1217 commonly via top-level `allow_shell = false`. \
1218 Work mode exposes shell by default with approval gating unless disabled. \
1219 Run `/config allow_shell true` for this session or add `--save` for future sessions; \
1220 the next turn will expose shell again"
1221 }
1222
1223 fn is_shell_tool_name(tool_name: &str) -> bool {
1224 matches!(
1225 tool_name,
1226 "exec_shell"
1227 | "exec_shell_wait"
1228 | "exec_shell_interact"
1229 | "task_shell_start"
1230 | "task_shell_wait"
1231 )
1232 }
1233
1234 pub(super) fn maybe_hydrate_requested_deferred_tool(
1235 tool_name: &str,
1236 tool_input: &Value,
1237 catalog: &[Tool],
1238 active_tools_at_batch_start: &HashSet<String>,
1239 hydrated_tools_this_batch: &mut HashSet<String>,
1240 ) -> Option<ToolResult> {
1241 let def = catalog.iter().find(|def| def.name == tool_name)?;
1242
1243 if !def.defer_loading.unwrap_or(false) || active_tools_at_batch_start.contains(tool_name) {
1244 return None;
1245 }
1246
1247 hydrated_tools_this_batch.insert(tool_name.to_string());
1248 if deferred_first_call_matches_schema(def, tool_input) {
1249 // Progressive disclosure keeps unused schemas out of the prefix; it
1250 // must not cost a well-formed call its turn. Every authority gate has
1251 // already run for this call, so executing it grants nothing new, and
1252 // the tool still activates at the tail for later requests.
1253 return None;
1254 }
1255 Some(deferred_tool_schema_hydration_result(def, tool_input))
1256 }
1257
1258 /// Whether a call to a tool whose schema the model has not yet been shown is
1259 /// shaped like that schema: an object carrying every required field and, when
1260 /// the schema declares properties, no field outside them. Known limitation:
1261 /// field types are left to the tool's own input validation, which reports a
1262 /// wrong type as an ordinary tool error after the schema has been activated.
1263 pub(crate) fn deferred_first_call_matches_schema(tool: &Tool, tool_input: &Value) -> bool {
1264 let Some(input) = tool_input.as_object() else {
1265 return false;
1266 };
1267 let expected = schema_fields(&tool.input_schema);
1268 let required = schema_required_fields(&tool.input_schema);
1269 required.iter().all(|field| input.contains_key(field))
1270 && (expected.is_empty() && input.is_empty()
1271 || !expected.is_empty()
1272 && input
1273 .keys()
1274 .all(|key| expected.iter().any(|field| &field.name == key)))
1275 }
1276
1277 #[cfg(test)]
1278 pub(super) fn preflight_requested_deferred_tool(
1279 tool_name: &str,
1280 tool_input: &Value,
1281 catalog: &[Tool],
1282 active_tools: &mut HashSet<String>,
1283 ) -> Option<ToolResult> {
1284 let active_tools_at_batch_start = active_tools.clone();
1285 let mut hydrated_tools_this_batch = HashSet::new();
1286 let result = maybe_hydrate_requested_deferred_tool(
1287 tool_name,
1288 tool_input,
1289 catalog,
1290 &active_tools_at_batch_start,
1291 &mut hydrated_tools_this_batch,
1292 );
1293 active_tools.extend(hydrated_tools_this_batch);
1294 result
1295 }
1296
1297 pub(crate) fn deferred_tool_schema_hydration_result(tool: &Tool, tool_input: &Value) -> ToolResult {
1298 let expected = schema_fields(&tool.input_schema);
1299 let required = schema_required_fields(&tool.input_schema);
1300 let received = received_field_names(tool_input);
1301 let missing = required
1302 .iter()
1303 .filter(|field| !received.contains(field))
1304 .cloned()
1305 .collect::<Vec<_>>();
1306 let unexpected = received
1307 .iter()
1308 .filter(|field| !expected.iter().any(|expected| &expected.name == *field))
1309 .cloned()
1310 .collect::<Vec<_>>();
1311 let corrections = likely_field_corrections(&received, &expected, &tool.name);
1312
1313 let mut lines = vec![
1314 format!("Tool `{}` was deferred and has now been loaded.", tool.name),
1315 String::new(),
1316 "The tool was not executed. Retry with the loaded schema.".to_string(),
1317 String::new(),
1318 "Expected fields:".to_string(),
1319 ];
1320 if expected.is_empty() {
1321 lines.push(" (none)".to_string());
1322 } else {
1323 for field in &expected {
1324 let required_marker = if required.contains(&field.name) {
1325 " required"
1326 } else {
1327 ""
1328 };
1329 lines.push(format!(
1330 " {}: {}{}",
1331 field.name, field.kind, required_marker
1332 ));
1333 }
1334 }
1335 lines.push(String::new());
1336 lines.push("Received fields:".to_string());
1337 if received.is_empty() {
1338 lines.push(" (none)".to_string());
1339 } else {
1340 lines.push(format!(" {}", received.join(", ")));
1341 }
1342 if !missing.is_empty() {
1343 lines.push(String::new());
1344 lines.push("Missing required fields:".to_string());
1345 lines.push(format!(" {}", missing.join(", ")));
1346 }
1347 if !unexpected.is_empty() {
1348 lines.push(String::new());
1349 lines.push("Unexpected fields:".to_string());
1350 lines.push(format!(" {}", unexpected.join(", ")));
1351 }
1352 if !corrections.is_empty() {
1353 lines.push(String::new());
1354 lines.push("Likely corrections:".to_string());
1355 for correction in &corrections {
1356 lines.push(format!(" {correction}"));
1357 }
1358 }
1359
1360 ToolResult::success(lines.join("\n")).with_metadata(json!({
1361 "event": "tool.schema_hydrated",
1362 "tool": tool.name,
1363 "executed": false,
1364 "retry_required": true,
1365 "reason": "deferred_tool_first_use",
1366 "deferred_tool_loaded": true,
1367 "tool_name": tool.name,
1368 "expected_fields": expected.iter().map(|field| field.name.clone()).collect::<Vec<_>>(),
1369 "received_fields": received,
1370 "missing_required_fields": missing,
1371 "unexpected_fields": unexpected,
1372 "likely_corrections": corrections,
1373 }))
1374 }
1375
1376 #[derive(Debug, Clone)]
1377 struct SchemaField {
1378 name: String,
1379 kind: String,
1380 }
1381
1382 fn schema_fields(schema: &Value) -> Vec<SchemaField> {
1383 let Some(properties) = schema.get("properties").and_then(Value::as_object) else {
1384 return Vec::new();
1385 };
1386 let mut fields = properties
1387 .iter()
1388 .map(|(name, spec)| SchemaField {
1389 name: name.clone(),
1390 kind: schema_type_label(spec),
1391 })
1392 .collect::<Vec<_>>();
1393 fields.sort_by(|a, b| a.name.cmp(&b.name));
1394 fields
1395 }
1396
1397 fn schema_required_fields(schema: &Value) -> Vec<String> {
1398 let mut required = schema
1399 .get("required")
1400 .and_then(Value::as_array)
1401 .into_iter()
1402 .flatten()
1403 .filter_map(|value| value.as_str().map(str::to_string))
1404 .collect::<Vec<_>>();
1405 required.sort();
1406 required
1407 }
1408
1409 fn schema_type_label(spec: &Value) -> String {
1410 let Some(kind) = spec.get("type").and_then(Value::as_str) else {
1411 return "value".to_string();
1412 };
1413 if let Some(values) = spec.get("enum").and_then(Value::as_array) {
1414 let labels = values.iter().filter_map(Value::as_str).collect::<Vec<_>>();
1415 if !labels.is_empty() {
1416 return format!("{kind} ({})", labels.join(" | "));
1417 }
1418 }
1419 kind.to_string()
1420 }
1421
1422 fn received_field_names(input: &Value) -> Vec<String> {
1423 let mut fields = input
1424 .as_object()
1425 .map(|object| object.keys().cloned().collect::<Vec<_>>())
1426 .unwrap_or_default();
1427 fields.sort();
1428 fields
1429 }
1430
1431 fn likely_field_corrections(
1432 received: &[String],
1433 expected: &[SchemaField],
1434 tool_name: &str,
1435 ) -> Vec<String> {
1436 let has_expected = |name: &str| expected.iter().any(|field| field.name == name);
1437 let has_received = |name: &str| received.iter().any(|field| field == name);
1438 let mut corrections = Vec::new();
1439
1440 if has_received("old_string") && has_expected("search") {
1441 corrections.push("old_string -> search".to_string());
1442 } else if has_received("old_str") && has_expected("search") {
1443 corrections.push("old_str -> search".to_string());
1444 }
1445 if has_received("new_string") && has_expected("replace") {
1446 corrections.push("new_string -> replace".to_string());
1447 } else if has_received("new_str") && has_expected("replace") {
1448 corrections.push("new_str -> replace".to_string());
1449 } else if has_received("replacement") && has_expected("replace") {
1450 corrections.push("replacement -> replace".to_string());
1451 }
1452 if matches!(tool_name, "checklist_update" | "todo_update") && has_received("todos") {
1453 corrections.push(
1454 "Use todo_write to replace the full list, or retry checklist_update/todo_update with id and status."
1455 .to_string(),
1456 );
1457 }
1458 // RLM source fields are easy to misname (#2659). rlm_open takes exactly one
1459 // of file_path / content / url / session_object; nudge common wrong names
1460 // toward those. The unified `rlm` tool carries the same fields for
1461 // action=open, so it gets the same correction.
1462 if matches!(tool_name, "rlm_open" | "rlm") {
1463 for wrong in [
1464 "prompt",
1465 "resident_file",
1466 "text",
1467 "body",
1468 "path",
1469 "file",
1470 "source",
1471 ] {
1472 if has_received(wrong)
1473 && !has_received("file_path")
1474 && !has_received("content")
1475 && !has_received("url")
1476 && !has_received("session_object")
1477 {
1478 corrections.push(format!("{wrong} -> file_path (local file), content (inline text), url, or session_object"));
1479 }
1480 }
1481 }
1482 corrections
1483 }
1484
1485 #[cfg(test)]
1486 pub(super) fn execute_tool_search(
1487 tool_name: &str,
1488 input: &serde_json::Value,
1489 catalog: &[Tool],
1490 active_tools: &mut HashSet<String>,
1491 ) -> Result<ToolResult, ToolError> {
1492 execute_tool_search_inner(tool_name, input, catalog, active_tools, None)
1493 }
1494
1495 /// Execute tool search while retaining activated schemas in the bounded
1496 /// conversation cache. Kept separate from the test-only pure search helper so existing
1497 /// pure catalog tests and compatibility callers do not need an engine session.
1498 pub(crate) fn execute_tool_search_with_cache(
1499 tool_name: &str,
1500 input: &serde_json::Value,
1501 catalog: &[Tool],
1502 active_tools: &mut HashSet<String>,
1503 cache: &mut crate::core::session::ToolActivationCache,
1504 ) -> Result<ToolResult, ToolError> {
1505 execute_tool_search_inner(tool_name, input, catalog, active_tools, Some(cache))
1506 }
1507
1508 fn execute_tool_search_inner(
1509 tool_name: &str,
1510 input: &serde_json::Value,
1511 catalog: &[Tool],
1512 active_tools: &mut HashSet<String>,
1513 cache: Option<&mut crate::core::session::ToolActivationCache>,
1514 ) -> Result<ToolResult, ToolError> {
1515 let query = required_str(input, "query")?;
1516 let match_kind = match tool_name {
1517 LEGACY_TOOL_SEARCH_REGEX_NAME => "regex",
1518 LEGACY_TOOL_SEARCH_BM25_NAME => "bm25",
1519 _ => optional_str(input, "match")?.unwrap_or("bm25"),
1520 };
1521 if !matches!(match_kind, "bm25" | "regex") {
1522 return Err(ToolError::invalid_input(format!(
1523 "Unsupported match algorithm '{match_kind}'. Expected one of: bm25, regex"
1524 )));
1525 }
1526 let max_results = usize::try_from(optional_u64(
1527 input,
1528 "max_results",
1529 TOOL_SEARCH_DEFAULT_MAX_RESULTS as u64,
1530 )?)
1531 .unwrap_or(TOOL_SEARCH_DEFAULT_MAX_RESULTS)
1532 .clamp(1, TOOL_SEARCH_MAX_RESULTS_LIMIT);
1533 let mut discovered = if match_kind == "regex" {
1534 discover_tools_with_regex(catalog, query, max_results)?
1535 } else {
1536 discover_tools_with_bm25_like(catalog, query, max_results)
1537 };
1538 let remaining_results = max_results.saturating_sub(discovered.len());
1539 let unavailable = if match_kind == "regex" {
1540 unavailable_core_action_tools_with_regex(catalog, query, remaining_results)?
1541 } else {
1542 unavailable_core_action_tools_with_bm25_like(catalog, query, remaining_results)
1543 };
1544
1545 let mut cache_rejected = Vec::new();
1546 if let Some(cache) = cache {
1547 let delta = cache.activate(catalog, &discovered);
1548 remove_evicted_cache_activations(catalog, active_tools, delta.evicted);
1549 cache_rejected = delta.rejected;
1550 discovered = delta.admitted;
1551 }
1552 for name in &discovered {
1553 active_tools.insert(name.clone());
1554 }
1555
1556 let references = discovered
1557 .iter()
1558 .map(|name| json!({"type": "tool_reference", "tool_name": name}))
1559 .collect::<Vec<_>>();
1560 let mut unavailable_references = unavailable
1561 .iter()
1562 .map(|fallback| {
1563 json!({
1564 "type": "unavailable_tool_reference",
1565 "tool_name": fallback.name,
1566 "reason": fallback.unavailable_reason,
1567 })
1568 })
1569 .collect::<Vec<_>>();
1570 unavailable_references.extend(cache_rejected.iter().map(|name| {
1571 json!({
1572 "type": "unavailable_tool_reference",
1573 "tool_name": name,
1574 "reason": "The tool schema exceeds the bounded conversation toolbox (8 cached tools, 16KiB of added serialized schemas). Narrow the search or use a smaller matching tool."
1575 })
1576 }));
1577
1578 let payload = json!({
1579 "type": "tool_search_tool_search_result",
1580 "tool_references": references,
1581 "unavailable_tool_references": unavailable_references.clone(),
1582 });
1583
1584 Ok(ToolResult {
1585 content: serde_json::to_string(&payload).unwrap_or_else(|_| payload.to_string()),
1586 success: true,
1587 metadata: Some(json!({
1588 "tool_references": discovered,
1589 "unavailable_tool_references": unavailable_references,
1590 })),
1591 })
1592 }
1593
1594 /// Describe-only `tool_search` for `execute_tools` programs: the same
1595 /// ranking as a direct search, answered with each match's name, description
1596 /// and input schema — and no activation, so the request's tool array and
1597 /// the session-pinned prefix stay exactly as they were.
1598 pub(super) fn describe_tools_for_program(
1599 input: &serde_json::Value,
1600 catalog: &[Tool],
1601 ) -> Result<ToolResult, ToolError> {
1602 let query = required_str(input, "query")?;
1603 let match_kind = optional_str(input, "match")?.unwrap_or("bm25");
1604 let max_results = usize::try_from(optional_u64(
1605 input,
1606 "max_results",
1607 TOOL_SEARCH_DEFAULT_MAX_RESULTS as u64,
1608 )?)
1609 .unwrap_or(TOOL_SEARCH_DEFAULT_MAX_RESULTS)
1610 .clamp(1, TOOL_SEARCH_MAX_RESULTS_LIMIT);
1611 let names = match match_kind {
1612 "regex" => discover_tools_with_regex(catalog, query, max_results)?,
1613 "bm25" => discover_tools_with_bm25_like(catalog, query, max_results),
1614 other => {
1615 return Err(ToolError::invalid_input(format!(
1616 "Unsupported match algorithm '{other}'. Expected one of: bm25, regex"
1617 )));
1618 }
1619 };
1620 let tools = names
1621 .iter()
1622 .filter_map(|name| catalog.iter().find(|tool| &tool.name == name))
1623 .map(|tool| {
1624 json!({
1625 "name": tool.name,
1626 "description": tool.description,
1627 "input_schema": tool.input_schema,
1628 })
1629 })
1630 .collect::<Vec<_>>();
1631 ToolResult::json(&json!({ "tools": tools }))
1632 .map_err(|error| ToolError::execution_failed(error.to_string()))
1633 }
1634
1635 /// Runs under the effective per-call policy, including an explicitly approved
1636 /// elevation. Shared-launcher platform limits still apply; this is not a
1637 /// persistent REPL. Code arrives on stdin so sandbox-private /tmp is harmless.
1638 pub(super) async fn execute_code_execution_tool(
1639 input: &serde_json::Value,
1640 workspace: &Path,
1641 context: &ToolContext,
1642 ) -> Result<ToolResult, ToolError> {
1643 let code = required_str(input, "code")?;
1644 let interpreter = crate::dependencies::Python::resolve().ok_or_else(|| {
1645 ToolError::execution_failed("code_execution: Python interpreter became unavailable")
1646 })?;
1647 let (program, mut args) = crate::dependencies::split_interpreter_spec(&interpreter);
1648 args.push("-".to_string());
1649 let budget = Duration::from_secs(120);
1650 let mut cmd =
1651 crate::tools::shell::sandboxed_runner_command(context, &program, args, workspace, budget)?;
1652 // Match the UTF-8 decoder below, including Windows Python's piped output.
1653 cmd.env("PYTHONIOENCODING", "utf-8");
1654 let output = tokio::time::timeout(
1655 budget,
1656 crate::process_tree::contained_output_with_input(&mut cmd, code.as_bytes().to_vec()),
1657 )
1658 .await
1659 .map_err(|_| ToolError::Timeout {
1660 seconds: budget.as_secs(),
1661 })
1662 .and_then(|res| res.map_err(|e| ToolError::execution_failed(e.to_string())))?;
1663
1664 let stdout = String::from_utf8_lossy(&output.stdout).to_string();
1665 let stderr = String::from_utf8_lossy(&output.stderr).to_string();
1666 let return_code = output.status.code().unwrap_or(-1);
1667 let success = output.status.success();
1668 let payload = json!({
1669 "type": "code_execution_result",
1670 "stdout": stdout,
1671 "stderr": stderr,
1672 "return_code": return_code,
1673 "content": [],
1674 });
1675
1676 Ok(ToolResult {
1677 content: serde_json::to_string(&payload).unwrap_or_else(|_| payload.to_string()),
1678 success,
1679 metadata: Some(payload),
1680 })
1681 }
1682
1683 #[cfg(test)]
1684 #[path = "tool_catalog/tests.rs"]
1685 mod synthetic_name_tests;
1686
1686 lines RUST