返回 CodeWhale
rlm.rs
根目录 / crates / tui / src / tools / rlm.rs
1 //! Compatibility persistent-RLM session tools.
2 //!
3 //! v0.8.33 replaces the old one-shot `rlm` tool with a head/hands surface:
4 //! `rlm_open` creates a named Python kernel over a large context,
5 //! `rlm_eval` runs bounded probes against it, `rlm_configure` adjusts runtime
6 //! feedback, and `rlm_close` tears it down.
7 //!
8 //! The normal Agent path now owns one session-persistent `repl` kernel. This
9 //! action-shaped surface stays registered for explicit compatibility and saved
10 //! transcript replay, but is hidden from new model turns. Its `rlm_*` aliases
11 //! force the action so old transcripts replay correctly — the pattern
12 //! `BashTool` established for `exec_shell*` in #4625.
13
14 use std::sync::Arc;
15 use std::time::{Duration, Instant};
16
17 use async_trait::async_trait;
18 use serde_json::{Value, json};
19
20 use crate::repl::PythonRuntime;
21 use crate::rlm::RlmBridge;
22 use crate::rlm::session::{
23 ContextMeta, OutputFeedback, RlmSession, derive_session_name, write_context_file,
24 };
25 use crate::tools::fetch_url::FetchUrlTool;
26 use crate::tools::handle::VarHandle;
27 use crate::tools::spec::{
28 ApprovalRequirement, ToolCapability, ToolContext, ToolError, ToolResult, ToolSpec,
29 };
30
31 /// Registered name of the persistent RLM session tool.
32 pub(crate) const RLM_TOOL_NAME: &str = "rlm";
33 const MAX_INLINE_CONTENT_CHARS: usize = 200_000;
34 const FULL_STDOUT_HEAD_CHARS: usize = 4_096;
35 const FULL_STDOUT_TAIL_CHARS: usize = 1_024;
36
37 /// When `rlm_eval` stdout exceeds this many characters the full body is
38 /// stored as a `var_handle` instead of inlined into the parent transcript.
39 /// The model retrieves the body via `handle_read` using the returned handle.
40 const STDOUT_HANDLE_THRESHOLD_CHARS: usize = 1_000;
41 const HARD_SUB_RLM_DEPTH_CAP: u32 = 3;
42
43 const ALL_ACTIONS: &[&str] = &["session_objects", "open", "eval", "configure", "close"];
44
45 fn rlm_kernel_error_result(
46 error: &str,
47 elapsed: Duration,
48 usage_batch: &crate::cost_status::RuntimeUsageBatch,
49 nested_events: &[Value],
50 ) -> ToolResult {
51 let mut metadata = json!({
52 // The registered tool is `rlm`; `eval` is its action. Naming a
53 // retired `rlm_eval` tool here taught the model a call it cannot
54 // make (2026-08-04 audit).
55 "tool": "rlm",
56 "action": "eval",
57 "duration_ms": elapsed.as_millis() as u64,
58 "kernel_error": true,
59 });
60 crate::cost_status::attach_child_usage_batch_metadata(&mut metadata, usage_batch);
61 ToolResult::error(
62 json!({
63 "tool": "rlm", "action": "eval", "error": error,
64 "nested_events": nested_events,
65 })
66 .to_string(),
67 )
68 .with_metadata(metadata)
69 }
70
71 /// Unified RLM session tool.
72 ///
73 /// One input schema and the existing per-session Python store. Provider RPCs
74 /// require the caller receipt attached by Core to the actual ToolContext.
75 pub struct RlmTool {
76 name: &'static str,
77 forced_action: Option<&'static str>,
78 }
79
80 impl RlmTool {
81 #[must_use]
82 pub fn new(name: &'static str) -> Self {
83 Self {
84 name,
85 forced_action: None,
86 }
87 }
88
89 #[cfg(test)]
90 #[must_use]
91 pub fn alias(name: &'static str, action: &'static str) -> Self {
92 Self {
93 name,
94 forced_action: Some(action),
95 }
96 }
97
98 fn resolve_action<'a>(&'a self, input: &'a Value) -> Result<&'a str, ToolError> {
99 let action = match self.forced_action {
100 Some(action) => action,
101 None => input.get("action").and_then(Value::as_str).ok_or_else(|| {
102 ToolError::invalid_input(format!(
103 "rlm: missing `action` (one of: {})",
104 ALL_ACTIONS.join(", ")
105 ))
106 })?,
107 };
108 if ALL_ACTIONS.contains(&action) {
109 Ok(action)
110 } else {
111 Err(ToolError::invalid_input(format!(
112 "rlm: invalid action `{action}` (one of: {})",
113 ALL_ACTIONS.join(", ")
114 )))
115 }
116 }
117
118 /// Without concrete input, open may fetch a URL. Input-specific approval
119 /// below keeps local/inline reads automatic and inherits fetch_url's
120 /// outbound-payload approval for URL sources.
121 fn action_requires_approval(action: &str) -> bool {
122 matches!(action, "eval" | "open")
123 }
124
125 /// Mirror of the legacy per-tool read-only contract (capability-derived):
126 /// `rlm_open` carries `ExecutesCode`, so only session_objects / configure /
127 /// close counted as read-only.
128 fn action_is_read_only(action: &str) -> bool {
129 matches!(action, "session_objects" | "configure" | "close")
130 }
131
132 fn action_capabilities(action: &str) -> Vec<ToolCapability> {
133 match action {
134 "session_objects" => vec![ToolCapability::ReadOnly],
135 "open" => vec![
136 ToolCapability::ReadOnly,
137 ToolCapability::Network,
138 ToolCapability::ExecutesCode,
139 ToolCapability::RequiresApproval,
140 ],
141 "eval" => vec![
142 ToolCapability::Network,
143 ToolCapability::ExecutesCode,
144 ToolCapability::RequiresApproval,
145 ],
146 // configure / close
147 _ => vec![ToolCapability::ReadOnly],
148 }
149 }
150 }
151
152 #[async_trait]
153 impl ToolSpec for RlmTool {
154 fn name(&self) -> &'static str {
155 self.name
156 }
157
158 fn model_visible(&self) -> bool {
159 // The normal Agent path owns a session-scoped `repl` kernel. Keep the
160 // old action fan-out registered for replay and explicit compatibility,
161 // but do not teach a second RLM workflow to new model turns.
162 false
163 }
164
165 fn description(&self) -> &'static str {
166 match self.forced_action {
167 Some("session_objects") => {
168 "List active prompt/history/session symbolic objects as compact cards. \
169 Pass one of the returned `id` values to `rlm_open` as \
170 `session_object` to inspect it inside an RLM REPL without copying the \
171 full prompt or transcript into the parent context."
172 }
173 Some("open") => {
174 "Open a persistent RLM context. Loads `file_path`, `content`, `url`, \
175 or `session_object` into a named Python kernel and returns only \
176 metadata: name, length, preview, and sha256. Use this for large or \
177 unfamiliar inputs so the parent transcript holds a handle, not the \
178 body."
179 }
180 Some("eval") => {
181 "Run one Python REPL block against a named RLM context. Returns a \
182 bounded projection of stdout/stderr plus metadata. If the code calls \
183 FINAL/finalize, the final value is stored as a var_handle retrievable \
184 with handle_read (if `handle_read` is not in your tool list, load it with \
185 `tool_search` first) instead of copied unbounded into the parent context. \
186 Large stdout/stderr payloads (>1k chars) are also stored as \
187 var_handles (returned in stdout_handle / stderr_handle) to keep the \
188 parent transcript lean. Batch child helpers require \
189 dependency_mode='independent'; use sub_query_sequence or a \
190 sequential loop for dependent work."
191 }
192 Some("configure") => {
193 "Configure a named RLM context: output feedback, child query timeout, \
194 recursive sub-RLM depth, and explicit session sharing."
195 }
196 Some("close") => {
197 "Close a named RLM context, tear down its Python kernel, and return \
198 usage/lifecycle metadata."
199 }
200 _ => {
201 "Persistent RLM sessions over large contexts. Actions: \"session_objects\" \
202 (list active prompt/history/session symbolic objects as compact cards), \
203 \"open\" (load file_path/content/url/session_object into a named Python \
204 kernel; returns only metadata so the parent transcript holds a handle, \
205 not the body), \"eval\" (run one bounded Python REPL block against a \
206 named context; approval required; FINAL/finalize values and large \
207 stdout/stderr become var_handles retrievable with handle_read; if \
208 `handle_read` is not in your tool list, load it with `tool_search` first), \
209 \"configure\" (output feedback, child timeout, sub-RLM depth, session \
210 sharing), \"close\" (tear down the kernel and return usage metadata)."
211 }
212 }
213 }
214
215 fn input_schema(&self) -> Value {
216 if let Some(action) = self.forced_action {
217 return legacy_action_schema(action);
218 }
219 json!({
220 "type": "object",
221 "properties": {
222 "action": {
223 "type": "string",
224 "enum": ALL_ACTIONS,
225 "description": "Action to perform."
226 },
227 "name": {
228 "type": "string",
229 "description": "RLM context name, unique within this parent session (action=open: optional, defaults to a slug from the source). Required for action=eval/configure/close."
230 },
231 "file_path": {
232 "type": "string",
233 "description": "Workspace-relative file to load (action=open; exactly one of file_path/content/url/session_object)."
234 },
235 "content": {
236 "type": "string",
237 "description": "Inline content to load. Capped at 200k chars. (action=open)"
238 },
239 "url": {
240 "type": "string",
241 "description": "HTTP/HTTPS URL to fetch (through the same path as Web action=\"fetch\") and load. (action=open)"
242 },
243 "session_object": {
244 "type": "string",
245 "description": "Stable symbolic active-session ref from action=session_objects, for example session://active/system_prompt or session://active/messages/0. (action=open)"
246 },
247 "code": {
248 "type": "string",
249 "description": "Raw Python executed against the context (no markdown fences). The loaded source is in scope as `content`; call FINAL(value)/finalize(...) to return a result handle. Example: print(len(content)). (action=eval)"
250 },
251 "output_feedback": {
252 "type": "string",
253 "enum": ["full", "metadata"],
254 "description": "(action=configure)"
255 },
256 "sub_query_timeout_secs": {
257 "type": "integer",
258 "description": "(action=configure)"
259 },
260 "sub_rlm_max_depth": {
261 "type": "integer",
262 "minimum": 0,
263 "maximum": 3,
264 "description": "(action=configure)"
265 },
266 "share_session": {
267 "type": "boolean",
268 "description": "(action=configure)"
269 }
270 },
271 "additionalProperties": false
272 })
273 }
274
275 fn capabilities(&self) -> Vec<ToolCapability> {
276 match self.forced_action {
277 Some(action) => Self::action_capabilities(action),
278 None => vec![
279 ToolCapability::Network,
280 ToolCapability::ExecutesCode,
281 ToolCapability::RequiresApproval,
282 ],
283 }
284 }
285
286 fn approval_requirement(&self) -> ApprovalRequirement {
287 match self.forced_action {
288 Some(action) if Self::action_requires_approval(action) => ApprovalRequirement::Required,
289 Some(_) => ApprovalRequirement::Auto,
290 None => ApprovalRequirement::Required,
291 }
292 }
293
294 fn approval_requirement_for(&self, input: &Value) -> ApprovalRequirement {
295 match self.resolve_action(input) {
296 Ok("open") if rlm_open_source_field(input, "url").is_some() => {
297 FetchUrlTool.approval_requirement_for(&json!({"url": input["url"]}))
298 }
299 Ok("open") => ApprovalRequirement::Auto,
300 Ok(action) if Self::action_requires_approval(action) => ApprovalRequirement::Required,
301 Ok(_) => ApprovalRequirement::Auto,
302 Err(_) => self.approval_requirement(),
303 }
304 }
305
306 fn is_read_only_for(&self, input: &Value) -> bool {
307 match self.resolve_action(input) {
308 Ok(action) => Self::action_is_read_only(action),
309 Err(_) => self.is_read_only(),
310 }
311 }
312
313 fn supports_parallel(&self) -> bool {
314 matches!(self.forced_action, Some("session_objects"))
315 }
316
317 fn supports_parallel_for(&self, input: &Value) -> bool {
318 matches!(self.resolve_action(input), Ok("session_objects"))
319 }
320
321 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
322 match self.resolve_action(&input)? {
323 "session_objects" => self.execute_session_objects(context).await,
324 "open" => self.execute_open(&input, context).await,
325 "eval" => self.execute_eval(&input, context).await,
326 "configure" => self.execute_configure(&input, context).await,
327 "close" => self.execute_close(&input, context).await,
328 action => Err(ToolError::invalid_input(format!(
329 "rlm: invalid action `{action}`"
330 ))),
331 }
332 }
333 }
334
335 impl RlmTool {
336 async fn execute_session_objects(
337 &self,
338 context: &ToolContext,
339 ) -> Result<ToolResult, ToolError> {
340 let snapshot = context.session_objects.as_ref().ok_or_else(|| {
341 ToolError::not_available("rlm_session_objects: active session snapshot unavailable")
342 })?;
343 ToolResult::json(&json!({
344 "objects": snapshot.object_cards(),
345 "open_with": {
346 "tool": "rlm",
347 "action": "open",
348 "field": "session_object",
349 "example": {
350 "name": "active_prompt",
351 "session_object": "session://active/system_prompt"
352 }
353 },
354 "redaction": format!(
355 "Large tool results and thinking blocks are represented by compact metadata in transcript objects; use returned handles and handle_read for bounded payload projections ({}).",
356 crate::tools::handle::HANDLE_READ_ACTIVATION_HINT
357 )
358 }))
359 .map_err(|e| ToolError::execution_failed(e.to_string()))
360 }
361
362 async fn execute_open(
363 &self,
364 input: &Value,
365 context: &ToolContext,
366 ) -> Result<ToolResult, ToolError> {
367 let deadline = context.turn_deadline.unwrap_or_else(|| {
368 tokio::time::Instant::now() + crate::tools::subagent::DEFAULT_CHILD_WALL_TIME
369 });
370 if tokio::time::Instant::now() >= deadline {
371 return Err(ToolError::execution_failed(
372 "RLM parent turn deadline exhausted before opening context",
373 ));
374 }
375 let open = async {
376 let source_count = rlm_open_source_count(input);
377 if source_count != 1 {
378 let mut msg = String::from(
379 "rlm_open: provide exactly one of `file_path` (local file), `content` (inline text), `url`, or `session_object`",
380 );
381 // "did you mean" for common misnamings (#2655).
382 if let Some(obj) = input.as_object() {
383 let seen: Vec<&str> = [
384 "prompt",
385 "resident_file",
386 "text",
387 "body",
388 "path",
389 "file",
390 "source",
391 ]
392 .into_iter()
393 .filter(|k| obj.contains_key(*k))
394 .collect();
395 if !seen.is_empty() {
396 msg.push_str(&format!(
397 ". Saw {seen:?} — did you mean file_path/content/url/session_object? (to evaluate against an existing context, pass its name to rlm action='eval', or use `session_object`)"
398 ));
399 }
400 }
401 return Err(ToolError::invalid_input(msg));
402 }
403
404 let (body, source_type, source_hint) = load_source(input, context).await?;
405 if body.trim().is_empty() {
406 return Err(ToolError::invalid_input(
407 "rlm_open: input is empty after loading",
408 ));
409 }
410
411 let name = input
412 .get("name")
413 .and_then(Value::as_str)
414 .map(str::trim)
415 .filter(|name| !name.is_empty())
416 .map(ToOwned::to_owned)
417 .unwrap_or_else(|| derive_session_name(source_hint.as_deref()));
418
419 {
420 let sessions = context.runtime.rlm_sessions.lock().await;
421 if sessions.contains_key(&name) {
422 return Err(ToolError::invalid_input(format!(
423 "rlm_open: context name `{name}` already exists"
424 )));
425 }
426 }
427
428 let context_path = write_context_file(&body).map_err(|e| {
429 ToolError::execution_failed(format!("rlm_open: failed to stage context: {e}"))
430 })?;
431 let context_path = tempfile::TempPath::try_from_path(context_path).map_err(|e| {
432 ToolError::execution_failed(format!("rlm_open: failed to own staged context: {e}"))
433 })?;
434 let kernel = PythonRuntime::spawn_with_context(&context_path)
435 .await
436 .map_err(|e| ToolError::execution_failed(format!("rlm_open: {e}")))?;
437 // PythonRuntime now owns cleanup; a dropped startup kept the TempPath.
438 let context_path = context_path.keep().map_err(|e| {
439 ToolError::execution_failed(format!("rlm_open: context handoff failed: {e}"))
440 })?;
441 let context_meta = ContextMeta::from_body(&body, source_type);
442 let session = RlmSession::new(name.clone(), kernel, context_meta.clone(), context_path);
443 let id = session.id.clone();
444
445 let mut sessions = context.runtime.rlm_sessions.lock().await;
446 sessions.insert(name.clone(), Arc::new(tokio::sync::Mutex::new(session)));
447
448 ToolResult::json(&json!({
449 "name": name,
450 "id": id,
451 "length": context_meta.length,
452 "type": context_meta.type_name,
453 "preview_500": context_meta.preview_500,
454 "sha256": context_meta.sha256,
455 }))
456 .map_err(|e| ToolError::execution_failed(e.to_string()))
457 };
458 tokio::select! {
459 biased;
460 () = async {
461 if let Some(cancel) = context.cancel_token.as_ref() {
462 cancel.cancelled().await;
463 } else {
464 std::future::pending::<()>().await;
465 }
466 } => Err(ToolError::cancelled("RLM originating turn cancelled while opening context")),
467 result = tokio::time::timeout_at(deadline, open) => result.map_err(|_| {
468 ToolError::execution_failed("RLM parent turn deadline exhausted opening context")
469 })?,
470 }
471 }
472
473 async fn execute_eval(
474 &self,
475 input: &Value,
476 context: &ToolContext,
477 ) -> Result<ToolResult, ToolError> {
478 crate::core::engine::tool_catalog::enforce_tool_denial(context, "rlm_eval", input)?;
479 let name = required_non_empty_str(input, "name")?;
480 let code = required_non_empty_str(input, "code").map_err(|_| {
481 ToolError::invalid_input(
482 "rlm_eval: `code` is required and runs raw Python against the RLM context (no markdown fences). \
483 Example: {\"name\": \"<ctx>\", \"code\": \"print(len(content))\"}; call FINAL(value) to return a result handle.",
484 )
485 })?;
486 let deadline = context.turn_deadline.unwrap_or_else(|| {
487 tokio::time::Instant::now() + crate::tools::subagent::DEFAULT_CHILD_WALL_TIME
488 });
489 if tokio::time::Instant::now() >= deadline {
490 return Err(ToolError::execution_failed(
491 "RLM parent turn deadline exhausted before execution",
492 ));
493 }
494 let original_cancel = context.cancel_token.clone();
495 let wait_for_cancel = || async {
496 if let Some(cancel) = original_cancel.as_ref() {
497 cancel.cancelled().await;
498 } else {
499 std::future::pending::<()>().await;
500 }
501 };
502 let mut session = tokio::select! {
503 biased;
504 () = wait_for_cancel() => return Err(ToolError::cancelled("RLM originating turn cancelled while waiting for context")),
505 result = tokio::time::timeout_at(deadline, async {
506 let session = get_session(context, name).await?;
507 Ok::<_, ToolError>(session.lock_owned().await)
508 }) => result.map_err(|_| {
509 ToolError::execution_failed("RLM parent turn deadline exhausted waiting for context")
510 })??,
511 };
512 let config = session.config.clone();
513
514 if let Some(caller) = context.rlm_caller.as_deref() {
515 caller.validate_context(context)?;
516 }
517 // The active round owns the existing interpreter. Dropping this
518 // future kills unknown running code rather than leaving it in the
519 // persistent map; only a completed round restores the same kernel.
520 let Some(mut kernel) = session.kernel.take() else {
521 return Err(ToolError::invalid_input(format!(
522 "rlm_eval: context `{name}` is closed"
523 )));
524 };
525
526 let started = Instant::now();
527 let (round, child_usage_batch, nested_events) = if let Some(caller) =
528 context.rlm_caller.as_deref()
529 {
530 let bridge = RlmBridge::new(
531 caller,
532 config.sub_rlm_max_depth.min(HARD_SUB_RLM_DEPTH_CAP),
533 Duration::from_secs(config.sub_query_timeout_secs),
534 )
535 .with_deadline(Some(deadline))
536 .with_gate(context.execution.nested_call_gate.clone());
537 let round_result = tokio::select! {
538 biased;
539 () = wait_for_cancel() => Err("RLM evaluation cancelled by its originating turn".into()),
540 result = tokio::time::timeout_at(bridge.deadline(), kernel.run(code, Some(&bridge))) => {
541 result.unwrap_or_else(|_| Err("RLM evaluation reached the parent turn deadline".into()))
542 },
543 };
544 let usage = bridge.usage_snapshot().await;
545 let round = match round_result {
546 Ok(round) => round,
547 Err(error) => {
548 // A bridge request may have completed and accrued usage
549 // before the Python kernel times out or closes stdout.
550 // Return a failed ToolResult (rather than a bare ToolError)
551 // so ToolCallComplete still carries the immutable child
552 // receipt and the runtime can durably account for it.
553 // Cancellation may leave Python running. Discard the
554 // owned interpreter rather than reusing unknown state.
555 session.kernel = None;
556 session.last_used_at = Instant::now();
557 return Ok(rlm_kernel_error_result(
558 &error.to_string(),
559 started.elapsed(),
560 &crate::cost_status::RuntimeUsageBatch {
561 decisions: Vec::new(),
562 records: usage.records,
563 drop_records: usage.drop_records,
564 dropped_records: usage.dropped_records,
565 },
566 &usage.nested_events,
567 ));
568 }
569 };
570 (
571 round,
572 crate::cost_status::RuntimeUsageBatch {
573 decisions: Vec::new(),
574 records: usage.records,
575 drop_records: usage.drop_records,
576 dropped_records: usage.dropped_records,
577 },
578 usage.nested_events,
579 )
580 } else {
581 let round = tokio::select! {
582 biased;
583 () = wait_for_cancel() => return Err(ToolError::cancelled("RLM evaluation cancelled by its originating turn")),
584 result = tokio::time::timeout_at(deadline, kernel.run(code, None::<&RlmBridge<'_>>)) => {
585 match result {
586 Ok(result) => result.map_err(|e| ToolError::execution_failed(format!("rlm_eval: {e}")))?,
587 Err(_) => return Err(ToolError::execution_failed("RLM evaluation reached the parent turn deadline")),
588 }
589 },
590 };
591 (
592 round,
593 crate::cost_status::RuntimeUsageBatch::default(),
594 Vec::new(),
595 )
596 };
597
598 session.kernel = Some(kernel);
599 session.rpc_count = session.rpc_count.saturating_add(round.rpc_count);
600 session.total_duration += round.elapsed;
601 session.last_used_at = Instant::now();
602
603 let final_handle = if let Some(value_json) = round.final_json.clone() {
604 session.final_count = session.final_count.saturating_add(1);
605 let handle_name = format!("final_{}", session.final_count);
606 let handle = {
607 let mut store = context.runtime.handle_store.lock().await;
608 match value_json {
609 Value::String(value) => {
610 store.insert_text(session.id.clone(), handle_name, value)
611 }
612 other => store.insert_json(session.id.clone(), handle_name, other),
613 }
614 };
615 Some(handle)
616 } else {
617 None
618 };
619
620 let had_error = round.has_error;
621 let rpc_count = round.rpc_count;
622 let duration_ms = round.elapsed.as_millis() as u64;
623 // Route large stdout/stderr into a var_handle to avoid bloat in
624 // the parent transcript. The model calls handle_read for bounded
625 // projections; a short inline note describes availability.
626 fn route_output(
627 text: &str,
628 feedback: &OutputFeedback,
629 store: &mut crate::tools::handle::HandleStore,
630 session_id: &str,
631 tag: &str,
632 ) -> (Option<String>, Option<crate::tools::handle::VarHandle>) {
633 let threshold = STDOUT_HANDLE_THRESHOLD_CHARS;
634 match (feedback, text.len()) {
635 (OutputFeedback::Full, len) if len <= threshold => {
636 (Some(preview_output(text)), None)
637 }
638 (OutputFeedback::Full, _) if !text.trim().is_empty() => {
639 // Store full body as a handle for out-of-band retrieval
640 let name = format!("{tag}_{}", 0); // single counter is fine
641 let handle = store.insert_text(session_id, name, text);
642 (
643 Some(format!(
644 "{} chars; retrieve via handle_read ({})",
645 text.len(),
646 crate::tools::handle::HANDLE_READ_ACTIVATION_HINT
647 )),
648 Some(handle),
649 )
650 }
651 _ => (None, None),
652 }
653 }
654
655 let (stdout_preview, stdout_handle) = route_output(
656 &round.full_stdout,
657 &config.output_feedback,
658 &mut *context.runtime.handle_store.lock().await,
659 &session.id,
660 "stdout",
661 );
662 let (stderr_preview, stderr_handle) = route_output(
663 &round.stderr,
664 &config.output_feedback,
665 &mut *context.runtime.handle_store.lock().await,
666 &session.id,
667 "stderr",
668 );
669
670 let mut output = json!({
671 "name": session.name,
672 "id": session.id,
673 "duration_ms": duration_ms,
674 "rpc_count": rpc_count,
675 "had_error": had_error,
676 "new_vars": [],
677 "nested_events": nested_events,
678 "final": final_handle,
679 });
680 if let Some(ref stdout_preview) = stdout_preview {
681 output["stdout_preview"] = json!(stdout_preview);
682 }
683 if let Some(ref stderr_preview) = stderr_preview {
684 output["stderr_preview"] = json!(stderr_preview);
685 }
686 if let (Some(h), Some(_)) = (stdout_handle, &stdout_preview) {
687 output["stdout_handle"] = json!(h);
688 }
689 if let (Some(h), Some(_)) = (stderr_handle, &stderr_preview) {
690 output["stderr_handle"] = json!(h);
691 }
692 if let Some(confidence) = round.final_confidence.clone() {
693 output["confidence"] = confidence;
694 }
695
696 let mut metadata = json!({
697 "tool": "rlm_eval",
698 "duration_ms": started.elapsed().as_millis() as u64,
699 });
700 // Every RLM provider call keeps its own dispatch timestamp and frozen
701 // quote. The preferred batch format prevents a fan-out from being
702 // retroactively priced as one aggregate call on the first route.
703 crate::cost_status::attach_child_usage_batch_metadata(&mut metadata, &child_usage_batch);
704
705 Ok(ToolResult::json(&output)
706 .map_err(|e| ToolError::execution_failed(e.to_string()))?
707 .with_metadata(metadata))
708 }
709
710 async fn execute_configure(
711 &self,
712 input: &Value,
713 context: &ToolContext,
714 ) -> Result<ToolResult, ToolError> {
715 let name = required_non_empty_str(input, "name")?;
716 // No cross-session authority exists in this surface. Refuse before
717 // mutating any other setting, rather than retaining an inert true bit.
718 if input.get("share_session").and_then(Value::as_bool) == Some(true) {
719 return Err(ToolError::invalid_input(
720 "rlm_configure: share_session=true is unsupported; contexts remain caller-session scoped",
721 ));
722 }
723 let session = get_session(context, name).await?;
724 let mut session = session.lock().await;
725
726 if let Some(value) = input.get("output_feedback").and_then(Value::as_str) {
727 session.config.output_feedback = match value {
728 "full" => OutputFeedback::Full,
729 "metadata" => OutputFeedback::Metadata,
730 other => {
731 return Err(ToolError::invalid_input(format!(
732 "rlm_configure: invalid output_feedback `{other}`"
733 )));
734 }
735 };
736 }
737 if let Some(timeout) = input.get("sub_query_timeout_secs").and_then(Value::as_u64) {
738 session.config.sub_query_timeout_secs = timeout.clamp(1, 600);
739 }
740 if let Some(depth) = input.get("sub_rlm_max_depth").and_then(Value::as_u64) {
741 session.config.sub_rlm_max_depth = (depth as u32).min(HARD_SUB_RLM_DEPTH_CAP);
742 }
743 if input.get("share_session").and_then(Value::as_bool) == Some(false) {
744 session.config.share_session = false;
745 }
746
747 ToolResult::json(&json!({
748 "name": session.name,
749 "current_config": session.config,
750 }))
751 .map_err(|e| ToolError::execution_failed(e.to_string()))
752 }
753
754 async fn execute_close(
755 &self,
756 input: &Value,
757 context: &ToolContext,
758 ) -> Result<ToolResult, ToolError> {
759 let name = required_non_empty_str(input, "name")?;
760 let removed = {
761 let mut sessions = context.runtime.rlm_sessions.lock().await;
762 sessions.remove(name)
763 };
764 let Some(session) = removed else {
765 return Err(ToolError::invalid_input(format!(
766 "rlm_close: unknown context `{name}`"
767 )));
768 };
769
770 let mut session = session.lock().await;
771 let kernel = session.kernel.take();
772 let output = json!({
773 "name": session.name,
774 "id": session.id,
775 "rpc_count": session.rpc_count,
776 "total_duration_ms": session.total_duration.as_millis() as u64,
777 "peak_var_count": session.peak_var_count,
778 "created_ms_ago": session.created_at.elapsed().as_millis() as u64,
779 "context_path": session.context_path,
780 });
781 drop(session);
782
783 if let Some(kernel) = kernel {
784 kernel.shutdown().await;
785 }
786
787 ToolResult::json(&output).map_err(|e| ToolError::execution_failed(e.to_string()))
788 }
789 }
790
791 /// The exact schema the legacy per-action tool exposed, kept so hidden alias
792 /// registrations report an identical contract to the pre-unification tools.
793 fn legacy_action_schema(action: &str) -> Value {
794 match action {
795 "session_objects" => json!({
796 "type": "object",
797 "properties": {}
798 }),
799 "open" => json!({
800 "type": "object",
801 "properties": {
802 "name": {
803 "type": "string",
804 "description": "Caller-chosen context name, unique within this parent session. Defaults to a slug from the source."
805 },
806 "file_path": {
807 "type": "string",
808 "description": "Workspace-relative file to load."
809 },
810 "content": {
811 "type": "string",
812 "description": "Inline content to load. Capped at 200k chars."
813 },
814 "url": {
815 "type": "string",
816 "description": "HTTP/HTTPS URL to fetch (through the same path as Web action=\"fetch\") and load."
817 },
818 "session_object": {
819 "type": "string",
820 "description": "Stable symbolic active-session ref from rlm_session_objects, for example session://active/system_prompt or session://active/messages/0."
821 }
822 }
823 }),
824 "eval" => json!({
825 "type": "object",
826 "required": ["name", "code"],
827 "properties": {
828 "name": { "type": "string", "description": "RLM context name returned by rlm_open." },
829 "code": { "type": "string", "description": "Raw Python executed against the context (no markdown fences). The loaded source is in scope as `content`; call FINAL(value)/finalize(...) to return a result handle. Example: print(len(content))." }
830 }
831 }),
832 "configure" => json!({
833 "type": "object",
834 "required": ["name"],
835 "properties": {
836 "name": { "type": "string" },
837 "output_feedback": { "type": "string", "enum": ["full", "metadata"] },
838 "sub_query_timeout_secs": { "type": "integer" },
839 "sub_rlm_max_depth": { "type": "integer", "minimum": 0, "maximum": 3 },
840 "share_session": { "type": "boolean" }
841 }
842 }),
843 // close
844 _ => json!({
845 "type": "object",
846 "required": ["name"],
847 "properties": {
848 "name": { "type": "string", "description": "RLM context name from rlm_open." }
849 }
850 }),
851 }
852 }
853
854 async fn load_source(
855 input: &Value,
856 context: &ToolContext,
857 ) -> Result<(String, String, Option<String>), ToolError> {
858 if let Some(path) = rlm_open_source_field(input, "file_path").map(str::trim) {
859 let resolved = context.resolve_path(path)?;
860 let body = tokio::fs::read_to_string(&resolved).await.map_err(|e| {
861 ToolError::execution_failed(format!("rlm_open: read {}: {e}", resolved.display()))
862 })?;
863 return Ok((body, "file".to_string(), Some(path.to_string())));
864 }
865
866 if let Some(content) = rlm_open_source_field(input, "content") {
867 if content.chars().count() > MAX_INLINE_CONTENT_CHARS {
868 return Err(ToolError::invalid_input(format!(
869 "rlm_open: inline content is {} chars (cap {MAX_INLINE_CONTENT_CHARS})",
870 content.chars().count()
871 )));
872 }
873 return Ok((content.to_string(), "content".to_string(), None));
874 }
875
876 if let Some(object_ref) = rlm_open_source_field(input, "session_object") {
877 let snapshot = context.session_objects.as_ref().ok_or_else(|| {
878 ToolError::not_available("rlm_open: active session snapshot unavailable")
879 })?;
880 let object = snapshot.resolve(object_ref).ok_or_else(|| {
881 ToolError::invalid_input(format!("rlm_open: unknown session object `{object_ref}`"))
882 })?;
883 return Ok((
884 object.body,
885 format!("session_object:{}", object.kind),
886 Some(object.id),
887 ));
888 }
889
890 let url = rlm_open_source_field(input, "url")
891 .map(str::trim)
892 .ok_or_else(|| ToolError::invalid_input("rlm_open: missing source"))?;
893 crate::core::engine::tool_catalog::enforce_tool_denial(context, "fetch_url", input)?;
894 let result = FetchUrlTool
895 .execute(json!({"url": url, "format": "raw"}), context)
896 .await?;
897 let parsed: Value = serde_json::from_str(&result.content).map_err(|e| {
898 ToolError::execution_failed(format!("rlm_open: fetch_url returned invalid JSON: {e}"))
899 })?;
900 let body = parsed
901 .get("content")
902 .and_then(Value::as_str)
903 .ok_or_else(|| ToolError::execution_failed("rlm_open: fetched body missing content"))?
904 .to_string();
905 let source_type = parsed
906 .get("content_type")
907 .and_then(Value::as_str)
908 .unwrap_or("url")
909 .to_string();
910 Ok((body, source_type, Some(url.to_string())))
911 }
912
913 fn rlm_open_source_count(input: &Value) -> usize {
914 ["file_path", "content", "url", "session_object"]
915 .iter()
916 .filter(|field| rlm_open_source_field(input, field).is_some())
917 .count()
918 }
919
920 fn rlm_open_source_field<'a>(input: &'a Value, field: &str) -> Option<&'a str> {
921 input
922 .get(field)
923 .and_then(Value::as_str)
924 .filter(|value| !value.trim().is_empty())
925 }
926
927 async fn get_session(
928 context: &ToolContext,
929 name: &str,
930 ) -> Result<Arc<tokio::sync::Mutex<RlmSession>>, ToolError> {
931 let sessions = context.runtime.rlm_sessions.lock().await;
932 sessions.get(name).cloned().ok_or_else(|| {
933 ToolError::invalid_input(format!(
934 "unknown RLM context `{name}`; open it first with rlm action='open'"
935 ))
936 })
937 }
938
939 fn required_non_empty_str<'a>(input: &'a Value, field: &str) -> Result<&'a str, ToolError> {
940 let value = input
941 .get(field)
942 .and_then(Value::as_str)
943 .ok_or_else(|| ToolError::missing_field(field))?
944 .trim();
945 if value.is_empty() {
946 return Err(ToolError::invalid_input(format!(
947 "rlm: `{field}` must not be empty"
948 )));
949 }
950 Ok(value)
951 }
952
953 fn preview_output(text: &str) -> String {
954 let total = text.chars().count();
955 if total <= FULL_STDOUT_HEAD_CHARS + FULL_STDOUT_TAIL_CHARS {
956 return text.to_string();
957 }
958 let head: String = text.chars().take(FULL_STDOUT_HEAD_CHARS).collect();
959 let tail: String = text
960 .chars()
961 .skip(total.saturating_sub(FULL_STDOUT_TAIL_CHARS))
962 .collect();
963 format!(
964 "{head}\n... [{} chars truncated, retrieve via handle_read when returned as a handle; {}] ...\n{tail}",
965 total.saturating_sub(FULL_STDOUT_HEAD_CHARS + FULL_STDOUT_TAIL_CHARS),
966 crate::tools::handle::HANDLE_READ_ACTIVATION_HINT
967 )
968 }
969
970 fn _assert_var_handle_shape(_: Option<VarHandle>) {}
971
972 #[cfg(test)]
973 mod tests {
974 use super::*;
975 use crate::rlm::session::SessionObjectSnapshot;
976 use crate::tools::handle::HandleReadTool;
977 use crate::tools::spec::ToolContext;
978 use codewhale_models::Role;
979 use codewhale_models::{ContentBlock, Message, SystemPrompt};
980 use std::path::PathBuf;
981
982 /// #6747: runtime text pointing at the deferred `handle_read` teaches
983 /// its activation path and names no hidden tool.
984 #[test]
985 fn handle_read_pointers_teach_activation_path() {
986 let long = "x".repeat(FULL_STDOUT_HEAD_CHARS + FULL_STDOUT_TAIL_CHARS + 10);
987 let preview = preview_output(&long);
988 let preview_footer = preview
989 .lines()
990 .find(|line| line.contains("chars truncated"))
991 .expect("truncation footer");
992 crate::tools::canonical_action::tests::assert_text_names_only_callable_tools(
993 "rlm preview footer",
994 preview_footer,
995 );
996 assert!(preview_footer.contains("`tool_search`"));
997 }
998
999 fn ctx() -> ToolContext {
1000 ToolContext::new(".")
1001 }
1002
1003 fn ctx_with_session_objects() -> ToolContext {
1004 ToolContext::new(".").with_session_objects(SessionObjectSnapshot::new(
1005 "session-1".to_string(),
1006 "deepseek-v4-pro".to_string(),
1007 PathBuf::from("."),
1008 Some(SystemPrompt::Text("You are CodeWhale.".to_string())),
1009 vec![
1010 Message {
1011 role: Role::User,
1012 content: vec![ContentBlock::Text {
1013 text: "Please inspect the RLM surface.".to_string(),
1014 cache_control: None,
1015 }],
1016 },
1017 Message {
1018 role: Role::Assistant,
1019 content: vec![ContentBlock::Text {
1020 text: "I will use symbolic session objects.".to_string(),
1021 cache_control: None,
1022 }],
1023 },
1024 ],
1025 ))
1026 }
1027
1028 #[test]
1029 fn schema_uses_new_tool_names() {
1030 assert_eq!(
1031 RlmTool::alias("rlm_session_objects", "session_objects").name(),
1032 "rlm_session_objects"
1033 );
1034 assert_eq!(RlmTool::alias("rlm_open", "open").name(), "rlm_open");
1035 assert_eq!(RlmTool::alias("rlm_eval", "eval").name(), "rlm_eval");
1036 assert_eq!(
1037 RlmTool::alias("rlm_configure", "configure").name(),
1038 "rlm_configure"
1039 );
1040 assert_eq!(RlmTool::alias("rlm_close", "close").name(), "rlm_close");
1041 }
1042
1043 #[test]
1044 fn rlm_tool_is_compatibility_only_not_model_visible() {
1045 let canonical = RlmTool::new("rlm");
1046 assert!(!canonical.model_visible());
1047 assert_eq!(canonical.name(), "rlm");
1048 let actions = canonical.input_schema()["properties"]["action"]["enum"]
1049 .as_array()
1050 .expect("action enum")
1051 .clone();
1052 for action in ["session_objects", "open", "eval", "configure", "close"] {
1053 assert!(
1054 actions.iter().any(|value| value.as_str() == Some(action)),
1055 "canonical schema must offer action {action}"
1056 );
1057 }
1058
1059 for alias in [
1060 RlmTool::alias("rlm_session_objects", "session_objects"),
1061 RlmTool::alias("rlm_open", "open"),
1062 RlmTool::alias("rlm_eval", "eval"),
1063 RlmTool::alias("rlm_configure", "configure"),
1064 RlmTool::alias("rlm_close", "close"),
1065 ] {
1066 assert!(
1067 !alias.model_visible(),
1068 "compatibility alias {} must stay hidden",
1069 alias.name()
1070 );
1071 }
1072 }
1073
1074 #[test]
1075 fn kernel_failure_result_retains_child_usage_receipt() {
1076 let route = crate::cost_status::EffectiveRouteEnvelope::capture(
1077 None,
1078 crate::config::ProviderKind::Deepseek,
1079 crate::config::ProviderKind::Deepseek.as_str(),
1080 "deepseek-v4-flash",
1081 Some(
1082 crate::config::ProviderKind::Deepseek
1083 .provider()
1084 .default_base_url(),
1085 ),
1086 chrono::Utc::now(),
1087 );
1088 let usage = codewhale_models::Usage {
1089 input_tokens: 23,
1090 output_tokens: 5,
1091 reasoning_replay_tokens: Some(7),
1092 ..Default::default()
1093 };
1094 let record = crate::cost_status::RuntimeUsageRecord {
1095 source_id: "rlm:test:request:0".to_string(),
1096 usage: crate::cost_status::EffectiveRouteUsage {
1097 route: route.clone(),
1098 usage: usage.clone(),
1099 },
1100 };
1101 let drop_record = crate::cost_status::RuntimeUsageDropRecord {
1102 reason: crate::cost_status::RuntimeUsageMissingReason::default(),
1103 source_id: "rlm:test:request:1".to_string(),
1104 route: route.clone(),
1105 };
1106 let result = rlm_kernel_error_result(
1107 "kernel stdout closed",
1108 Duration::from_millis(11),
1109 &crate::cost_status::RuntimeUsageBatch {
1110 decisions: Vec::new(),
1111 records: vec![record],
1112 drop_records: vec![drop_record],
1113 dropped_records: 1,
1114 },
1115 &[],
1116 );
1117
1118 assert!(!result.success);
1119 let metadata = result
1120 .metadata
1121 .expect("usage metadata on failed tool result");
1122 let batch = crate::cost_status::child_usage_records_from_metadata(&metadata)
1123 .expect("preferred routed batch");
1124 assert_eq!(batch.dropped_records, 1);
1125 assert_eq!(batch.records.len(), 1);
1126 assert_eq!(batch.drop_records.len(), 1);
1127 assert_eq!(batch.records[0].usage.route, route);
1128 assert_eq!(batch.records[0].usage.usage, usage);
1129 assert_eq!(batch.drop_records[0].route, route);
1130 }
1131
1132 #[test]
1133 fn rlm_eval_requires_approval() {
1134 let tool = RlmTool::alias("rlm_eval", "eval");
1135 assert_eq!(tool.approval_requirement(), ApprovalRequirement::Required);
1136 assert!(
1137 tool.capabilities()
1138 .contains(&ToolCapability::RequiresApproval)
1139 );
1140
1141 // Evaluation requires approval; concrete open inputs are classified below.
1142 let canonical = RlmTool::new("rlm");
1143 assert_eq!(
1144 canonical.approval_requirement_for(&json!({"action": "eval"})),
1145 ApprovalRequirement::Required
1146 );
1147 assert_eq!(
1148 canonical.approval_requirement_for(&json!({"action": "open"})),
1149 ApprovalRequirement::Auto
1150 );
1151 assert_eq!(
1152 canonical.approval_requirement_for(&json!({"action": "session_objects"})),
1153 ApprovalRequirement::Auto
1154 );
1155 }
1156
1157 #[test]
1158 fn rlm_open_requires_outbound_approval_but_keeps_local_reads_automatic() {
1159 for tool in [RlmTool::new("rlm"), RlmTool::alias("rlm_open", "open")] {
1160 assert_eq!(tool.approval_requirement(), ApprovalRequirement::Required);
1161 for source in [
1162 json!({"url": "https://example.com/document"}),
1163 json!({"url": " https://example.com/document ", "content": ""}),
1164 ] {
1165 let mut input = source;
1166 input["action"] = json!("open");
1167 assert_eq!(
1168 tool.approval_requirement_for(&input),
1169 ApprovalRequirement::Required
1170 );
1171 }
1172 for source in [
1173 json!({"content": "local fixture"}),
1174 json!({"file_path": "fixture.txt"}),
1175 json!({"session_object": "fixture-object"}),
1176 json!({"content": "local fixture", "url": " "}),
1177 ] {
1178 let mut input = source;
1179 input["action"] = json!("open");
1180 assert_eq!(
1181 tool.approval_requirement_for(&input),
1182 ApprovalRequirement::Auto
1183 );
1184 }
1185 }
1186 }
1187
1188 #[test]
1189 fn read_only_and_parallel_flags_match_legacy_contract() {
1190 // Legacy: session_objects was parallel-friendly read-only; open carried
1191 // ExecutesCode (not read-only). Open now classifies the concrete source.
1192 let session_objects = RlmTool::alias("rlm_session_objects", "session_objects");
1193 assert!(session_objects.supports_parallel());
1194 assert!(session_objects.is_read_only_for(&json!({})));
1195
1196 let open = RlmTool::alias("rlm_open", "open");
1197 assert!(!open.is_read_only_for(&json!({})));
1198 assert_eq!(open.approval_requirement(), ApprovalRequirement::Required);
1199
1200 let canonical = RlmTool::new("rlm");
1201 assert!(canonical.supports_parallel_for(&json!({"action": "session_objects"})));
1202 assert!(!canonical.supports_parallel_for(&json!({"action": "eval"})));
1203 assert!(canonical.is_read_only_for(&json!({"action": "configure"})));
1204 assert!(!canonical.is_read_only_for(&json!({"action": "open"})));
1205 assert!(!canonical.is_read_only_for(&json!({"action": "eval"})));
1206 }
1207
1208 #[test]
1209 fn canonical_rejects_unknown_or_missing_action() {
1210 let tool = RlmTool::new("rlm");
1211 let err = tool
1212 .resolve_action(&json!({}))
1213 .expect_err("missing action must fail");
1214 assert!(err.to_string().contains("missing `action`"));
1215 let err = tool
1216 .resolve_action(&json!({"action": "explode"}))
1217 .expect_err("unknown action must fail");
1218 assert!(err.to_string().contains("invalid action"));
1219 }
1220
1221 #[test]
1222 fn rlm_open_source_count_ignores_empty_string_defaults() {
1223 assert_eq!(
1224 rlm_open_source_count(
1225 &json!({"name": "url-doc", "file_path": "", "content": "", "url": "https://example.com/doc"})
1226 ),
1227 1
1228 );
1229 assert_eq!(
1230 rlm_open_source_count(
1231 &json!({"name": "inline-doc", "file_path": "", "content": "body", "url": ""})
1232 ),
1233 1
1234 );
1235 assert_eq!(
1236 rlm_open_source_count(&json!({"content": "body", "url": "https://example.com/doc"})),
1237 2
1238 );
1239 assert_eq!(
1240 rlm_open_source_count(
1241 &json!({"content": "body", "session_object": "session://active/system_prompt"})
1242 ),
1243 2
1244 );
1245 }
1246
1247 #[tokio::test]
1248 async fn rlm_session_objects_lists_active_prompt_object() {
1249 let ctx = ctx_with_session_objects();
1250 let result = RlmTool::alias("rlm_session_objects", "session_objects")
1251 .execute(json!({}), &ctx)
1252 .await
1253 .expect("list session objects");
1254 let body: Value = serde_json::from_str(&result.content).expect("json");
1255 let objects = body["objects"].as_array().expect("objects array");
1256
1257 assert!(objects.iter().any(|object| {
1258 object["id"] == "session://active/system_prompt" && object["kind"] == "system_prompt"
1259 }));
1260 assert!(objects.iter().any(|object| {
1261 object["id"] == "session://active/messages/0" && object["kind"] == "message"
1262 }));
1263 }
1264
1265 #[tokio::test]
1266 async fn rlm_open_loads_active_session_prompt_object() {
1267 let ctx = ctx_with_session_objects();
1268 let open = RlmTool::alias("rlm_open", "open")
1269 .execute(
1270 json!({"name": "active_prompt", "session_object": "session://active/system_prompt"}),
1271 &ctx,
1272 )
1273 .await
1274 .expect("open prompt object");
1275 let open_json: Value = serde_json::from_str(&open.content).expect("open json");
1276 assert_eq!(open_json["type"], "session_object:system_prompt");
1277 assert!(
1278 open_json["preview_500"]
1279 .as_str()
1280 .unwrap()
1281 .contains("CodeWhale")
1282 );
1283
1284 RlmTool::alias("rlm_close", "close")
1285 .execute(json!({"name": "active_prompt"}), &ctx)
1286 .await
1287 .expect("close");
1288 }
1289
1290 #[tokio::test]
1291 async fn rlm_open_loads_transcript_message_object() {
1292 let ctx = ctx_with_session_objects();
1293 let open = RlmTool::alias("rlm_open", "open")
1294 .execute(
1295 json!({"name": "first_message", "session_object": "session://active/messages/0"}),
1296 &ctx,
1297 )
1298 .await
1299 .expect("open transcript slice");
1300 let open_json: Value = serde_json::from_str(&open.content).expect("open json");
1301 assert_eq!(open_json["type"], "session_object:message");
1302 assert!(
1303 open_json["preview_500"]
1304 .as_str()
1305 .unwrap()
1306 .contains("RLM surface")
1307 );
1308
1309 RlmTool::alias("rlm_close", "close")
1310 .execute(json!({"name": "first_message"}), &ctx)
1311 .await
1312 .expect("close");
1313 }
1314
1315 #[tokio::test]
1316 async fn rlm_open_ignores_blank_source_defaults_from_schema_fillers() {
1317 let ctx = ctx();
1318 RlmTool::alias("rlm_open", "open")
1319 .execute(
1320 json!({"name": "blank-defaults", "file_path": "", "content": "body", "url": ""}),
1321 &ctx,
1322 )
1323 .await
1324 .expect("open with blank sibling source fields");
1325
1326 RlmTool::alias("rlm_close", "close")
1327 .execute(json!({"name": "blank-defaults"}), &ctx)
1328 .await
1329 .expect("close");
1330 }
1331
1332 #[tokio::test]
1333 async fn rlm_open_misnamed_source_field_gets_did_you_mean_hint() {
1334 // #2655: a wrong source field name yields actionable guidance, not just
1335 // the canonical "provide exactly one" message.
1336 let ctx = ctx();
1337 let err = RlmTool::alias("rlm_open", "open")
1338 .execute(json!({"name": "doc", "prompt": "summarize this"}), &ctx)
1339 .await
1340 .expect_err("misnamed source field should fail");
1341 let msg = err.to_string();
1342 assert!(msg.contains("file_path"), "names the real fields: {msg}");
1343 assert!(
1344 msg.contains("`url`, or `session_object`"),
1345 "names session_object in the valid source field list: {msg}"
1346 );
1347 assert!(msg.contains("prompt"), "echoes the wrong field: {msg}");
1348 }
1349
1350 #[tokio::test]
1351 async fn rlm_eval_missing_code_explains_raw_python() {
1352 // #2655: the missing-code error should teach the tool, with an example.
1353 let ctx = ctx();
1354 let err = RlmTool::alias("rlm_eval", "eval")
1355 .execute(json!({"name": "doc"}), &ctx)
1356 .await
1357 .expect_err("missing code should fail");
1358 let msg = err.to_string();
1359 assert!(msg.contains("raw Python"), "explains it runs Python: {msg}");
1360 assert!(
1361 msg.contains("print(len(content))") || msg.contains("FINAL"),
1362 "includes an example: {msg}"
1363 );
1364 }
1365
1366 #[test]
1367 fn rlm_eval_schema_names_the_runtime_content_variable() {
1368 let schema = RlmTool::alias("rlm_eval", "eval").input_schema();
1369 let description = schema["properties"]["code"]["description"]
1370 .as_str()
1371 .expect("rlm_eval code description");
1372
1373 assert!(description.contains("`content`"));
1374 assert!(description.contains("print(len(content))"));
1375 assert!(!description.contains("SOURCE"));
1376 }
1377
1378 #[tokio::test]
1379 async fn nested_rlm_receipts_survive_tool_result_save_and_reopen_even_on_kernel_failure() {
1380 use wiremock::matchers::method;
1381 use wiremock::{Mock, MockServer, ResponseTemplate};
1382 let server = MockServer::start().await;
1383 Mock::given(method("POST")).respond_with(ResponseTemplate::new(200).set_body_json(json!({
1384 "id": "nested-model-response", "object": "chat.completion",
1385 "model": "deepseek-v4-flash",
1386 "choices": [{"index": 0, "message": {"role": "assistant",
1387 "content": "```repl\nFINAL('durable nested answer')\n```"}, "finish_reason": "stop"}],
1388 "usage": {"prompt_tokens": 7, "completion_tokens": 9, "total_tokens": 16}
1389 }))).expect(2).mount(&server).await;
1390 let mut config = crate::core::engine::tests::rlm_host::fixture_config("deepseek-v4-flash");
1391 let identity = config.active_provider_identity().unwrap();
1392 config.provider_config_for_mut(&identity).unwrap().base_url = Some(server.uri());
1393 let client = Arc::new(crate::client::CodewhaleClient::new(&config).unwrap());
1394 let tool = RlmTool::new("rlm");
1395 let temp = tempfile::tempdir().unwrap();
1396 let mut context = ToolContext::new(temp.path());
1397 context.execution.nested_call_gate =
1398 Some(crate::tools::codemode::NestedCallGate::admitting_for_test());
1399 context = crate::core::engine::tests::rlm_host::context_for_replies(
1400 &config,
1401 "deepseek-v4-flash",
1402 &context,
1403 client,
1404 );
1405 tool.execute(
1406 json!({"action": "open", "name": "receipts", "content": "fixture context"}),
1407 &context,
1408 )
1409 .await
1410 .unwrap();
1411 for (failed, code) in [
1412 (false, "print(rlm_query('nested context'))"),
1413 (true, "print(rlm_query('nested context')); _os._exit(2)"),
1414 ] {
1415 let result = tool
1416 .execute(
1417 json!({"action": "eval", "name": "receipts", "code": code}),
1418 &context,
1419 )
1420 .await
1421 .unwrap();
1422 assert_eq!(result.success, !failed);
1423 let output: Value = serde_json::from_str(&result.content).unwrap();
1424 let events = output["nested_events"]
1425 .as_array()
1426 .expect("retained nested events");
1427 assert_eq!(
1428 events
1429 .iter()
1430 .filter(|entry| entry["kind"] == "code")
1431 .count(),
1432 1
1433 );
1434 assert!(events.iter().any(|entry| {
1435 entry["content"]
1436 .as_str()
1437 .is_some_and(|line| line.contains("FINAL('durable nested answer')"))
1438 }));
1439 assert!(events.iter().any(|entry| {
1440 entry["content"]
1441 .as_str()
1442 .is_some_and(|line| line.contains("RLM finished: Final"))
1443 }));
1444 let messages = vec![
1445 Message {
1446 role: Role::Assistant,
1447 content: vec![ContentBlock::ToolUse {
1448 id: "rlm-call".into(),
1449 name: "rlm".into(),
1450 input: json!({"action": "eval", "name": "receipts", "code": code}),
1451 execution_id: Some("rlm-execution".into()),
1452 caller: None,
1453 thought_signature: None,
1454 }],
1455 },
1456 Message {
1457 role: Role::User,
1458 content: vec![ContentBlock::ToolResult {
1459 tool_use_id: "rlm-call".into(),
1460 execution_id: Some("rlm-execution".into()),
1461 content: result.content,
1462 is_error: Some(failed),
1463 content_blocks: None,
1464 }],
1465 },
1466 ];
1467 let manager =
1468 crate::session_manager::SessionManager::new(temp.path().join("sessions")).unwrap();
1469 let saved = crate::session_manager::create_saved_session(
1470 &messages,
1471 "deepseek-v4-flash",
1472 temp.path(),
1473 16,
1474 None,
1475 );
1476 manager.save_session(&saved).unwrap();
1477 let reopened = manager.load_session(&saved.metadata.id).unwrap();
1478 let ContentBlock::ToolResult { content, .. } = &reopened.messages[1].content[0] else {
1479 panic!("saved tool result");
1480 };
1481 let replay: Value = serde_json::from_str(content).unwrap();
1482 assert_eq!(&replay["nested_events"], &output["nested_events"]);
1483 }
1484 assert!(
1485 get_session(&context, "receipts")
1486 .await
1487 .unwrap()
1488 .lock()
1489 .await
1490 .kernel
1491 .is_none(),
1492 "a failed interpreter must not be reused"
1493 );
1494 }
1495
1496 #[tokio::test]
1497 async fn persistent_open_obeys_original_cancel_and_deadline_without_installing_a_kernel() {
1498 let temp = tempfile::tempdir().unwrap();
1499 let tool = RlmTool::new("rlm");
1500 for cancelled in [true, false] {
1501 let mut context = ToolContext::new(temp.path());
1502 let cancel = tokio_util::sync::CancellationToken::new();
1503 context.cancel_token = Some(cancel.clone());
1504 context.turn_deadline = Some(
1505 tokio::time::Instant::now()
1506 + if cancelled {
1507 Duration::from_secs(30)
1508 } else {
1509 Duration::from_millis(40)
1510 },
1511 );
1512 let sessions = context.runtime.rlm_sessions.lock().await;
1513 let input = json!({"action": "open", "name": "never-installed", "content": "fixture"});
1514 let mut open = tool.execute(input, &context);
1515 assert!(
1516 tokio::time::timeout(Duration::from_millis(10), &mut open)
1517 .await
1518 .is_err()
1519 );
1520 if cancelled {
1521 cancel.cancel();
1522 }
1523 let error = tokio::time::timeout(Duration::from_secs(1), &mut open)
1524 .await
1525 .expect("original cancellation and deadline both release a contended open")
1526 .unwrap_err();
1527 if cancelled {
1528 assert!(matches!(error, ToolError::Cancelled { .. }));
1529 } else {
1530 assert!(
1531 error
1532 .to_string()
1533 .contains("deadline exhausted opening context")
1534 );
1535 }
1536 assert!(
1537 sessions.is_empty(),
1538 "no interpreter is installed while admission waits"
1539 );
1540 drop(open);
1541 drop(sessions);
1542 assert!(context.runtime.rlm_sessions.lock().await.is_empty());
1543 }
1544 }
1545
1546 #[tokio::test]
1547 async fn parent_deadline_bounds_rlm_session_lock_waits() {
1548 let temp = tempfile::tempdir().unwrap();
1549 let mut context = ToolContext::new(temp.path());
1550 let tool = RlmTool::new("rlm");
1551 tool.execute(
1552 json!({"action": "open", "name": "contended", "content": "fixture"}),
1553 &context,
1554 )
1555 .await
1556 .unwrap();
1557 let session = get_session(&context, "contended").await.unwrap();
1558 let marker = temp.path().join("must-not-exist");
1559 let code = format!(
1560 "open({}, 'w').write('ran')",
1561 serde_json::to_string(&marker.to_string_lossy()).unwrap()
1562 );
1563 for registry_locked in [true, false] {
1564 context.turn_deadline = Some(tokio::time::Instant::now() + Duration::from_millis(50));
1565 let registry_guard = if registry_locked {
1566 Some(context.runtime.rlm_sessions.lock().await)
1567 } else {
1568 None
1569 };
1570 let session_guard = if registry_locked {
1571 None
1572 } else {
1573 Some(session.lock().await)
1574 };
1575 let error = tokio::time::timeout(
1576 Duration::from_secs(2),
1577 tool.execute(
1578 json!({"action": "eval", "name": "contended", "code": code}),
1579 &context,
1580 ),
1581 )
1582 .await
1583 .expect("lock waits must obey the parent deadline")
1584 .unwrap_err();
1585 assert!(
1586 error
1587 .to_string()
1588 .contains("deadline exhausted waiting for context")
1589 );
1590 assert!(!marker.exists());
1591 drop(session_guard);
1592 drop(registry_guard);
1593 }
1594 tool.execute(json!({"action": "close", "name": "contended"}), &context)
1595 .await
1596 .unwrap();
1597 }
1598
1599 #[tokio::test]
1600 async fn expired_parent_deadline_prevents_rlm_python_side_effects() {
1601 let temp = tempfile::tempdir().unwrap();
1602 let mut context = ToolContext::new(temp.path());
1603 let tool = RlmTool::new("rlm");
1604 tool.execute(
1605 json!({"action": "open", "name": "expired", "content": "fixture"}),
1606 &context,
1607 )
1608 .await
1609 .unwrap();
1610 context.turn_deadline = Some(tokio::time::Instant::now());
1611 let marker = temp.path().join("must-not-exist");
1612 let code = format!(
1613 "open({}, 'w').write('ran')",
1614 serde_json::to_string(&marker.to_string_lossy()).unwrap()
1615 );
1616 let error = tool
1617 .execute(
1618 json!({"action": "eval", "name": "expired", "code": code}),
1619 &context,
1620 )
1621 .await
1622 .unwrap_err();
1623 assert!(error.to_string().contains("deadline exhausted"));
1624 assert!(!marker.exists());
1625 tool.execute(json!({"action": "close", "name": "expired"}), &context)
1626 .await
1627 .unwrap();
1628 }
1629
1630 #[tokio::test]
1631 async fn rlm_session_open_eval_close_lifecycle() {
1632 let ctx = ctx();
1633 RlmTool::alias("rlm_open", "open")
1634 .execute(
1635 json!({"name": "sample", "content": "alpha\nbeta\ngamma"}),
1636 &ctx,
1637 )
1638 .await
1639 .expect("open");
1640
1641 let eval = RlmTool::alias("rlm_eval", "eval")
1642 .execute(json!({"name": "sample", "code": "print('ok')"}), &ctx)
1643 .await
1644 .expect("eval");
1645 let eval_json: Value = serde_json::from_str(&eval.content).expect("eval json");
1646 let stdout_preview = eval_json["stdout_preview"]
1647 .as_str()
1648 .expect("stdout_preview")
1649 .replace("\r\n", "\n");
1650 assert_eq!(stdout_preview, "ok\n");
1651
1652 let close = RlmTool::alias("rlm_close", "close")
1653 .execute(json!({"name": "sample"}), &ctx)
1654 .await
1655 .expect("close");
1656 assert!(close.content.contains("sample"));
1657 }
1658
1659 #[tokio::test]
1660 async fn rlm_canonical_action_routing_runs_full_lifecycle() {
1661 // The visible surface: one `rlm` tool, action-parameterized.
1662 let ctx = ctx();
1663 let tool = RlmTool::new("rlm");
1664 tool.execute(
1665 json!({"action": "open", "name": "canonical", "content": "body"}),
1666 &ctx,
1667 )
1668 .await
1669 .expect("open via canonical action");
1670
1671 let eval = tool
1672 .execute(
1673 json!({"action": "eval", "name": "canonical", "code": "print('ok')"}),
1674 &ctx,
1675 )
1676 .await
1677 .expect("eval via canonical action");
1678 let eval_json: Value = serde_json::from_str(&eval.content).expect("eval json");
1679 let stdout_preview = eval_json["stdout_preview"]
1680 .as_str()
1681 .expect("stdout_preview")
1682 .replace("\r\n", "\n");
1683 assert_eq!(stdout_preview, "ok\n");
1684
1685 let close = tool
1686 .execute(json!({"action": "close", "name": "canonical"}), &ctx)
1687 .await
1688 .expect("close via canonical action");
1689 assert!(close.content.contains("canonical"));
1690 }
1691
1692 #[tokio::test]
1693 async fn rlm_eval_final_returns_handle() {
1694 let ctx = ctx();
1695 RlmTool::alias("rlm_open", "open")
1696 .execute(json!({"name": "finals", "content": "body"}), &ctx)
1697 .await
1698 .expect("open");
1699
1700 let eval = RlmTool::alias("rlm_eval", "eval")
1701 .execute(
1702 json!({"name": "finals", "code": "finalize('done', confidence=0.8)"}),
1703 &ctx,
1704 )
1705 .await
1706 .expect("eval");
1707 let eval_json: Value = serde_json::from_str(&eval.content).expect("eval json");
1708 assert_eq!(eval_json["final"]["kind"], "var_handle");
1709 assert_eq!(eval_json["final"]["name"], "final_1");
1710 assert_eq!(eval_json["confidence"], 0.8);
1711
1712 RlmTool::alias("rlm_close", "close")
1713 .execute(json!({"name": "finals"}), &ctx)
1714 .await
1715 .expect("close");
1716 }
1717
1718 #[tokio::test]
1719 async fn rlm_eval_final_preserves_json_handle() {
1720 let ctx = ctx();
1721 RlmTool::alias("rlm_open", "open")
1722 .execute(json!({"name": "json-final", "content": "body"}), &ctx)
1723 .await
1724 .expect("open");
1725
1726 let eval = RlmTool::alias("rlm_eval", "eval")
1727 .execute(
1728 json!({"name": "json-final", "code": "finalize({'answer': 42, 'items': ['a', 'b']})"}),
1729 &ctx,
1730 )
1731 .await
1732 .expect("eval");
1733 let eval_json: Value = serde_json::from_str(&eval.content).expect("eval json");
1734 assert_eq!(eval_json["final"]["kind"], "var_handle");
1735 assert_eq!(eval_json["final"]["type"], "dict");
1736 assert_eq!(eval_json["final"]["length"], 2);
1737
1738 let read = HandleReadTool
1739 .execute(
1740 json!({"handle": eval_json["final"].clone(), "jsonpath": "$.items[*]"}),
1741 &ctx,
1742 )
1743 .await
1744 .expect("read final handle");
1745 let read_json: Value = serde_json::from_str(&read.content).expect("read json");
1746 assert_eq!(read_json["matches"], json!(["a", "b"]));
1747
1748 RlmTool::alias("rlm_close", "close")
1749 .execute(json!({"name": "json-final"}), &ctx)
1750 .await
1751 .expect("close");
1752 }
1753
1754 #[tokio::test]
1755 async fn rlm_configure_metadata_omits_stdout() {
1756 let ctx = ctx();
1757 RlmTool::alias("rlm_open", "open")
1758 .execute(json!({"name": "quiet", "content": "body"}), &ctx)
1759 .await
1760 .expect("open");
1761 RlmTool::alias("rlm_configure", "configure")
1762 .execute(
1763 json!({"name": "quiet", "output_feedback": "metadata", "sub_rlm_max_depth": 99}),
1764 &ctx,
1765 )
1766 .await
1767 .expect("configure");
1768
1769 let eval = RlmTool::alias("rlm_eval", "eval")
1770 .execute(json!({"name": "quiet", "code": "print('hidden')"}), &ctx)
1771 .await
1772 .expect("eval");
1773 let eval_json: Value = serde_json::from_str(&eval.content).expect("eval json");
1774 assert!(eval_json.get("stdout_preview").is_none());
1775
1776 RlmTool::alias("rlm_close", "close")
1777 .execute(json!({"name": "quiet"}), &ctx)
1778 .await
1779 .expect("close");
1780 }
1781 #[tokio::test]
1782 async fn shared_session_refusal_is_atomic_and_persistent_kernel_owns_no_caller() {
1783 let workspace = tempfile::tempdir().unwrap();
1784 let base = ToolContext::new(workspace.path());
1785 let model = "captured-kernel-model";
1786 let config = crate::core::engine::tests::rlm_host::fixture_config(model);
1787 let mock = Arc::new(crate::llm_client::mock::MockLlmClient::new(Vec::new()));
1788 let context = crate::core::engine::tests::rlm_host::context_for_replies(
1789 &config,
1790 model,
1791 &base,
1792 mock.clone(),
1793 );
1794 let weak_caller = Arc::downgrade(context.rlm_caller.as_ref().unwrap());
1795 let sessions = context.runtime.rlm_sessions.clone();
1796 let tool = RlmTool::new("rlm");
1797 tool.execute(
1798 json!({"action":"open", "name":"lifetime", "content":"long local context"}),
1799 &context,
1800 )
1801 .await
1802 .unwrap();
1803 let before = serde_json::to_value(
1804 get_session(&context, "lifetime")
1805 .await
1806 .unwrap()
1807 .lock()
1808 .await
1809 .config
1810 .clone(),
1811 )
1812 .unwrap();
1813 let error = tool.execute(json!({"action":"configure", "name":"lifetime", "share_session":true, "sub_query_timeout_secs":1, "output_feedback":"metadata"}), &context).await.unwrap_err();
1814 assert!(
1815 error
1816 .to_string()
1817 .contains("share_session=true is unsupported")
1818 );
1819 let after = serde_json::to_value(
1820 get_session(&context, "lifetime")
1821 .await
1822 .unwrap()
1823 .lock()
1824 .await
1825 .config
1826 .clone(),
1827 )
1828 .unwrap();
1829 assert_eq!(
1830 before, after,
1831 "unsupported session authority refuses before any settings change"
1832 );
1833 drop(context);
1834 assert!(
1835 weak_caller.upgrade().is_none(),
1836 "persistent kernels must not retain model authority or their borrowed callback"
1837 );
1838 assert_eq!(
1839 sessions.lock().await.len(),
1840 1,
1841 "the actual kernel map is still live"
1842 );
1843 assert_eq!(mock.call_count(), 0);
1844 sessions.lock().await.clear();
1845 }
1846 #[tokio::test]
1847 async fn persistent_eval_drop_and_original_cancel_kill_running_kernel_without_retaining_caller()
1848 {
1849 for cancel_original in [false, true] {
1850 let workspace = tempfile::tempdir().unwrap();
1851 let base = ToolContext::new(workspace.path());
1852 let model = "captured-kernel-model";
1853 let config = crate::core::engine::tests::rlm_host::fixture_config(model);
1854 let mock = Arc::new(crate::llm_client::mock::MockLlmClient::new(Vec::new()));
1855 let context = crate::core::engine::tests::rlm_host::context_for_replies(
1856 &config,
1857 model,
1858 &base,
1859 mock.clone(),
1860 );
1861 let weak_caller = Arc::downgrade(context.rlm_caller.as_ref().unwrap());
1862 let cancel = context.cancel_token.as_ref().unwrap().clone();
1863 let tool = RlmTool::new("rlm");
1864 tool.execute(
1865 json!({"action":"open", "name":"running", "content":"local long context"}),
1866 &context,
1867 )
1868 .await
1869 .unwrap();
1870 let marker = workspace.path().join("started.txt");
1871 let effect = workspace.path().join("must-not-run.txt");
1872 let code = format!(
1873 "import pathlib, time\nstarted = pathlib.Path({})\nstarted_temporary = started.with_suffix('.tmp')\nstarted_temporary.write_text(_ctx_file)\nstarted_temporary.replace(started)\ntime.sleep(1)\npathlib.Path({}).write_text('effect after cancellation')",
1874 serde_json::to_string(&marker.to_string_lossy()).unwrap(),
1875 serde_json::to_string(&effect.to_string_lossy()).unwrap(),
1876 );
1877 let input = json!({"action":"eval", "name":"running", "code":code});
1878 let mut call = Box::pin(tool.execute(input, &context));
1879 tokio::time::timeout(Duration::from_secs(2), async {
1880 tokio::select! {
1881 result = &mut call => panic!("running eval finished before its marker: {result:?}"),
1882 () = async {
1883 while !marker.exists() {
1884 tokio::time::sleep(Duration::from_millis(5)).await;
1885 }
1886 } => {},
1887 }
1888 }).await.expect("actual persistent Python round starts");
1889 let context_path = PathBuf::from(std::fs::read_to_string(&marker).unwrap());
1890 assert!(context_path.exists());
1891 if cancel_original {
1892 cancel.cancel();
1893 let result = tokio::time::timeout(Duration::from_secs(1), &mut call)
1894 .await
1895 .expect("origin cancellation hands back promptly")
1896 .unwrap();
1897 assert!(!result.success);
1898 assert!(result.content.contains("cancelled by its originating turn"));
1899 }
1900 drop(call);
1901 let session = get_session(&context, "running").await.unwrap();
1902 assert!(session.lock().await.kernel.is_none());
1903 assert!(
1904 !context_path.exists(),
1905 "dropped interpreter releases its owned context file"
1906 );
1907 drop(context);
1908 assert!(weak_caller.upgrade().is_none());
1909 tokio::time::sleep(Duration::from_millis(1200)).await;
1910 assert!(
1911 !effect.exists(),
1912 "unknown Python code cannot outlive dropped/cancelled eval"
1913 );
1914 assert_eq!(mock.call_count(), 0);
1915 }
1916 }
1917 }
1918
1918 lines RUST