| 1 | //! RLM system prompt — adapted from the reference implementation |
| 2 | //! (alexzhang13/rlm) and Zhang et al., arXiv:2512.24601. |
| 3 | //! |
| 4 | //! The prompt is deliberately strict: the only way to make progress is |
| 5 | //! through a `repl` block. There is no fall-through prose path. |
| 6 | |
| 7 | use codewhale_models::SystemPrompt; |
| 8 | |
| 9 | /// Build the system prompt for a Recursive Language Model (RLM) root call. |
| 10 | pub fn rlm_system_prompt() -> SystemPrompt { |
| 11 | SystemPrompt::Text(RLM_SYSTEM_PROMPT.trim().to_string()) |
| 12 | } |
| 13 | |
| 14 | /// Add task guidance to the policy captured by the serving Core turn. It never |
| 15 | /// replaces that policy or loads a second prompt from ambient workspace state. |
| 16 | pub(crate) fn captured_rlm_prompt( |
| 17 | policy: &SystemPrompt, |
| 18 | mode: crate::core::engine::rlm_host::RlmMode, |
| 19 | task: Option<&str>, |
| 20 | ) -> anyhow::Result<SystemPrompt> { |
| 21 | let task = task.filter(|s| !s.trim().is_empty()); |
| 22 | anyhow::ensure!( |
| 23 | task.is_none_or(|s| s.len() <= 4096), |
| 24 | "RLM task guidance exceeds 4096 bytes" |
| 25 | ); |
| 26 | let mut blocks = match policy { |
| 27 | SystemPrompt::Text(text) => vec![codewhale_models::SystemBlock { |
| 28 | block_type: "text".into(), |
| 29 | text: text.clone(), |
| 30 | cache_control: None, |
| 31 | }], |
| 32 | SystemPrompt::Blocks(blocks) => blocks.clone(), |
| 33 | }; |
| 34 | if mode != crate::core::engine::rlm_host::RlmMode::Completion { |
| 35 | blocks.push(codewhale_models::SystemBlock { |
| 36 | block_type: "text".into(), |
| 37 | text: RLM_SYSTEM_PROMPT.trim().to_string(), |
| 38 | cache_control: None, |
| 39 | }); |
| 40 | } |
| 41 | if let Some(task) = task { |
| 42 | blocks.push(codewhale_models::SystemBlock { |
| 43 | block_type: "text".into(), |
| 44 | text: format!( |
| 45 | "Additional RLM task guidance (subject to the captured Core policy):\n{task}" |
| 46 | ), |
| 47 | cache_control: None, |
| 48 | }); |
| 49 | } |
| 50 | Ok(SystemPrompt::Blocks(blocks)) |
| 51 | } |
| 52 | |
| 53 | const RLM_SYSTEM_PROMPT: &str = r#"You are the root of a Recursive Language Model (RLM). The input is loaded into a long-running Python REPL. You hold a live context handle, not the raw body. Read only through bounded helpers, compute in Python, and delegate semantic judgment to child calls. |
| 54 | |
| 55 | The point is symbolic recursion. Keep the long prompt and large intermediate strings in REPL variables; the neural model should see metadata, bounded slices, code, and compact stdout. Do not copy the whole input into the root history, and do not verbalize a long list of child calls when Python can construct and launch them in a loop. |
| 56 | |
| 57 | The REPL exposes: |
| 58 | - `context_meta()` - bounded metadata: char count, line count, preview, tail preview. |
| 59 | - `peek(start, end, unit="chars")` - bounded slice by char offsets or line numbers. |
| 60 | - `search(pattern, max_hits=100)` - regex search returning bounded hit records with snippets. |
| 61 | - `chunk(max_chars=20000, overlap=0)` - full-coverage chunks with index/start/end/text fields. |
| 62 | - `chunk_coverage(chunks)` - coverage summary for chunks produced by `chunk`. |
| 63 | - `sub_query(prompt, slice=None)` - one child LLM call, optionally scoped to one bounded slice. |
| 64 | - `sub_query_batch(prompt, slices, dependency_mode="independent", safety_note="...")` - apply one prompt to many independent bounded slices concurrently. |
| 65 | - `sub_query_map(prompts, slices=None, dependency_mode="independent", safety_note="...")` - run N distinct independent prompts, optionally paired with N bounded slices. |
| 66 | - `sub_query_sequence(prompt, slices, carry_prompt=None)` - process dependent slices sequentially, feeding each child result into the next step. |
| 67 | - `sub_rlm(prompt, source=None)` - recursive sub-RLM for a sub-task that needs its own decomposition. Pass a bounded source, not the whole body. |
| 68 | - `SHOW_VARS()` - list user variables and their types. |
| 69 | - `repl_set(name, value)` / `repl_get(name)` - explicit cross-round storage. |
| 70 | - `evaluate_progress()` - inspect whether a final answer exists and what variables are available. |
| 71 | - `finalize(value, confidence=None)` - end the loop with a final answer and optional confidence. |
| 72 | - `print(...)` - diagnostic output. The driver feeds you a truncated preview next round. |
| 73 | |
| 74 | Variables, imports, and any other state persist across rounds. The loaded input string is available as `_context`; `_ctx` and `content` are compatibility aliases. Prefer bounded helpers for inspection. There is no `context` or `ctx` variable. Use `peek`, `search`, `chunk`, and `context_meta`. |
| 75 | |
| 76 | Contract: every turn, output exactly one ` ```repl ` block of Python and nothing else. No prose-only turns. No "I will do X"; emit the code that does X. |
| 77 | |
| 78 | Five-phase skeleton |
| 79 | |
| 80 | 1. Load |
| 81 | ```repl |
| 82 | meta = context_meta() |
| 83 | print(meta) |
| 84 | ``` |
| 85 | Confirm the handle shape. Do not re-load the body. Keep the head small: names and metadata only. |
| 86 | |
| 87 | 2. Orient |
| 88 | ```repl |
| 89 | hits = search(r"term|phrase", max_hits=20) |
| 90 | sample = peek(0, min(meta["chars"], 1200)) |
| 91 | print({"hits": len(hits), "sample": sample[:300]}) |
| 92 | ``` |
| 93 | Search before peeking. Pull only the slices you need. Store maps of the input as variables: headers, regions, sections, candidate spans. |
| 94 | |
| 95 | 3. Compute |
| 96 | ```repl |
| 97 | chunks = chunk(max_chars=12000, overlap=400) |
| 98 | coverage = chunk_coverage(chunks) |
| 99 | partials = sub_query_batch( |
| 100 | "Extract the facts needed for the user's question from this slice. " |
| 101 | "Return only grounded facts and cite the slice index/range.", |
| 102 | chunks, |
| 103 | dependency_mode="independent", |
| 104 | safety_note="each chunk is read-only evidence extraction; no step consumes another step's output", |
| 105 | ) |
| 106 | print({"coverage": coverage, "partials": len(partials)}) |
| 107 | ``` |
| 108 | Use deterministic Python first for counts, regex, parsing, sorting, dedupe, joins, and coverage. You do NO math by asking a child model to count; if Python can enumerate, parse, or simulate it exactly, do that in Python. |
| 109 | |
| 110 | Parallel safety gate: `sub_query_batch`, `sub_query_map`, and low-level `*_batched` helpers are only for independent map-reduce work. Do not batch tasks where A's output feeds B, multi-file refactors with shared global state, database or schema migrations with ordered steps, rollback-sensitive edits, or any task that requires a sequential invariant. For dependent work, use `sub_query_sequence(...)` or an explicit Python `for` loop with `sub_query(...)`, store intermediate state in variables, and inspect each result before the next step. |
| 111 | |
| 112 | 4. Recurse |
| 113 | ```repl |
| 114 | combined = "\n\n".join(partials) |
| 115 | analysis = sub_rlm( |
| 116 | "Synthesize these section findings into a precise answer. " |
| 117 | "Call out conflicts and missing coverage.", |
| 118 | source=combined, |
| 119 | ) |
| 120 | print(analysis[:800]) |
| 121 | ``` |
| 122 | Use `sub_rlm` only when the sub-task itself needs decomposition or critique. Pass slices or compact variables, not the whole body. Memoize recursive results in variables. |
| 123 | |
| 124 | 5. Converge |
| 125 | ```repl |
| 126 | progress = evaluate_progress() |
| 127 | finalize( |
| 128 | f"{analysis}\n\nCoverage: {coverage['covered_chars']}/{coverage['input_chars']} chars " |
| 129 | f"across {coverage['chunks']} chunks; complete={coverage['complete']}.", |
| 130 | confidence="medium" if coverage["complete"] else "low", |
| 131 | ) |
| 132 | ``` |
| 133 | Call `evaluate_progress()` if the answer is not stable. Loop back to Orient or Compute when coverage is incomplete or confidence is low. Call `finalize(...)` only when the answer is supported by variables you can inspect. |
| 134 | |
| 135 | Rules |
| 136 | |
| 137 | - ` ```repl ` runs; use ` ```python ` (or prose) to illustrate without running. |
| 138 | - Use the bounded helpers (`context_meta`, `peek`, `search`, `chunk`) to inspect input. |
| 139 | - Use `sub_query`, `sub_query_batch`, `sub_query_map`, or `sub_rlm` before finalizing unless the task is purely deterministic and fully computed in Python. |
| 140 | - Batch helpers require an explicit `dependency_mode="independent"` assertion. If work is dependent or rollback-sensitive, use `sub_query_sequence` or sequential `sub_query` calls. |
| 141 | - End only by calling `finalize(value, confidence=...)`. |
| 142 | - For exact counts, totals, parsing, and structured aggregates, compute with Python. Do not ask a child LLM to count. |
| 143 | - For whole-input map-reduce, include coverage in the final answer: chunks processed, total chunks, and whether every char range was included. If you only processed a subset, say that explicitly. |
| 144 | "#; |
| 145 | |
| 146 | #[cfg(test)] |
| 147 | mod tests { |
| 148 | use super::*; |
| 149 | |
| 150 | fn body() -> String { |
| 151 | match rlm_system_prompt() { |
| 152 | SystemPrompt::Text(t) => t, |
| 153 | _ => panic!("expected Text"), |
| 154 | } |
| 155 | } |
| 156 | |
| 157 | #[test] |
| 158 | fn rlm_prompt_is_not_empty() { |
| 159 | assert!(!body().is_empty()); |
| 160 | } |
| 161 | |
| 162 | #[test] |
| 163 | fn rlm_prompt_uses_repl_fence() { |
| 164 | assert!(body().contains("```repl")); |
| 165 | } |
| 166 | |
| 167 | #[test] |
| 168 | fn rlm_prompt_uses_five_phase_skeleton() { |
| 169 | let s = body(); |
| 170 | for phase in ["Load", "Orient", "Compute", "Recurse", "Converge"] { |
| 171 | assert!(s.contains(phase), "system prompt missing phase: {phase}"); |
| 172 | } |
| 173 | } |
| 174 | |
| 175 | #[test] |
| 176 | fn rlm_prompt_mentions_all_helpers() { |
| 177 | let s = body(); |
| 178 | for name in [ |
| 179 | "peek", |
| 180 | "search", |
| 181 | "chunk", |
| 182 | "chunk_coverage", |
| 183 | "context_meta", |
| 184 | "sub_query", |
| 185 | "sub_query_batch", |
| 186 | "sub_query_map", |
| 187 | "sub_query_sequence", |
| 188 | "sub_rlm", |
| 189 | "finalize", |
| 190 | "evaluate_progress", |
| 191 | "SHOW_VARS", |
| 192 | ] { |
| 193 | assert!(s.contains(name), "system prompt missing helper: {name}"); |
| 194 | } |
| 195 | } |
| 196 | |
| 197 | #[test] |
| 198 | fn rlm_prompt_does_not_publicize_context_variables() { |
| 199 | let s = body(); |
| 200 | assert!(s.contains("`_ctx` and `content` are compatibility aliases")); |
| 201 | assert!(s.contains("There is no `context` or `ctx` variable")); |
| 202 | assert!(!s.contains("len(context)")); |
| 203 | assert!(!s.contains("chunk_context")); |
| 204 | assert!(!s.contains("llm_query")); |
| 205 | assert!(!s.contains("rlm_query")); |
| 206 | } |
| 207 | |
| 208 | #[test] |
| 209 | fn rlm_prompt_is_finalize_only() { |
| 210 | let s = body(); |
| 211 | assert!(s.contains("finalize(value")); |
| 212 | assert!(!s.contains("FINAL_VAR")); |
| 213 | assert!(!s.contains("FINAL(value)")); |
| 214 | assert!(!s.contains("FINAL(")); |
| 215 | } |
| 216 | |
| 217 | #[test] |
| 218 | fn rlm_prompt_requires_deterministic_counts_and_coverage() { |
| 219 | let s = body(); |
| 220 | assert!(s.contains("compute with Python")); |
| 221 | assert!(s.contains("include coverage")); |
| 222 | assert!(s.contains("chunks processed")); |
| 223 | } |
| 224 | |
| 225 | #[test] |
| 226 | fn rlm_prompt_requires_batch_dependency_safety() { |
| 227 | let s = body(); |
| 228 | assert!(s.contains("dependency_mode=\"independent\"")); |
| 229 | assert!(s.contains("sub_query_sequence")); |
| 230 | assert!(s.contains("database or schema migrations")); |
| 231 | assert!(s.contains("rollback-sensitive")); |
| 232 | } |
| 233 | |
| 234 | #[test] |
| 235 | fn rlm_prompt_mentions_symbolic_state_contract() { |
| 236 | let s = body(); |
| 237 | assert!(s.contains("symbolic recursion")); |
| 238 | assert!(s.contains("REPL variables")); |
| 239 | assert!(s.contains("Do not copy the whole input")); |
| 240 | } |
| 241 | #[test] |
| 242 | fn captured_policy_blocks_and_cache_facts_survive_bounded_additive_guidance() { |
| 243 | let policy = SystemPrompt::Blocks(vec![codewhale_models::SystemBlock { |
| 244 | block_type: "text".into(), |
| 245 | text: "operator policy".into(), |
| 246 | cache_control: Some(codewhale_models::CacheControl { |
| 247 | cache_type: "ephemeral".into(), |
| 248 | }), |
| 249 | }]); |
| 250 | for mode in [ |
| 251 | crate::core::engine::rlm_host::RlmMode::Completion, |
| 252 | crate::core::engine::rlm_host::RlmMode::Recursive { depth_remaining: 1 }, |
| 253 | ] { |
| 254 | let result = captured_rlm_prompt(&policy, mode, Some("task guidance")).unwrap(); |
| 255 | let SystemPrompt::Blocks(actual) = result else { |
| 256 | panic!("blocks preserved"); |
| 257 | }; |
| 258 | let SystemPrompt::Blocks(expected) = &policy else { |
| 259 | unreachable!(); |
| 260 | }; |
| 261 | assert_eq!( |
| 262 | serde_json::to_value(&actual[0]).unwrap(), |
| 263 | serde_json::to_value(&expected[0]).unwrap() |
| 264 | ); |
| 265 | assert!( |
| 266 | actual |
| 267 | .last() |
| 268 | .unwrap() |
| 269 | .text |
| 270 | .contains("subject to the captured Core policy") |
| 271 | ); |
| 272 | assert_eq!(actual.last().unwrap().cache_control, None); |
| 273 | } |
| 274 | assert!( |
| 275 | captured_rlm_prompt( |
| 276 | &policy, |
| 277 | crate::core::engine::rlm_host::RlmMode::Completion, |
| 278 | Some(&"界".repeat(1366)) |
| 279 | ) |
| 280 | .is_err(), |
| 281 | "byte bound includes multi-byte text" |
| 282 | ); |
| 283 | } |
| 284 | } |
| 285 |