返回 CodeWhale
integration_mock_llm.rs
根目录 / crates / tui / tests / integration / integration_mock_llm.rs
1 //! Integration tests for the [`MockLlmClient`](mock::MockLlmClient).
2 //!
3 //! These tests exercise the [`LlmClient`](llm_client::LlmClient) trait surface
4 //! directly. They verify that the mock client itself behaves correctly under
5 //! the patterns the runtime relies on:
6 //!
7 //! - **Streaming turn loop** — events arrive in order, `MessageStop` terminates
8 //! the stream.
9 //! - **Reasoning replay** (issue #69 / V4 §5.1.1) — when the runtime sends a
10 //! second turn after a tool round, it MUST replay prior `reasoning_content`.
11 //! Catches the HTTP 400 path that broke v0.4.9-v0.5.1.
12 //! - **Tool-call round-trip** — assistant emits `tool_calls`, runtime executes,
13 //! tool result is appended, next turn streams text.
14 //! - **Multiple tool calls in one round** — assistant returns N tool_calls;
15 //! the request payload preserves their ordering.
16 //! - **Compaction-style non-streaming call** — `create_message` returns a
17 //! queued `MessageResponse` without going through the streaming path.
18 //! - **Sub-agent style turn** — child mailbox receives a parent prompt and
19 //! replies; trait boundary is the same.
20 //! - **Capacity-gate observation** — runtime can probe estimated request size
21 //! and decline to dispatch; the mock surfaces capture-side hooks for that.
22 //!
23 //! # Why trait-level (not engine-level)
24 //!
25 //! As of v0.6.7 the engine (`crates/tui/src/core/engine.rs`) holds a concrete
26 //! `Option<CodewhaleClient>` — the [`LlmClient`] trait is implemented but no
27 //! consumer takes `Arc<dyn LlmClient>` or generic `<C: LlmClient>`. Wiring the
28 //! mock into a full engine turn-loop therefore requires a separate refactor:
29 //! every `Option<CodewhaleClient>` consumer (engine, registry, rlm, review,
30 //! cycle_manager, compaction, subagent) must move to `Arc<dyn LlmClient>`.
31 //!
32 //! Per the v0.7.0 mock-LLM issue (the parent of this file): "If the engine's
33 //! API surfaces are too tangled to mock cleanly … document that as BLOCKED with
34 //! what wiring needs to change. In that case still commit any partial work
35 //! that lands cleanly." Full engine integration coverage remains blocked on
36 //! that seam; this file keeps the blocker documented instead of carrying
37 //! ignored placeholder tests.
38 //!
39 //! Once `Arc<dyn LlmClient>` lands, add engine-level tests that reuse this mock.
40
41 use futures_util::StreamExt;
42
43 // Bring in the production model types verbatim — no other crate sources are
44 // needed because the mock is self-contained against `models.rs`.
45
46 // Mirror the real `llm_client` module hierarchy so that `mock.rs`'s
47 // `super::{LlmClient, StreamEventBox}` paths resolve. We re-declare a local
48 // `LlmClient` trait + `StreamEventBox` alias that match the production shape
49 // 1:1 (the public surface that ships in the binary). The mock implements
50 // this local trait, which is structurally identical to the production trait.
51 //
52 // The helper file lives under `tests/support/` so cargo does not try to
53 // compile it as its own test binary.
54
55 use crate::llm_client::LlmClient;
56 use crate::llm_client::mock::{MockLlmClient, canned};
57 use codewhale_models::Role;
58 use codewhale_models::{ContentBlock, Delta, Message, MessageRequest, StreamEvent, Usage};
59
60 // === Helpers ===============================================================
61
62 fn user_message(text: &str) -> Message {
63 Message {
64 role: Role::User,
65 content: vec![ContentBlock::Text {
66 text: text.to_string(),
67 cache_control: None,
68 }],
69 }
70 }
71
72 fn assistant_thinking(thinking: &str, text: &str) -> Message {
73 Message {
74 role: Role::Assistant,
75 content: vec![
76 ContentBlock::Thinking {
77 thinking: thinking.to_string(),
78 signature: None,
79 state: None,
80 },
81 ContentBlock::Text {
82 text: text.to_string(),
83 cache_control: None,
84 },
85 ],
86 }
87 }
88
89 fn assistant_tool_call(id: &str, name: &str, input: serde_json::Value) -> Message {
90 Message {
91 role: Role::Assistant,
92 content: vec![ContentBlock::ToolUse {
93 execution_id: None,
94 id: id.to_string(),
95 name: name.to_string(),
96 input,
97 caller: None,
98 thought_signature: None,
99 }],
100 }
101 }
102
103 fn tool_result_message(tool_use_id: &str, content: &str) -> Message {
104 Message {
105 role: Role::User,
106 content: vec![ContentBlock::ToolResult {
107 execution_id: None,
108 tool_use_id: tool_use_id.to_string(),
109 content: content.to_string(),
110 is_error: None,
111 content_blocks: None,
112 }],
113 }
114 }
115
116 fn make_request(messages: Vec<Message>) -> MessageRequest {
117 MessageRequest {
118 model: "deepseek-v4-pro".to_string(),
119 messages,
120 max_tokens: 4096,
121 system: None,
122 tools: None,
123 tool_choice: None,
124 metadata: None,
125 thinking: None,
126 reasoning_effort: Some("high".to_string()),
127 stream: Some(true),
128 temperature: None,
129 top_p: None,
130 }
131 }
132
133 async fn drain_stream_text(
134 mock: &MockLlmClient,
135 request: MessageRequest,
136 ) -> (String, Option<String>) {
137 let mut stream = mock
138 .create_message_stream(request)
139 .await
140 .expect("stream open");
141 let mut text = String::new();
142 let mut stop_reason: Option<String> = None;
143 while let Some(ev) = stream.next().await {
144 match ev.expect("event") {
145 StreamEvent::ContentBlockDelta {
146 delta: Delta::TextDelta { text: t },
147 ..
148 } => text.push_str(&t),
149 StreamEvent::MessageDelta { delta, .. } => {
150 stop_reason = delta.stop_reason;
151 }
152 StreamEvent::MessageStop => break,
153 _ => {}
154 }
155 }
156 (text, stop_reason)
157 }
158
159 // === 1. Full turn loop with streaming =======================================
160
161 #[tokio::test]
162 async fn full_turn_loop_streams_text_chunks() {
163 // Two text deltas + finish reason — exercises the canonical streaming
164 // turn-loop path the engine drives.
165 let turn = vec![
166 canned::message_start("msg_1"),
167 canned::text_block_start(0),
168 canned::text_delta(0, "Hello, "),
169 canned::text_delta(0, "world!"),
170 canned::block_stop(0),
171 canned::message_delta("end_turn", Some(Usage::default())),
172 canned::message_stop(),
173 ];
174 let mock = MockLlmClient::new(vec![turn]);
175
176 let request = make_request(vec![user_message("greet me")]);
177 let (text, stop) = drain_stream_text(&mock, request).await;
178
179 assert_eq!(text, "Hello, world!");
180 assert_eq!(stop.as_deref(), Some("end_turn"));
181 assert_eq!(mock.call_count(), 1);
182 assert_eq!(mock.captured_requests().len(), 1);
183 }
184
185 // === 2. Reasoning replay (V4 thinking-mode HTTP-400 regression) =============
186
187 #[tokio::test]
188 async fn reasoning_replay_required_on_subsequent_turn() {
189 // Turn 1: assistant emits thinking + tool_call. Turn 2: text reply.
190 let turn1 = vec![
191 canned::message_start("r1"),
192 canned::thinking_delta(0, "I should call list_dir."),
193 canned::tool_use_block_start(1, "call_a", "list_dir"),
194 canned::tool_input_delta(1, r#"{"path":"/tmp"}"#),
195 canned::block_stop(1),
196 canned::message_delta("tool_use", None),
197 canned::message_stop(),
198 ];
199 let turn2 = vec![
200 canned::message_start("r2"),
201 canned::text_block_start(0),
202 canned::text_delta(0, "I see /tmp."),
203 canned::block_stop(0),
204 canned::message_delta("end_turn", None),
205 canned::message_stop(),
206 ];
207 let mock = MockLlmClient::new(vec![turn1, turn2]);
208
209 // === Round 1: user prompt -> assistant tool_call ===
210 let req1 = make_request(vec![user_message("list /tmp")]);
211 let _ = mock.create_message_stream(req1).await.unwrap().next().await;
212 // (we don't drain — capture is what matters here)
213
214 // === Round 2: runtime composes the next request including the prior
215 // assistant turn's reasoning_content. The mock can verify that any
216 // ContentBlock::Thinking the runtime preserves is present in the next
217 // outgoing request — the very payload shape that broke v0.4.9-v0.5.1.
218 let next_messages = vec![
219 user_message("list /tmp"),
220 assistant_thinking("I should call list_dir.", ""),
221 assistant_tool_call("call_a", "list_dir", serde_json::json!({ "path": "/tmp" })),
222 tool_result_message("call_a", "/tmp/file1\n/tmp/file2"),
223 ];
224 let req2 = make_request(next_messages);
225 let _ = mock.create_message_stream(req2).await.unwrap();
226
227 // The mock captured both requests. Assert the SECOND request preserves
228 // the prior assistant message's Thinking block — i.e. the runtime did
229 // not strip reasoning_content before re-sending. (V4 thinking-mode tool
230 // turns reject HTTP 400 if reasoning_content is missing.)
231 let captured = mock.captured_requests();
232 assert_eq!(captured.len(), 2);
233
234 let req2 = &captured[1];
235 let assistant_with_thinking = req2
236 .messages
237 .iter()
238 .find(|m| {
239 m.role == "assistant"
240 && m.content
241 .iter()
242 .any(|b| matches!(b, ContentBlock::Thinking { .. }))
243 })
244 .expect("turn 2 request must replay assistant Thinking content");
245
246 let thinking_text = assistant_with_thinking
247 .content
248 .iter()
249 .find_map(|b| match b {
250 ContentBlock::Thinking { thinking, .. } => Some(thinking.clone()),
251 _ => None,
252 })
253 .expect("Thinking block present");
254 assert_eq!(
255 thinking_text, "I should call list_dir.",
256 "reasoning_content must be replayed verbatim across tool-call rounds"
257 );
258 }
259
260 // === 3. Tool-call round-trip ================================================
261
262 #[tokio::test]
263 async fn tool_call_round_trip_streams_args_then_continues() {
264 // Turn 1 emits a tool_use block with chunked input JSON.
265 let turn1 = vec![
266 canned::message_start("rt1"),
267 canned::tool_use_block_start(0, "call_x", "read_file"),
268 canned::tool_input_delta(0, r#"{"path":"#),
269 canned::tool_input_delta(0, r#""README.md"}"#),
270 canned::block_stop(0),
271 canned::message_delta("tool_use", None),
272 canned::message_stop(),
273 ];
274 let turn2 = vec![
275 canned::message_start("rt2"),
276 canned::text_block_start(0),
277 canned::text_delta(0, "README starts with: # deepseek-tui"),
278 canned::block_stop(0),
279 canned::message_delta("end_turn", None),
280 canned::message_stop(),
281 ];
282 let mock = MockLlmClient::new(vec![turn1, turn2]);
283
284 // Round 1
285 let mut s1 = mock
286 .create_message_stream(make_request(vec![user_message("read README.md")]))
287 .await
288 .unwrap();
289
290 let mut tool_use_seen = false;
291 let mut json_seen = String::new();
292 while let Some(ev) = s1.next().await {
293 match ev.unwrap() {
294 StreamEvent::ContentBlockStart { content_block, .. } => {
295 use codewhale_models::ContentBlockStart;
296 if let ContentBlockStart::ToolUse { name, .. } = content_block {
297 assert_eq!(name, "read_file");
298 tool_use_seen = true;
299 }
300 }
301 StreamEvent::ContentBlockDelta {
302 delta: Delta::InputJsonDelta { partial_json },
303 ..
304 } => json_seen.push_str(&partial_json),
305 StreamEvent::MessageStop => break,
306 _ => {}
307 }
308 }
309 assert!(tool_use_seen);
310 let parsed: serde_json::Value =
311 serde_json::from_str(&json_seen).expect("valid JSON after concat");
312 assert_eq!(parsed["path"], "README.md");
313
314 // Round 2 — runtime sends back a tool_result and the mock replies with
315 // the final assistant text turn.
316 let req2 = make_request(vec![
317 user_message("read README.md"),
318 assistant_tool_call(
319 "call_x",
320 "read_file",
321 serde_json::json!({ "path": "README.md" }),
322 ),
323 tool_result_message("call_x", "# deepseek-tui\n..."),
324 ]);
325 let (text, stop) = drain_stream_text(&mock, req2).await;
326 assert!(text.contains("# deepseek-tui"));
327 assert_eq!(stop.as_deref(), Some("end_turn"));
328 }
329
330 // === 4. Multiple tool calls in one round (parallel ordering) ================
331
332 #[tokio::test]
333 async fn parallel_tool_calls_preserve_ordering_in_turn_payload() {
334 // Assistant returns two tool_calls in a single turn (indices 0 and 1).
335 // The runtime is free to execute them in parallel; this test asserts that
336 // the canonical event ordering survives a single-turn replay.
337 let turn = vec![
338 canned::message_start("p1"),
339 canned::tool_use_block_start(0, "call_one", "list_dir"),
340 canned::tool_input_delta(0, r#"{"path":"a"}"#),
341 canned::block_stop(0),
342 canned::tool_use_block_start(1, "call_two", "list_dir"),
343 canned::tool_input_delta(1, r#"{"path":"b"}"#),
344 canned::block_stop(1),
345 canned::message_delta("tool_use", None),
346 canned::message_stop(),
347 ];
348 let mock = MockLlmClient::new(vec![turn]);
349
350 let mut stream = mock
351 .create_message_stream(make_request(vec![user_message("list both")]))
352 .await
353 .unwrap();
354
355 let mut starts: Vec<(u32, String)> = Vec::new();
356 while let Some(ev) = stream.next().await {
357 if let StreamEvent::ContentBlockStart {
358 index,
359 content_block,
360 } = ev.unwrap()
361 {
362 use codewhale_models::ContentBlockStart;
363 if let ContentBlockStart::ToolUse { id, .. } = content_block {
364 starts.push((index, id));
365 }
366 }
367 }
368
369 assert_eq!(starts.len(), 2);
370 assert_eq!(starts[0], (0, "call_one".to_string()));
371 assert_eq!(starts[1], (1, "call_two".to_string()));
372 }
373
374 // === 5. Compaction-style non-streaming call =================================
375
376 #[tokio::test]
377 async fn compaction_non_streaming_returns_queued_message_response() {
378 use codewhale_models::MessageResponse;
379
380 let mock = MockLlmClient::new(vec![]);
381 mock.push_message_response(MessageResponse {
382 id: "compact_msg".to_string(),
383 r#type: "message".to_string(),
384 role: "assistant".to_string(),
385 content: vec![ContentBlock::Text {
386 text: "## Summary\n- Step 1\n- Step 2".to_string(),
387 cache_control: None,
388 }],
389 model: "deepseek-v4-pro".to_string(),
390 stop_reason: Some("end_turn".to_string()),
391 stop_sequence: None,
392 container: None,
393 usage: Usage::default(),
394 });
395
396 // The runtime's compaction path uses create_message (not stream).
397 let req = MessageRequest {
398 stream: Some(false),
399 ..make_request(vec![user_message("summarize")])
400 };
401 let resp = mock.create_message(req).await.unwrap();
402
403 let text = match &resp.content[0] {
404 ContentBlock::Text { text, .. } => text.clone(),
405 _ => panic!("expected text content"),
406 };
407 assert!(text.contains("Summary"));
408 assert_eq!(resp.id, "compact_msg");
409 assert_eq!(mock.call_count(), 1);
410 }
411
412 // === 6. Sub-agent style turn ================================================
413 //
414 // The next turn after an `agent` summary must re-verify the claimed
415 // side effect before reporting success.
416
417 #[tokio::test]
418 async fn v4_parent_reverifies_subagent_file_self_report_before_claiming_success() {
419 let tmp = tempfile::tempdir().expect("tempdir");
420 let missing = tmp.path().join("child-claimed-write.txt");
421 assert!(!missing.exists(), "fixture path must start missing");
422 let missing_path = missing.display().to_string();
423
424 let parent = MockLlmClient::new(vec![vec![
425 canned::message_start("parent_verify"),
426 canned::thinking_delta(0, "Verify the child's file-write self-report first."),
427 canned::tool_use_block_start(1, "verify_file", "read_file"),
428 canned::tool_input_delta(1, &serde_json::json!({ "path": &missing_path }).to_string()),
429 canned::block_stop(1),
430 canned::message_delta("tool_use", None),
431 canned::message_stop(),
432 ]])
433 .with_model("deepseek-v4-pro");
434 let tool_summary = format!(
435 "[sub-agent result summarized for parent context]\n\
436 Child results are self-reports; verify side effects with tools like read_file or list_dir before claiming success.\n\
437 - agent_filecheck (implementer) status=Completed\n result: Wrote {missing_path} successfully."
438 );
439
440 let mut stream = parent
441 .create_message_stream(make_request(vec![
442 user_message("Use a child to create the file, then report back."),
443 assistant_tool_call(
444 "agent_call",
445 "agent",
446 serde_json::json!({
447 "prompt": "Create the requested file and report the result.",
448 "role": "implementer"
449 }),
450 ),
451 tool_result_message("agent_call", &tool_summary),
452 ]))
453 .await
454 .unwrap();
455
456 let mut text_before_verification = String::new();
457 let mut tool_name = None;
458 let mut tool_input = String::new();
459 while let Some(ev) = stream.next().await {
460 match ev.unwrap() {
461 StreamEvent::ContentBlockStart { content_block, .. } => {
462 use codewhale_models::ContentBlockStart;
463 if let ContentBlockStart::ToolUse { name, .. } = content_block {
464 tool_name = Some(name);
465 }
466 }
467 StreamEvent::ContentBlockDelta { delta, .. } => match delta {
468 Delta::InputJsonDelta { partial_json } => tool_input.push_str(&partial_json),
469 Delta::TextDelta { text } => text_before_verification.push_str(&text),
470 _ => {}
471 },
472 StreamEvent::MessageStop => break,
473 _ => {}
474 }
475 }
476
477 assert_eq!(text_before_verification, "");
478 assert_eq!(tool_name.as_deref(), Some("read_file"));
479 let parsed: serde_json::Value = serde_json::from_str(&tool_input).expect("tool input JSON");
480 assert_eq!(parsed["path"], missing_path);
481 }
482
483 // === 7. Request capture observation =========================================
484 //
485 // The mock surfaces request captures BEFORE the response stream is opened, so
486 // trait-level tests can verify that captured requests are observable per-call
487 // rather than buffered across calls.
488
489 #[tokio::test]
490 async fn capacity_gate_can_observe_request_before_response_streams() {
491 let turn = vec![canned::simple_text_turn("ok")];
492 let mock = MockLlmClient::new(turn);
493
494 // Build a "near-limit" request — many user messages.
495 let mut messages = Vec::new();
496 for i in 0..200 {
497 messages.push(user_message(&format!("m{i}")));
498 }
499 let req = make_request(messages);
500
501 // BEFORE the runtime drains the stream, the mock has already captured
502 // the request. The capacity controller can inspect this and short-circuit
503 // the dispatch if the estimated token cost exceeds the soft cap.
504 let stream_future = mock.create_message_stream(req);
505 let mut stream = stream_future.await.unwrap();
506
507 assert_eq!(mock.captured_requests().len(), 1);
508 let captured = mock.last_request().unwrap();
509 assert_eq!(captured.messages.len(), 200);
510 // Verify the capacity gate could compute a "should defer" decision based
511 // on raw message count + payload size of the captured request.
512 let total_chars: usize = captured
513 .messages
514 .iter()
515 .flat_map(|m| m.content.iter())
516 .map(|b| match b {
517 ContentBlock::Text { text, .. } => text.len(),
518 _ => 0,
519 })
520 .sum();
521 assert!(
522 total_chars > 100,
523 "synthetic over-cap request should have non-trivial size"
524 );
525
526 // Drain to keep the mock state consistent.
527 while stream.next().await.is_some() {}
528 }
529
530 // === 8. Compaction defaults (#402 P0) ======================================
531
532 #[test]
533 fn compaction_config_defaults_are_enabled_for_session_survivability() {
534 // The production CompactionConfig is gated behind a `#[path = ...]` module
535 // that isn't wired here, but we can test the principle: the
536 // `should_compact` function and `CompactionConfig` live in the same crate.
537 // Re-import from the production module to verify the default.
538 //
539 // We test via the mock pathway: the non-streaming compaction call (test 5
540 // above) already exercises `create_message` with `stream: Some(false)`,
541 // which is the code path `compact_messages` uses. Combined with the
542 // capacity controller's `TargetedContextRefresh`, the enabled-by-default
543 // compaction config means long sessions auto-compact before hitting the
544 // context window limit.
545 //
546 // This test is a smoke check that the defaults compile and are correct.
547 // The production `CompactionConfig::default()` is exercised by
548 // `compaction::tests::should_compact_respects_enabled_flag` etc.
549 let config =
550 codewhale_models::compaction_threshold_for_model_at_percent("deepseek-v4-pro", 80.0);
551 // Verify the threshold is reasonable (> 0 and < context window).
552 assert!(config > 0, "compaction threshold must be positive");
553 assert!(config < 1_000_000, "compaction threshold must be below 1M");
554 }
555
555 lines RUST