返回 CodeWhale
tests.rs
根目录 / crates / tui / src / core / engine / preview / tests.rs
1 use super::*;
2 use crate::core::ops::TurnSpec;
3
4 fn tool(name: &str, deferred: bool) -> Tool {
5 Tool {
6 tool_type: None,
7 name: name.to_string(),
8 description: format!("{name} description"),
9 input_schema: json!({"type": "object", "properties": {}}),
10 allowed_callers: None,
11 defer_loading: Some(deferred),
12 input_examples: None,
13 strict: None,
14 cache_control: None,
15 }
16 }
17
18 #[test]
19 fn active_catalog_hash_tracks_membership_order_and_schema() {
20 let base = vec![tool("Bash", false), tool("File", false)];
21 let baseline = active_tool_catalog_sha256(&base);
22
23 assert_eq!(baseline, active_tool_catalog_sha256(&base.clone()));
24
25 let mut reordered = base.clone();
26 reordered.swap(0, 1);
27 assert_ne!(baseline, active_tool_catalog_sha256(&reordered));
28
29 let mut fewer = base.clone();
30 fewer.pop();
31 assert_ne!(baseline, active_tool_catalog_sha256(&fewer));
32
33 let mut retyped = base.clone();
34 retyped[0].input_schema = json!({"type": "object", "required": ["cmd"]});
35 assert_ne!(baseline, active_tool_catalog_sha256(&retyped));
36 }
37
38 #[test]
39 fn preview_and_production_share_one_input_estimate_and_send_decision() {
40 let messages = vec![Message {
41 role: Role::User,
42 content: vec![ContentBlock::Text {
43 text: "near the context ceiling".to_string(),
44 cache_control: None,
45 }],
46 }];
47 let system = SystemPrompt::Text("stable system".to_string());
48
49 // Production sends stored history and nothing else, and preview describes
50 // that same list, so a single estimate drives the manifest number and the
51 // overflow/no-exact-outbound decision. There is no second synthetic list
52 // to charge a separate framing overhead for.
53 let estimate = crate::compaction::estimate_input_tokens_conservative(&messages, Some(&system));
54
55 let ceiling = estimate - 1;
56 assert_eq!(
57 crate::request_manifest::production_input_headroom(Some(ceiling), estimate),
58 Some(-1)
59 );
60 assert!(crate::request_manifest::production_input_budget_exceeded(
61 Some(ceiling),
62 estimate
63 ));
64 assert!(!crate::request_manifest::production_input_budget_exceeded(
65 Some(estimate),
66 estimate
67 ));
68 }
69
70 #[test]
71 fn standard_and_full_are_reported_collapsed_from_the_real_shaper() {
72 let catalog = vec![tool("Bash", false), tool("agent", false), tool("Web", true)];
73 let always_load = std::collections::HashSet::new();
74 assert!(
75 standard_and_full_collapse(&catalog, &always_load),
76 "Standard and Full apply no narrowing today, so they must report collapsed"
77 );
78 }
79
80 #[tokio::test]
81 async fn pending_shell_completion_makes_the_body_unavailable_without_draining_it() {
82 let config = deepseek_config();
83 let identity = deepseek_identity();
84 let (mut engine, _handle, _tmp) = preview_engine(&config);
85 engine.config.features.disable(Feature::Mcp);
86 let owner_session_id = engine.session.id.clone();
87
88 {
89 let mut manager = engine.shell_manager.lock().expect("shell manager");
90 let command = if cfg!(windows) {
91 "Start-Sleep -Seconds 30"
92 } else {
93 "sleep 30"
94 };
95 manager
96 .execute_with_options_env_for_session(
97 command,
98 None,
99 30_000,
100 true,
101 None,
102 false,
103 None,
104 std::collections::HashMap::new(),
105 &owner_session_id,
106 )
107 .expect("background shell starts");
108 assert!(manager.may_have_undelivered_completion_for_session(&owner_session_id));
109 }
110
111 let planned = plan(&config, &identity, false, "inspect the next request").await;
112 let manifest = engine
113 .build_request_manifest(inputs(false, Some(planned), "inspect the next request"))
114 .await;
115 let unavailable = match &manifest.body {
116 Availability::Unavailable(unavailable) => unavailable,
117 Availability::Exact(_) => panic!("pending shell completion must fail closed"),
118 };
119 assert_eq!(
120 unavailable.reason,
121 UnavailableReason::RuntimeTransformsBeforeSend
122 );
123 assert!(
124 unavailable
125 .detail
126 .as_deref()
127 .is_some_and(|detail| detail.contains("background shell completion")),
128 "{unavailable:?}"
129 );
130
131 let mut manager = engine.shell_manager.lock().expect("shell manager");
132 assert!(
133 manager.may_have_undelivered_completion_for_session(&owner_session_id),
134 "preview must not drain or report the completion"
135 );
136 let _ = manager.kill_running();
137 let _ = manager.drain_finished_jobs_with_evidence();
138 }
139
140 #[tokio::test]
141 async fn running_direct_child_fails_closed_without_consuming_or_mutating_state() {
142 let config = deepseek_config();
143 let identity = deepseek_identity();
144 let (mut engine, _handle, tmp) = preview_engine(&config);
145 engine.config.features.disable(Feature::Mcp);
146 let active_session_id = engine.session.id.clone();
147 let mut before = {
148 let mut manager = engine.subagent_manager.write().await;
149 let agent_id = manager.insert_test_running_direct_child("preview_pending", tmp.path());
150 manager.assign_test_session_owner(&agent_id, &active_session_id);
151 serde_json::to_value(manager.list()).expect("manager snapshot")
152 };
153 if let Some(rows) = before.as_array_mut() {
154 for row in rows {
155 row.as_object_mut()
156 .expect("agent object")
157 .remove("duration_ms");
158 }
159 }
160 let delivered_before = engine.delivered_subagent_completion_ids.clone();
161
162 let planned = plan(&config, &identity, false, "inspect while child runs").await;
163 let manifest = engine
164 .build_request_manifest(inputs(false, Some(planned), "inspect while child runs"))
165 .await;
166 let unavailable = match &manifest.body {
167 Availability::Unavailable(unavailable) => unavailable,
168 Availability::Exact(_) => panic!("running child must fail closed"),
169 };
170 assert_eq!(
171 unavailable.reason,
172 UnavailableReason::RuntimeTransformsBeforeSend
173 );
174 assert!(
175 unavailable
176 .detail
177 .as_deref()
178 .is_some_and(|detail| detail.contains("running or undelivered sub-agent"))
179 );
180
181 let mut after = {
182 let manager = engine.subagent_manager.read().await;
183 assert!(manager.may_transform_next_parent_request_for_session(
184 &active_session_id,
185 &engine.delivered_subagent_completion_ids,
186 ));
187 serde_json::to_value(manager.list()).expect("manager snapshot")
188 };
189 if let Some(rows) = after.as_array_mut() {
190 for row in rows {
191 row.as_object_mut()
192 .expect("agent object")
193 .remove("duration_ms");
194 }
195 }
196 assert_eq!(before, after, "preview must not mutate child state");
197 assert_eq!(
198 delivered_before, engine.delivered_subagent_completion_ids,
199 "preview must not claim child delivery"
200 );
201 }
202
203 #[tokio::test]
204 async fn terminal_undelivered_child_fails_closed_without_claiming_delivery() {
205 let config = deepseek_config();
206 let identity = deepseek_identity();
207 let (mut engine, _handle, tmp) = preview_engine(&config);
208 engine.config.features.disable(Feature::Mcp);
209 let active_session_id = engine.session.id.clone();
210 let agent_id = {
211 let mut manager = engine.subagent_manager.write().await;
212 let agent_id = manager.insert_test_terminal_direct_child("preview_terminal", tmp.path());
213 manager.assign_test_session_owner(&agent_id, &active_session_id);
214 agent_id
215 };
216
217 let planned = plan(&config, &identity, false, "inspect settled child").await;
218 let manifest = engine
219 .build_request_manifest(inputs(false, Some(planned), "inspect settled child"))
220 .await;
221 assert!(matches!(manifest.body, Availability::Unavailable(_)));
222 assert!(
223 !engine.delivered_subagent_completion_ids.contains(&agent_id),
224 "preview must not claim terminal delivery"
225 );
226 let manager = engine.subagent_manager.read().await;
227 assert!(manager.may_transform_next_parent_request_for_session(
228 &active_session_id,
229 &engine.delivered_subagent_completion_ids,
230 ));
231 assert!(matches!(
232 manager
233 .get_result(&agent_id)
234 .expect("terminal child")
235 .status,
236 crate::tools::subagent::SubAgentStatus::Completed
237 ));
238 }
239
240 #[test]
241 fn turn_metadata_uses_planned_cross_route_limits_not_installed_limits() {
242 let config = deepseek_config();
243 let (mut engine, _handle, _tmp) = preview_engine(&config);
244 // Pressure advice is only surfaced when the user opts out of automatic
245 // maintenance. The route-budget assertion still applies in that mode.
246 engine.config.compaction.enabled = false;
247 engine.api_provider = ProviderKind::Deepseek;
248 let installed_limits = codewhale_config::route::RouteLimits {
249 context_tokens: Some(4_096),
250 input_tokens: None,
251 output_tokens: Some(512),
252 };
253 engine.active_route_limits = Some(installed_limits);
254 // Large enough to be critical for the installed 4K route, but safely
255 // below the warning threshold for the planned 123K route.
256 engine.session.messages.push(Message {
257 role: Role::User,
258 content: vec![ContentBlock::Text {
259 text: "x".repeat(20_000),
260 cache_control: None,
261 }],
262 });
263 let prompt_context = NextTurnPromptContext::for_planned_turn(
264 ProviderKind::Openrouter,
265 "qwen/qwen3.6-flash".to_string(),
266 Some(codewhale_config::route::RouteLimits {
267 context_tokens: Some(123_456),
268 input_tokens: None,
269 output_tokens: Some(4_096),
270 }),
271 AppMode::Agent,
272 None,
273 GoalStatus::Active,
274 None,
275 false,
276 None,
277 );
278 let system_prompt = engine.compose_stable_system_prompt(&prompt_context);
279 assert_eq!(
280 engine.context_pressure_line(
281 "cross-route budget",
282 &prompt_context,
283 system_prompt.as_ref()
284 ),
285 None,
286 "the planned 123K route must not inherit the installed route's pressure"
287 );
288 let installed_context = NextTurnPromptContext::for_planned_turn(
289 ProviderKind::Deepseek,
290 "deepseek-v4-flash".to_string(),
291 Some(installed_limits),
292 AppMode::Agent,
293 None,
294 GoalStatus::Active,
295 None,
296 false,
297 None,
298 );
299 let pressure = engine
300 .context_pressure_line("cross-route budget", &installed_context, None)
301 .unwrap();
302 assert!(pressure.contains("Context pressure: critical"));
303 assert!(pressure.contains("Estimated input:"));
304 assert!(pressure.contains("Making room automatically is off"));
305 let message = engine.user_text_message_from_snapshot(
306 "cross-route budget".to_string(),
307 &prompt_context.model,
308 true,
309 None,
310 false,
311 UserInputProvenance::ExternalUser,
312 TurnMetadataSnapshot {
313 prompt_context: &prompt_context,
314 system_prompt: system_prompt.as_ref(),
315 approval_mode: ApprovalMode::Suggest,
316 working_set: &engine.session.working_set,
317 policy_narrowing: None,
318 },
319 );
320 let metadata = message
321 .content
322 .iter()
323 .filter_map(|block| match block {
324 ContentBlock::Text { text, .. } => Some(text.as_str()),
325 _ => None,
326 })
327 .next_back()
328 .expect("turn metadata text");
329 assert!(
330 !metadata.contains("Context pressure:"),
331 "planned route metadata must remain below warning: {metadata}"
332 );
333 assert!(!metadata.contains("123456 tokens"), "{metadata}");
334 assert!(!metadata.contains("4096 tokens"), "{metadata}");
335 }
336
337 /// #perf-r5: the pressure-line helper must estimate the history IN PLACE and
338 /// add the composer text arithmetically. Guards two things at once:
339 ///
340 /// 1. Equivalence — the arithmetic form must equal the naive
341 /// "clone + push + estimate" reference for non-trivial inputs (Unicode
342 /// multi-byte content included, since Text blocks count *chars* for the
343 /// conservative estimator but the delta path counts... the same rule as
344 /// `estimate_tokens_for_message`: bytes/4).
345 /// 2. The contract that empty/no-op composer text costs nothing extra.
346 #[test]
347 fn context_pressure_delta_matches_clone_and_push_reference() {
348 let config = deepseek_config();
349 let (mut engine, _handle, _tmp) = preview_engine(&config);
350 engine.api_provider = ProviderKind::Deepseek;
351 let installed_limits = codewhale_config::route::RouteLimits {
352 context_tokens: Some(64_000),
353 input_tokens: None,
354 output_tokens: Some(512),
355 };
356 engine.active_route_limits = Some(installed_limits);
357 // Multi-byte content on purpose: chars().count() != len() here, so an
358 // arity mistake between the byte rule (estimator) would surface.
359 engine.session.messages.push(Message {
360 role: Role::User,
361 content: vec![ContentBlock::Text {
362 text: "héllo wörld — ünïcode ✓ ".repeat(500),
363 cache_control: None,
364 }],
365 });
366 engine.session.messages.push(Message {
367 role: Role::Assistant,
368 content: vec![ContentBlock::Thinking {
369 thinking: "step".repeat(100),
370 signature: None,
371 state: None,
372 }],
373 });
374 // Replayed-reasoning case (#perf-r5 fresh-eyes fix): an assistant message
375 // carrying BOTH thinking and a tool call keeps its reasoning content in
376 // every subsequent request — the estimator counts those bytes, and this
377 // was the exact arm the delta helper originally missed. Both parity
378 // variants of the thinking byte-count are exercised below.
379 engine.session.messages.push(Message {
380 role: Role::Assistant,
381 content: vec![
382 ContentBlock::Thinking {
383 thinking: "replayed".repeat(300), // 8 bytes per unit -> even count
384 signature: None,
385 state: None,
386 },
387 ContentBlock::ToolUse {
388 execution_id: None,
389 id: "call_1".to_string(),
390 name: "bash".to_string(),
391 input: json!({"command": "echo hello"}),
392 caller: None,
393 thought_signature: None,
394 },
395 ],
396 });
397 engine.session.messages.push(Message {
398 role: Role::Assistant,
399 content: vec![
400 ContentBlock::Thinking {
401 thinking: "odd replay".to_string(), // 11 bytes / 4 = 2 (even)... use odd total
402 signature: None,
403 state: None,
404 },
405 ContentBlock::ToolUse {
406 execution_id: None,
407 id: "call_2".to_string(),
408 name: "read".to_string(),
409 input: json!({"path": "x"}), // 13-byte JSON -> 3
410 caller: None,
411 thought_signature: None,
412 },
413 ],
414 });
415 let _prompt_context = NextTurnPromptContext::for_planned_turn(
416 ProviderKind::Deepseek,
417 "deepseek-v4-flash".to_string(),
418 Some(installed_limits),
419 AppMode::Agent,
420 None,
421 GoalStatus::Active,
422 None,
423 false,
424 None,
425 );
426 let _ = &_prompt_context;
427
428 // Naive reference implementation: clone the transcript, push a
429 // hypothetical user message, run the full conservative estimator.
430 let reference = |engine: &Engine, text: &str| -> usize {
431 let mut messages: Vec<Message> =
432 crate::prompt_zones::AppendLog::clone(&engine.session.messages).into();
433 if !text.trim().is_empty() {
434 messages.push(Message {
435 role: Role::User,
436 content: vec![ContentBlock::Text {
437 text: text.to_string(),
438 cache_control: None,
439 }],
440 });
441 }
442 crate::compaction::estimate_input_tokens_conservative(&messages, None)
443 };
444
445 for text in [
446 "",
447 " ",
448 "short",
449 "a much longer composer draft with punctuation…",
450 ] {
451 let via_pressure_line_input = engine.active_input_tokens_with_current_text(text, None);
452 assert_eq!(
453 via_pressure_line_input,
454 reference(&engine, text),
455 "delta arithmetic diverged from clone+push+estimate for {text:?}"
456 );
457 }
458 }
459
460 /// #perf-r5 guard: billed input above the threshold must report pressure with
461 /// a provably-empty history — proving the short-circuit answers from billing
462 /// alone without consulting message contents.
463 #[test]
464 fn billed_pressure_above_threshold_answers_from_billing_alone() {
465 let config = CompactionConfig {
466 enabled: true,
467 token_threshold: 1_000,
468 ..Default::default()
469 };
470 let pressure = crate::compaction::compaction_pressure_reached_with_billed(
471 &[], // empty history: only billing can prove pressure
472 None,
473 &config,
474 Some(2_000),
475 );
476 assert!(pressure, "billed 2000 >= threshold 1000 must be pressure");
477 }
478
479 /// #perf-r5 guard: under-threshold billing keeps the old max() semantics —
480 /// an estimate above the trigger still fires even when billing is quiet.
481 #[test]
482 fn billed_below_threshold_still_fires_on_estimate() {
483 let config = CompactionConfig {
484 enabled: true,
485 token_threshold: 100,
486 ..Default::default()
487 };
488 let big = Message {
489 role: Role::User,
490 content: vec![ContentBlock::Text {
491 text: "x".repeat(4 * 200),
492 cache_control: None,
493 }],
494 };
495 let pressure = crate::compaction::compaction_pressure_reached_with_billed(
496 std::slice::from_ref(&big),
497 None,
498 &config,
499 Some(10), // below threshold; must not short-circuit to false either
500 );
501 assert!(pressure, "estimate 200 (+1.0 framing) >= 100 must fire");
502 }
503
504 /// #perf-r5 guard: a direct `session.messages` overwrite (the SyncSession
505 /// restore path) must advance `messages_revision` so the token-estimate
506 /// cache invalidates instead of serving the pre-sync value.
507 #[test]
508 fn sync_restore_bumps_messages_revision_for_estimate_cache() {
509 use crate::core::engine::token_estimate_cache::TokenEstimateCache;
510
511 let config = deepseek_config();
512 let (mut engine, _handle, _tmp) = preview_engine(&config);
513 engine.session.add_message(Message {
514 role: Role::User,
515 content: vec![ContentBlock::Text {
516 text: "before restore".to_string(),
517 cache_control: None,
518 }],
519 });
520 let revision_before = engine.session.messages_revision;
521 let mut cache = TokenEstimateCache::new();
522 let stale = cache.lookup_or_compute(
523 revision_before,
524 engine.session.system_prompt.as_ref(),
525 &engine.session.messages,
526 );
527
528 // Simulate the restore's direct field assignment.
529 engine.session.messages = Vec::new().into();
530 engine.session.bump_messages_revision();
531
532 assert_ne!(
533 engine.session.messages_revision, revision_before,
534 "restore must bump the revision the estimate cache keys on"
535 );
536 let fresh = cache.lookup_or_compute(
537 engine.session.messages_revision,
538 engine.session.system_prompt.as_ref(),
539 &engine.session.messages,
540 );
541 assert_ne!(
542 fresh, stale,
543 "cache must recompute after a restore-driven revision bump"
544 );
545 }
546
547 #[tokio::test]
548 async fn compaction_preview_uses_the_planned_routes_system_prompt() {
549 let config = deepseek_config();
550 let (mut engine, _handle, _tmp) = preview_engine(&config);
551 engine.config.features.disable(Feature::Mcp);
552
553 let messages: Vec<Message> = (0..30)
554 .map(|index| Message {
555 role: if index % 2 == 0 {
556 Role::User
557 } else {
558 Role::Assistant
559 },
560 content: vec![ContentBlock::Text {
561 text: "x".repeat(10_000),
562 cache_control: None,
563 }],
564 })
565 .collect();
566 let installed_prompt = SystemPrompt::Text("installed route".to_string());
567 let planned_prompt = SystemPrompt::Text("planned-route-system ".repeat(1_500));
568 engine.session.system_prompt = Some(installed_prompt.clone());
569
570 let installed_pressure =
571 crate::compaction::estimate_input_tokens_for_pressure(&messages, Some(&installed_prompt));
572 let planned_pressure =
573 crate::compaction::estimate_input_tokens_for_pressure(&messages, Some(&planned_prompt));
574 assert!(planned_pressure > installed_pressure);
575 let compaction = crate::compaction::CompactionConfig {
576 enabled: true,
577 token_threshold: installed_pressure + (planned_pressure - installed_pressure) / 2,
578 ..Default::default()
579 };
580
581 let planned_reasons = engine
582 .preview_runtime_transforms(&messages, Some(&planned_prompt), &compaction)
583 .await;
584 assert!(
585 planned_reasons.contains(&"making room would summarize the conversation first"),
586 "the planned route prompt crosses the compaction threshold: {planned_reasons:?}"
587 );
588
589 let installed_reasons = engine
590 .preview_runtime_transforms(&messages, Some(&installed_prompt), &compaction)
591 .await;
592 assert!(
593 !installed_reasons.contains(&"making room would summarize the conversation first"),
594 "the installed route prompt is the below-threshold control: {installed_reasons:?}"
595 );
596 }
597
598 #[tokio::test]
599 async fn planned_route_builds_subagent_catalog_without_installed_client() {
600 let config = deepseek_config();
601 let identity = deepseek_identity();
602 let (mut engine, _handle, _tmp) = preview_engine(&config);
603 engine.config.features.disable(Feature::Mcp);
604 let _ = engine.config.features.enable(Feature::Subagents);
605 engine.config.subagents_enabled = true;
606 engine.codewhale_client = None;
607 let planned = plan(&config, &identity, false, "planned child route").await;
608 let route = planned.route.validate().expect("planned route validates");
609 let planned_model = route.model.clone();
610 let policy = TurnAuthority::from_effective_fields(
611 AppMode::Agent,
612 false,
613 false,
614 false,
615 ApprovalMode::Suggest,
616 );
617 let build = engine
618 .build_turn_tool_registry_and_catalog(
619 &policy,
620 &[],
621 None,
622 SubAgentWiring::Inert,
623 McpAccess::PassiveSnapshot,
624 TurnRouteContext {
625 provider: route.identity.provider,
626 model: route.model.clone(),
627 capabilities: route.candidate.capabilities(),
628 limits: crate::route_budget::known_route_limits(route.candidate.limits()),
629 client: Some(route.client),
630 api_config: route.config,
631 locale_tag: engine.config.locale_tag.clone(),
632 role_models: engine.subagent_role_models(),
633 auto_model: false,
634 reasoning_effort: planned.effective_reasoning_effort,
635 reasoning_effort_auto: planned.auto_controls_reasoning,
636 },
637 "",
638 )
639 .await;
640 assert!(
641 build
642 .surface
643 .catalog
644 .iter()
645 .any(|tool| tool.name == "agent"),
646 "the planned route client must make sub-agent tools available"
647 );
648 assert_eq!(
649 build.subagent_runtime_model.as_deref(),
650 Some(planned_model.as_str()),
651 "the child runtime must carry the planned route model"
652 );
653 }
654
655 /// Auto routing with no hypothetical prompt: every route-derived fact is
656 /// structurally absent, and the flag is never cleared just because the
657 /// session happens to have an installed route.
658 #[tokio::test]
659 async fn auto_route_without_a_prompt_omits_every_final_fact() {
660 let tmp = tempfile::tempdir().expect("tempdir");
661 let (mut engine, _handle) = Engine::new(
662 EngineConfig {
663 workspace: tmp.path().to_path_buf(),
664 ..Default::default()
665 },
666 &crate::config::Config::default(),
667 );
668
669 let manifest = engine
670 .build_request_manifest(PreviewRequestInputs {
671 mode: AppMode::Agent,
672 allow_shell: false,
673 trust_mode: false,
674 auto_approve: false,
675 approval_mode: ApprovalMode::Suggest,
676 allowed_tools: None,
677 dynamic_tools: Vec::new(),
678 provenance: UserInputProvenance::ExternalUser,
679 requested_model: "auto".to_string(),
680 requested_reasoning: "auto".to_string(),
681 auto_model: true,
682 hypothetical_prompt_supplied: false,
683 next_turn: None,
684 unresolved: PreviewUnresolved::AutoRouteNeedsPrompt,
685 })
686 .await;
687
688 assert!(manifest.route.exact().is_none());
689 assert!(manifest.tools.exact().is_none());
690 assert!(manifest.body.exact().is_none());
691 assert_eq!(manifest.session.requested_model.as_str(), "auto");
692 assert!(manifest.session.auto_model_routing);
693 assert!(!manifest.session.hypothetical_prompt_supplied);
694
695 let json = manifest.to_json();
696 for forbidden in [
697 "provider_id",
698 "wire_model",
699 "endpoint_fingerprint",
700 "body_sha256",
701 "tool_surface_budget",
702 "billing",
703 ] {
704 assert!(!json.contains(forbidden), "{forbidden} leaked:\n{json}");
705 }
706 }
707
708 fn deepseek_config() -> crate::config::Config {
709 let providers = crate::config::ProvidersConfig {
710 deepseek: crate::config::ProviderConfig {
711 api_key: Some("sk-test-deepseek".to_string()),
712 model: Some("deepseek-chat".to_string()),
713 ..crate::config::ProviderConfig::default()
714 },
715 ..crate::config::ProvidersConfig::default()
716 };
717 crate::config::Config {
718 provider: Some("deepseek".to_string()),
719 providers: Some(providers),
720 ..crate::config::Config::default()
721 }
722 }
723
724 fn deepseek_identity() -> crate::config::ProviderIdentity {
725 crate::config::Config::default()
726 .builtin_provider_identity(ProviderKind::Deepseek)
727 .unwrap()
728 }
729
730 /// Run the *production* route planner, provider-free: with `auto_model`
731 /// the classifier short-circuits to the inventory heuristic under `cfg!(test)`.
732 async fn plan(
733 config: &crate::config::Config,
734 identity: &crate::config::ProviderIdentity,
735 auto_model: bool,
736 prompt: &str,
737 ) -> crate::turn_route_plan::PlannedTurnRoute {
738 plan_for(
739 config,
740 identity,
741 ProviderKind::Deepseek,
742 "deepseek-chat",
743 auto_model,
744 prompt,
745 )
746 .await
747 }
748
749 async fn plan_for(
750 config: &crate::config::Config,
751 identity: &crate::config::ProviderIdentity,
752 provider: ProviderKind,
753 model: &str,
754 auto_model: bool,
755 prompt: &str,
756 ) -> crate::turn_route_plan::PlannedTurnRoute {
757 plan_with_reasoning(
758 config,
759 identity,
760 provider,
761 model,
762 auto_model,
763 if auto_model {
764 crate::reasoning_preference::ReasoningEffort::Auto
765 } else {
766 crate::reasoning_preference::ReasoningEffort::High
767 },
768 prompt,
769 )
770 .await
771 }
772
773 /// `plan_for` with the requested reasoning tier under test control. The
774 /// exact-route matrix needs `off` to observe a route that normalizes it
775 /// (direct Moonshot K3 sends `low`).
776 async fn plan_with_reasoning(
777 config: &crate::config::Config,
778 identity: &crate::config::ProviderIdentity,
779 provider: ProviderKind,
780 model: &str,
781 auto_model: bool,
782 reasoning_effort: crate::reasoning_preference::ReasoningEffort,
783 prompt: &str,
784 ) -> crate::turn_route_plan::PlannedTurnRoute {
785 crate::turn_route_plan::plan_turn_route(crate::turn_route_plan::TurnRoutePlanRequest {
786 route_config: config,
787 app_route_identity: identity,
788 api_provider: provider,
789 app_model: model,
790 auto_model,
791 reasoning_effort,
792 mode: AppMode::Agent,
793 content: prompt,
794 auto_router_context: "",
795 should_auto_resolve: auto_model,
796 allow_auto_router_response_cache: false,
797 preflight_required: false,
798 auto_compact_user_configured: false,
799 auto_compact: true,
800 auto_compact_threshold_percent: 80.0,
801 })
802 .await
803 .expect("the shared planner resolves a configured route")
804 }
805
806 fn preview_engine(config: &crate::config::Config) -> (Engine, EngineHandle, tempfile::TempDir) {
807 let tmp = tempfile::tempdir().expect("tempdir");
808 let (engine, handle) = Engine::new(
809 EngineConfig {
810 workspace: tmp.path().to_path_buf(),
811 ..Default::default()
812 },
813 config,
814 );
815 (engine, handle, tmp)
816 }
817
818 /// The engine plus its handle: a turn is admitted only while its event
819 /// consumer is alive, as in every production embedding, so callers keep the
820 /// handle for the engine's lifetime even though these fixtures read no events.
821 fn wire_preview_engine(
822 config: &crate::config::Config,
823 ) -> (Engine, crate::core::engine::EngineHandle, tempfile::TempDir) {
824 let tmp = tempfile::tempdir().expect("tempdir");
825 let (mut engine, handle) = Engine::new(
826 EngineConfig {
827 workspace: tmp.path().to_path_buf(),
828 max_steps: 1,
829 snapshots_enabled: false,
830 terminal_chrome_enabled: false,
831 ..Default::default()
832 },
833 config,
834 );
835 engine.config.features.disable(Feature::Mcp);
836 engine.config.subagents_enabled = false;
837 (engine, handle, tmp)
838 }
839
840 fn inputs(
841 auto_model: bool,
842 planned: Option<crate::turn_route_plan::PlannedTurnRoute>,
843 prompt: &str,
844 ) -> PreviewRequestInputs {
845 PreviewRequestInputs {
846 mode: AppMode::Agent,
847 allow_shell: false,
848 trust_mode: false,
849 auto_approve: false,
850 approval_mode: ApprovalMode::Suggest,
851 allowed_tools: None,
852 dynamic_tools: Vec::new(),
853 provenance: UserInputProvenance::ExternalUser,
854 requested_model: if auto_model {
855 "auto".to_string()
856 } else {
857 "deepseek-chat".to_string()
858 },
859 requested_reasoning: if auto_model { "auto" } else { "high" }.to_string(),
860 auto_model,
861 hypothetical_prompt_supplied: true,
862 next_turn: planned.map(|planned| {
863 let prompt_context = NextTurnPromptContext::for_planned_turn(
864 planned.route.identity.provider,
865 planned.route.model.clone(),
866 crate::route_budget::known_route_limits(planned.route.candidate.limits()),
867 AppMode::Agent,
868 None,
869 GoalStatus::Active,
870 None,
871 false,
872 None,
873 );
874 Box::new(PreviewNextTurn {
875 content: prompt.to_string(),
876 route: Box::new(planned.route),
877 prompt_context,
878 reasoning_effort: planned.effective_reasoning_effort,
879 reasoning_effort_auto: planned.auto_controls_reasoning,
880 auto_route_source: planned
881 .auto_selection
882 .as_ref()
883 .map(|selection| selection.source.label().to_string()),
884 routing_source: planned.routing_source,
885 compaction: planned.compaction,
886 })
887 }),
888 unresolved: PreviewUnresolved::NoPrompt,
889 }
890 }
891
892 /// Typed controls for the preview/wire parity fixture. Defaults mirror the
893 /// ordinary active DeepSeek turn; individual tests override only the
894 /// production context they are proving.
895 struct PreviewWireFixture {
896 goal_objective: Option<String>,
897 goal_status: GoalStatus,
898 translation_enabled: bool,
899 verbosity: Option<String>,
900 requested_model: Option<String>,
901 requested_reasoning: Option<String>,
902 }
903
904 impl Default for PreviewWireFixture {
905 fn default() -> Self {
906 Self {
907 goal_objective: None,
908 goal_status: GoalStatus::Active,
909 translation_enabled: false,
910 verbosity: None,
911 requested_model: None,
912 requested_reasoning: None,
913 }
914 }
915 }
916
917 async fn assert_preview_matches_first_wire_body(
918 engine: &mut Engine,
919 server: &wiremock::MockServer,
920 planned: crate::turn_route_plan::PlannedTurnRoute,
921 prompt: &str,
922 fixture: PreviewWireFixture,
923 ) -> (RequestManifest, serde_json::Value) {
924 let PreviewWireFixture {
925 goal_objective,
926 goal_status,
927 translation_enabled,
928 verbosity,
929 requested_model,
930 requested_reasoning,
931 } = fixture;
932 let production_route = planned.route.clone();
933 let compaction = planned.compaction.clone();
934 let reasoning_effort = planned.effective_reasoning_effort.clone();
935 let reasoning_effort_auto = planned.auto_controls_reasoning;
936 let mut preview_inputs = inputs(false, Some(planned), prompt);
937 if let Some(requested_model) = requested_model {
938 preview_inputs.requested_model = requested_model;
939 }
940 if let Some(requested_reasoning) = requested_reasoning {
941 preview_inputs.requested_reasoning = requested_reasoning;
942 }
943 let next = preview_inputs.next_turn.as_mut().expect("planned preview");
944 next.prompt_context = NextTurnPromptContext::for_planned_turn(
945 production_route.identity.provider,
946 production_route.model.clone(),
947 crate::route_budget::known_route_limits(production_route.candidate.limits()),
948 AppMode::Agent,
949 goal_objective.clone(),
950 goal_status,
951 None,
952 translation_enabled,
953 verbosity.clone(),
954 );
955 let manifest = engine.build_request_manifest(preview_inputs).await;
956 let preview_hash = manifest
957 .body
958 .exact()
959 .expect("preview body is exact")
960 .body_sha256
961 .clone();
962
963 let _ = engine
964 .handle_send_message(TurnSpec {
965 content: prompt.to_string(),
966 mode: AppMode::Agent,
967 route: Box::new(production_route),
968 compaction: Box::new(compaction),
969 initial_routed_usage: Box::new(crate::cost_status::RuntimeUsageBatch::default()),
970 goal_objective,
971 goal_token_budget: None,
972 goal_status,
973 reasoning_effort,
974 reasoning_effort_auto,
975 auto_model: false,
976 allow_shell: false,
977 trust_mode: false,
978 auto_approve: false,
979 approval_mode: ApprovalMode::Suggest,
980 translation_enabled,
981 allowed_tools: None,
982 dynamic_tools: Vec::new(),
983 hook_executor: None,
984 verbosity,
985 provenance: UserInputProvenance::ExternalUser,
986 images: Vec::new(),
987 max_output_tokens: None,
988 submission_id: None,
989 })
990 .await;
991
992 let requests = server
993 .received_requests()
994 .await
995 .expect("wire mock records requests");
996 assert_eq!(requests.len(), 1, "the fixture must make one provider call");
997 let first_wire_body: serde_json::Value =
998 serde_json::from_slice(&requests[0].body).expect("first HTTP body is JSON");
999 let first_wire_hash =
1000 crate::hashing::sha256_hex(crate::client::canonical_json(&first_wire_body).as_bytes());
1001 assert_eq!(
1002 preview_hash, first_wire_hash,
1003 "preview hash must match the body captured at the HTTP boundary"
1004 );
1005 (manifest, first_wire_body)
1006 }
1007
1008 #[tokio::test]
1009 async fn graph_backed_todo_is_not_reinjected_into_the_first_http_body() {
1010 use wiremock::matchers::method;
1011 use wiremock::{Mock, MockServer, ResponseTemplate};
1012
1013 let server = MockServer::start().await;
1014 Mock::given(method("POST"))
1015 .respond_with(
1016 ResponseTemplate::new(200)
1017 .insert_header("content-type", "text/event-stream")
1018 .set_body_string("data: [DONE]\n\n"),
1019 )
1020 .mount(&server)
1021 .await;
1022 let mut config = deepseek_config();
1023 config
1024 .providers
1025 .as_mut()
1026 .expect("providers")
1027 .deepseek
1028 .base_url = Some(server.uri());
1029 let identity = deepseek_identity();
1030 let (mut engine, _handle, _tmp) = wire_preview_engine(&config);
1031 let graph_todos = crate::tools::todo::TodoListSnapshot {
1032 items: vec![crate::tools::todo::TodoItem {
1033 id: 1,
1034 content: "preserve this graph-authoritative Work item".to_string(),
1035 status: crate::tools::todo::TodoStatus::InProgress,
1036 }],
1037 completion_pct: 0,
1038 in_progress_id: Some(1),
1039 };
1040 let work = crate::work_graph::new_shared_work_runtime(
1041 engine.config.todos.clone(),
1042 engine.config.plan_state.clone(),
1043 );
1044 work.restore(
1045 "preview-graph-work",
1046 None,
1047 &graph_todos,
1048 &crate::tools::plan::PlanSnapshot::default(),
1049 )
1050 .expect("restore graph-backed Work state");
1051 *engine.config.todos.lock().await = crate::tools::todo::TodoList::new();
1052 assert!(
1053 engine.config.todos.lock().await.snapshot().is_empty(),
1054 "legacy projection is intentionally stale for this authority test"
1055 );
1056 engine.config.runtime_services.work = Some(work);
1057
1058 let prompt = "inspect the request without restating the To-do list";
1059 let planned = plan(&config, &identity, false, prompt).await;
1060 let (_, first_wire_body) = assert_preview_matches_first_wire_body(
1061 &mut engine,
1062 &server,
1063 planned,
1064 prompt,
1065 PreviewWireFixture::default(),
1066 )
1067 .await;
1068 let body_text = first_wire_body.to_string();
1069 assert!(
1070 !body_text.contains("<codewhale:work_state>")
1071 && !body_text.contains("preserve this graph-authoritative Work item"),
1072 "provider requests must not receive a synthetic per-step To-do tail: {body_text}"
1073 );
1074 }
1075
1076 #[tokio::test]
1077 async fn exhausted_active_goal_remains_previewable() {
1078 let config = deepseek_config();
1079 let identity = deepseek_identity();
1080 let (mut engine, _handle, _tmp) = preview_engine(&config);
1081 engine.config.features.disable(Feature::Mcp);
1082 sync_goal_state_from_host(
1083 &engine.config.goal_state,
1084 Some("finish the release"),
1085 Some(100),
1086 GoalStatus::Active,
1087 );
1088 engine
1089 .config
1090 .goal_state
1091 .lock()
1092 .expect("goal state")
1093 .record_usage(100, 0);
1094
1095 let prompt = "continue the release";
1096 let planned = plan(&config, &identity, false, prompt).await;
1097 let manifest = engine
1098 .build_request_manifest(inputs(false, Some(planned), prompt))
1099 .await;
1100 assert!(manifest.route.exact().is_some());
1101 assert!(manifest.tools.exact().is_some());
1102 assert!(manifest.body.exact().is_some());
1103 }
1104
1105 #[tokio::test]
1106 async fn resumed_goal_with_raised_budget_becomes_previewable_again() {
1107 let config = deepseek_config();
1108 let identity = deepseek_identity();
1109 let (mut engine, _handle, _tmp) = preview_engine(&config);
1110 engine.config.features.disable(Feature::Mcp);
1111 sync_goal_state_from_host(
1112 &engine.config.goal_state,
1113 Some("finish the release"),
1114 Some(100),
1115 GoalStatus::Active,
1116 );
1117 {
1118 let mut state = engine.config.goal_state.lock().expect("goal state");
1119 state.record_usage(100, 0);
1120 state
1121 .mark_paused(GoalPauseReason::BudgetLimit)
1122 .expect("pause goal");
1123 }
1124 sync_goal_state_from_host(
1125 &engine.config.goal_state,
1126 Some("finish the release"),
1127 Some(200),
1128 GoalStatus::Active,
1129 );
1130
1131 let prompt = "continue under the raised budget";
1132 let planned = plan(&config, &identity, false, prompt).await;
1133 let manifest = engine
1134 .build_request_manifest(inputs(false, Some(planned), prompt))
1135 .await;
1136 assert!(manifest.body.exact().is_some());
1137 }
1138
1139 #[tokio::test]
1140 async fn lowering_active_goal_budget_below_used_tokens_keeps_preview_open() {
1141 let config = deepseek_config();
1142 let identity = deepseek_identity();
1143 let (mut engine, _handle, _tmp) = preview_engine(&config);
1144 engine.config.features.disable(Feature::Mcp);
1145 sync_goal_state_from_host(
1146 &engine.config.goal_state,
1147 Some("finish the release"),
1148 Some(200),
1149 GoalStatus::Active,
1150 );
1151 engine
1152 .config
1153 .goal_state
1154 .lock()
1155 .expect("goal state")
1156 .record_usage(100, 0);
1157 sync_goal_state_from_host(
1158 &engine.config.goal_state,
1159 Some("finish the release"),
1160 Some(50),
1161 GoalStatus::Active,
1162 );
1163
1164 let prompt = "continue after lowering the budget";
1165 let planned = plan(&config, &identity, false, prompt).await;
1166 let manifest = engine
1167 .build_request_manifest(inputs(false, Some(planned), prompt))
1168 .await;
1169 assert!(manifest.route.exact().is_some());
1170 assert!(manifest.tools.exact().is_some());
1171 assert!(manifest.body.exact().is_some());
1172 }
1173
1174 #[tokio::test]
1175 async fn translation_prompt_context_matches_captured_first_production_body() {
1176 use wiremock::matchers::method;
1177 use wiremock::{Mock, MockServer, ResponseTemplate};
1178
1179 let server = MockServer::start().await;
1180 Mock::given(method("POST"))
1181 .respond_with(
1182 ResponseTemplate::new(200)
1183 .insert_header("content-type", "text/event-stream")
1184 .set_body_string("data: [DONE]\n\n"),
1185 )
1186 .mount(&server)
1187 .await;
1188 let mut config = deepseek_config();
1189 config
1190 .providers
1191 .as_mut()
1192 .expect("providers")
1193 .deepseek
1194 .base_url = Some(server.uri());
1195 let identity = deepseek_identity();
1196 let (mut engine, _handle, _tmp) = wire_preview_engine(&config);
1197 engine.config.translation_enabled = false;
1198 let planned = plan(&config, &identity, false, "/translate explain this").await;
1199 let _ = assert_preview_matches_first_wire_body(
1200 &mut engine,
1201 &server,
1202 planned,
1203 "/translate explain this",
1204 PreviewWireFixture {
1205 translation_enabled: true,
1206 verbosity: Some("concise".to_string()),
1207 ..Default::default()
1208 },
1209 )
1210 .await;
1211 }
1212
1213 #[tokio::test]
1214 async fn paused_detach_goal_context_matches_captured_first_production_body() {
1215 use wiremock::matchers::method;
1216 use wiremock::{Mock, MockServer, ResponseTemplate};
1217
1218 let server = MockServer::start().await;
1219 Mock::given(method("POST"))
1220 .respond_with(
1221 ResponseTemplate::new(200)
1222 .insert_header("content-type", "text/event-stream")
1223 .set_body_string("data: [DONE]\n\n"),
1224 )
1225 .mount(&server)
1226 .await;
1227 let mut config = deepseek_config();
1228 config
1229 .providers
1230 .as_mut()
1231 .expect("providers")
1232 .deepseek
1233 .base_url = Some(server.uri());
1234 let identity = deepseek_identity();
1235 let (mut engine, _handle, _tmp) = wire_preview_engine(&config);
1236 engine.config.goal_objective = Some("stale paused objective".to_string());
1237 sync_goal_state_from_host(
1238 &engine.config.goal_state,
1239 Some("stale paused objective"),
1240 None,
1241 GoalStatus::Active,
1242 );
1243 let prompt = "answer only this new question\n\nCodewhale paused custom slash command context:\nThe user is not resuming that paused command.";
1244 let planned = plan(&config, &identity, false, prompt).await;
1245 let (_, first_wire_body) = assert_preview_matches_first_wire_body(
1246 &mut engine,
1247 &server,
1248 planned,
1249 prompt,
1250 PreviewWireFixture::default(),
1251 )
1252 .await;
1253 assert!(
1254 !first_wire_body
1255 .to_string()
1256 .contains("stale paused objective"),
1257 "detached paused goal leaked onto the first wire body"
1258 );
1259 }
1260
1261 #[tokio::test]
1262 async fn anthropic_preview_matches_the_first_native_messages_wire_body() {
1263 use wiremock::matchers::{method, path};
1264 use wiremock::{Mock, MockServer, ResponseTemplate};
1265
1266 let server = MockServer::start().await;
1267 Mock::given(method("POST"))
1268 .and(path("/v1/messages"))
1269 .respond_with(
1270 ResponseTemplate::new(200)
1271 .insert_header("content-type", "text/event-stream")
1272 .set_body_string("data: {\"type\":\"message_stop\"}\n\n"),
1273 )
1274 .mount(&server)
1275 .await;
1276 let model = "claude-sonnet-4-6";
1277 let config = crate::config::Config {
1278 provider: Some("anthropic".to_string()),
1279 providers: Some(crate::config::ProvidersConfig {
1280 anthropic: crate::config::ProviderConfig {
1281 api_key: Some("test-anthropic-key".to_string()),
1282 base_url: Some(server.uri()),
1283 model: Some(model.to_string()),
1284 ..crate::config::ProviderConfig::default()
1285 },
1286 ..crate::config::ProvidersConfig::default()
1287 }),
1288 ..crate::config::Config::default()
1289 };
1290 let identity = config
1291 .builtin_provider_identity(ProviderKind::Anthropic)
1292 .unwrap();
1293 let (mut engine, _handle, _tmp) = wire_preview_engine(&config);
1294 let prompt = "inspect the native Messages payload";
1295 let planned = plan_for(
1296 &config,
1297 &identity,
1298 ProviderKind::Anthropic,
1299 model,
1300 false,
1301 prompt,
1302 )
1303 .await;
1304 let (_, first_wire_body) = assert_preview_matches_first_wire_body(
1305 &mut engine,
1306 &server,
1307 planned,
1308 prompt,
1309 PreviewWireFixture::default(),
1310 )
1311 .await;
1312
1313 assert!(first_wire_body.get("system").is_some());
1314 assert!(first_wire_body.get("messages").is_some());
1315 assert!(first_wire_body.get("input").is_none());
1316 }
1317
1318 // ---------------------------------------------------------------------
1319 // #4707 — provider-free exact-route request/receipt matrix.
1320 //
1321 // The four route wire-truths are already pinned at the client boundary
1322 // (`client.rs`, `client/chat.rs`). What was missing is the *join*: that the
1323 // manifest a user reads from `/preview-request` describes the very bytes
1324 // those routes put on the wire. Each case below runs the production
1325 // planner, previews, then sends one turn through a local capture server
1326 // with the semantic endpoint left exact, and asserts the manifest against
1327 // the captured body — hash, sizes, route facts, and the requested→effective
1328 // reasoning triple.
1329 // ---------------------------------------------------------------------
1330
1331 /// One exact provider route in the matrix.
1332 struct MatrixRoute {
1333 /// Test-facing name; also the failure-message prefix.
1334 name: &'static str,
1335 provider: ProviderKind,
1336 provider_key: &'static str,
1337 base_url: &'static str,
1338 model: &'static str,
1339 requested_reasoning: crate::reasoning_preference::ReasoningEffort,
1340 requested_reasoning_label: &'static str,
1341 /// Reasoning-control keys the manifest must report, in receipt order.
1342 expect_control_keys: &'static [&'static str],
1343 /// Effort actually on the wire — `None` when the route publishes a
1344 /// thinking toggle with no granularity. Never a fabricated tier.
1345 expect_wire_effort: Option<&'static str>,
1346 expect_wire_effort_source: Option<&'static str>,
1347 /// The output-cap key this route writes, and the one it must not.
1348 expect_output_cap_key: &'static str,
1349 expect_absent_output_cap_key: &'static str,
1350 }
1351
1352 fn glm_5_2_zai_coding() -> MatrixRoute {
1353 MatrixRoute {
1354 name: "GLM-5.2 @ Z.ai coding",
1355 provider: ProviderKind::Zai,
1356 provider_key: "zai",
1357 base_url: crate::config::DEFAULT_ZAI_BASE_URL,
1358 model: crate::config::ZAI_GLM_5_2_MODEL,
1359 requested_reasoning: crate::reasoning_preference::ReasoningEffort::High,
1360 requested_reasoning_label: "high",
1361 expect_control_keys: &["reasoning_effort", "thinking"],
1362 expect_wire_effort: Some("high"),
1363 expect_wire_effort_source: Some("reasoning_effort"),
1364 expect_output_cap_key: "max_tokens",
1365 expect_absent_output_cap_key: "max_completion_tokens",
1366 }
1367 }
1368
1369 fn glm_5_turbo_zai() -> MatrixRoute {
1370 MatrixRoute {
1371 name: "GLM-5-Turbo @ Z.ai",
1372 provider: ProviderKind::Zai,
1373 provider_key: "zai",
1374 base_url: crate::config::DEFAULT_ZAI_BASE_URL,
1375 model: crate::config::ZAI_GLM_5_TURBO_MODEL,
1376 requested_reasoning: crate::reasoning_preference::ReasoningEffort::High,
1377 requested_reasoning_label: "high",
1378 // No invented granularity: the toggle ships, the tier does not.
1379 expect_control_keys: &["thinking"],
1380 expect_wire_effort: None,
1381 expect_wire_effort_source: None,
1382 expect_output_cap_key: "max_tokens",
1383 expect_absent_output_cap_key: "max_completion_tokens",
1384 }
1385 }
1386
1387 fn kimi_k3_moonshot_direct() -> MatrixRoute {
1388 MatrixRoute {
1389 name: "kimi-k3 @ api.moonshot.ai",
1390 provider: ProviderKind::Moonshot,
1391 provider_key: "moonshot",
1392 base_url: crate::config::DEFAULT_MOONSHOT_BASE_URL,
1393 model: crate::config::MOONSHOT_KIMI_K3_MODEL,
1394 // The visible normalization: `off` is not a tier this route has.
1395 requested_reasoning: crate::reasoning_preference::ReasoningEffort::Off,
1396 requested_reasoning_label: "off",
1397 expect_control_keys: &["reasoning_effort"],
1398 expect_wire_effort: Some("low"),
1399 expect_wire_effort_source: Some("reasoning_effort"),
1400 expect_output_cap_key: "max_completion_tokens",
1401 expect_absent_output_cap_key: "max_tokens",
1402 }
1403 }
1404
1405 fn k3_kimi_code() -> MatrixRoute {
1406 MatrixRoute {
1407 name: "k3 @ api.kimi.com/coding/v1",
1408 provider: ProviderKind::Moonshot,
1409 provider_key: "moonshot",
1410 base_url: crate::config::DEFAULT_KIMI_CODE_BASE_URL,
1411 model: crate::config::KIMI_CODE_K3_MODEL,
1412 requested_reasoning: crate::reasoning_preference::ReasoningEffort::Off,
1413 requested_reasoning_label: "off",
1414 expect_control_keys: &["thinking"],
1415 expect_wire_effort: Some("low"),
1416 expect_wire_effort_source: Some("thinking.effort"),
1417 expect_output_cap_key: "max_tokens",
1418 expect_absent_output_cap_key: "max_completion_tokens",
1419 }
1420 }
1421
1422 fn minimax_m3() -> MatrixRoute {
1423 MatrixRoute {
1424 name: "MiniMax-M3 @ api.minimax.io",
1425 provider: ProviderKind::Minimax,
1426 provider_key: "minimax",
1427 base_url: crate::config::DEFAULT_MINIMAX_BASE_URL,
1428 model: crate::config::DEFAULT_MINIMAX_MODEL,
1429 requested_reasoning: crate::reasoning_preference::ReasoningEffort::High,
1430 requested_reasoning_label: "high",
1431 expect_control_keys: &["thinking", "reasoning_split"],
1432 expect_wire_effort: None,
1433 expect_wire_effort_source: None,
1434 expect_output_cap_key: "max_completion_tokens",
1435 expect_absent_output_cap_key: "max_tokens",
1436 }
1437 }
1438
1439 fn matrix_routes() -> Vec<MatrixRoute> {
1440 vec![
1441 glm_5_2_zai_coding(),
1442 glm_5_turbo_zai(),
1443 kimi_k3_moonshot_direct(),
1444 k3_kimi_code(),
1445 minimax_m3(),
1446 ]
1447 }
1448
1449 fn matrix_config(route: &MatrixRoute) -> crate::config::Config {
1450 let entry = crate::config::ProviderConfig {
1451 api_key: Some(format!("sk-test-{}-matrix-key", route.provider_key)),
1452 base_url: Some(route.base_url.to_string()),
1453 model: Some(route.model.to_string()),
1454 ..crate::config::ProviderConfig::default()
1455 };
1456 let mut providers = crate::config::ProvidersConfig::default();
1457 match route.provider {
1458 ProviderKind::Zai => providers.zai = entry,
1459 ProviderKind::Moonshot => providers.moonshot = entry,
1460 ProviderKind::Minimax => providers.minimax = entry,
1461 other => panic!("{}: unhandled matrix provider {other:?}", route.name),
1462 }
1463 crate::config::Config {
1464 provider: Some(route.provider_key.to_string()),
1465 providers: Some(providers),
1466 ..crate::config::Config::default()
1467 }
1468 }
1469
1470 fn matrix_identity(route: &MatrixRoute) -> crate::config::ProviderIdentity {
1471 matrix_config(route)
1472 .resolve_provider_pin_identity(route.provider_key)
1473 .unwrap()
1474 }
1475
1476 /// Plan the exact route through production, then redirect only the
1477 /// *transport* at the local capture server. The endpoint identity the
1478 /// route shaper reads is untouched, so the captured body is the body
1479 /// `api.z.ai` / `api.moonshot.ai` / `api.kimi.com` / `api.minimax.io`
1480 /// would have received.
1481 async fn matrix_planned_route(
1482 route: &MatrixRoute,
1483 config: &crate::config::Config,
1484 transport_base_url: Option<&str>,
1485 prompt: &str,
1486 ) -> crate::turn_route_plan::PlannedTurnRoute {
1487 let identity = matrix_identity(route);
1488 let mut planned = plan_with_reasoning(
1489 config,
1490 &identity,
1491 route.provider,
1492 route.model,
1493 false,
1494 route.requested_reasoning,
1495 prompt,
1496 )
1497 .await;
1498 if let Some(transport_base_url) = transport_base_url {
1499 let validated = planned
1500 .route
1501 .clone()
1502 .validate()
1503 .expect("the matrix route validates into a concrete client");
1504 let mut client = validated.client.clone();
1505 client.set_test_chat_transport_base_url(transport_base_url.to_string());
1506 planned.route = crate::route_runtime::ValidatedRuntimeRoute {
1507 client,
1508 ..validated
1509 }
1510 .into_resolved();
1511 }
1512 planned
1513 }
1514
1515 /// Non-system messages on a Chat Completions body — the manifest counts
1516 /// the system region separately, so `message_count` must exclude it.
1517 fn non_system_message_count(body: &serde_json::Value) -> usize {
1518 body.get("messages")
1519 .and_then(serde_json::Value::as_array)
1520 .map(|messages| {
1521 messages
1522 .iter()
1523 .filter(|message| {
1524 message.get("role").and_then(serde_json::Value::as_str) != Some("system")
1525 })
1526 .count()
1527 })
1528 .unwrap_or_default()
1529 }
1530
1531 async fn assert_matrix_route(route: &MatrixRoute) {
1532 use wiremock::matchers::method;
1533 use wiremock::{Mock, MockServer, ResponseTemplate};
1534
1535 let name = route.name;
1536 let server = MockServer::start().await;
1537 Mock::given(method("POST"))
1538 .respond_with(
1539 ResponseTemplate::new(200)
1540 .insert_header("content-type", "text/event-stream")
1541 .set_body_string("data: [DONE]\n\n"),
1542 )
1543 .mount(&server)
1544 .await;
1545
1546 let config = matrix_config(route);
1547 let (mut engine, _handle, _tmp) = wire_preview_engine(&config);
1548 let prompt = "inspect the exact next request for this route";
1549 let uri = server.uri();
1550 let planned = matrix_planned_route(route, &config, Some(uri.as_str()), prompt).await;
1551 let planned_provider = planned.route.identity.provider;
1552 let planned_base_url = planned.route.candidate.endpoint().base_url.clone();
1553 let planned_model = planned.route.model.clone();
1554 let expected_wire_model = crate::config::wire_model_for_provider_route(
1555 planned_provider,
1556 &planned_base_url,
1557 &planned_model,
1558 );
1559 assert_eq!(
1560 planned_base_url.trim_end_matches('/'),
1561 route.base_url.trim_end_matches('/'),
1562 "{name}: the planner must keep the exact configured endpoint"
1563 );
1564
1565 let (manifest, body) = assert_preview_matches_first_wire_body(
1566 &mut engine,
1567 &server,
1568 planned,
1569 prompt,
1570 PreviewWireFixture {
1571 requested_model: Some(route.model.to_string()),
1572 requested_reasoning: Some(route.requested_reasoning_label.to_string()),
1573 ..Default::default()
1574 },
1575 )
1576 .await;
1577
1578 // --- Route facts -------------------------------------------------
1579 let facts = manifest
1580 .route
1581 .exact()
1582 .expect("a configured fixed route is exact");
1583 assert_eq!(facts.provider_id.as_str(), route.provider_key, "{name}");
1584 assert_eq!(facts.wire_model.as_str(), expected_wire_model, "{name}");
1585 assert_eq!(facts.dialect, "chat-completions", "{name}");
1586 assert_eq!(facts.routing_source, "active-fixed-route", "{name}");
1587 assert_eq!(
1588 body.get("model").and_then(serde_json::Value::as_str),
1589 Some(expected_wire_model.as_str()),
1590 "{name}: the manifest's wire model must be the model on the wire: {body}"
1591 );
1592
1593 // --- Body identity, byte accounting, and size estimates ----------
1594 let facts_body = manifest.body.exact().expect("a prompted body is exact");
1595 let canonical = crate::client::canonical_json(&body);
1596 assert_eq!(
1597 facts_body.body_sha256,
1598 crate::hashing::sha256_hex(canonical.as_bytes()),
1599 "{name}: manifest body hash must equal the captured wire body hash"
1600 );
1601 assert_eq!(
1602 facts_body.body_canonical_json_bytes,
1603 canonical.len(),
1604 "{name}: canonical body size must describe the captured body"
1605 );
1606 assert_eq!(
1607 facts_body.system_canonical_json_bytes
1608 + facts_body.tool_schema_canonical_json_bytes
1609 + facts_body.message_canonical_json_bytes
1610 + facts_body.framing_canonical_json_bytes,
1611 facts_body.body_canonical_json_bytes,
1612 "{name}: the four accounting classes must sum to the body"
1613 );
1614 assert!(
1615 facts_body.system_canonical_json_bytes > 0,
1616 "{name}: this route sends a system region"
1617 );
1618 assert!(
1619 facts_body.tool_schema_canonical_json_bytes > 0,
1620 "{name}: this route sends tool schemas"
1621 );
1622 assert!(
1623 facts_body.message_canonical_json_bytes > 0,
1624 "{name}: this route sends messages"
1625 );
1626 assert_eq!(
1627 facts_body.message_count,
1628 non_system_message_count(&body),
1629 "{name}: message_count counts the non-system messages on the wire"
1630 );
1631 assert!(
1632 facts_body.tool_result_canonical_json_bytes <= facts_body.message_canonical_json_bytes,
1633 "{name}: tool results are a subset of messages"
1634 );
1635 assert!(
1636 facts_body.attachment_canonical_json_bytes <= facts_body.message_canonical_json_bytes,
1637 "{name}: attachments are a subset of messages"
1638 );
1639 assert_eq!(
1640 facts_body.tool_schema_wire_sha256,
1641 body.get("tools").map(|tools| {
1642 crate::hashing::sha256_hex(crate::client::canonical_json(tools).as_bytes())
1643 }),
1644 "{name}: the tool-schema digest must be over the schemas on the wire"
1645 );
1646 assert!(
1647 facts_body.estimates.system > 0 && facts_body.estimates.tool_schemas > 0,
1648 "{name}: per-class estimates are derived from the same wire regions"
1649 );
1650 assert!(
1651 facts_body.estimates.total_conservative > 0,
1652 "{name}: a whole-body estimate is available"
1653 );
1654
1655 // --- Output cap: exactly the key this route writes ----------------
1656 let wire_cap = body
1657 .get(route.expect_output_cap_key)
1658 .and_then(serde_json::Value::as_u64);
1659 assert!(
1660 wire_cap.is_some(),
1661 "{name}: expected `{}` on the wire: {body}",
1662 route.expect_output_cap_key
1663 );
1664 assert!(
1665 body.get(route.expect_absent_output_cap_key).is_none(),
1666 "{name}: `{}` must not be on the wire: {body}",
1667 route.expect_absent_output_cap_key
1668 );
1669 assert_eq!(
1670 facts_body.wire_output_cap_tokens, wire_cap,
1671 "{name}: the reported output cap is the one literally on the wire"
1672 );
1673
1674 // --- requested → effective reasoning ------------------------------
1675 assert_eq!(
1676 manifest.session.requested_model.as_str(),
1677 route.model,
1678 "{name}"
1679 );
1680 assert_eq!(
1681 manifest.session.requested_reasoning.as_str(),
1682 route.requested_reasoning_label,
1683 "{name}"
1684 );
1685 assert_eq!(
1686 facts_body.reasoning_resolution,
1687 ReasoningResolution::Explicit,
1688 "{name}: a fixed route with an explicitly requested tier"
1689 );
1690 assert_eq!(
1691 facts_body.reasoning_wire_control_keys, route.expect_control_keys,
1692 "{name}: reasoning-control keys, against the captured body {body}"
1693 );
1694 assert_eq!(
1695 facts_body
1696 .reasoning_wire_effort
1697 .as_ref()
1698 .map(|effort| effort.as_str()),
1699 route.expect_wire_effort,
1700 "{name}: wire effort, against the captured body {body}"
1701 );
1702 assert_eq!(
1703 facts_body.reasoning_wire_effort_source.as_deref(),
1704 route.expect_wire_effort_source,
1705 "{name}"
1706 );
1707 // Every reported control key is genuinely present on the wire, and the
1708 // reported effort is genuinely readable at the reported key path.
1709 for key in &facts_body.reasoning_wire_control_keys {
1710 assert!(
1711 body.get(key).is_some(),
1712 "{name}: reported control key `{key}` is not on the wire: {body}"
1713 );
1714 }
1715 match (
1716 facts_body.reasoning_wire_effort_source.as_deref(),
1717 route.expect_wire_effort,
1718 ) {
1719 (Some("reasoning_effort"), Some(effort)) => assert_eq!(
1720 body.get("reasoning_effort")
1721 .and_then(serde_json::Value::as_str),
1722 Some(effort),
1723 "{name}: {body}"
1724 ),
1725 (Some(path), Some(effort)) => {
1726 let pointer = format!("/{}", path.replace('.', "/"));
1727 assert_eq!(
1728 body.pointer(&pointer).and_then(serde_json::Value::as_str),
1729 Some(effort),
1730 "{name}: {body}"
1731 );
1732 }
1733 (None, None) => assert!(
1734 body.get("reasoning_effort").is_none(),
1735 "{name}: no effort was reported, so none may be on the wire: {body}"
1736 ),
1737 (source, effort) => panic!("{name}: inconsistent effort receipt {source:?}/{effort:?}"),
1738 }
1739
1740 // --- Provider-authoritative usage ---------------------------------
1741 // A preview describes a request that has not been sent. Unknown stays
1742 // unknown; it never becomes a zero.
1743 assert!(
1744 matches!(
1745 &facts_body.provider_reported_usage,
1746 Availability::Unavailable(unavailable)
1747 if unavailable.reason == UnavailableReason::ProviderRequestNotExecuted
1748 ),
1749 "{name}: preview must not claim provider usage"
1750 );
1751 let json = manifest.to_json();
1752 assert!(
1753 !json.contains("\"input_tokens\""),
1754 "{name}: no fabricated usage counters reach the surface:\n{json}"
1755 );
1756 }
1757
1758 #[tokio::test]
1759 async fn matrix_glm_5_2_zai_coding_preview_matches_the_first_wire_body() {
1760 assert_matrix_route(&glm_5_2_zai_coding()).await;
1761 }
1762
1763 #[tokio::test]
1764 async fn matrix_glm_5_turbo_zai_preview_matches_the_first_wire_body() {
1765 assert_matrix_route(&glm_5_turbo_zai()).await;
1766 }
1767
1768 #[tokio::test]
1769 async fn matrix_kimi_k3_moonshot_direct_preview_matches_the_first_wire_body() {
1770 assert_matrix_route(&kimi_k3_moonshot_direct()).await;
1771 }
1772
1773 #[tokio::test]
1774 async fn matrix_k3_kimi_code_preview_matches_the_first_wire_body() {
1775 assert_matrix_route(&k3_kimi_code()).await;
1776 }
1777
1778 #[tokio::test]
1779 async fn matrix_minimax_m3_preview_matches_the_first_wire_body() {
1780 assert_matrix_route(&minimax_m3()).await;
1781 }
1782
1783 /// The active tool-catalog hash is a *catalog identity*, not a wire fact:
1784 /// the same catalog under the same posture must hash the same on every
1785 /// route, however differently each dialect then shapes those schemas.
1786 /// (Unit-level membership/order/schema sensitivity is pinned by
1787 /// `active_catalog_hash_tracks_membership_order_and_schema`.)
1788 #[tokio::test]
1789 async fn matrix_routes_share_one_active_tool_catalog_hash() {
1790 let workspace = tempfile::tempdir().expect("tempdir");
1791 let prompt = "describe the shared tool catalog";
1792 let mut observed: Vec<(&'static str, usize, String, String)> = Vec::new();
1793
1794 for route in matrix_routes() {
1795 let config = matrix_config(&route);
1796 let (mut engine, _handle) = Engine::new(
1797 EngineConfig {
1798 workspace: workspace.path().to_path_buf(),
1799 max_steps: 1,
1800 snapshots_enabled: false,
1801 terminal_chrome_enabled: false,
1802 ..Default::default()
1803 },
1804 &config,
1805 );
1806 engine.config.features.disable(Feature::Mcp);
1807 engine.config.subagents_enabled = false;
1808
1809 let planned = matrix_planned_route(&route, &config, None, prompt).await;
1810 let manifest = engine
1811 .build_request_manifest(inputs(false, Some(planned), prompt))
1812 .await;
1813 let tools = manifest
1814 .tools
1815 .exact()
1816 .expect("MCP is off, so the tool surface is exact");
1817 assert!(
1818 tools.standard_and_full_surfaces_collapsed,
1819 "{}: this fixture's catalog fits both budgets, so the surface \
1820 budget label cannot change catalog membership",
1821 route.name
1822 );
1823 observed.push((
1824 route.name,
1825 tools.active_tool_count,
1826 tools.active_tool_catalog_sha256.clone(),
1827 tools.tool_surface_budget.clone(),
1828 ));
1829 }
1830
1831 assert_eq!(observed.len(), 5, "every matrix route is represented");
1832 let (first_name, first_count, first_hash, _) = observed[0].clone();
1833 for (name, count, hash, _) in &observed {
1834 assert_eq!(
1835 *count, first_count,
1836 "{name} vs {first_name}: the matrix fixture holds the tool surface constant"
1837 );
1838 assert_eq!(
1839 hash, &first_hash,
1840 "{name} vs {first_name}: one catalog must hash to one identity across routes"
1841 );
1842 }
1843
1844 // …and the routes really are distinct in capability posture: the shared
1845 // hash is a genuine cross-route agreement, not five copies of one
1846 // route. GLM-5.2 publishes a `Full` tool surface budget while
1847 // GLM-5-Turbo publishes `Standard`, and the catalog identity is
1848 // unchanged by that difference.
1849 let budgets: std::collections::BTreeSet<&str> = observed
1850 .iter()
1851 .map(|(_, _, _, budget)| budget.as_str())
1852 .collect();
1853 assert!(
1854 budgets.len() > 1,
1855 "the matrix spans routes with different surface budgets: {budgets:?}"
1856 );
1857 }
1858
1859 /// Provider-authoritative usage is never a preview fact, and it is never a
1860 /// zero standing in for "not measured". It becomes knowable only when a
1861 /// response reports it, through the same `parse_usage` seam the turn loop
1862 /// uses.
1863 #[tokio::test]
1864 async fn provider_reported_usage_is_unavailable_until_a_response_reports_it() {
1865 use wiremock::matchers::method;
1866 use wiremock::{Mock, MockServer, ResponseTemplate};
1867
1868 let usage = json!({"prompt_tokens": 137, "completion_tokens": 24, "total_tokens": 161});
1869 let stream = format!(
1870 "data: {}\n\ndata: {}\n\ndata: [DONE]\n\n",
1871 json!({
1872 "choices": [{"index": 0, "delta": {"content": "ok"}}],
1873 }),
1874 json!({
1875 "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
1876 "usage": usage,
1877 })
1878 );
1879
1880 let server = MockServer::start().await;
1881 Mock::given(method("POST"))
1882 .respond_with(
1883 ResponseTemplate::new(200)
1884 .insert_header("content-type", "text/event-stream")
1885 .set_body_string(stream),
1886 )
1887 .mount(&server)
1888 .await;
1889
1890 let mut config = deepseek_config();
1891 config
1892 .providers
1893 .as_mut()
1894 .expect("providers")
1895 .deepseek
1896 .base_url = Some(server.uri());
1897 let identity = deepseek_identity();
1898 let (mut engine, _handle, _tmp) = wire_preview_engine(&config);
1899
1900 let prompt = "count the tokens this turn will report";
1901 let planned = plan(&config, &identity, false, prompt).await;
1902 let production_route = planned.route.clone();
1903 let compaction = planned.compaction.clone();
1904 let reasoning_effort = planned.effective_reasoning_effort.clone();
1905 let reasoning_effort_auto = planned.auto_controls_reasoning;
1906
1907 let manifest = engine
1908 .build_request_manifest(inputs(false, Some(planned), prompt))
1909 .await;
1910 let body = manifest.body.exact().expect("a prompted body is exact");
1911 assert!(
1912 matches!(
1913 &body.provider_reported_usage,
1914 Availability::Unavailable(unavailable)
1915 if unavailable.reason == UnavailableReason::ProviderRequestNotExecuted
1916 ),
1917 "no request occurred, so there is nothing the provider reported"
1918 );
1919 assert_eq!(
1920 engine.session.total_usage.input_tokens, 0,
1921 "and nothing has been recorded yet"
1922 );
1923 assert_eq!(engine.session.total_usage.output_tokens, 0);
1924
1925 let _ = engine
1926 .handle_send_message(TurnSpec {
1927 content: prompt.to_string(),
1928 mode: AppMode::Agent,
1929 route: Box::new(production_route),
1930 compaction: Box::new(compaction),
1931 initial_routed_usage: Box::new(crate::cost_status::RuntimeUsageBatch::default()),
1932 goal_objective: None,
1933 goal_token_budget: None,
1934 goal_status: GoalStatus::Active,
1935 reasoning_effort,
1936 reasoning_effort_auto,
1937 auto_model: false,
1938 allow_shell: false,
1939 trust_mode: false,
1940 auto_approve: false,
1941 approval_mode: ApprovalMode::Suggest,
1942 translation_enabled: false,
1943 allowed_tools: None,
1944 dynamic_tools: Vec::new(),
1945 hook_executor: None,
1946 verbosity: None,
1947 provenance: UserInputProvenance::ExternalUser,
1948 images: Vec::new(),
1949 max_output_tokens: None,
1950 submission_id: None,
1951 })
1952 .await;
1953
1954 // The completed turn's counts are exactly what `parse_usage` reads off
1955 // the reported usage object — no rounding, no substituted estimate.
1956 let parsed = crate::client::parse_usage(Some(&usage));
1957 assert_eq!(u64::from(parsed.input_tokens), 137);
1958 assert_eq!(u64::from(parsed.output_tokens), 24);
1959 assert_eq!(
1960 (
1961 engine.session.total_usage.input_tokens,
1962 engine.session.total_usage.output_tokens
1963 ),
1964 (
1965 u64::from(parsed.input_tokens),
1966 u64::from(parsed.output_tokens)
1967 ),
1968 "the turn records the provider-authoritative counts, not an estimate"
1969 );
1970 let reported = crate::request_manifest::ProviderReportedUsage {
1971 input_tokens: engine.session.total_usage.input_tokens,
1972 output_tokens: engine.session.total_usage.output_tokens,
1973 };
1974 assert_eq!(reported.input_tokens, 137);
1975 assert_eq!(reported.output_tokens, 24);
1976 }
1977
1978 /// A fixed route with a hypothetical prompt describes the next turn
1979 /// exactly: route, tools, and body are all published, and the prompt is
1980 /// part of the hashed body.
1981 #[tokio::test]
1982 async fn fixed_route_with_a_prompt_describes_the_exact_next_turn() {
1983 let mut config = deepseek_config();
1984 config
1985 .providers
1986 .as_mut()
1987 .expect("providers")
1988 .deepseek
1989 .context_window = Some(123_456);
1990 let identity = deepseek_identity();
1991 let (mut engine, _handle, _tmp) = preview_engine(&config);
1992 engine.config.features.disable(Feature::Mcp);
1993 engine.active_route_limits = Some(codewhale_config::route::RouteLimits {
1994 context_tokens: Some(4_096),
1995 input_tokens: Some(3_000),
1996 output_tokens: Some(512),
1997 });
1998
1999 let planned = plan(&config, &identity, false, "refactor the parser").await;
2000 let planned_limits = crate::route_budget::known_route_limits(planned.route.candidate.limits());
2001 let expected_input_budget = context_input_budget_for_route(
2002 planned.route.identity.provider,
2003 &planned.route.model,
2004 planned_limits,
2005 0,
2006 );
2007 let expected_wire_output = crate::route_budget::effective_max_output_tokens_for_route(
2008 planned.route.identity.provider,
2009 &planned.route.model,
2010 planned_limits,
2011 );
2012 let manifest = engine
2013 .build_request_manifest(inputs(false, Some(planned), "refactor the parser"))
2014 .await;
2015
2016 let route = manifest.route.exact().expect("a fixed route is exact");
2017 assert_eq!(route.provider_id.as_str(), "deepseek");
2018 assert_eq!(route.routing_source, "active-fixed-route");
2019 assert_eq!(route.dialect, "chat-completions");
2020 assert_eq!(route.caller_entrypoint, "streaming");
2021 assert_eq!(route.body_stream_field, Some(true));
2022 assert_eq!(route.context_limit_tokens, 123_456);
2023 assert_eq!(
2024 route.context_limit_source,
2025 crate::route_runtime::ContextWindowSource::Configured
2026 );
2027 assert_eq!(
2028 route.route_input_limit_tokens,
2029 planned_limits.and_then(|limits| limits.input_tokens)
2030 );
2031 assert_eq!(
2032 route.route_output_limit_tokens,
2033 planned_limits.and_then(|limits| limits.output_tokens)
2034 );
2035 assert!(!route.wire_model.is_redacted());
2036 assert!(
2037 manifest.tools.exact().is_some(),
2038 "MCP is off in this engine"
2039 );
2040
2041 let body = manifest.body.exact().expect("a prompted body is exact");
2042 assert_eq!(body.input_budget_ceiling_tokens, expected_input_budget);
2043 assert_eq!(
2044 body.wire_output_cap_tokens,
2045 Some(u64::from(expected_wire_output))
2046 );
2047 assert_eq!(body.body_sha256.len(), 64);
2048 assert!(
2049 body.message_count >= 1,
2050 "the hypothetical prompt is a message"
2051 );
2052 assert!(body.local_system_tools_component_sha256.is_some());
2053 assert!(manifest.session.hypothetical_prompt_supplied);
2054
2055 // The prompt is genuinely part of the request being described.
2056 let other_planned = plan(&config, &identity, false, "write the release notes").await;
2057 let other = engine
2058 .build_request_manifest(inputs(
2059 false,
2060 Some(other_planned),
2061 "write the release notes",
2062 ))
2063 .await;
2064 assert_ne!(
2065 body.body_sha256,
2066 other.body.exact().expect("exact").body_sha256,
2067 "a different next prompt must produce a different body hash"
2068 );
2069 }
2070
2071 /// The engine can describe an Auto route receipt supplied by a trusted
2072 /// host without consulting installed state. The human preview command
2073 /// deliberately never obtains such a receipt, because doing so would call
2074 /// the provider-backed classifier.
2075 #[tokio::test]
2076 async fn host_supplied_auto_route_receipt_matches_the_production_planner() {
2077 let config = deepseek_config();
2078 let identity = deepseek_identity();
2079 let (mut engine, _handle, _tmp) = preview_engine(&config);
2080 engine.config.features.disable(Feature::Mcp);
2081
2082 let planned = plan(&config, &identity, true, "explain this stack trace").await;
2083 let planned_provider = planned.effective_provider;
2084 let planned_identity = planned.effective_provider_identity.clone();
2085 let planned_model = planned.route.model.clone();
2086 let planned_base_url = planned.route.candidate.endpoint().base_url.clone();
2087 assert!(
2088 planned.auto_controls_reasoning,
2089 "the helper requests auto reasoning for its auto-model fixture"
2090 );
2091
2092 let manifest = engine
2093 .build_request_manifest(inputs(true, Some(planned), "explain this stack trace"))
2094 .await;
2095
2096 let route = manifest
2097 .route
2098 .exact()
2099 .expect("auto + prompt resolves a route");
2100 assert_eq!(route.provider_id.as_str(), planned_provider.as_str());
2101 assert_eq!(route.routing_source, "auto-provider-classifier");
2102 assert_eq!(planned_identity, "deepseek");
2103 assert_eq!(
2104 route.wire_model.as_str(),
2105 crate::config::wire_model_for_provider_route(
2106 planned_provider,
2107 &planned_base_url,
2108 &planned_model,
2109 ),
2110 "the wire model is the planner's model after route remapping — not \
2111 the model the session happens to have installed"
2112 );
2113 assert_eq!(
2114 manifest.session.requested_model.as_str(),
2115 "auto",
2116 "the manifest never reports the resolved model as the user's selection"
2117 );
2118
2119 let body = match &manifest.body {
2120 Availability::Exact(body) => body,
2121 Availability::Unavailable(unavailable) => {
2122 panic!("auto + prompt should have an exact body: {unavailable:?}")
2123 }
2124 };
2125 assert_ne!(
2126 body.reasoning_resolution,
2127 ReasoningResolution::Explicit,
2128 "an auto-routed turn never claims an explicit user tier"
2129 );
2130
2131 // The hypothetical prompt is part of the hashed body on the auto path
2132 // too, not only on the fixed one.
2133 let other = plan(&config, &identity, true, "rename one local variable").await;
2134 let other = engine
2135 .build_request_manifest(inputs(true, Some(other), "rename one local variable"))
2136 .await;
2137 assert_ne!(
2138 body.body_sha256,
2139 other.body.exact().expect("exact").body_sha256
2140 );
2141 }
2142
2143 /// The passive path must not create an MCP pool, connect a server, or
2144 /// emit a UI event — it reports the tool surface unavailable instead.
2145 #[tokio::test]
2146 async fn preview_tool_snapshot_has_no_mcp_or_event_side_effects() {
2147 let tmp = tempfile::tempdir().expect("tempdir");
2148 let config = crate::config::Config {
2149 provider: Some("deepseek".to_string()),
2150 ..crate::config::Config::default()
2151 };
2152 let (mut engine, handle) = Engine::new(
2153 EngineConfig {
2154 workspace: tmp.path().to_path_buf(),
2155 ..Default::default()
2156 },
2157 &config,
2158 );
2159 let _ = engine.config.features.enable(Feature::Mcp);
2160
2161 let policy = TurnAuthority::from_effective_fields(
2162 AppMode::Agent,
2163 false,
2164 false,
2165 false,
2166 ApprovalMode::Suggest,
2167 );
2168 let build = engine
2169 .build_turn_tool_registry_and_catalog(
2170 &policy,
2171 &[],
2172 None,
2173 SubAgentWiring::Inert,
2174 McpAccess::PassiveSnapshot,
2175 TurnRouteContext {
2176 provider: engine.api_provider,
2177 model: engine.session.model.clone(),
2178 capabilities: engine.active_route_capabilities,
2179 limits: engine.active_route_limits,
2180 client: engine.codewhale_client.clone(),
2181 api_config: Box::new(engine.api_config.clone()),
2182 locale_tag: engine.config.locale_tag.clone(),
2183 role_models: engine.subagent_role_models(),
2184 auto_model: false,
2185 reasoning_effort: None,
2186 reasoning_effort_auto: false,
2187 },
2188 "",
2189 )
2190 .await;
2191
2192 assert!(
2193 engine.mcp_pool.is_none(),
2194 "a passive snapshot must not create the MCP pool"
2195 );
2196 assert!(
2197 matches!(build.mcp, McpToolState::Unavailable { .. }),
2198 "with no connected pool the MCP tool state is unavailable, not empty"
2199 );
2200 assert!(build.mcp.server_count().is_none());
2201 drop(handle);
2202 }
2203
2204 /// The reviewed blocker: with MCP enabled but nothing connected, the
2205 /// preview built a catalog with zero MCP tools, prepared a body from it,
2206 /// and published that body as `Exact` — a hash of a request no turn would
2207 /// ever send. The body must inherit the tool surface's typed reason.
2208 #[tokio::test]
2209 async fn unavailable_mcp_state_makes_the_body_unavailable_too() {
2210 let config = deepseek_config();
2211 let identity = deepseek_identity();
2212 let (mut engine, _handle, _tmp) = preview_engine(&config);
2213 // MCP on, pool never started: a real turn would connect and could
2214 // discover tools this catalog does not contain.
2215 let _ = engine.config.features.enable(Feature::Mcp);
2216
2217 let planned = plan(&config, &identity, false, "refactor the parser").await;
2218 let manifest = engine
2219 .build_request_manifest(inputs(false, Some(planned), "refactor the parser"))
2220 .await;
2221
2222 assert!(
2223 manifest.tools.exact().is_none(),
2224 "an unconnected MCP pool is not a snapshottable tool surface"
2225 );
2226 assert!(
2227 manifest.body.exact().is_none(),
2228 "a body built from a tool surface missing its MCP contribution \
2229 must not be published as exact"
2230 );
2231 assert!(
2232 manifest.route.exact().is_some(),
2233 "the route does not depend on the MCP contribution and stays exact"
2234 );
2235
2236 // No body fact — hash, byte count, or local component fingerprint —
2237 // reaches either surface.
2238 let json = manifest.to_json();
2239 for forbidden in [
2240 "body_sha256",
2241 "local_system_tools_component_sha256",
2242 "tool_schema_wire_sha256",
2243 "body_canonical_json_bytes",
2244 "estimated_input_headroom_tokens",
2245 ] {
2246 assert!(!json.contains(forbidden), "{forbidden} leaked:\n{json}");
2247 }
2248 assert!(json.contains("mcp-state-not-snapshottable"), "{json}");
2249 assert!(engine.mcp_pool.is_none(), "no pool was created by looking");
2250 }
2251
2252 /// A preview is an inspection. Every piece of engine state a turn would
2253 /// have written must be byte-identical afterwards — including the ones the
2254 /// earlier implementation wrote and restored around an `.await`.
2255 #[tokio::test]
2256 async fn building_a_manifest_writes_no_engine_state() {
2257 let config = deepseek_config();
2258 let identity = deepseek_identity();
2259 let (mut engine, _handle, _tmp) = preview_engine(&config);
2260 engine.config.features.disable(Feature::Mcp);
2261 engine.config.allowed_tools = Some(vec!["Bash".to_string()]);
2262 engine.session.add_message(Message {
2263 role: Role::User,
2264 content: vec![ContentBlock::Text {
2265 text: "an earlier turn".to_string(),
2266 cache_control: None,
2267 }],
2268 });
2269
2270 let allowed_before = engine.config.allowed_tools.clone();
2271 let disallowed_before = engine.config.disallowed_tools.clone();
2272 let messages_before = engine.messages_with_turn_metadata();
2273 let model_before = engine.session.model.clone();
2274 let system_prompt_before = system_prompt_hash(engine.session.system_prompt.as_ref());
2275 let system_hash_before = engine.session.last_system_prompt_hash;
2276 let working_set_before = engine
2277 .session
2278 .working_set
2279 .summary_block(&engine.config.workspace);
2280 let provider_before = engine.api_provider;
2281 let mode_before = engine.current_mode;
2282 let narrowing_before = format!("{:?}", engine.last_policy_narrowing);
2283 let turn_counter_before = engine.turn_counter;
2284
2285 // A *different* command-scoped gate than the installed one, and a
2286 // prompt that mentions a path so the working set would move if the
2287 // preview observed it on the session rather than on a clone.
2288 let mut preview_inputs = inputs(
2289 false,
2290 Some(plan(&config, &identity, false, "inspect src/lib.rs").await),
2291 "inspect src/lib.rs",
2292 );
2293 preview_inputs.allowed_tools = Some(vec!["Read".to_string()]);
2294 let manifest = engine.build_request_manifest(preview_inputs).await;
2295 assert!(manifest.body.exact().is_some(), "fixture should be exact");
2296
2297 assert_eq!(engine.config.allowed_tools, allowed_before, "tool gate");
2298 assert_eq!(engine.config.disallowed_tools, disallowed_before);
2299 assert_eq!(
2300 engine.messages_with_turn_metadata(),
2301 messages_before,
2302 "history"
2303 );
2304 assert_eq!(engine.session.model, model_before);
2305 assert_eq!(
2306 system_prompt_hash(engine.session.system_prompt.as_ref()),
2307 system_prompt_before
2308 );
2309 assert_eq!(engine.session.last_system_prompt_hash, system_hash_before);
2310 assert_eq!(
2311 engine
2312 .session
2313 .working_set
2314 .summary_block(&engine.config.workspace),
2315 working_set_before,
2316 "the hypothetical message is observed on a clone, never on the session"
2317 );
2318 assert_eq!(engine.api_provider, provider_before);
2319 assert_eq!(engine.current_mode, mode_before);
2320 assert_eq!(
2321 format!("{:?}", engine.last_policy_narrowing),
2322 narrowing_before
2323 );
2324 assert_eq!(engine.turn_counter, turn_counter_before);
2325 assert!(engine.mcp_pool.is_none());
2326 }
2327
2328 /// The gate is a parameter, so it shapes the previewed catalog without
2329 /// ever being installed.
2330 #[tokio::test]
2331 async fn the_previewed_tool_gate_applies_without_being_installed() {
2332 let config = deepseek_config();
2333 let identity = deepseek_identity();
2334 let (mut engine, _handle, _tmp) = preview_engine(&config);
2335 engine.config.features.disable(Feature::Mcp);
2336
2337 let wide = engine
2338 .build_request_manifest(inputs(
2339 false,
2340 Some(plan(&config, &identity, false, "do the thing").await),
2341 "do the thing",
2342 ))
2343 .await;
2344
2345 let mut narrow_inputs = inputs(
2346 false,
2347 Some(plan(&config, &identity, false, "do the thing").await),
2348 "do the thing",
2349 );
2350 narrow_inputs.allowed_tools = Some(vec!["Read".to_string()]);
2351 let narrow = engine.build_request_manifest(narrow_inputs).await;
2352
2353 let wide_tools = wide.tools.exact().expect("exact");
2354 let narrow_tools = narrow.tools.exact().expect("exact");
2355 assert!(
2356 narrow_tools.active_tool_count < wide_tools.active_tool_count,
2357 "the passed gate must narrow the previewed catalog: {} vs {}",
2358 narrow_tools.active_tool_count,
2359 wide_tools.active_tool_count
2360 );
2361 assert_eq!(
2362 narrow.session.allowed_tool_gate_count,
2363 Some(1),
2364 "and the session section reports the gate that was previewed"
2365 );
2366 assert_eq!(
2367 engine.config.allowed_tools, None,
2368 "…while the engine keeps its own"
2369 );
2370 }
2371
2372 /// A plan failure still happened *because of* a supplied prompt. Reporting
2373 /// otherwise tells the user to pass the flag they just passed.
2374 #[tokio::test]
2375 async fn a_failed_plan_still_reports_that_a_prompt_was_supplied() {
2376 let (mut engine, _handle, _tmp) = preview_engine(&crate::config::Config::default());
2377 let mut failed = inputs(false, None, "");
2378 failed.unresolved = PreviewUnresolved::PlanFailed(
2379 "no API key configured for route 'my-gateway' at /home/someone/.config".to_string(),
2380 );
2381
2382 let manifest = engine.build_request_manifest(failed).await;
2383 assert!(manifest.session.hypothetical_prompt_supplied);
2384 assert!(manifest.route.exact().is_none());
2385
2386 let rendered = manifest.render();
2387 assert!(
2388 !rendered.contains("Pass `--prompt <text>`"),
2389 "the user already did:\n{rendered}"
2390 );
2391 // …and the raw host text never reaches a surface verbatim.
2392 for surface in [rendered, manifest.to_json()] {
2393 assert!(!surface.contains("my-gateway'"), "{surface}");
2394 assert!(!surface.contains("/home/someone"), "{surface}");
2395 }
2396 }
2397
2398 /// Pending runtime injections are *counted*, never consumed, and they make
2399 /// the body unavailable rather than silently absent from it.
2400 #[tokio::test]
2401 async fn pending_runtime_injections_make_the_body_unavailable_without_consuming_them() {
2402 let config = deepseek_config();
2403 let identity = deepseek_identity();
2404 let (mut engine, _handle, _tmp) = preview_engine(&config);
2405 engine.config.features.disable(Feature::Mcp);
2406 engine.pending_lsp_blocks.push(crate::lsp::DiagnosticBlock {
2407 file: std::path::PathBuf::from("src/lib.rs"),
2408 items: Vec::new(),
2409 });
2410
2411 let manifest = engine
2412 .build_request_manifest(inputs(
2413 false,
2414 Some(plan(&config, &identity, false, "fix it").await),
2415 "fix it",
2416 ))
2417 .await;
2418
2419 assert!(
2420 manifest.body.exact().is_none(),
2421 "the turn loop would inject diagnostics before the first request"
2422 );
2423 assert!(manifest.route.exact().is_some());
2424 assert_eq!(
2425 engine.pending_lsp_blocks.len(),
2426 1,
2427 "inspecting must not flush the pending blocks"
2428 );
2429 assert!(
2430 manifest
2431 .to_json()
2432 .contains("runtime-transforms-before-send"),
2433 "{}",
2434 manifest.to_json()
2435 );
2436 }
2437
2437 lines RUST