| 1 | //! Scripted-provider fixture for the one context-pressure number (0.10.1 |
| 2 | //! item 9). |
| 3 | //! |
| 4 | //! A tool-heavy turn crosses the auto-compaction threshold mid-turn, compacts, |
| 5 | //! then the next turn switches route and endpoint and continues. At every |
| 6 | //! model request the engine actually sent, the context meter, the |
| 7 | //! auto-compaction gate, the compaction preflight, and the `/context` headline |
| 8 | //! must read the same number — and it must be the pressure estimate, not the |
| 9 | //! 1.5x-inflated overflow guard, which `/context` shows only as a labeled |
| 10 | //! secondary line. |
| 11 | |
| 12 | use std::time::Duration; |
| 13 | |
| 14 | use codewhale_models::MessageRequest; |
| 15 | use serde_json::json; |
| 16 | use tempfile::tempdir; |
| 17 | |
| 18 | use super::{build_context_report, format_context_report, format_context_summary}; |
| 19 | use crate::compaction::{ |
| 20 | CompactionConfig, compaction_pressure_reached_with_billed, estimate_input_tokens_conservative, |
| 21 | estimate_input_tokens_for_pressure, |
| 22 | }; |
| 23 | use crate::config::Config; |
| 24 | use crate::core::engine::{Engine, EngineConfig}; |
| 25 | use crate::core::events::{Event, TurnOutcomeStatus}; |
| 26 | use crate::core::ops::{Op, TurnSpec, UserInputProvenance}; |
| 27 | use crate::llm_client::mock::{MockLlmClient, canned}; |
| 28 | use crate::route_runtime::{ResolvedRuntimeRoute, resolve_runtime_route}; |
| 29 | use crate::test_support::{EnvVarGuard, lock_test_env}; |
| 30 | use crate::tui::app::App; |
| 31 | |
| 32 | const THRESHOLD: usize = 40_000; |
| 33 | const PRIVATE_BASE_URL: &str = "https://private-fixture.test/v1"; |
| 34 | const PRIVATE_MODEL: &str = "private-fixture-deployment"; |
| 35 | const PRIVATE_WINDOW: u32 = 200_000; |
| 36 | |
| 37 | fn private_route_config() -> Config { |
| 38 | Config { |
| 39 | provider: Some("custom".to_string()), |
| 40 | providers: Some(crate::config::ProvidersConfig { |
| 41 | custom: std::collections::HashMap::from([( |
| 42 | "custom".to_string(), |
| 43 | crate::config::ProviderConfig { |
| 44 | kind: Some("openai-compatible".to_string()), |
| 45 | api_key: Some("test-private-key".to_string()), |
| 46 | base_url: Some(PRIVATE_BASE_URL.to_string()), |
| 47 | model: Some(PRIVATE_MODEL.to_string()), |
| 48 | context_window: Some(PRIVATE_WINDOW), |
| 49 | ..Default::default() |
| 50 | }, |
| 51 | )]), |
| 52 | ..Default::default() |
| 53 | }), |
| 54 | ..Default::default() |
| 55 | } |
| 56 | } |
| 57 | |
| 58 | fn resolve(config: &Config, model: &str) -> ResolvedRuntimeRoute { |
| 59 | resolve_runtime_route( |
| 60 | config, |
| 61 | config.active_provider_identity().unwrap().provider, |
| 62 | Some(model), |
| 63 | ) |
| 64 | .expect("resolve route") |
| 65 | } |
| 66 | |
| 67 | fn fixture_compaction() -> CompactionConfig { |
| 68 | CompactionConfig { |
| 69 | token_threshold: THRESHOLD, |
| 70 | ..CompactionConfig::default() |
| 71 | } |
| 72 | } |
| 73 | |
| 74 | fn turn_op(content: &str, route: &ResolvedRuntimeRoute) -> Op { |
| 75 | let compaction = fixture_compaction(); |
| 76 | Op::SendMessage(TurnSpec { |
| 77 | max_output_tokens: None, |
| 78 | content: content.to_string(), |
| 79 | images: Vec::new(), |
| 80 | mode: codewhale_config::AppMode::Agent, |
| 81 | route: Box::new(route.clone()), |
| 82 | compaction: Box::new(compaction), |
| 83 | initial_routed_usage: Box::default(), |
| 84 | goal_objective: None, |
| 85 | goal_token_budget: None, |
| 86 | goal_status: crate::tools::goal::GoalStatus::Active, |
| 87 | reasoning_effort: None, |
| 88 | reasoning_effort_auto: false, |
| 89 | auto_model: false, |
| 90 | allow_shell: true, |
| 91 | trust_mode: false, |
| 92 | auto_approve: true, |
| 93 | approval_mode: codewhale_execpolicy::ApprovalMode::Suggest, |
| 94 | translation_enabled: false, |
| 95 | allowed_tools: None, |
| 96 | dynamic_tools: Vec::new(), |
| 97 | hook_executor: None, |
| 98 | verbosity: None, |
| 99 | provenance: UserInputProvenance::ExternalUser, |
| 100 | submission_id: None, |
| 101 | }) |
| 102 | } |
| 103 | |
| 104 | fn tool_step(step: usize, payload_chars: usize) -> Vec<codewhale_models::StreamEvent> { |
| 105 | vec![ |
| 106 | canned::message_start(&format!("response-{step}")), |
| 107 | canned::text_block_start(0), |
| 108 | canned::text_delta(0, &format!("Step {step}: {}", "x".repeat(payload_chars))), |
| 109 | canned::block_stop(0), |
| 110 | canned::tool_use_block_start(1, &format!("read-{step}"), "File"), |
| 111 | canned::tool_input_delta(1, r#"{"action":"read","path":"README.md"}"#), |
| 112 | canned::block_stop(1), |
| 113 | canned::message_delta("tool_use", None), |
| 114 | canned::message_stop(), |
| 115 | ] |
| 116 | } |
| 117 | |
| 118 | #[derive(Debug, Default)] |
| 119 | struct TurnEvents { |
| 120 | auto_compactions: usize, |
| 121 | receipts: Vec<(String, Option<u64>)>, |
| 122 | } |
| 123 | |
| 124 | async fn run_turn(handle: &crate::core::engine::EngineHandle, op: Op) -> TurnEvents { |
| 125 | handle.send(op).await.expect("send turn"); |
| 126 | let mut events = TurnEvents::default(); |
| 127 | let mut rx = handle.rx_event.write().await; |
| 128 | loop { |
| 129 | match tokio::time::timeout(Duration::from_secs(30), rx.recv()) |
| 130 | .await |
| 131 | .expect("engine event before timeout") |
| 132 | .expect("engine event channel open") |
| 133 | { |
| 134 | Event::CompactionCompleted { |
| 135 | auto: true, |
| 136 | message, |
| 137 | post_input_tokens, |
| 138 | .. |
| 139 | } => { |
| 140 | events.auto_compactions += 1; |
| 141 | events.receipts.push((message, post_input_tokens)); |
| 142 | } |
| 143 | Event::CompactionFailed { message, .. } => { |
| 144 | panic!("fixture compaction must succeed: {message}") |
| 145 | } |
| 146 | Event::TurnComplete { status, error, .. } => { |
| 147 | assert_eq!(status, TurnOutcomeStatus::Completed, "{error:?}"); |
| 148 | return events; |
| 149 | } |
| 150 | _ => {} |
| 151 | } |
| 152 | } |
| 153 | } |
| 154 | |
| 155 | /// Mirror the engine's installed route and the exact request it sent into a |
| 156 | /// TUI `App`, then read every surface. Returns the one agreed number. |
| 157 | fn assert_one_pressure_number( |
| 158 | app: &mut App, |
| 159 | label: &str, |
| 160 | request: &MessageRequest, |
| 161 | route: &ResolvedRuntimeRoute, |
| 162 | ) -> usize { |
| 163 | app.api_provider = route.identity.provider; |
| 164 | app.model = route.model.clone(); |
| 165 | app.active_route_limits = crate::route_budget::known_route_limits(route.candidate.limits()); |
| 166 | app.active_context_window_source = route.context_window.source; |
| 167 | // Install the transcript the way the host applies an engine session |
| 168 | // projection (`apply_engine_session_projection`): the meter's per-message |
| 169 | // cache is dropped before the rewritten history lands, so a compaction |
| 170 | // cannot leave stale per-index counts behind. |
| 171 | app.context_token_cache.borrow_mut().clear(); |
| 172 | app.set_api_messages(std::sync::Arc::new(request.messages.clone())); |
| 173 | app.system_prompt = request.system.clone(); |
| 174 | // Mock usage bills no prompt tokens; the number is the estimate alone. |
| 175 | app.last_billed_input_tokens = None; |
| 176 | |
| 177 | let messages = &request.messages; |
| 178 | let system = request.system.as_ref(); |
| 179 | |
| 180 | // Auto-compaction gate. |
| 181 | let gate = estimate_input_tokens_for_pressure(messages, system); |
| 182 | let compaction = fixture_compaction(); |
| 183 | assert_eq!( |
| 184 | compaction_pressure_reached_with_billed(messages, system, &compaction, None), |
| 185 | gate >= THRESHOLD, |
| 186 | "{label}: gate decision must follow the gate number {gate}" |
| 187 | ); |
| 188 | |
| 189 | // Compaction preflight (live input the turn loop measures). |
| 190 | let preflight = crate::core::turn::TurnContext::new(8) |
| 191 | .live_input_tokens_for_compaction(messages, system, None) |
| 192 | .expect("non-empty request"); |
| 193 | |
| 194 | // Context meter (footer). |
| 195 | let (meter, meter_window, meter_percent) = |
| 196 | crate::tui::ui::context_usage_snapshot(app).expect("meter reading"); |
| 197 | |
| 198 | // `/context` headline. |
| 199 | let report = build_context_report(app); |
| 200 | |
| 201 | assert_eq!(preflight, gate as u64, "{label}: preflight vs gate"); |
| 202 | assert_eq!(meter, gate as i64, "{label}: meter vs gate"); |
| 203 | assert_eq!( |
| 204 | report.active_context_estimated_tokens, gate, |
| 205 | "{label}: /context headline vs gate" |
| 206 | ); |
| 207 | assert_eq!( |
| 208 | report.context_window_tokens, |
| 209 | Some(meter_window), |
| 210 | "{label}: /context window vs meter window" |
| 211 | ); |
| 212 | let report_percent = report.budget_used_percent.expect("window known"); |
| 213 | assert!( |
| 214 | (report_percent - meter_percent).abs() < 1e-9, |
| 215 | "{label}: /context {report_percent}% vs meter {meter_percent}%" |
| 216 | ); |
| 217 | |
| 218 | // The inflated figure is only the labeled secondary overflow-guard line. |
| 219 | let guard = estimate_input_tokens_conservative(messages, system); |
| 220 | assert!( |
| 221 | guard > gate, |
| 222 | "{label}: fixture must separate the estimators" |
| 223 | ); |
| 224 | assert_eq!(report.overflow_guard_estimated_tokens, Some(guard)); |
| 225 | for text in [ |
| 226 | format_context_report(&report), |
| 227 | format_context_summary(&report), |
| 228 | ] { |
| 229 | assert!( |
| 230 | text.contains(&format!("Estimated active context: {gate} tokens")), |
| 231 | "{label}: {text}" |
| 232 | ); |
| 233 | assert!( |
| 234 | text.contains(&format!("Overflow guard: {guard} tokens")), |
| 235 | "{label}: {text}" |
| 236 | ); |
| 237 | } |
| 238 | |
| 239 | // Once the provider bills a prompt above the local estimate, every |
| 240 | // surface lifts to the bill together — the footer meter included. |
| 241 | let billed = gate + 7_000; |
| 242 | app.last_billed_input_tokens = Some(u32::try_from(billed).expect("fixture bill")); |
| 243 | let billed_u64 = billed as u64; |
| 244 | assert_eq!( |
| 245 | compaction_pressure_reached_with_billed(messages, system, &compaction, Some(billed_u64)), |
| 246 | billed >= THRESHOLD, |
| 247 | "{label}: billed gate decision" |
| 248 | ); |
| 249 | let billed_preflight = crate::core::turn::TurnContext::new(8) |
| 250 | .live_input_tokens_for_compaction( |
| 251 | messages, |
| 252 | system, |
| 253 | Some(u32::try_from(billed).expect("fixture bill")), |
| 254 | ) |
| 255 | .expect("non-empty request"); |
| 256 | let (billed_meter, _, _) = crate::tui::ui::context_usage_snapshot(app).expect("meter reading"); |
| 257 | let billed_report = build_context_report(app); |
| 258 | assert_eq!(billed_preflight, billed_u64, "{label}: billed preflight"); |
| 259 | assert_eq!(billed_meter, billed as i64, "{label}: billed meter"); |
| 260 | assert_eq!( |
| 261 | billed_report.active_context_estimated_tokens, billed, |
| 262 | "{label}: billed /context headline" |
| 263 | ); |
| 264 | app.last_billed_input_tokens = None; |
| 265 | gate |
| 266 | } |
| 267 | |
| 268 | /// The `~before → ~after tokens` pair a compaction receipt prints. |
| 269 | fn receipt_token_pair(message: &str) -> (usize, usize) { |
| 270 | let (_, tail) = message.split_once("), ~").expect("receipt token clause"); |
| 271 | let (before, rest) = tail.split_once(" → ~").expect("receipt arrow"); |
| 272 | let (after, _) = rest.split_once(" tokens").expect("receipt tokens"); |
| 273 | ( |
| 274 | before.parse().expect("before tokens"), |
| 275 | after.parse().expect("after tokens"), |
| 276 | ) |
| 277 | } |
| 278 | |
| 279 | fn streaming(requests: &[MessageRequest]) -> Vec<MessageRequest> { |
| 280 | requests |
| 281 | .iter() |
| 282 | .filter(|request| request.stream == Some(true)) |
| 283 | .cloned() |
| 284 | .collect() |
| 285 | } |
| 286 | |
| 287 | #[test] |
| 288 | fn one_pressure_number_across_mid_turn_compaction_and_route_switch() { |
| 289 | let _env = lock_test_env(); |
| 290 | let home = tempdir().expect("home"); |
| 291 | let _codewhale_home = EnvVarGuard::set("CODEWHALE_HOME", home.path()); |
| 292 | let _user_home = EnvVarGuard::set("HOME", home.path()); |
| 293 | let _user_profile = EnvVarGuard::set("USERPROFILE", home.path()); |
| 294 | let runtime = tokio::runtime::Builder::new_current_thread() |
| 295 | .enable_all() |
| 296 | .build() |
| 297 | .expect("runtime"); |
| 298 | runtime.block_on(async { |
| 299 | let workspace = tempdir().expect("workspace"); |
| 300 | std::fs::write(workspace.path().join("README.md"), "verified fixture evidence") |
| 301 | .expect("write fixture"); |
| 302 | |
| 303 | let default_config = Config::default(); |
| 304 | let route_a = resolve(&default_config, crate::config::DEFAULT_TEXT_MODEL); |
| 305 | let private_config = private_route_config(); |
| 306 | let route_b = resolve(&private_config, PRIVATE_MODEL); |
| 307 | assert_ne!( |
| 308 | route_a.candidate.endpoint().base_url, |
| 309 | route_b.candidate.endpoint().base_url, |
| 310 | "the second turn must switch endpoint" |
| 311 | ); |
| 312 | assert_ne!(route_a.model, route_b.model, "and route"); |
| 313 | |
| 314 | // Turn 1: tool-heavy, crosses the threshold mid-turn. |
| 315 | let mock = std::sync::Arc::new(MockLlmClient::new(Vec::new())); |
| 316 | for step in 0..8 { |
| 317 | mock.push_turn(tool_step(step, 32_000)); |
| 318 | } |
| 319 | mock.push_turn(canned::simple_text_turn("All reads verified on route A.")); |
| 320 | // Turn 2: continues on the new route and endpoint. |
| 321 | mock.push_turn(tool_step(100, 400)); |
| 322 | mock.push_turn(canned::simple_text_turn("Continued on route B.")); |
| 323 | for checkpoint in 0..6 { |
| 324 | mock.push_message_response( |
| 325 | serde_json::from_value(json!({ |
| 326 | "id": format!("summary-{checkpoint}"), |
| 327 | "type": "message", |
| 328 | "role": "assistant", |
| 329 | "content": [{"type": "text", "text": format!( |
| 330 | "Current objective: finish the README reads. Checkpoint {checkpoint}: earlier reads verified; continue the remaining reads, then report." |
| 331 | )}], |
| 332 | "model": "mock-model", |
| 333 | "usage": {"input_tokens": 0, "output_tokens": 0} |
| 334 | })) |
| 335 | .expect("summary response"), |
| 336 | ); |
| 337 | } |
| 338 | |
| 339 | let engine_config = EngineConfig { |
| 340 | workspace: workspace.path().to_path_buf(), |
| 341 | snapshots_enabled: false, |
| 342 | subagents_enabled: false, |
| 343 | ..EngineConfig::default() |
| 344 | }; |
| 345 | let (engine, handle) = |
| 346 | Engine::new_with_model_client(engine_config, &default_config, mock.clone()); |
| 347 | let task = tokio::spawn(engine.run()); |
| 348 | |
| 349 | let turn_one = run_turn( |
| 350 | &handle, |
| 351 | turn_op("Read README.md repeatedly and verify it.", &route_a), |
| 352 | ) |
| 353 | .await; |
| 354 | let after_turn_one = mock.captured_requests().len(); |
| 355 | let turn_two = run_turn(&handle, turn_op("Continue on the new route.", &route_b)).await; |
| 356 | handle.send(Op::Shutdown).await.expect("shutdown"); |
| 357 | task.await.expect("engine task"); |
| 358 | |
| 359 | let requests = mock.captured_requests(); |
| 360 | let turn_one_requests = streaming(&requests[..after_turn_one]); |
| 361 | let turn_two_requests = streaming(&requests[after_turn_one..]); |
| 362 | assert_eq!(turn_one_requests.len(), 9, "one request per scripted step"); |
| 363 | assert_eq!(turn_two_requests.len(), 2, "turn two continues"); |
| 364 | assert!( |
| 365 | turn_one.auto_compactions >= 1, |
| 366 | "turn one must compact mid-turn: {turn_one:?}" |
| 367 | ); |
| 368 | assert_eq!( |
| 369 | turn_two.auto_compactions, 0, |
| 370 | "turn two stays under the threshold: {turn_two:?}" |
| 371 | ); |
| 372 | for request in &turn_two_requests { |
| 373 | assert_eq!(request.model, PRIVATE_MODEL, "turn two uses route B"); |
| 374 | } |
| 375 | |
| 376 | let mut app = crate::test_support::test_app_with_options( |
| 377 | crate::test_support::test_tui_options(workspace.path()), |
| 378 | ); |
| 379 | let mut readings = Vec::new(); |
| 380 | for (index, request) in turn_one_requests.iter().enumerate() { |
| 381 | readings.push(assert_one_pressure_number( |
| 382 | &mut app, |
| 383 | &format!("turn 1 request {index}"), |
| 384 | request, |
| 385 | &route_a, |
| 386 | )); |
| 387 | } |
| 388 | // The compaction happened mid-turn: the transcript shrank between two |
| 389 | // requests of the same turn, and the pressure number fell with it. |
| 390 | let shrink = turn_one_requests |
| 391 | .windows(2) |
| 392 | .position(|pair| pair[1].messages.len() < pair[0].messages.len()) |
| 393 | .expect("a mid-turn compaction shrinks the next request"); |
| 394 | assert!( |
| 395 | readings[shrink + 1] < readings[shrink], |
| 396 | "pressure falls across the compaction: {readings:?}" |
| 397 | ); |
| 398 | assert!( |
| 399 | readings[..=shrink].iter().any(|tokens| *tokens + 16_000 >= THRESHOLD), |
| 400 | "the turn approached the threshold before compacting: {readings:?}" |
| 401 | ); |
| 402 | |
| 403 | let (_, window_a, _) = crate::tui::ui::context_usage_snapshot(&app).expect("meter"); |
| 404 | for (index, request) in turn_two_requests.iter().enumerate() { |
| 405 | assert_one_pressure_number( |
| 406 | &mut app, |
| 407 | &format!("turn 2 request {index}"), |
| 408 | request, |
| 409 | &route_b, |
| 410 | ); |
| 411 | } |
| 412 | let (_, window_b, _) = crate::tui::ui::context_usage_snapshot(&app).expect("meter"); |
| 413 | assert_eq!(window_b, PRIVATE_WINDOW, "meter follows the switched route"); |
| 414 | assert_ne!(window_a, window_b, "the route switch changes the window"); |
| 415 | |
| 416 | // Compaction receipts report the same pressure number: the printed |
| 417 | // `after` equals `post_input_tokens`, which is the reading of the |
| 418 | // first request sent after that compaction. The printed `before` |
| 419 | // crossed the gate's threshold. |
| 420 | let shrinks: Vec<usize> = turn_one_requests |
| 421 | .windows(2) |
| 422 | .enumerate() |
| 423 | .filter(|(_, pair)| pair[1].messages.len() < pair[0].messages.len()) |
| 424 | .map(|(index, _)| index + 1) |
| 425 | .collect(); |
| 426 | assert_eq!( |
| 427 | shrinks.len(), |
| 428 | turn_one.receipts.len(), |
| 429 | "one shrink per receipt: {:?}", |
| 430 | turn_one.receipts |
| 431 | ); |
| 432 | for ((message, post_input_tokens), next) in turn_one.receipts.iter().zip(&shrinks) { |
| 433 | let (before, after) = receipt_token_pair(message); |
| 434 | assert_eq!(Some(after as u64), *post_input_tokens, "{message}"); |
| 435 | assert_eq!(after, readings[*next], "receipt vs next request: {message}"); |
| 436 | assert!(before >= THRESHOLD, "receipt before crossed the gate: {message}"); |
| 437 | } |
| 438 | }); |
| 439 | } |
| 440 |