| 1 | //! Keep the host from idling into sleep while a turn is in flight. |
| 2 | //! |
| 3 | //! A suspended host cannot run the engine, so nothing here survives a real |
| 4 | //! suspend — `core::engine::streaming::sleep_gap_detected` already reports that |
| 5 | //! case and the engine re-issues the request (issue #2990). What this module |
| 6 | //! prevents is the avoidable one: an unattended machine idling into sleep in |
| 7 | //! the middle of a turn, which is how a long turn gets lost with no error at |
| 8 | //! all. |
| 9 | //! |
| 10 | //! Scope, stated so nobody expects more than it does: |
| 11 | //! |
| 12 | //! - It holds the platform's *idle-sleep* assertion only. An explicit `sleep`/ |
| 13 | //! `pmset sleepnow`, a closed lid, or a low battery still wins — refusing |
| 14 | //! those is the machine owner's call, not a running task's. |
| 15 | //! - It is held for the duration of a turn and released on drop, so the host's |
| 16 | //! power behaviour outside a turn is untouched. |
| 17 | //! - It follows the same gate as the rest of the host-facing chrome |
| 18 | //! (`EngineConfig::terminal_chrome_enabled`): an interactive TUI turn holds |
| 19 | //! it, while headless hosts — `exec`, app-server, CI — never do. A dedicated |
| 20 | //! `[tui]` opt-out key is not implemented yet; the headless gate is the |
| 21 | //! escape hatch today. |
| 22 | //! - Windows is not implemented. `SetThreadExecutionState` is thread-affine — |
| 23 | //! the release has to happen on the thread that set it, which a guard that |
| 24 | //! travels with a turn cannot promise. Rather than ship an untested holder |
| 25 | //! that might silently never release, this is a no-op there for now. |
| 26 | //! |
| 27 | //! Release is `Drop` and never cached: a leaked inhibitor would keep a laptop |
| 28 | //! awake forever, which is worse than the problem this solves. |
| 29 | //! |
| 30 | //! The inhibitor is a `tokio::process` child, because the guard lives inside |
| 31 | //! `Engine::run_turn` on the runtime: the spawn registers with the runtime, |
| 32 | //! `kill_on_drop` sends the release signal, and the runtime reaps the child — |
| 33 | //! no `wait` runs inline on a worker (#6149). `hold` therefore has to be |
| 34 | //! called from within a Tokio runtime context. |
| 35 | //! |
| 36 | //! `Drop` never runs when the process is killed, crashes or calls |
| 37 | //! `process::exit`, so each inhibitor is also tied to this process's lifetime |
| 38 | //! independently of the guard: on macOS `caffeinate -w <pid>` exits with the |
| 39 | //! process it watches, and on Linux the pipe below closes with it. |
| 40 | //! |
| 41 | //! On Linux the lock is held by `systemd-inhibit` around a child of its own, |
| 42 | //! and a kill is never forwarded to that grandchild. The command is therefore |
| 43 | //! `cat` reading a pipe this guard holds: dropping the guard closes the pipe, |
| 44 | //! `cat` exits on EOF, and `systemd-inhibit` follows — nothing is left behind. |
| 45 | |
| 46 | #[cfg(unix)] |
| 47 | use tokio::process::Child; |
| 48 | // Only the macOS and Linux inhibitors spawn anything; every other Unix |
| 49 | // (Android, the BSDs, illumos) is a no-op and would see these as dead. |
| 50 | #[cfg(any(target_os = "macos", target_os = "linux"))] |
| 51 | use std::process::Stdio; |
| 52 | #[cfg(any(target_os = "macos", target_os = "linux"))] |
| 53 | use tokio::process::Command; |
| 54 | |
| 55 | /// An idle-sleep assertion held for as long as this value lives. |
| 56 | pub struct SleepGuard { |
| 57 | /// The platform inhibitor process, when one was started. `None` means the |
| 58 | /// platform has no implementation, or the process could not be started — |
| 59 | /// keeping the host awake is best-effort and must never fail a turn. |
| 60 | /// Dropping it is the release: the child is spawned with `kill_on_drop`. |
| 61 | #[cfg(unix)] |
| 62 | child: Option<Child>, |
| 63 | } |
| 64 | |
| 65 | impl SleepGuard { |
| 66 | /// Hold the host awake until the returned guard drops. Call it from the |
| 67 | /// Tokio runtime: the inhibitor is a `tokio::process` child. |
| 68 | #[must_use] |
| 69 | pub fn hold() -> Self { |
| 70 | #[cfg(unix)] |
| 71 | { |
| 72 | Self { |
| 73 | child: start_inhibitor(), |
| 74 | } |
| 75 | } |
| 76 | #[cfg(not(unix))] |
| 77 | { |
| 78 | Self {} |
| 79 | } |
| 80 | } |
| 81 | |
| 82 | /// The inhibitor's process id, for diagnostics and tests. Absent when the |
| 83 | /// platform is a no-op or the process did not start. |
| 84 | #[cfg(all(test, unix))] |
| 85 | pub(crate) fn inhibitor_pid(&self) -> Option<u32> { |
| 86 | self.child.as_ref().and_then(Child::id) |
| 87 | } |
| 88 | } |
| 89 | |
| 90 | #[cfg(unix)] |
| 91 | impl Drop for SleepGuard { |
| 92 | fn drop(&mut self) { |
| 93 | // Releasing is dropping the child: `kill_on_drop` sends the signal |
| 94 | // here, synchronously, and the runtime reaps the process afterwards. |
| 95 | // Explicit so the field's purpose is code rather than a lint waiver. |
| 96 | drop(self.child.take()); |
| 97 | } |
| 98 | } |
| 99 | |
| 100 | /// `-i` prevents idle sleep. `Drop` kills caffeinate, and macOS releases the |
| 101 | /// assertion with the process. `-w` watches this process too: a crash, kill |
| 102 | /// or `process::exit` skips `Drop`, and the reparented caffeinate would |
| 103 | /// otherwise keep the host awake indefinitely. |
| 104 | #[cfg(target_os = "macos")] |
| 105 | fn start_inhibitor() -> Option<Child> { |
| 106 | let pid = std::process::id().to_string(); |
| 107 | spawn("caffeinate", &caffeinate_args(&pid)) |
| 108 | } |
| 109 | |
| 110 | #[cfg(target_os = "macos")] |
| 111 | fn caffeinate_args(watched_pid: &str) -> [&str; 3] { |
| 112 | ["-i", "-w", watched_pid] |
| 113 | } |
| 114 | |
| 115 | /// `--what=idle` only: an explicit suspend or a closed lid is still honoured. |
| 116 | /// `cat` on the guard's pipe is the command whose lifetime holds the block |
| 117 | /// open: it exits on EOF when the guard drops, which no signal sent to |
| 118 | /// `systemd-inhibit` could make a `sleep infinity` grandchild do. |
| 119 | #[cfg(target_os = "linux")] |
| 120 | fn start_inhibitor() -> Option<Child> { |
| 121 | spawn( |
| 122 | "systemd-inhibit", |
| 123 | &[ |
| 124 | "--what=idle", |
| 125 | "--why=Codewhale turn in flight", |
| 126 | "--mode=block", |
| 127 | "cat", |
| 128 | ], |
| 129 | ) |
| 130 | } |
| 131 | |
| 132 | /// Everything else Unix (BSD, illumos, …) has no inhibitor this module knows. |
| 133 | #[cfg(all(unix, not(any(target_os = "macos", target_os = "linux"))))] |
| 134 | fn start_inhibitor() -> Option<Child> { |
| 135 | None |
| 136 | } |
| 137 | |
| 138 | #[cfg(any(target_os = "macos", target_os = "linux"))] |
| 139 | fn spawn(program: &str, args: &[&str]) -> Option<Child> { |
| 140 | Command::new(program) |
| 141 | .args(args) |
| 142 | // The pipe is never written to: closing it when the guard drops is |
| 143 | // what ends an inhibitor's own child (see the Linux inhibitor). |
| 144 | .stdin(Stdio::piped()) |
| 145 | .stdout(Stdio::null()) |
| 146 | .stderr(Stdio::null()) |
| 147 | // Killing the inhibitor is what releases the assertion; the runtime |
| 148 | // reaps the child afterwards, so nothing here waits inline. |
| 149 | .kill_on_drop(true) |
| 150 | .spawn() |
| 151 | .ok() |
| 152 | } |
| 153 | |
| 154 | #[cfg(all(test, unix))] |
| 155 | mod tests { |
| 156 | use super::*; |
| 157 | |
| 158 | /// Whether the process is still there. `kill(pid, 0)` asks the kernel |
| 159 | /// without touching the process, so this cannot perturb the guard. |
| 160 | fn alive(pid: u32) -> bool { |
| 161 | // SAFETY: signal 0 performs the permission/existence check only. |
| 162 | unsafe { libc::kill(pid as libc::pid_t, 0) == 0 } |
| 163 | } |
| 164 | |
| 165 | /// The kill is sent on drop and the runtime reaps the child on the next |
| 166 | /// `SIGCHLD`, so "gone" is a short poll rather than an instant fact. If |
| 167 | /// the pid were reused by a new process in that window the test would be |
| 168 | /// racing itself, which is why callers assert on the guard's own child. |
| 169 | async fn released(pid: u32) -> bool { |
| 170 | let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(5); |
| 171 | while alive(pid) { |
| 172 | if tokio::time::Instant::now() >= deadline { |
| 173 | return false; |
| 174 | } |
| 175 | tokio::time::sleep(std::time::Duration::from_millis(10)).await; |
| 176 | } |
| 177 | true |
| 178 | } |
| 179 | |
| 180 | #[tokio::test] |
| 181 | #[cfg(any(target_os = "macos", target_os = "linux"))] |
| 182 | async fn the_inhibitor_lives_exactly_as_long_as_the_guard() { |
| 183 | let guard = SleepGuard::hold(); |
| 184 | let pid = guard |
| 185 | .inhibitor_pid() |
| 186 | .expect("this platform starts an inhibitor"); |
| 187 | assert!(alive(pid), "the inhibitor must be running while held"); |
| 188 | |
| 189 | drop(guard); |
| 190 | |
| 191 | assert!( |
| 192 | released(pid).await, |
| 193 | "a released guard must not leave an inhibitor keeping the host awake" |
| 194 | ); |
| 195 | } |
| 196 | |
| 197 | /// Linux: `systemd-inhibit` holds the lock around a child of its own and a |
| 198 | /// kill never reaches that grandchild — the guard's pipe is what ends it. |
| 199 | /// Without logind the inhibitor exits at once and the list is empty, so |
| 200 | /// this proves something only where an inhibitor really runs. |
| 201 | #[tokio::test] |
| 202 | #[cfg(target_os = "linux")] |
| 203 | async fn a_released_guard_leaves_no_grandchild_behind() { |
| 204 | let guard = SleepGuard::hold(); |
| 205 | let pid = guard |
| 206 | .inhibitor_pid() |
| 207 | .expect("this platform starts an inhibitor"); |
| 208 | // Give the inhibitor a moment to fork its command. |
| 209 | tokio::time::sleep(std::time::Duration::from_millis(200)).await; |
| 210 | let grandchildren: Vec<u32> = |
| 211 | tokio::fs::read_to_string(format!("/proc/{pid}/task/{pid}/children")) |
| 212 | .await |
| 213 | .unwrap_or_default() |
| 214 | .split_whitespace() |
| 215 | .filter_map(|child| child.parse().ok()) |
| 216 | .collect(); |
| 217 | |
| 218 | drop(guard); |
| 219 | |
| 220 | assert!(released(pid).await, "the inhibitor itself must be gone"); |
| 221 | for grandchild in grandchildren { |
| 222 | assert!( |
| 223 | released(grandchild).await, |
| 224 | "process {grandchild} outlived the guard: the inhibitor's command must end with the guard's pipe" |
| 225 | ); |
| 226 | } |
| 227 | } |
| 228 | |
| 229 | /// macOS: when the owning process dies without running `Drop`, the |
| 230 | /// inhibitor must exit on its own. A stand-in owner is killed here, since |
| 231 | /// the test cannot kill its own process. |
| 232 | #[tokio::test] |
| 233 | #[cfg(target_os = "macos")] |
| 234 | async fn the_inhibitor_exits_when_its_owner_dies_without_dropping_the_guard() { |
| 235 | let mut owner = Command::new("sleep") |
| 236 | .arg("60") |
| 237 | .kill_on_drop(true) |
| 238 | .spawn() |
| 239 | .expect("start a stand-in owner process"); |
| 240 | let owner_pid = owner.id().expect("owner pid").to_string(); |
| 241 | let mut inhibitor = spawn("caffeinate", &caffeinate_args(&owner_pid)) |
| 242 | .expect("start caffeinate watching the stand-in owner"); |
| 243 | let pid = inhibitor.id().expect("inhibitor pid"); |
| 244 | assert!( |
| 245 | alive(pid), |
| 246 | "the inhibitor must be running while its owner lives" |
| 247 | ); |
| 248 | |
| 249 | // The owner dies without anything killing the inhibitor, as when the |
| 250 | // process is SIGKILLed and the guard's `Drop` never runs. |
| 251 | owner.kill().await.expect("kill the stand-in owner"); |
| 252 | |
| 253 | // Waiting (rather than probing the pid) also reaps the child; if it |
| 254 | // never exits, dropping it at the end of the test kills it. |
| 255 | let exited = |
| 256 | tokio::time::timeout(std::time::Duration::from_secs(5), inhibitor.wait()).await; |
| 257 | assert!( |
| 258 | exited.is_ok(), |
| 259 | "an inhibitor whose owner died must not keep the host awake" |
| 260 | ); |
| 261 | } |
| 262 | |
| 263 | /// macOS: the guard a turn really takes must watch this process. The |
| 264 | /// test above proves `-w` ends caffeinate with a stand-in owner; this one |
| 265 | /// proves `hold` passes it, and passes our own pid. |
| 266 | #[tokio::test] |
| 267 | #[cfg(target_os = "macos")] |
| 268 | async fn a_held_guard_watches_the_process_that_holds_it() { |
| 269 | let guard = SleepGuard::hold(); |
| 270 | let pid = guard |
| 271 | .inhibitor_pid() |
| 272 | .expect("this platform starts an inhibitor"); |
| 273 | let ps = std::process::Command::new("ps") |
| 274 | .args(["-o", "args=", "-p", &pid.to_string()]) |
| 275 | .output() |
| 276 | .expect("run ps"); |
| 277 | let args = String::from_utf8_lossy(&ps.stdout); |
| 278 | assert_eq!( |
| 279 | args.trim(), |
| 280 | format!("caffeinate -i -w {}", std::process::id()), |
| 281 | "the inhibitor must end with the process whose turn it holds" |
| 282 | ); |
| 283 | } |
| 284 | |
| 285 | #[tokio::test] |
| 286 | #[cfg(any(target_os = "macos", target_os = "linux"))] |
| 287 | async fn holding_twice_holds_two_independent_inhibitors() { |
| 288 | // Turns are serialized, but nothing here should assume it: two guards |
| 289 | // must not share one process, or the first drop would release both. |
| 290 | let first = SleepGuard::hold(); |
| 291 | let second = SleepGuard::hold(); |
| 292 | let (a, b) = ( |
| 293 | first.inhibitor_pid().expect("first inhibitor"), |
| 294 | second.inhibitor_pid().expect("second inhibitor"), |
| 295 | ); |
| 296 | assert_ne!(a, b, "each guard owns its own inhibitor process"); |
| 297 | drop(first); |
| 298 | assert!(released(a).await, "the first guard released only its own"); |
| 299 | assert!(alive(b), "the second guard still holds the host awake"); |
| 300 | } |
| 301 | } |
| 302 |