返回 CodeWhale
sleep_guard.rs
根目录 / crates / runtime / src / sleep_guard.rs
1 //! Keep the host from idling into sleep while a turn is in flight.
2 //!
3 //! A suspended host cannot run the engine, so nothing here survives a real
4 //! suspend — `core::engine::streaming::sleep_gap_detected` already reports that
5 //! case and the engine re-issues the request (issue #2990). What this module
6 //! prevents is the avoidable one: an unattended machine idling into sleep in
7 //! the middle of a turn, which is how a long turn gets lost with no error at
8 //! all.
9 //!
10 //! Scope, stated so nobody expects more than it does:
11 //!
12 //! - It holds the platform's *idle-sleep* assertion only. An explicit `sleep`/
13 //! `pmset sleepnow`, a closed lid, or a low battery still wins — refusing
14 //! those is the machine owner's call, not a running task's.
15 //! - It is held for the duration of a turn and released on drop, so the host's
16 //! power behaviour outside a turn is untouched.
17 //! - It follows the same gate as the rest of the host-facing chrome
18 //! (`EngineConfig::terminal_chrome_enabled`): an interactive TUI turn holds
19 //! it, while headless hosts — `exec`, app-server, CI — never do. A dedicated
20 //! `[tui]` opt-out key is not implemented yet; the headless gate is the
21 //! escape hatch today.
22 //! - Windows is not implemented. `SetThreadExecutionState` is thread-affine —
23 //! the release has to happen on the thread that set it, which a guard that
24 //! travels with a turn cannot promise. Rather than ship an untested holder
25 //! that might silently never release, this is a no-op there for now.
26 //!
27 //! Release is `Drop` and never cached: a leaked inhibitor would keep a laptop
28 //! awake forever, which is worse than the problem this solves.
29 //!
30 //! The inhibitor is a `tokio::process` child, because the guard lives inside
31 //! `Engine::run_turn` on the runtime: the spawn registers with the runtime,
32 //! `kill_on_drop` sends the release signal, and the runtime reaps the child —
33 //! no `wait` runs inline on a worker (#6149). `hold` therefore has to be
34 //! called from within a Tokio runtime context.
35 //!
36 //! `Drop` never runs when the process is killed, crashes or calls
37 //! `process::exit`, so each inhibitor is also tied to this process's lifetime
38 //! independently of the guard: on macOS `caffeinate -w <pid>` exits with the
39 //! process it watches, and on Linux the pipe below closes with it.
40 //!
41 //! On Linux the lock is held by `systemd-inhibit` around a child of its own,
42 //! and a kill is never forwarded to that grandchild. The command is therefore
43 //! `cat` reading a pipe this guard holds: dropping the guard closes the pipe,
44 //! `cat` exits on EOF, and `systemd-inhibit` follows — nothing is left behind.
45
46 #[cfg(unix)]
47 use tokio::process::Child;
48 // Only the macOS and Linux inhibitors spawn anything; every other Unix
49 // (Android, the BSDs, illumos) is a no-op and would see these as dead.
50 #[cfg(any(target_os = "macos", target_os = "linux"))]
51 use std::process::Stdio;
52 #[cfg(any(target_os = "macos", target_os = "linux"))]
53 use tokio::process::Command;
54
55 /// An idle-sleep assertion held for as long as this value lives.
56 pub struct SleepGuard {
57 /// The platform inhibitor process, when one was started. `None` means the
58 /// platform has no implementation, or the process could not be started —
59 /// keeping the host awake is best-effort and must never fail a turn.
60 /// Dropping it is the release: the child is spawned with `kill_on_drop`.
61 #[cfg(unix)]
62 child: Option<Child>,
63 }
64
65 impl SleepGuard {
66 /// Hold the host awake until the returned guard drops. Call it from the
67 /// Tokio runtime: the inhibitor is a `tokio::process` child.
68 #[must_use]
69 pub fn hold() -> Self {
70 #[cfg(unix)]
71 {
72 Self {
73 child: start_inhibitor(),
74 }
75 }
76 #[cfg(not(unix))]
77 {
78 Self {}
79 }
80 }
81
82 /// The inhibitor's process id, for diagnostics and tests. Absent when the
83 /// platform is a no-op or the process did not start.
84 #[cfg(all(test, unix))]
85 pub(crate) fn inhibitor_pid(&self) -> Option<u32> {
86 self.child.as_ref().and_then(Child::id)
87 }
88 }
89
90 #[cfg(unix)]
91 impl Drop for SleepGuard {
92 fn drop(&mut self) {
93 // Releasing is dropping the child: `kill_on_drop` sends the signal
94 // here, synchronously, and the runtime reaps the process afterwards.
95 // Explicit so the field's purpose is code rather than a lint waiver.
96 drop(self.child.take());
97 }
98 }
99
100 /// `-i` prevents idle sleep. `Drop` kills caffeinate, and macOS releases the
101 /// assertion with the process. `-w` watches this process too: a crash, kill
102 /// or `process::exit` skips `Drop`, and the reparented caffeinate would
103 /// otherwise keep the host awake indefinitely.
104 #[cfg(target_os = "macos")]
105 fn start_inhibitor() -> Option<Child> {
106 let pid = std::process::id().to_string();
107 spawn("caffeinate", &caffeinate_args(&pid))
108 }
109
110 #[cfg(target_os = "macos")]
111 fn caffeinate_args(watched_pid: &str) -> [&str; 3] {
112 ["-i", "-w", watched_pid]
113 }
114
115 /// `--what=idle` only: an explicit suspend or a closed lid is still honoured.
116 /// `cat` on the guard's pipe is the command whose lifetime holds the block
117 /// open: it exits on EOF when the guard drops, which no signal sent to
118 /// `systemd-inhibit` could make a `sleep infinity` grandchild do.
119 #[cfg(target_os = "linux")]
120 fn start_inhibitor() -> Option<Child> {
121 spawn(
122 "systemd-inhibit",
123 &[
124 "--what=idle",
125 "--why=Codewhale turn in flight",
126 "--mode=block",
127 "cat",
128 ],
129 )
130 }
131
132 /// Everything else Unix (BSD, illumos, …) has no inhibitor this module knows.
133 #[cfg(all(unix, not(any(target_os = "macos", target_os = "linux"))))]
134 fn start_inhibitor() -> Option<Child> {
135 None
136 }
137
138 #[cfg(any(target_os = "macos", target_os = "linux"))]
139 fn spawn(program: &str, args: &[&str]) -> Option<Child> {
140 Command::new(program)
141 .args(args)
142 // The pipe is never written to: closing it when the guard drops is
143 // what ends an inhibitor's own child (see the Linux inhibitor).
144 .stdin(Stdio::piped())
145 .stdout(Stdio::null())
146 .stderr(Stdio::null())
147 // Killing the inhibitor is what releases the assertion; the runtime
148 // reaps the child afterwards, so nothing here waits inline.
149 .kill_on_drop(true)
150 .spawn()
151 .ok()
152 }
153
154 #[cfg(all(test, unix))]
155 mod tests {
156 use super::*;
157
158 /// Whether the process is still there. `kill(pid, 0)` asks the kernel
159 /// without touching the process, so this cannot perturb the guard.
160 fn alive(pid: u32) -> bool {
161 // SAFETY: signal 0 performs the permission/existence check only.
162 unsafe { libc::kill(pid as libc::pid_t, 0) == 0 }
163 }
164
165 /// The kill is sent on drop and the runtime reaps the child on the next
166 /// `SIGCHLD`, so "gone" is a short poll rather than an instant fact. If
167 /// the pid were reused by a new process in that window the test would be
168 /// racing itself, which is why callers assert on the guard's own child.
169 async fn released(pid: u32) -> bool {
170 let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(5);
171 while alive(pid) {
172 if tokio::time::Instant::now() >= deadline {
173 return false;
174 }
175 tokio::time::sleep(std::time::Duration::from_millis(10)).await;
176 }
177 true
178 }
179
180 #[tokio::test]
181 #[cfg(any(target_os = "macos", target_os = "linux"))]
182 async fn the_inhibitor_lives_exactly_as_long_as_the_guard() {
183 let guard = SleepGuard::hold();
184 let pid = guard
185 .inhibitor_pid()
186 .expect("this platform starts an inhibitor");
187 assert!(alive(pid), "the inhibitor must be running while held");
188
189 drop(guard);
190
191 assert!(
192 released(pid).await,
193 "a released guard must not leave an inhibitor keeping the host awake"
194 );
195 }
196
197 /// Linux: `systemd-inhibit` holds the lock around a child of its own and a
198 /// kill never reaches that grandchild — the guard's pipe is what ends it.
199 /// Without logind the inhibitor exits at once and the list is empty, so
200 /// this proves something only where an inhibitor really runs.
201 #[tokio::test]
202 #[cfg(target_os = "linux")]
203 async fn a_released_guard_leaves_no_grandchild_behind() {
204 let guard = SleepGuard::hold();
205 let pid = guard
206 .inhibitor_pid()
207 .expect("this platform starts an inhibitor");
208 // Give the inhibitor a moment to fork its command.
209 tokio::time::sleep(std::time::Duration::from_millis(200)).await;
210 let grandchildren: Vec<u32> =
211 tokio::fs::read_to_string(format!("/proc/{pid}/task/{pid}/children"))
212 .await
213 .unwrap_or_default()
214 .split_whitespace()
215 .filter_map(|child| child.parse().ok())
216 .collect();
217
218 drop(guard);
219
220 assert!(released(pid).await, "the inhibitor itself must be gone");
221 for grandchild in grandchildren {
222 assert!(
223 released(grandchild).await,
224 "process {grandchild} outlived the guard: the inhibitor's command must end with the guard's pipe"
225 );
226 }
227 }
228
229 /// macOS: when the owning process dies without running `Drop`, the
230 /// inhibitor must exit on its own. A stand-in owner is killed here, since
231 /// the test cannot kill its own process.
232 #[tokio::test]
233 #[cfg(target_os = "macos")]
234 async fn the_inhibitor_exits_when_its_owner_dies_without_dropping_the_guard() {
235 let mut owner = Command::new("sleep")
236 .arg("60")
237 .kill_on_drop(true)
238 .spawn()
239 .expect("start a stand-in owner process");
240 let owner_pid = owner.id().expect("owner pid").to_string();
241 let mut inhibitor = spawn("caffeinate", &caffeinate_args(&owner_pid))
242 .expect("start caffeinate watching the stand-in owner");
243 let pid = inhibitor.id().expect("inhibitor pid");
244 assert!(
245 alive(pid),
246 "the inhibitor must be running while its owner lives"
247 );
248
249 // The owner dies without anything killing the inhibitor, as when the
250 // process is SIGKILLed and the guard's `Drop` never runs.
251 owner.kill().await.expect("kill the stand-in owner");
252
253 // Waiting (rather than probing the pid) also reaps the child; if it
254 // never exits, dropping it at the end of the test kills it.
255 let exited =
256 tokio::time::timeout(std::time::Duration::from_secs(5), inhibitor.wait()).await;
257 assert!(
258 exited.is_ok(),
259 "an inhibitor whose owner died must not keep the host awake"
260 );
261 }
262
263 /// macOS: the guard a turn really takes must watch this process. The
264 /// test above proves `-w` ends caffeinate with a stand-in owner; this one
265 /// proves `hold` passes it, and passes our own pid.
266 #[tokio::test]
267 #[cfg(target_os = "macos")]
268 async fn a_held_guard_watches_the_process_that_holds_it() {
269 let guard = SleepGuard::hold();
270 let pid = guard
271 .inhibitor_pid()
272 .expect("this platform starts an inhibitor");
273 let ps = std::process::Command::new("ps")
274 .args(["-o", "args=", "-p", &pid.to_string()])
275 .output()
276 .expect("run ps");
277 let args = String::from_utf8_lossy(&ps.stdout);
278 assert_eq!(
279 args.trim(),
280 format!("caffeinate -i -w {}", std::process::id()),
281 "the inhibitor must end with the process whose turn it holds"
282 );
283 }
284
285 #[tokio::test]
286 #[cfg(any(target_os = "macos", target_os = "linux"))]
287 async fn holding_twice_holds_two_independent_inhibitors() {
288 // Turns are serialized, but nothing here should assume it: two guards
289 // must not share one process, or the first drop would release both.
290 let first = SleepGuard::hold();
291 let second = SleepGuard::hold();
292 let (a, b) = (
293 first.inhibitor_pid().expect("first inhibitor"),
294 second.inhibitor_pid().expect("second inhibitor"),
295 );
296 assert_ne!(a, b, "each guard owns its own inhibitor process");
297 drop(first);
298 assert!(released(a).await, "the first guard released only its own");
299 assert!(alive(b), "the second guard still holds the host awake");
300 }
301 }
302
302 lines RUST