返回 CodeWhale
job_control_guard.rs
根目录 / crates / tui / src / tui / ui / job_control_guard.rs
1 //! Job-control suspend/resume handshake for the TUI (#6169).
2 //!
3 //! A full-screen TUI that never handles job control poisons the shell it was
4 //! started from: once the process group reads a controlling tty it does not
5 //! own, the kernel stops it (SIGTTIN — or the user stops it with SIGTSTP) with
6 //! every mode it enabled still active — raw mode, mouse reporting, bracketed
7 //! paste, the alternate screen — and whatever shell is in the foreground is
8 //! then fed raw SGR/CUP escape fragments. The startup check
9 //! `terminal::require_foreground_terminal_owner` guards exactly one moment;
10 //! this module is the runtime half of the same contract: restore on stop,
11 //! rebuild on continue, like vim/less/htop.
12 //!
13 //! Shape (mirrors `fatal_signal_guard`, deliberately narrow):
14 //!
15 //! - **One handler serves both SIGTSTP and SIGTTIN.** The SIGTTIN case is the
16 //! one that bites in #6169: a handler that *returns* turns the background
17 //! read into `EIO`, which the input pump reports as a dead tty (the forbidden
18 //! EIO spin). A handler that never returns cannot: it always finishes with
19 //! `raise(SIGSTOP)`, which is uncatchable and unblockable.
20 //! - **The handler does exactly three things**, all async-signal-safe: write
21 //! the fixed restore bytes from `fatal_signal_guard::FATAL_RESTORE_BYTES`
22 //! (one byte table serves both process death and suspension — no second
23 //! table), `tcsetattr(TCSANOW)` from the cooked snapshot taken at install
24 //! time, and stop. **No crossterm call**: crossterm's raw-mode state lives
25 //! behind a mutex that the stopped input-pump thread may be holding, so
26 //! `disable_raw_mode()` here can deadlock inside a handler (and the process
27 //! is unstoppable while it does). No `tracing`, no allocation, no lock.
28 //! - **The SIGCONT handler is a single atomic store.** Re-entering modes and
29 //! repainting happen on the event-loop thread in normal context, where
30 //! crossterm is safe to call.
31 //! - **The resume action is skipped** while another owner has the terminal (the
32 //! child-handoff pause) or while this process group is still in the
33 //! background (a plain `bg`): re-entering raw mode and the alternate screen
34 //! then would steal the shell's tty.
35 //!
36 //! Out of scope by design (see the issue's maintainer discussion): a
37 //! `tcgetpgrp` pre-poll guard and the `restart_detached` liveness lie. Both are
38 //! check-then-act, and neither can close the race — the background `read(2)`
39 //! itself is the atomic foreground test.
40 //!
41 //! Not recoverable: SIGKILL while stopped (no handler runs), exactly as with
42 //! the fatal guard. Kill-switch: `CODEWHALE_DISABLE_JOB_CONTROL_GUARD=1`.
43
44 #[cfg(unix)]
45 use std::sync::OnceLock;
46 use std::sync::atomic::{AtomicBool, AtomicU8, Ordering};
47
48 #[cfg(unix)]
49 use super::fatal_signal_guard::FATAL_RESTORE_BYTES;
50
51 /// The teardown the stop handler writes. Deliberately the fatal guard's byte
52 /// string: one table, so a mode added to one restore path cannot be missing
53 /// from the other.
54 #[cfg(unix)]
55 pub(crate) const SUSPEND_RESTORE_BYTES: &[u8] = FATAL_RESTORE_BYTES;
56
57 /// A stop handler has run (and the process has been stopped by SIGSTOP).
58 const STOPPED_UNDER_HANDLER: u8 = 0b01;
59 /// SIGCONT arrived after such a stop: the terminal must be rebuilt.
60 const CONT_SEEN: u8 = 0b10;
61 /// Both bits: a handler-driven suspend is waiting for its resume repaint.
62 const PENDING_RESUME: u8 = STOPPED_UNDER_HANDLER | CONT_SEEN;
63
64 /// Suspend handshake state. Only ever touched by a single atomic op per site,
65 /// so the signal handlers stay async-signal-safe.
66 static SUSPEND_STATE: AtomicU8 = AtomicU8::new(0);
67
68 /// True once the restore bytes have been written for the current suspend
69 /// cycle. Cleared by [`mark_resumed`], so each new suspend restores again.
70 /// Guards both the byte write and the `tcsetattr` — repeated stops inside one
71 /// suspend cycle (SIGTTIN, SIGCONT, SIGTTIN again while still backgrounded)
72 /// must not repeat either.
73 static RESTORED: AtomicBool = AtomicBool::new(false);
74
75 /// Cooked termios snapshot taken at install time, before `enable_raw_mode`.
76 ///
77 /// A `OnceLock` read from a signal handler is safe here for the same reason it
78 /// is in `fatal_signal_guard`: the write happens once, on the main thread,
79 /// before any worker exists, and is never written again.
80 #[cfg(unix)]
81 static ORIGINAL_TERMIOS: OnceLock<libc::termios> = OnceLock::new();
82
83 /// The cooked snapshot above, for the fatal-signal guard: one snapshot, taken
84 /// once before raw mode, serves every signal-time restore.
85 #[cfg(unix)]
86 pub(crate) fn original_termios() -> Option<&'static libc::termios> {
87 ORIGINAL_TERMIOS.get()
88 }
89
90 /// Install the job-control guard. POSIX only; no-op elsewhere.
91 ///
92 /// Call once on the main thread, after the foreground-ownership check (the
93 /// termios snapshot below needs the still-cooked tty) and **before**
94 /// `enable_raw_mode()` — so every mode the TUI goes on to enable has a handler
95 /// that can undo it.
96 pub(crate) fn install_job_control_guard() {
97 #[cfg(unix)]
98 {
99 if std::env::var("CODEWHALE_DISABLE_JOB_CONTROL_GUARD")
100 .is_ok_and(|value| value == "1" || value == "true")
101 {
102 tracing::debug!("Job-control guard disabled by CODEWHALE_DISABLE_JOB_CONTROL_GUARD");
103 return;
104 }
105 // Piped/embedded surfaces must never receive escape bytes, and have no
106 // job control to participate in.
107 if unsafe { libc::isatty(libc::STDOUT_FILENO) } == 0 {
108 tracing::debug!("Job-control guard skipped: stdout is not a TTY");
109 return;
110 }
111 // Snapshot the cooked attributes now. In the handler this is the only
112 // way back: crossterm's own raw-mode teardown is behind a lock.
113 let mut original: libc::termios = unsafe { std::mem::zeroed() };
114 // SAFETY: `original` is a fully owned, properly sized `termios`.
115 if unsafe { libc::tcgetattr(libc::STDIN_FILENO, &mut original) } != 0 {
116 tracing::warn!(
117 "Job-control guard: tcgetattr(stdin) failed; terminal attributes will not be restored on suspend"
118 );
119 } else {
120 let _ = ORIGINAL_TERMIOS.set(original);
121 }
122 for signal in [libc::SIGTSTP, libc::SIGTTIN] {
123 // SAFETY: `signal` is a stop-class signal whose default action is
124 // "stop"; the handler is async-signal-safe by construction.
125 unsafe { install_stop_handler(signal) };
126 }
127 // SAFETY: SIGCONT's default action is to continue, which we replace.
128 unsafe { install_continue_handler() };
129 // The SIGTTIN path runs our stop handler while the group is in the
130 // BACKGROUND, and `tcsetattr` from a background group raises SIGTTOU
131 // (default action: stop) — the process would stop inside the handler
132 // before `raise(SIGSTOP)`, and the first `fg` would resume the rest of
133 // the handler and immediately stop again. Ignoring SIGTTOU is the
134 // standard full-screen-program disposition (vim does the same) and is
135 // the only way the background restore can complete. This is set in the
136 // installer (normal context), never in a handler.
137 unsafe { libc::signal(libc::SIGTTOU, libc::SIG_IGN) };
138 tracing::debug!(
139 "Job-control guard installed (TSTP/TTIN -> restore + SIGSTOP, CONT -> resume)"
140 );
141 }
142 #[cfg(not(unix))]
143 {
144 // Windows consoles have no job control; the console-mode cleanup path
145 // in `terminal.rs` covers the equivalent "mode leak" surface.
146 }
147 }
148
149 /// True while a handler-driven suspend is waiting for its resume repaint.
150 ///
151 /// This is the event loop's first check each iteration. Deliberately a *peek*,
152 /// not a take: the state is only cleared by [`mark_resumed`] once the terminal
153 /// has actually been rebuilt. Consuming it here would lose the resume when the
154 /// action has to be deferred (a child owns the tty, or we are still a
155 /// background process group) — and a resume that is lost is a clobbered screen.
156 pub(crate) fn take_resume() -> bool {
157 SUSPEND_STATE.load(Ordering::Acquire) & PENDING_RESUME == PENDING_RESUME
158 }
159
160 /// Acknowledge a completed resume: the next suspend restores from scratch.
161 ///
162 /// Narrow `fetch_and` rather than a plain store so bits set by a handler racing
163 /// with this call are not erased outright. A suspend landing inside the
164 /// load/acknowledge window can at worst lose one repaint; it can never lose the
165 /// stop, nor leak a mode, because the handler restores *before* stopping.
166 pub(crate) fn mark_resumed() {
167 let _ = SUSPEND_STATE.fetch_and(!PENDING_RESUME, Ordering::AcqRel);
168 RESTORED.store(false, Ordering::Release);
169 }
170
171 /// Write the restore bytes, then `tcsetattr`, then stop — nothing else.
172 ///
173 /// # Safety
174 ///
175 /// Async-signal-safe by construction: `write(2)`, `tcsetattr(3)`, `raise(2)`
176 /// and two atomic flag updates on plain `static`s. No crossterm, no `tracing`,
177 /// no allocation, no lock, no `OnceLock` *initialization* (only a read of one
178 /// that was filled before any thread existed).
179 #[cfg(unix)]
180 unsafe extern "C" fn stop_handler(_signal: libc::c_int) {
181 unsafe {
182 // One restore per suspend cycle, whatever order the stops arrive in.
183 if !RESTORED.swap(true, Ordering::AcqRel) {
184 let mut written: usize = 0;
185 while written < SUSPEND_RESTORE_BYTES.len() {
186 let n = libc::write(
187 libc::STDOUT_FILENO,
188 SUSPEND_RESTORE_BYTES.as_ptr().add(written) as *const libc::c_void,
189 SUSPEND_RESTORE_BYTES.len() - written,
190 );
191 if n <= 0 {
192 break;
193 }
194 written += n as usize;
195 }
196 if written == 0 {
197 let _ = libc::write(
198 libc::STDERR_FILENO,
199 SUSPEND_RESTORE_BYTES.as_ptr() as *const libc::c_void,
200 SUSPEND_RESTORE_BYTES.len(),
201 );
202 }
203 // TCSANOW: never TCSADRAIN/TCSAFLUSH, which can block on output a
204 // stopped peer will never drain.
205 if let Some(original) = ORIGINAL_TERMIOS.get() {
206 let _ = libc::tcsetattr(libc::STDIN_FILENO, libc::TCSANOW, original);
207 }
208 }
209
210 // Record that the stop is ours, then stop for real. SIGSTOP cannot be
211 // caught or blocked, so the handler always ends here: SIGTTIN is never
212 // allowed to return into the pump as `EIO`.
213 SUSPEND_STATE.fetch_or(STOPPED_UNDER_HANDLER, Ordering::Release);
214 libc::raise(libc::SIGSTOP);
215 }
216 }
217
218 /// One atomic store. Nothing else — every mode change happens on the event loop
219 /// thread, in normal context.
220 #[cfg(unix)]
221 unsafe extern "C" fn continue_handler(_signal: libc::c_int) {
222 SUSPEND_STATE.fetch_or(CONT_SEEN, Ordering::Release);
223 }
224
225 /// Install [`stop_handler`] for one stop-class signal via `sigaction`.
226 ///
227 /// # Safety
228 ///
229 /// `signal` must be a signal whose default action is "stop".
230 #[cfg(unix)]
231 unsafe fn install_stop_handler(signal: libc::c_int) {
232 unsafe {
233 // Zero the whole struct then set our two fields; the remaining members
234 // (empty signal mask, per-OS plumbing) are exactly what a zeroed
235 // default means, and the wrapper fills in what it owns. SA_RESTART
236 // keeps the input pump's interrupted read restarted after resume.
237 let mut action: libc::sigaction = std::mem::zeroed();
238 action.sa_sigaction = stop_handler as *const () as libc::sighandler_t;
239 action.sa_flags = libc::SA_RESTART;
240 if libc::sigaction(signal, &action, std::ptr::null_mut()) != 0 {
241 tracing::warn!(signal, "job-control guard install failed");
242 }
243 }
244 }
245
246 /// # Safety
247 ///
248 /// Installed only from [`install_job_control_guard`], on the main thread.
249 #[cfg(unix)]
250 unsafe fn install_continue_handler() {
251 unsafe {
252 let mut action: libc::sigaction = std::mem::zeroed();
253 action.sa_sigaction = continue_handler as *const () as libc::sighandler_t;
254 action.sa_flags = libc::SA_RESTART;
255 if libc::sigaction(libc::SIGCONT, &action, std::ptr::null_mut()) != 0 {
256 tracing::warn!(signal = libc::SIGCONT, "job-control guard install failed");
257 }
258 }
259 }
260
261 #[cfg(all(test, unix))]
262 mod tests {
263 use super::*;
264
265 /// Serializes the state-machine test against any other test that might poke
266 /// the same statics. There is exactly one such test today; the lock keeps
267 /// that true if a second one is added.
268 static STATE_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
269
270 #[test]
271 fn job_control_restore_bytes_are_the_fatal_guard_table() {
272 // One byte table for death and suspension: if this ever forks into a
273 // second table, a mode can be restored on one path and leaked on the
274 // other.
275 assert_eq!(SUSPEND_RESTORE_BYTES, FATAL_RESTORE_BYTES);
276 // The teardown the suspend path cannot do without.
277 let bytes = String::from_utf8_lossy(SUSPEND_RESTORE_BYTES).to_string();
278 for mode in [
279 "?1000l", // mouse tracking
280 "?1002l", // button-event mouse tracking
281 "?1003l", // any-motion mouse tracking
282 "?1006l", // SGR mouse encoding
283 "?2004l", // bracketed paste
284 "?1049l", // alternate screen
285 ] {
286 assert!(
287 bytes.contains(mode),
288 "suspend restore must reset {mode}; got: {bytes:?}"
289 );
290 }
291 }
292
293 #[test]
294 fn job_control_state_bits_are_disjoint() {
295 // The resume test is a mask compare; aliasing bits would make a stop
296 // with no SIGCONT look resumable.
297 assert_eq!(STOPPED_UNDER_HANDLER & CONT_SEEN, 0);
298 assert_eq!(PENDING_RESUME, STOPPED_UNDER_HANDLER | CONT_SEEN);
299 }
300
301 #[test]
302 fn job_control_state_machine_is_ordered_and_idempotent() {
303 let _guard = STATE_LOCK.lock().unwrap_or_else(|err| err.into_inner());
304 SUSPEND_STATE.store(0, Ordering::Release);
305 RESTORED.store(false, Ordering::Release);
306
307 // Nothing suspended: no resume.
308 assert!(!take_resume());
309 mark_resumed();
310 assert!(!take_resume());
311
312 // A bare SIGCONT (somebody else's `kill -CONT`) is not a resume: no
313 // handler-driven stop is on record.
314 SUSPEND_STATE.fetch_or(CONT_SEEN, Ordering::Release);
315 assert!(!take_resume());
316 mark_resumed();
317 assert!(!take_resume());
318
319 // The real handshake: the stop handler records the stop, SIGCONT
320 // records the continue, the loop sees both.
321 SUSPEND_STATE.fetch_or(STOPPED_UNDER_HANDLER, Ordering::Release);
322 assert!(!take_resume(), "stopped without SIGCONT is not resumable");
323 SUSPEND_STATE.fetch_or(CONT_SEEN, Ordering::Release);
324 assert!(take_resume());
325 // Peeking is idempotent: the action may have to be deferred, so the
326 // state must survive until it actually runs.
327 assert!(take_resume());
328
329 // Acknowledging resumes the cycle: no stale resume remains.
330 mark_resumed();
331 assert!(!take_resume());
332 assert!(!RESTORED.load(Ordering::Acquire));
333 mark_resumed();
334 assert!(!take_resume());
335 }
336 }
337
337 lines RUST