返回 CodeWhale
repo.rs
根目录 / crates / tui / src / snapshot / repo.rs
1 //! Side-git repository wrapper for workspace snapshots.
2 //!
3 //! `SnapshotRepo` shells out to the system `git` binary (we deliberately
4 //! avoid `git2` to dodge its LGPL surface). The two paths that matter:
5 //!
6 //! - `git_dir` → `<snapshot state dir>/<project_hash>/<worktree_hash>/.git`
7 //! - `work_tree` → the user's actual workspace
8 //!
9 //! Every git invocation passes both `--git-dir` AND `--work-tree`. That is
10 //! the single biggest safety mechanism: it guarantees we never accidentally
11 //! mutate the user's own `.git` directory. If git can't find the side
12 //! repo, the command fails fast instead of falling back to "current
13 //! directory".
14
15 use std::cell::RefCell;
16 use std::collections::{HashMap, HashSet};
17 use std::io;
18 use std::path::{Component, Path, PathBuf};
19 use std::process::Output;
20 use std::time::{Duration, SystemTime, UNIX_EPOCH};
21
22 use wait_timeout::ChildExt as _;
23
24 use crate::dependencies::ExternalTool;
25
26 use super::paths::{ensure_snapshot_dir, snapshot_git_dir};
27
28 /// Identifier for a snapshot — the underlying git commit id.
29 ///
30 /// The field is private: [`SnapshotId::parse`] is the only way to build one,
31 /// so every value handed to `git` as a revision is a full SHA-1 or SHA-256
32 /// hex object id and can never be read as an option or a revision expression.
33 #[derive(Debug, Clone, PartialEq, Eq)]
34 pub struct SnapshotId(String);
35
36 impl SnapshotId {
37 /// Accept exactly a full hex object id: 40 (SHA-1) or 64 (SHA-256)
38 /// ASCII hex digits. Anything else is `InvalidInput`.
39 pub fn parse(id: &str) -> io::Result<Self> {
40 if Self::is_well_formed(id) {
41 Ok(Self(id.to_string()))
42 } else {
43 Err(io::Error::new(
44 io::ErrorKind::InvalidInput,
45 "snapshot id must be a full hexadecimal commit id",
46 ))
47 }
48 }
49
50 /// Whether `id` would be accepted by [`SnapshotId::parse`].
51 pub fn is_well_formed(id: &str) -> bool {
52 matches!(id.len(), 40 | 64) && id.bytes().all(|b| b.is_ascii_hexdigit())
53 }
54
55 /// Take the id string out.
56 pub fn into_string(self) -> String {
57 self.0
58 }
59
60 /// Borrow the SHA as a string slice.
61 pub fn as_str(&self) -> &str {
62 &self.0
63 }
64 }
65
66 /// A single snapshot record (one row in `git log`).
67 #[derive(Debug, Clone)]
68 pub struct Snapshot {
69 /// Commit SHA inside the side repo.
70 pub id: SnapshotId,
71 /// Subject line — the label passed to [`SnapshotRepo::snapshot`].
72 pub label: String,
73 /// Root tree of the snapshot commit. Unlike [`Self::id`] it survives the
74 /// survivor-chain rebuild every prune performs (the rebuild re-commits
75 /// the same tree under a new commit id), so it is the durable identity a
76 /// recorded restore point is resolved by.
77 pub tree: SnapshotId,
78 /// Author timestamp (Unix seconds).
79 pub timestamp: i64,
80 /// Session this snapshot belongs to, when recorded (encoded as a
81 /// `[sid=...] ` label prefix). `None` for legacy snapshots taken
82 /// before session tagging existed.
83 pub session_id: Option<String>,
84 }
85
86 /// One path that differs between two snapshots
87 /// ([`SnapshotRepo::path_changes_between`]).
88 #[derive(Debug, Clone, PartialEq, Eq)]
89 pub struct SnapshotPathChange {
90 /// Workspace-relative path, as git names it.
91 pub path: String,
92 /// git's status letter: `A` added, `D` deleted, `M` modified, `T` type
93 /// changed.
94 pub status: char,
95 /// Lines added and removed; `None` for a binary file.
96 pub added: Option<u64>,
97 pub removed: Option<u64>,
98 }
99
100 /// What a file-scoped restore did to one path, relative to the working tree
101 /// it was applied to.
102 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
103 pub enum PathRestoreAction {
104 /// Both the snapshot and the working tree had the path; its content came
105 /// back from the snapshot.
106 Modified,
107 /// The snapshot had the path and the working tree no longer did, so the
108 /// restore recreated the file.
109 Recreated,
110 /// The working tree had the path and the snapshot did not, so the restore
111 /// removed the file.
112 Removed,
113 }
114
115 impl PathRestoreAction {
116 /// Stable wire name, also used by the runtime API response.
117 pub fn as_str(self) -> &'static str {
118 match self {
119 Self::Modified => "modified",
120 Self::Recreated => "recreated",
121 Self::Removed => "removed",
122 }
123 }
124 }
125
126 /// The safety snapshot a path restore writes first, or one already taken.
127 enum RestoreBackup<'a> {
128 Take(&'a str),
129 Existing(&'a SnapshotId),
130 }
131
132 /// Report of what [`SnapshotRepo::restore_paths`] did to one path.
133 #[derive(Debug, Clone)]
134 pub struct PathRestoreOutcome {
135 /// Workspace-relative path that was restored.
136 pub path: PathBuf,
137 /// How the working tree changed.
138 pub action: PathRestoreAction,
139 }
140
141 /// A snapshot as it was just written: its commit and the root tree it holds.
142 ///
143 /// The commit id can be rewritten by the prune that follows every snapshot
144 /// once the repo holds more than [`crate::snapshot::DEFAULT_MAX_SNAPSHOTS`]
145 /// (see `rebuild_survivor_chain`); the tree id cannot, because a tree is
146 /// content-addressed and the rebuild re-commits the same tree.
147 #[derive(Debug, Clone, PartialEq, Eq)]
148 pub struct TakenSnapshot {
149 pub id: SnapshotId,
150 pub tree: SnapshotId,
151 }
152
153 /// Wrapper around the per-workspace side-git repo.
154 pub struct SnapshotRepo {
155 git_dir: PathBuf,
156 work_tree: PathBuf,
157 }
158
159 const STALE_TMP_PACK_AGE: Duration = Duration::from_secs(60 * 60);
160
161 /// Advisory lock file inside the side repo's `.git`. Every session, turn and
162 /// sub-agent working in one workspace shares one side repo, and its index,
163 /// HEAD and object store are only consistent when snapshot, restore, prune
164 /// and gc run one at a time: a gc in one session otherwise deleted another
165 /// session's new, not yet referenced commit and left HEAD on a missing
166 /// object.
167 const SNAPSHOT_LOCK_FILE: &str = "codewhale-snapshot.lock";
168
169 /// Longest a snapshot, restore or prune waits for another writer's lock
170 /// before failing (the turn then shows the snapshot-failure notice).
171 const SNAPSHOT_LOCK_WAIT: Duration = Duration::from_secs(30);
172
173 /// How often a waiting writer retries the lock.
174 const SNAPSHOT_LOCK_POLL: Duration = Duration::from_millis(25);
175
176 thread_local! {
177 /// Side repos whose write lock this thread already holds, so a locked
178 /// operation that calls another (restore takes a safety snapshot, a
179 /// snapshot runs the size prune) does not wait on its own lock.
180 static HELD_SNAPSHOT_LOCKS: RefCell<Vec<PathBuf>> = const { RefCell::new(Vec::new()) };
181 }
182
183 /// Maximum total snapshot storage in megabytes before pruning kicks in at
184 /// snapshot time. Keeps the side repo from blowing up the user's disk during
185 /// long-running or high-churn sessions (#1112).
186 const MAX_SNAPSHOT_SIZE_MB: u64 = 500;
187
188 const BYTES_PER_MB: u64 = 1024 * 1024;
189
190 /// Grace margin below `MAX_SNAPSHOT_SIZE_MB` used as the prune target
191 /// so the repo doesn't hit the limit again one snapshot later.
192 const PRUNE_TARGET_MB: u64 = 400;
193
194 /// Default workspace-size ceiling above which snapshots self-disable
195 /// on first use (2 GB of non-excluded content). Reports from users with
196 /// multi-hundred-GB project directories — datasets, model weights,
197 /// docker image dumps that fall outside the built-in excludes —
198 /// surfaced that `git add -A` on first init would hang the TUI for
199 /// minutes-to-hours while indexing the workspace. Snapshots are a
200 /// rollback safety net, not a backup tool; bailing out on workspaces
201 /// that big is the right tradeoff. Users with legitimate large
202 /// monorepos can raise `[snapshots] max_workspace_gb` (or set it to
203 /// `0` to disable the cap entirely).
204 pub const DEFAULT_MAX_WORKSPACE_BYTES_FOR_SNAPSHOT: u64 = 2 * 1024 * 1024 * 1024;
205
206 /// Hard cap on the number of file entries the bounded size estimator
207 /// will inspect before declaring the workspace "too large". Protects
208 /// against a workspace with millions of tiny files (no individual
209 /// file is large, but `git add -A` would still take forever).
210 pub const SIZE_WALK_MAX_ENTRIES: usize = 200_000;
211
212 /// Which snapshot gate refused a workspace. The recovery differs per gate —
213 /// raising `[snapshots] max_workspace_gb` lifts only [`WorkspaceGate::TooLarge`]
214 /// — so callers must not offer one gate's remedy for another's failure.
215 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
216 pub enum WorkspaceGate {
217 /// Snapshot-eligible content exceeds the configured byte cap.
218 TooLarge,
219 /// The bounded walk hit [`SIZE_WALK_MAX_ENTRIES`]. This bound is
220 /// independent of the byte cap: `max_workspace_gb = 0` does not lift it.
221 TooManyEntries,
222 }
223
224 /// Leading text of the `io::Error` each gate produces. `core::turn` matches on
225 /// these to pick the right consequence/recovery notice, so they are one
226 /// declaration shared by producer and matcher rather than two literals.
227 pub const GATE_TOO_LARGE_MARKER: &str = "workspace too large for snapshots";
228 pub const GATE_TOO_MANY_ENTRIES_MARKER: &str = "workspace has too many files for snapshots";
229 pub const GATE_UNSAFE_LOCATION_MARKER: &str = "workspace snapshots are disabled";
230
231 /// Display a workspace path in gate diagnostics. The diagnostic names the
232 /// path the caller passed in — canonicalization is for filesystem and
233 /// security logic, not for the message. On Windows `Path::canonicalize`
234 /// rewrites more than the verbatim (`\\?\`) prefix (case, 8.3 names), so a
235 /// canonical spelling can never be relied on to match what users name; the
236 /// verbatim prefix is still stripped when present for readability.
237 fn display_workspace_for_gate(workspace: &Path) -> String {
238 let raw = workspace.display().to_string();
239 raw.strip_prefix(r"\\?\")
240 .or_else(|| raw.strip_prefix("//?/"))
241 .unwrap_or(&raw)
242 .to_string()
243 }
244
245 impl WorkspaceGate {
246 /// One-line English diagnostic for logs, `/undo`, and the gate matcher.
247 /// The user-facing consequence and recovery are localized by the notice
248 /// surfaces; this string must not restate them.
249 fn describe(self, cap_bytes: u64, workspace: &Path) -> String {
250 let workspace = display_workspace_for_gate(workspace);
251 match self {
252 Self::TooLarge => format!(
253 "{GATE_TOO_LARGE_MARKER}: over {} bytes of snapshot-eligible content in {workspace}",
254 cap_bytes,
255 ),
256 Self::TooManyEntries => format!(
257 "{GATE_TOO_MANY_ENTRIES_MARKER}: over {SIZE_WALK_MAX_ENTRIES} snapshot-eligible entries in {workspace}"
258 ),
259 }
260 }
261 }
262
263 /// Top-level directory and extension patterns that the snapshot path
264 /// already excludes via `BUILTIN_EXCLUDES`. The estimator skips these
265 /// up front so the size walk reflects what would actually land in the
266 /// snapshot commit. Kept narrow to common build-output dirs — anything
267 /// else falls back to the `.gitignore` filter.
268 const SIZE_WALK_SKIP_DIRS: &[&str] = &[
269 "node_modules",
270 "target",
271 "dist",
272 "build",
273 ".build",
274 ".next",
275 ".nuxt",
276 ".svelte-kit",
277 ".turbo",
278 ".parcel-cache",
279 "vendor",
280 ".cargo",
281 ".rustup",
282 ".npm",
283 ".bun",
284 ".yarn",
285 ".pnpm-store",
286 ".cache",
287 ".venv",
288 "venv",
289 ".tox",
290 "__pycache__",
291 ".mypy_cache",
292 ".pytest_cache",
293 ".ruff_cache",
294 ".gradle",
295 ".m2",
296 ".local",
297 ".git",
298 ];
299
300 const BUILTIN_EXCLUDES: &str = "\
301 # CodeWhale built-in snapshot exclusions
302 node_modules/
303 target/
304 dist/
305 build/
306 .build/
307 .next/
308 .nuxt/
309 .svelte-kit/
310 .turbo/
311 .parcel-cache/
312 vendor/
313 .cargo/
314 .rustup/
315 .npm/
316 .bun/
317 .yarn/
318 .pnpm-store/
319 .cache/
320 .venv/
321 venv/
322 .tox/
323 __pycache__/
324 *.pyc
325 .mypy_cache/
326 .pytest_cache/
327 .ruff_cache/
328 .gradle/
329 .m2/
330 .local/
331 .DS_Store
332
333 # Binary and generated artifacts. Snapshots are source rollback checkpoints,
334 # not a full binary backup; keeping these out avoids side-repo bloat.
335 *.exe
336 *.dll
337 *.so
338 *.dylib
339 *.wasm
340 *.o
341 *.obj
342 *.class
343 *.pdb
344 *.dSYM
345 *.zip
346 *.tar
347 *.tar.gz
348 *.tgz
349 *.tar.bz2
350 *.tar.xz
351 *.7z
352 *.rar
353 *.iso
354 *.dmg
355 *.bin
356 *.mp4
357 *.mov
358 *.mkv
359 *.avi
360 *.webm
361 *.mp3
362 *.wav
363 *.flac
364 *.aac
365 ";
366
367 impl SnapshotRepo {
368 /// Open an existing snapshot repo for `workspace` without creating or
369 /// initializing anything on disk.
370 ///
371 /// This is useful for read-only UI surfaces that want to report checkpoint
372 /// availability without paying the first-init size walk or surprising the
373 /// user by creating a side repo from a view action.
374 pub fn open_existing(workspace: &Path) -> io::Result<Option<Self>> {
375 let work_tree = workspace
376 .canonicalize()
377 .unwrap_or_else(|_| workspace.to_path_buf());
378 let git_dir = snapshot_git_dir(&work_tree)?;
379 if !git_dir.exists() || !git_dir.join("HEAD").exists() {
380 return Ok(None);
381 }
382 Ok(Some(Self { git_dir, work_tree }))
383 }
384
385 /// Open or initialize the snapshot repo for `workspace`.
386 ///
387 /// On first use this:
388 /// 1. Creates the `.git` dir under the resolved snapshot store.
389 /// 2. Runs `git init --bare=false --quiet`.
390 /// 3. Sets a fixed `user.name` / `user.email` so commits don't pick up
391 /// the user's global git identity (we don't want our snapshots to
392 /// look like they came from the user).
393 pub fn open_or_init(workspace: &Path) -> io::Result<Self> {
394 Self::open_or_init_with_cap(workspace, DEFAULT_MAX_WORKSPACE_BYTES_FOR_SNAPSHOT)
395 }
396
397 /// Variant of [`Self::open_or_init`] that accepts an explicit
398 /// workspace-size cap. `cap_bytes = 0` disables the cap entirely
399 /// (always snapshot, regardless of size).
400 ///
401 /// When the workspace exceeds the cap and the side repo hasn't
402 /// been initialized yet, returns `Err(InvalidInput)` with a
403 /// "workspace too large" reason. Subsequent calls (after the user
404 /// shrinks the workspace or raises the cap via config) succeed.
405 pub fn open_or_init_with_cap(workspace: &Path, cap_bytes: u64) -> io::Result<Self> {
406 let work_tree = workspace
407 .canonicalize()
408 .unwrap_or_else(|_| workspace.to_path_buf());
409 if let Some(reason) = unsafe_workspace_snapshot_reason(
410 &work_tree,
411 crate::config::effective_home_dir().as_deref(),
412 ) {
413 return Err(io::Error::new(
414 io::ErrorKind::InvalidInput,
415 format!(
416 "{GATE_UNSAFE_LOCATION_MARKER} for {reason}: {}",
417 display_workspace_for_gate(workspace)
418 ),
419 ));
420 }
421
422 let git_dir = ensure_snapshot_dir(&work_tree)?.join(".git");
423
424 // A `.git` without HEAD is an init that never finished (or only a
425 // peer's lock file): initialize it rather than use it half-made.
426 let needs_init = !git_dir.join("HEAD").exists();
427 if needs_init {
428 // First-init size guard. Skipping this on subsequent opens
429 // is intentional: paying a workspace walk on every snapshot
430 // would defeat the purpose of the cap, and a workspace
431 // that fit on first init is allowed to grow within the
432 // existing repo's `MAX_SNAPSHOT_SIZE_MB` budget. Users on
433 // workspaces that grew past the cap mid-session get the
434 // existing aggressive-pruning path in `snapshot()`.
435 if let Err(gate) =
436 estimate_workspace_size_bounded(&work_tree, cap_bytes, SIZE_WALK_MAX_ENTRIES)
437 {
438 return Err(io::Error::new(
439 io::ErrorKind::InvalidInput,
440 gate.describe(cap_bytes, workspace),
441 ));
442 }
443 // The lock file lives in `.git`, so create it first; `git init`
444 // accepts an existing `.git` directory.
445 std::fs::create_dir_all(&git_dir)?;
446 }
447 let repo = Self { git_dir, work_tree };
448 if needs_init {
449 // Two sessions opening a new workspace at once must not both run
450 // `git init` and the identity config: a config write that loses
451 // git's own config lock is ignored, and a snapshot committed
452 // before the identity lands would carry the user's.
453 repo.with_write_lock(|| {
454 if repo.git_dir.join("HEAD").exists() {
455 return Ok(());
456 }
457 repo.init_side_repo()
458 })?;
459 }
460
461 write_builtin_excludes(&repo.git_dir)?;
462 if let Err(err) = cleanup_stale_pack_temps(&repo.git_dir, STALE_TMP_PACK_AGE) {
463 tracing::debug!(
464 target: "snapshot",
465 "failed to clean stale snapshot tmp_pack files: {err}"
466 );
467 }
468 Ok(repo)
469 }
470
471 /// `git init` the side repo and pin its config. Runs under the write lock.
472 fn init_side_repo(&self) -> io::Result<()> {
473 let (git_dir, work_tree) = (&self.git_dir, &self.work_tree);
474 let parent = git_dir.parent().ok_or_else(|| {
475 io::Error::new(io::ErrorKind::InvalidInput, "snapshot dir has no parent")
476 })?;
477 // `git init` here uses the parent directory as the work tree
478 // and stores metadata in `.git`. We then continue to use
479 // explicit `--git-dir` / `--work-tree` flags for every other
480 // command so behaviour is invariant of cwd.
481 let mut git =
482 crate::dependencies::Git::command().ok_or_else(|| io_other("git not found on PATH"))?;
483 let init = git.arg("init").arg("--quiet").arg(parent);
484 let init = run_bounded_git(init, "init")
485 .map_err(|e| io_other(format!("failed to run git init: {e}")))?;
486 if !init.status.success() {
487 return Err(io_other(format!(
488 "git init failed: {}",
489 String::from_utf8_lossy(&init.stderr).trim()
490 )));
491 }
492
493 // Pin a stable identity so snapshot commits are recognisable
494 // and don't bleed into the user's git config.
495 let _ = run_git(
496 git_dir,
497 work_tree,
498 &["config", "user.name", "deepseek-snapshots"],
499 );
500 let _ = run_git(
501 git_dir,
502 work_tree,
503 &["config", "user.email", "snapshots@codewhale.local"],
504 );
505 // Don't auto-gc on every commit; we manage pruning ourselves.
506 let _ = run_git(git_dir, work_tree, &["config", "gc.auto", "0"]);
507 // Ignore CRLF rewriting — we want byte-for-byte fidelity.
508 let _ = run_git(git_dir, work_tree, &["config", "core.autocrlf", "false"]);
509 Ok(())
510 }
511
512 /// Take a snapshot of the current working tree.
513 ///
514 /// Internally: `git add -A`, `git write-tree`, `git commit-tree`, then
515 /// `git update-ref HEAD <commit>`.
516 /// `git add -A` honours the user's workspace ignore rules while staging
517 /// into the side repo's index.
518 ///
519 /// Before committing, checks whether the snapshot directory exceeds
520 /// [`MAX_SNAPSHOT_SIZE_MB`] and prunes the oldest snapshots if it does.
521 ///
522 /// Returns the snapshot's commit SHA.
523 #[allow(dead_code)] // convenience entry kept for tests and legacy callers; production writes go through snapshot_with_session
524 pub fn snapshot(&self, label: &str) -> io::Result<SnapshotId> {
525 self.snapshot_with_session(label, None)
526 }
527
528 /// Take a snapshot, tagging it with the owning session id.
529 ///
530 /// The session id is encoded into the commit message as a `[sid=...] `
531 /// label prefix. [`Self::list`] decodes it back into
532 /// [`Snapshot::session_id`] and strips the prefix from the visible
533 /// label, so existing listing surfaces keep showing the plain label.
534 /// Legacy snapshots taken through [`Self::snapshot`] carry no prefix
535 /// and decode with `session_id == None`.
536 pub fn snapshot_with_session(
537 &self,
538 label: &str,
539 session_id: Option<&str>,
540 ) -> io::Result<SnapshotId> {
541 self.take_snapshot(label, session_id).map(|taken| taken.id)
542 }
543
544 /// [`Self::snapshot_with_session`], also reporting the root tree the
545 /// snapshot holds — the identity a recorded restore point keeps.
546 pub fn take_snapshot(
547 &self,
548 label: &str,
549 session_id: Option<&str>,
550 ) -> io::Result<TakenSnapshot> {
551 self.with_write_lock(|| self.take_snapshot_locked(label, session_id))
552 }
553
554 fn take_snapshot_locked(
555 &self,
556 label: &str,
557 session_id: Option<&str>,
558 ) -> io::Result<TakenSnapshot> {
559 // Guard against disk blowup (#1112): if the snapshot directory has
560 // grown beyond the limit, prune aggressively before adding more.
561 // When the prune actually destroys restore points the user is told
562 // once per workspace — losing undo history to a log line is the S5
563 // failure mode (2026-08-04 snapshot hunt).
564 if let Ok(removed) = self.prune_size_pressure(
565 MAX_SNAPSHOT_SIZE_MB * BYTES_PER_MB,
566 PRUNE_TARGET_MB * BYTES_PER_MB,
567 ) && removed > 0
568 {
569 notify_snapshot_history_pruned_once(&self.work_tree, removed);
570 }
571 // A HEAD that names a missing commit would make every `commit-tree
572 // -p` below fail ("is not a valid object"), silently ending undo.
573 self.repair_broken_head()?;
574 let parent = run_git(
575 &self.git_dir,
576 &self.work_tree,
577 &["rev-parse", "--verify", "--quiet", "HEAD^{commit}"],
578 )?;
579 let parent = parent
580 .status
581 .success()
582 .then(|| String::from_utf8_lossy(&parent.stdout).trim().to_string())
583 .filter(|s| !s.is_empty());
584
585 self.stage_work_tree()?;
586 let (tree, sha) = match self.commit_staged_tree(parent.as_deref(), label, session_id) {
587 Ok(done) => done,
588 Err(first) => {
589 // An index naming objects that no longer exist (left by an
590 // interrupted or racing gc) breaks every later snapshot:
591 // `add -A` does not rehash files whose stat data is
592 // unchanged, and `write-tree` hands back the index's cached
593 // tree id even when that tree is gone, so `commit-tree` then
594 // fails. Rebuild the index from the work tree once.
595 let reset = run_git(&self.git_dir, &self.work_tree, &["read-tree", "--empty"])?;
596 if !reset.status.success() {
597 return Err(first);
598 }
599 self.stage_work_tree()?;
600 self.commit_staged_tree(parent.as_deref(), label, session_id)?
601 }
602 };
603
604 self.move_head(Some(&sha), parent.as_deref())?;
605
606 let id = SnapshotId::parse(&sha).map_err(|_| {
607 io_other(format!(
608 "git commit-tree returned a malformed commit id: {sha:?}"
609 ))
610 })?;
611 let tree = SnapshotId::parse(&tree).map_err(|_| {
612 io_other(format!(
613 "git write-tree returned a malformed tree id: {tree:?}"
614 ))
615 })?;
616 Ok(TakenSnapshot { id, tree })
617 }
618
619 /// `write-tree` the staged index and `commit-tree` it onto `parent`,
620 /// returning the tree and commit ids.
621 fn commit_staged_tree(
622 &self,
623 parent: Option<&str>,
624 label: &str,
625 session_id: Option<&str>,
626 ) -> io::Result<(String, String)> {
627 let tree = run_git(&self.git_dir, &self.work_tree, &["write-tree"])?;
628 if !tree.status.success() {
629 return Err(io_other(format!(
630 "git write-tree failed: {}",
631 String::from_utf8_lossy(&tree.stderr).trim()
632 )));
633 }
634 let tree = String::from_utf8_lossy(&tree.stdout).trim().to_string();
635
636 let mut args = vec!["commit-tree".to_string(), tree.clone()];
637 if let Some(parent) = parent {
638 args.push("-p".to_string());
639 args.push(parent.to_string());
640 }
641 args.push("-m".to_string());
642 args.push(Self::encode_session_label(label, session_id));
643 let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect();
644
645 // `commit-tree` creates marker commits even when the tree matches its
646 // parent, and it does not run user/global commit hooks.
647 let commit = run_git(&self.git_dir, &self.work_tree, &arg_refs)?;
648 if !commit.status.success() {
649 return Err(io_other(format!(
650 "git commit-tree failed: {}",
651 String::from_utf8_lossy(&commit.stderr).trim()
652 )));
653 }
654 let sha = String::from_utf8_lossy(&commit.stdout).trim().to_string();
655 Ok((tree, sha))
656 }
657
658 /// Repair a side repo whose HEAD names a commit that no longer exists
659 /// (an interrupted gc or prune, a copied or partially deleted
660 /// `~/.codewhale/snapshots` directory). Left alone, every later snapshot
661 /// fails on `commit-tree -p <missing>` and /undo is dead without a word.
662 ///
663 /// The broken ref is deleted so the next snapshot starts a fresh history.
664 /// Restore points before the break cannot be recovered; the caller tells
665 /// the user. Returns `true` when a repair happened.
666 pub fn repair_broken_head(&self) -> io::Result<bool> {
667 self.with_write_lock(|| self.repair_broken_head_locked())
668 }
669
670 fn repair_broken_head_locked(&self) -> io::Result<bool> {
671 let commit = run_git(
672 &self.git_dir,
673 &self.work_tree,
674 &["rev-parse", "--verify", "--quiet", "HEAD^{commit}"],
675 )?;
676 if commit.status.success() {
677 return Ok(false);
678 }
679 // An unborn branch (fresh repo) names nothing: nothing to repair.
680 let named = run_git(
681 &self.git_dir,
682 &self.work_tree,
683 &["rev-parse", "--verify", "--quiet", "HEAD"],
684 )?;
685 if !named.status.success() {
686 return Ok(false);
687 }
688 let missing = String::from_utf8_lossy(&named.stdout).trim().to_string();
689 // A lookup can fail for a moment while another session sharing this
690 // repo runs gc or repack. Only a commit that is really absent
691 // justifies touching HEAD: deleting it discards every restore point.
692 if self.is_commit(&missing)? {
693 return Ok(false);
694 }
695 self.reset_missing_head(&missing)
696 }
697
698 /// Move HEAD off `missing`, a commit id it named when this repair looked.
699 /// Both moves are compare-and-swap against `missing`: a writer that does
700 /// not take the snapshot lock (an older build) may have published a valid
701 /// snapshot since, and neither the reflog reset nor the delete may then
702 /// overwrite it. A refused reset is reported, never followed by a delete.
703 fn reset_missing_head(&self, missing: &str) -> io::Result<bool> {
704 // The newest reflog entry that still names a commit keeps the
705 // restore points before the break.
706 if let Some(recovered) = self.newest_reflog_commit(missing)? {
707 self.move_head(Some(&recovered), Some(missing))
708 .map_err(|error| {
709 io_other(format!(
710 "snapshot history HEAD pointed at missing commit {missing} and was not reset to {recovered}: {error}"
711 ))
712 })?;
713 tracing::warn!(
714 target: "snapshot",
715 "snapshot history HEAD pointed at missing commit {missing}; reset to {recovered} from the reflog"
716 );
717 return Ok(false);
718 }
719 self.move_head(None, Some(missing)).map_err(|error| {
720 io_other(format!(
721 "snapshot history HEAD points at missing commit {missing} and could not be reset: {error}"
722 ))
723 })?;
724 tracing::warn!(
725 target: "snapshot",
726 "snapshot history HEAD pointed at missing commit {missing}; started a fresh history"
727 );
728 Ok(true)
729 }
730
731 fn is_commit(&self, oid: &str) -> io::Result<bool> {
732 let object = format!("{oid}^{{commit}}");
733 Ok(
734 run_git(&self.git_dir, &self.work_tree, &["cat-file", "-e", &object])?
735 .status
736 .success(),
737 )
738 }
739
740 /// The newest commit recorded in HEAD's reflogs (its branch's, then
741 /// HEAD's own) that still exists, skipping `missing`.
742 fn newest_reflog_commit(&self, missing: &str) -> io::Result<Option<String>> {
743 let branch = run_git(&self.git_dir, &self.work_tree, &["symbolic-ref", "HEAD"])?;
744 let mut logs = Vec::new();
745 if branch.status.success() {
746 let name = String::from_utf8_lossy(&branch.stdout).trim().to_string();
747 logs.push(self.git_dir.join("logs").join(name));
748 }
749 logs.push(self.git_dir.join("logs").join("HEAD"));
750 for log in logs {
751 let Ok(text) = std::fs::read_to_string(&log) else {
752 continue;
753 };
754 // Each line is `<old> <new> <who> <when>\t<message>`.
755 for line in text.lines().rev() {
756 let mut fields = line.split(' ');
757 let (Some(old), Some(new)) = (fields.next(), fields.next()) else {
758 continue;
759 };
760 for oid in [new, old] {
761 if oid.len() >= 40
762 && oid != missing
763 && oid.bytes().any(|b| b != b'0')
764 && self.is_commit(oid)?
765 {
766 return Ok(Some(oid.to_string()));
767 }
768 }
769 }
770 }
771 Ok(None)
772 }
773
774 /// Point the side repo's HEAD branch at a commit id that does not exist.
775 #[cfg(test)]
776 pub(crate) fn point_head_at_missing_commit_for_test(&self) {
777 let branch = run_git(&self.git_dir, &self.work_tree, &["symbolic-ref", "HEAD"])
778 .expect("symbolic-ref");
779 let branch = String::from_utf8_lossy(&branch.stdout).trim().to_string();
780 let path = self.git_dir.join(&branch);
781 std::fs::create_dir_all(path.parent().expect("ref parent")).expect("ref dir");
782 std::fs::write(path, "1111111111111111111111111111111111111111\n").expect("write ref");
783 }
784
785 /// Prefix a snapshot label with its owning session id, if any.
786 fn encode_session_label(label: &str, session_id: Option<&str>) -> String {
787 match session_id {
788 Some(sid) if !sid.is_empty() => format!("[sid={sid}] {label}"),
789 _ => label.to_string(),
790 }
791 }
792
793 /// Split a possibly session-tagged label back into `(session_id, label)`.
794 ///
795 /// Returns `(None, label)` for untagged labels. The decoded label is
796 /// the original one without the `[sid=...] ` prefix, so consumers that
797 /// match on `pre-turn:`/`tool:`/`redo:` prefixes keep working unchanged.
798 fn decode_session_label(label: &str) -> (Option<String>, String) {
799 let Some(rest) = label.strip_prefix("[sid=") else {
800 return (None, label.to_string());
801 };
802 let Some(end) = rest.find("] ") else {
803 return (None, label.to_string());
804 };
805 let sid = &rest[..end];
806 let plain = &rest[end + 2..];
807 if sid.is_empty() || plain.is_empty() {
808 return (None, label.to_string());
809 }
810 (Some(sid.to_string()), plain.to_string())
811 }
812 /// Size-pressure prune (#1112): if the side repo exceeds `max_bytes`,
813 /// drop the oldest half of the snapshots, repeatedly, until the store is
814 /// at or under `target_bytes` or only the protected restore points are
815 /// left. Returns the number of snapshots destroyed, so the caller can
816 /// tell the user their undo history shrank (S5).
817 ///
818 /// The prune goes oldest first by count and always keeps the newest
819 /// snapshot plus each session's newest `pre-turn:` and `post-turn:`
820 /// boundaries, so every session's running turn (and the one before it)
821 /// stays restorable. It used to
822 /// prune by age starting at one second, which on the first pass dropped
823 /// every snapshot older than a second and then wiped the rest: a workspace
824 /// whose side repo sat above the cap lost all undo history on every
825 /// snapshot.
826 fn prune_size_pressure(&self, max_bytes: u64, target_bytes: u64) -> io::Result<usize> {
827 self.with_write_lock(|| {
828 let current_bytes = dir_size_bytes(&self.git_dir)?;
829 if current_bytes <= max_bytes {
830 return Ok(0);
831 }
832 tracing::warn!(
833 target: "snapshot",
834 current_mb = current_bytes / BYTES_PER_MB,
835 limit_mb = max_bytes / BYTES_PER_MB,
836 "snapshot storage over limit — pruning the oldest snapshots"
837 );
838 let mut removed_total: usize = 0;
839 loop {
840 let snapshots = self.list(usize::MAX)?;
841 let mut survivors = size_pressure_survivors(&snapshots, snapshots.len() / 2);
842 if survivors.len() >= snapshots.len() {
843 // Halving kept only protected points: cut to those alone.
844 survivors = size_pressure_survivors(&snapshots, 0);
845 }
846 if survivors.is_empty() || survivors.len() >= snapshots.len() {
847 break;
848 }
849 self.rebuild_survivor_chain(&survivors, &snapshots[0].id)?;
850 self.reclaim_unreachable();
851 removed_total = removed_total.saturating_add(snapshots.len() - survivors.len());
852 let new_size = dir_size_bytes(&self.git_dir)?;
853 if new_size <= target_bytes {
854 tracing::info!(
855 target: "snapshot",
856 new_size_mb = new_size / BYTES_PER_MB,
857 "pruned snapshot storage back under limit"
858 );
859 break;
860 }
861 }
862 Ok(removed_total)
863 })
864 }
865
866 /// Restore the workspace to the state at `id`.
867 ///
868 /// Requires a durable safety snapshot before changing files. A failed
869 /// restore attempts to put those files back; if that also fails, the
870 /// error identifies the retained safety snapshot for recovery. This is
871 /// not atomic against an external editor changing the live workspace.
872 /// File/directory transitions are refused before checkout because the
873 /// replaced directory may contain files excluded from the backup.
874 /// We never touch the user's own `.git`.
875 pub fn restore(&self, id: &SnapshotId) -> io::Result<()> {
876 self.with_write_lock(|| self.restore_locked(id))
877 }
878
879 fn restore_locked(&self, id: &SnapshotId) -> io::Result<()> {
880 // The backup label is deliberately not an undo/revert-turn candidate.
881 let target_short = &id.as_str()[..id.as_str().len().min(12)];
882 let backup = self
883 .snapshot_with_session(&format!("pre-restore:{target_short}"), None)
884 .map_err(|error| {
885 io_other(format!(
886 "pre-restore safety snapshot failed; no workspace files were changed: {error}"
887 ))
888 })?;
889 let current_paths = self.tree_paths(backup.as_str())?;
890 let target_paths = self.tree_paths(id.as_str())?;
891 for rel in &target_paths {
892 match std::fs::symlink_metadata(self.work_tree.join(rel)) {
893 Ok(metadata) if metadata.is_dir() => {
894 return Err(io_other(format!(
895 "'{}' requires a file/directory transition; nothing was restored",
896 rel.display()
897 )));
898 }
899 Ok(_) if !current_paths.contains(rel) => {
900 return Err(io_other(format!(
901 "'{}' was excluded from the safety snapshot; nothing was restored",
902 rel.display()
903 )));
904 }
905 // Unix reports a path under a live file as NotADirectory;
906 // Windows reports NotFound, so look for the file itself.
907 Err(error)
908 if error.kind() == io::ErrorKind::NotADirectory
909 || (error.kind() == io::ErrorKind::NotFound
910 && ancestor_is_not_a_directory(&self.work_tree, rel)) =>
911 {
912 return Err(io_other(format!(
913 "'{}' requires a file/directory transition; nothing was restored",
914 rel.display()
915 )));
916 }
917 Err(error) if error.kind() != io::ErrorKind::NotFound => return Err(error),
918 _ => {}
919 }
920 }
921 if let Err(error) = self.restore_tree(id, &current_paths, &target_paths) {
922 let recovery = match self.restore_tree(&backup, &target_paths, &current_paths) {
923 Ok(()) => "previous snapshot files were restored".to_string(),
924 Err(rollback) => format!("rollback also failed: {rollback}"),
925 };
926 return Err(io_other(format!(
927 "restore failed: {error}; {recovery}; safety snapshot {} retains the previous files",
928 backup.as_str()
929 )));
930 }
931 Ok(())
932 }
933
934 fn restore_tree(
935 &self,
936 id: &SnapshotId,
937 current_paths: &HashSet<PathBuf>,
938 target_paths: &HashSet<PathBuf>,
939 ) -> io::Result<()> {
940 // An empty target (the first snapshot of an empty directory) has no
941 // path for `:/` to match, and git refuses the checkout outright; there
942 // is nothing to write back, only the later files to remove.
943 if !target_paths.is_empty() {
944 let checkout = run_git(
945 &self.git_dir,
946 &self.work_tree,
947 &["checkout", "--end-of-options", id.as_str(), "--", ":/"],
948 )?;
949 if !checkout.status.success() {
950 return Err(io_other(format!(
951 "git checkout failed: {}",
952 String::from_utf8_lossy(&checkout.stderr).trim()
953 )));
954 }
955 }
956 self.remove_paths_missing_from_target(current_paths, target_paths)
957 }
958
959 /// File restore never traverses symlinks, directories, or Git metadata.
960 /// Validate every existing component before reading, backing up or writing.
961 pub fn validate_restore_file(&self, rel: &Path) -> io::Result<bool> {
962 if !is_safe_relative_path(rel)
963 || rel
964 .components()
965 .any(|part| is_git_metadata_name(part.as_os_str()))
966 {
967 return Err(io::Error::new(
968 io::ErrorKind::InvalidInput,
969 format!(
970 "refusing to restore unsafe path '{}': restore requires a regular workspace file",
971 rel.display()
972 ),
973 ));
974 }
975 let mut path = self.work_tree.clone();
976 for part in rel.components() {
977 path.push(part);
978 match std::fs::symlink_metadata(&path) {
979 Ok(meta)
980 if meta.file_type().is_symlink()
981 || (path == self.work_tree.join(rel) && !meta.is_file())
982 || (path != self.work_tree.join(rel) && !meta.is_dir()) =>
983 {
984 return Err(io::Error::new(
985 io::ErrorKind::InvalidInput,
986 "restore refuses directories, symlinks and non-regular files",
987 ));
988 }
989 Ok(_) => {}
990 Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false),
991 Err(error) => return Err(error),
992 }
993 }
994 Ok(true)
995 }
996
997 fn snapshot_contains_regular_file(&self, id: &SnapshotId, rel: &Path) -> io::Result<bool> {
998 self.snapshot_file_blob(id, rel).map(|blob| blob.is_some())
999 }
1000
1001 /// The blob id `rel` holds in snapshot (commit or tree) `id`, or `None`
1002 /// when the snapshot does not contain it. Anything other than a regular
1003 /// file is `InvalidInput`: file-scoped restores never write symlinks or
1004 /// directories.
1005 fn snapshot_file_blob(&self, id: &SnapshotId, rel: &Path) -> io::Result<Option<String>> {
1006 let entry = run_git(
1007 &self.git_dir,
1008 &self.work_tree,
1009 &[
1010 "--literal-pathspecs",
1011 "ls-tree",
1012 "-z",
1013 "--end-of-options",
1014 id.as_str(),
1015 "--",
1016 rel.to_str()
1017 .ok_or_else(|| io_other("restore path must be UTF-8"))?,
1018 ],
1019 )?;
1020 if !entry.status.success() {
1021 return Err(io_other(format!(
1022 "Failed to inspect snapshot file: {}",
1023 String::from_utf8_lossy(&entry.stderr).trim()
1024 )));
1025 }
1026 if entry.stdout.is_empty() {
1027 return Ok(None);
1028 }
1029 let line = String::from_utf8_lossy(&entry.stdout);
1030 let blob = line
1031 .strip_prefix("100644 blob ")
1032 .or_else(|| line.strip_prefix("100755 blob "))
1033 .and_then(|rest| rest.split('\t').next())
1034 .filter(|sha| SnapshotId::is_well_formed(sha))
1035 .ok_or_else(|| {
1036 io::Error::new(
1037 io::ErrorKind::InvalidInput,
1038 "snapshot path is not a regular file",
1039 )
1040 })?;
1041 Ok(Some(blob.to_string()))
1042 }
1043
1044 /// Whether the working-tree bytes of `rel` are exactly what snapshot `id`
1045 /// holds for it (both absent counts as a match).
1046 ///
1047 /// Compares content hashes directly instead of `git diff <id>`: the side
1048 /// repo's index only lists paths present at the last snapshot, and a diff
1049 /// against a commit reports a path the index lost as deleted even when
1050 /// the file is on disk.
1051 pub fn path_matches_snapshot(&self, id: &SnapshotId, rel: &Path) -> io::Result<bool> {
1052 let in_work = self.validate_restore_file(rel)?;
1053 match (self.snapshot_file_blob(id, rel)?, in_work) {
1054 (None, false) => Ok(true),
1055 (None, true) | (Some(_), false) => Ok(false),
1056 (Some(blob), true) => {
1057 let rel_str = rel
1058 .to_str()
1059 .ok_or_else(|| io_other("restore path must be UTF-8"))?;
1060 let abs = self.work_tree.join(rel);
1061 let abs = abs
1062 .to_str()
1063 .ok_or_else(|| io_other("restore path must be UTF-8"))?;
1064 // Git for Windows cannot open the `\\?\` verbatim form that
1065 // `canonicalize` gives the work tree.
1066 let abs = match abs.strip_prefix(r"\\?\") {
1067 Some(rest) => match rest.strip_prefix(r"UNC\") {
1068 Some(unc) => format!(r"\\{unc}"),
1069 None => rest.to_string(),
1070 },
1071 None => abs.to_string(),
1072 };
1073 let hashed = run_git(
1074 &self.git_dir,
1075 &self.work_tree,
1076 &["hash-object", &format!("--path={rel_str}"), "--", &abs],
1077 )?;
1078 if !hashed.status.success() {
1079 return Err(io_other(format!(
1080 "git hash-object failed: {}",
1081 String::from_utf8_lossy(&hashed.stderr).trim()
1082 )));
1083 }
1084 Ok(String::from_utf8_lossy(&hashed.stdout).trim() == blob)
1085 }
1086 }
1087 }
1088
1089 /// Paths whose content differs between snapshots `from` and `to` (commit
1090 /// or tree ids), with renames reported as a delete plus an add so each
1091 /// side's path is restorable on its own.
1092 pub fn changed_paths_between(
1093 &self,
1094 from: &SnapshotId,
1095 to: &SnapshotId,
1096 ) -> io::Result<Vec<PathBuf>> {
1097 let diff = run_git(
1098 &self.git_dir,
1099 &self.work_tree,
1100 &[
1101 "diff",
1102 "--no-renames",
1103 "--name-only",
1104 "-z",
1105 "--end-of-options",
1106 from.as_str(),
1107 to.as_str(),
1108 "--",
1109 ],
1110 )?;
1111 if !diff.status.success() {
1112 return Err(io_other(format!(
1113 "git diff --name-only failed: {}",
1114 String::from_utf8_lossy(&diff.stderr).trim()
1115 )));
1116 }
1117 let mut paths: Vec<PathBuf> = parse_nul_paths(&diff.stdout).into_iter().collect();
1118 paths.sort();
1119 Ok(paths)
1120 }
1121
1122 /// Whether `rel` has the same content (or is absent) in both snapshots.
1123 pub fn path_same_in_snapshots(
1124 &self,
1125 a: &SnapshotId,
1126 b: &SnapshotId,
1127 rel: &Path,
1128 ) -> io::Result<bool> {
1129 Ok(self.snapshot_file_blob(a, rel)? == self.snapshot_file_blob(b, rel)?)
1130 }
1131
1132 /// Return whether `rel` differs between snapshot `id` and the current
1133 /// working tree.
1134 ///
1135 /// This is the single-path counterpart of
1136 /// [`Self::work_tree_matches_snapshot`]: it answers "would restoring just
1137 /// this file change anything?", which is what file-scoped revert
1138 /// cursoring needs. A path that exists in neither the snapshot nor the
1139 /// working tree does not differ.
1140 pub fn path_differs_from_snapshot(&self, id: &SnapshotId, rel: &Path) -> io::Result<bool> {
1141 let in_work = self.validate_restore_file(rel)?;
1142 let in_target = self.snapshot_contains_regular_file(id, rel)?;
1143 match (in_target, in_work) {
1144 // Neither side has it: nothing to restore and nothing to remove.
1145 (false, false) => Ok(false),
1146 // The snapshot has it and the working tree lost it.
1147 (true, false) => Ok(true),
1148 // The path was created after the snapshot.
1149 (false, true) => Ok(true),
1150 (true, true) => {
1151 let rel = rel.to_string_lossy().into_owned();
1152 let diff = run_git(
1153 &self.git_dir,
1154 &self.work_tree,
1155 &[
1156 "--literal-pathspecs",
1157 "diff",
1158 "--quiet",
1159 "--end-of-options",
1160 id.as_str(),
1161 "--",
1162 rel.as_str(),
1163 ],
1164 )?;
1165 git_diff_matches(diff).map(|matches| !matches)
1166 }
1167 }
1168 }
1169
1170 /// Restore only `rel_paths` from snapshot `id`.
1171 ///
1172 /// This is the file-scoped counterpart of [`Self::restore`]. The
1173 /// difference that matters: the whole-tree `git checkout <sha> -- :/` is
1174 /// replaced by a pathspec-limited checkout, so a working-tree path outside
1175 /// `rel_paths` is never written or deleted. The safety backup reads the workspace.
1176 ///
1177 /// A path the snapshot does not track is removed from the working tree
1178 /// (that is how a file created after the snapshot is reverted), and a path
1179 /// the snapshot tracks but the working tree lost is recreated. A path that
1180 /// exists in neither side produces no outcome at all, rather than a
1181 /// report claiming a change that did not happen.
1182 #[cfg(test)]
1183 pub fn restore_paths(
1184 &self,
1185 id: &SnapshotId,
1186 rel_paths: &[PathBuf],
1187 ) -> io::Result<Vec<PathRestoreOutcome>> {
1188 self.restore_paths_checked(id, rel_paths, || Ok(()))
1189 }
1190
1191 pub fn restore_file_if_unchanged(
1192 &self,
1193 id: &SnapshotId,
1194 rel: &Path,
1195 expected_hash: &str,
1196 ) -> io::Result<Vec<PathRestoreOutcome>> {
1197 let verify = || {
1198 let actual = if self.validate_restore_file(rel)? {
1199 let bytes = std::fs::read(self.work_tree.join(rel))?;
1200 format!("sha256:{}", crate::hashing::sha256_hex(bytes))
1201 } else {
1202 "absent".to_string()
1203 };
1204 if actual != expected_hash {
1205 return Err(io::Error::new(
1206 io::ErrorKind::WouldBlock,
1207 "The file changed after the selected change record. Refresh and review it before restoring; nothing was changed.",
1208 ));
1209 }
1210 Ok(())
1211 };
1212 verify()?;
1213 self.restore_paths_checked(id, &[rel.to_path_buf()], verify)
1214 }
1215
1216 fn restore_paths_checked(
1217 &self,
1218 id: &SnapshotId,
1219 rel_paths: &[PathBuf],
1220 preflight: impl FnOnce() -> io::Result<()>,
1221 ) -> io::Result<Vec<PathRestoreOutcome>> {
1222 let target_short = &id.as_str()[..id.as_str().len().min(12)];
1223 let plan: Vec<(PathBuf, SnapshotId)> = rel_paths
1224 .iter()
1225 .map(|rel| (rel.clone(), id.clone()))
1226 .collect();
1227 self.restore_path_plan(
1228 &plan,
1229 &format!("pre-restore:{target_short}"),
1230 false,
1231 preflight,
1232 )
1233 }
1234
1235 /// Restore each `(path, source)` pair of `plan` from its own snapshot
1236 /// (commit or tree id), and nothing else.
1237 ///
1238 /// The whole plan is validated, then one mandatory `backup_label` safety
1239 /// snapshot is written, then `preflight` runs immediately before the
1240 /// first mutation. A path its source does not hold is removed; with
1241 /// `prune_emptied_dirs` the directories that removal leaves empty go too
1242 /// (a turn-scoped undo removes the directories a dropped turn created),
1243 /// otherwise they stay (a file-scoped revert names a file, not a tree).
1244 pub fn restore_path_plan(
1245 &self,
1246 plan: &[(PathBuf, SnapshotId)],
1247 backup_label: &str,
1248 prune_emptied_dirs: bool,
1249 preflight: impl FnOnce() -> io::Result<()>,
1250 ) -> io::Result<Vec<PathRestoreOutcome>> {
1251 self.restore_path_plan_backed_up(
1252 plan,
1253 RestoreBackup::Take(backup_label),
1254 prune_emptied_dirs,
1255 preflight,
1256 )
1257 }
1258
1259 /// [`Self::restore_path_plan`] with a safety snapshot the caller already
1260 /// took (`backup`, a commit id) instead of a new one. `preflight` must
1261 /// prove every planned path is still as `backup` holds it, so the backup
1262 /// is as good as one taken now; reusing it keeps a second snapshot, and
1263 /// the prune that comes with it, out of the window between planning and
1264 /// the first write.
1265 pub fn restore_path_plan_with_backup(
1266 &self,
1267 plan: &[(PathBuf, SnapshotId)],
1268 backup: &SnapshotId,
1269 prune_emptied_dirs: bool,
1270 preflight: impl FnOnce() -> io::Result<()>,
1271 ) -> io::Result<Vec<PathRestoreOutcome>> {
1272 self.restore_path_plan_backed_up(
1273 plan,
1274 RestoreBackup::Existing(backup),
1275 prune_emptied_dirs,
1276 preflight,
1277 )
1278 }
1279
1280 fn restore_path_plan_backed_up(
1281 &self,
1282 plan: &[(PathBuf, SnapshotId)],
1283 backup: RestoreBackup<'_>,
1284 prune_emptied_dirs: bool,
1285 preflight: impl FnOnce() -> io::Result<()>,
1286 ) -> io::Result<Vec<PathRestoreOutcome>> {
1287 self.with_write_lock(|| {
1288 self.restore_path_plan_locked(plan, backup, prune_emptied_dirs, preflight)
1289 })
1290 }
1291
1292 fn restore_path_plan_locked(
1293 &self,
1294 plan: &[(PathBuf, SnapshotId)],
1295 backup: RestoreBackup<'_>,
1296 prune_emptied_dirs: bool,
1297 preflight: impl FnOnce() -> io::Result<()>,
1298 ) -> io::Result<Vec<PathRestoreOutcome>> {
1299 if plan.is_empty() {
1300 return Ok(Vec::new());
1301 }
1302 // Validate the entire request before any mutation or backup. A snapshot
1303 // directory entry must not turn a file action into recursive checkout.
1304 let mut pre_state = Vec::with_capacity(plan.len());
1305 for (rel, id) in plan {
1306 let in_work = self.validate_restore_file(rel)?;
1307 let in_target = self.snapshot_contains_regular_file(id, rel)?;
1308 pre_state.push((rel.clone(), id.clone(), in_target, in_work));
1309 }
1310
1311 // A durable backup is required for this destructive API. Ignored
1312 // files cannot be removed/overwritten if the snapshot cannot retain them.
1313 let backup = match backup {
1314 RestoreBackup::Take(label) => self.snapshot_with_session(label, None)?,
1315 RestoreBackup::Existing(id) => id.clone(),
1316 };
1317 for (rel, _, _, in_work) in &pre_state {
1318 if *in_work && !self.snapshot_contains_regular_file(&backup, rel)? {
1319 return Err(io_other(
1320 "File was excluded from the safety snapshot; nothing was restored",
1321 ));
1322 }
1323 self.validate_restore_file(rel)?;
1324 }
1325
1326 // Recheck after the potentially slow safety snapshot, immediately
1327 // before checkout/removal. New editor work is retained in the backup.
1328 preflight()?;
1329
1330 // One pathspec-limited checkout per source snapshot, in plan order.
1331 let mut sources: Vec<&SnapshotId> = Vec::new();
1332 for (_, id, in_target, _) in &pre_state {
1333 if *in_target && !sources.contains(&id) {
1334 sources.push(id);
1335 }
1336 }
1337 for source in sources {
1338 let tracked: Vec<String> = pre_state
1339 .iter()
1340 .filter(|(_, id, in_target, _)| *in_target && id == source)
1341 .map(|(rel, _, _, _)| rel.to_string_lossy().into_owned())
1342 .collect();
1343 let mut args: Vec<String> = vec![
1344 "--literal-pathspecs".to_string(),
1345 "checkout".to_string(),
1346 "--end-of-options".to_string(),
1347 source.as_str().to_string(),
1348 "--".to_string(),
1349 ];
1350 args.extend(tracked);
1351 let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect();
1352 let checkout = run_git(&self.git_dir, &self.work_tree, &arg_refs)?;
1353 if !checkout.status.success() {
1354 return Err(io_other(format!(
1355 "git checkout failed: {} (safety snapshot {} holds the previous files)",
1356 String::from_utf8_lossy(&checkout.stderr).trim(),
1357 backup.as_str()
1358 )));
1359 }
1360 }
1361
1362 let mut outcomes = Vec::new();
1363 for (rel, _, in_target, was_in_work) in pre_state {
1364 match (in_target, was_in_work) {
1365 (true, true) => outcomes.push(PathRestoreOutcome {
1366 path: rel,
1367 action: PathRestoreAction::Modified,
1368 }),
1369 (true, false) => outcomes.push(PathRestoreOutcome {
1370 path: rel,
1371 action: PathRestoreAction::Recreated,
1372 }),
1373 (false, true) => {
1374 let path = self.work_tree.join(&rel);
1375 self.validate_restore_file(&rel)?;
1376 std::fs::remove_file(&path).map_err(|error| {
1377 io_other(format!(
1378 "removing '{}' failed: {error} (safety snapshot {} holds the previous files)",
1379 rel.display(),
1380 backup.as_str()
1381 ))
1382 })?;
1383 if prune_emptied_dirs {
1384 self.prune_empty_parent_dirs(path.parent());
1385 }
1386 outcomes.push(PathRestoreOutcome {
1387 path: rel,
1388 action: PathRestoreAction::Removed,
1389 });
1390 }
1391 // Already in the snapshot's state.
1392 (false, false) => {}
1393 }
1394 }
1395 Ok(outcomes)
1396 }
1397
1398 /// Return whether the current workspace matches the given snapshot's
1399 /// tracked file content.
1400 ///
1401 /// This is intentionally narrower than a full "workspace identical"
1402 /// claim: it compares the current working tree against the snapshot's
1403 /// tracked paths via git's diff machinery. That is sufficient for
1404 /// `/undo` cursoring — if the diff is empty, restoring this snapshot
1405 /// again would be a no-op, so the caller should continue scanning
1406 /// older snapshots.
1407 pub fn work_tree_matches_snapshot(&self, id: &SnapshotId) -> io::Result<bool> {
1408 let diff = run_git(
1409 &self.git_dir,
1410 &self.work_tree,
1411 &[
1412 "diff",
1413 "--quiet",
1414 "--end-of-options",
1415 id.as_str(),
1416 "--",
1417 ":/",
1418 ],
1419 )?;
1420 git_diff_matches(diff)
1421 }
1422
1423 /// Paths that differ between snapshots `from` and `to`, in git's order,
1424 /// one [`SnapshotPathChange`] each: its `status` is git's `A`/`M`/`D`/`T`
1425 /// letter and its line counts are `None` for a binary file. Paths come
1426 /// back as git stores them (`-z`), control characters included, so a
1427 /// caller that prints one must escape it. Both trees are read from the side repo; neither the
1428 /// work tree nor the index is touched. At most `limit` paths are
1429 /// returned; the flag says whether more differed.
1430 pub fn path_changes_between(
1431 &self,
1432 from: &SnapshotId,
1433 to: &SnapshotId,
1434 limit: usize,
1435 ) -> io::Result<(Vec<SnapshotPathChange>, bool)> {
1436 let run = |format: &str| -> io::Result<String> {
1437 let output = run_git(
1438 &self.git_dir,
1439 &self.work_tree,
1440 &[
1441 "diff",
1442 "--no-renames",
1443 "--no-ext-diff",
1444 "--no-textconv",
1445 format,
1446 "-z",
1447 "--end-of-options",
1448 from.as_str(),
1449 to.as_str(),
1450 ],
1451 )?;
1452 if !output.status.success() {
1453 return Err(io_other(format!(
1454 "git diff {format} failed: {}",
1455 String::from_utf8_lossy(&output.stderr).trim()
1456 )));
1457 }
1458 Ok(String::from_utf8_lossy(&output.stdout).into_owned())
1459 };
1460 // `--numstat -z`: `added\tremoved\tpath\0`, `-` for a binary side.
1461 let numstat = run("--numstat")?;
1462 let mut counts: HashMap<String, (Option<u64>, Option<u64>)> = HashMap::new();
1463 for record in numstat.split('\0').filter(|record| !record.is_empty()) {
1464 let mut fields = record.splitn(3, '\t');
1465 let (Some(added), Some(removed), Some(path)) =
1466 (fields.next(), fields.next(), fields.next())
1467 else {
1468 continue;
1469 };
1470 counts.insert(path.to_string(), (added.parse().ok(), removed.parse().ok()));
1471 }
1472 // `--name-status -z`: `status\0path\0` pairs.
1473 let name_status = run("--name-status")?;
1474 let mut fields = name_status.split('\0').filter(|field| !field.is_empty());
1475 let mut changes = Vec::new();
1476 let mut truncated = false;
1477 while let (Some(status), Some(path)) = (fields.next(), fields.next()) {
1478 if changes.len() == limit {
1479 truncated = true;
1480 break;
1481 }
1482 let (added, removed) = counts.get(path).copied().unwrap_or((None, None));
1483 changes.push(SnapshotPathChange {
1484 path: path.to_string(),
1485 status: status.chars().next().unwrap_or('M'),
1486 added,
1487 removed,
1488 });
1489 }
1490 Ok((changes, truncated))
1491 }
1492
1493 fn tree_paths(&self, treeish: &str) -> io::Result<HashSet<PathBuf>> {
1494 let ls = run_git(
1495 &self.git_dir,
1496 &self.work_tree,
1497 &[
1498 "ls-tree",
1499 "-r",
1500 "-z",
1501 "--name-only",
1502 "--end-of-options",
1503 treeish,
1504 ],
1505 )?;
1506 if !ls.status.success() {
1507 return Err(io_other(format!(
1508 "git ls-tree failed: {}",
1509 String::from_utf8_lossy(&ls.stderr).trim()
1510 )));
1511 }
1512 Ok(parse_nul_paths(&ls.stdout))
1513 }
1514
1515 fn remove_paths_missing_from_target(
1516 &self,
1517 current_paths: &HashSet<PathBuf>,
1518 target_paths: &HashSet<PathBuf>,
1519 ) -> io::Result<()> {
1520 let removals: Vec<&PathBuf> = current_paths
1521 .difference(target_paths)
1522 .filter(|rel| is_safe_relative_path(rel))
1523 .collect();
1524 // The removal list comes from the side repo, not the live tree. A
1525 // directory may have been replaced by a symlink since the backup, and
1526 // `remove_file` would follow it and delete outside the workspace.
1527 // Refuse the whole removal before deleting anything.
1528 for rel in &removals {
1529 self.removal_parent_is_real(rel)?;
1530 }
1531 for rel in removals {
1532 // Checked again next to the delete: the tree is live.
1533 if !self.removal_parent_is_real(rel)? {
1534 continue;
1535 }
1536 let path = self.work_tree.join(rel);
1537 let metadata = match std::fs::symlink_metadata(&path) {
1538 Ok(metadata) => metadata,
1539 Err(error) if error.kind() == io::ErrorKind::NotFound => continue,
1540 Err(error) => return Err(error),
1541 };
1542 if metadata.file_type().is_dir() {
1543 // A file-to-directory transition can make this path a
1544 // required parent of files just restored from the target.
1545 if target_paths.iter().any(|target| target.starts_with(rel)) {
1546 continue;
1547 }
1548 std::fs::remove_dir(&path)?;
1549 } else {
1550 std::fs::remove_file(&path)?;
1551 }
1552 self.prune_empty_parent_dirs(path.parent());
1553 }
1554 Ok(())
1555 }
1556
1557 /// Whether every directory between the work tree and `rel` is a real
1558 /// directory. `Ok(false)`: one is missing, so `rel` is gone too. An
1559 /// ancestor that is a symlink or a file is an `InvalidInput` refusal.
1560 fn removal_parent_is_real(&self, rel: &Path) -> io::Result<bool> {
1561 let mut dir = self.work_tree.clone();
1562 for part in rel.parent().into_iter().flat_map(Path::components) {
1563 dir.push(part);
1564 match std::fs::symlink_metadata(&dir) {
1565 Ok(meta) if meta.file_type().is_dir() => {}
1566 Ok(_) => {
1567 return Err(io::Error::new(
1568 io::ErrorKind::InvalidInput,
1569 format!(
1570 "restore refuses to remove '{}': '{}' is no longer a directory (a symlink could lead outside the workspace)",
1571 rel.display(),
1572 dir.strip_prefix(&self.work_tree).unwrap_or(&dir).display()
1573 ),
1574 ));
1575 }
1576 Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false),
1577 Err(error) => return Err(error),
1578 }
1579 }
1580 Ok(true)
1581 }
1582
1583 fn prune_empty_parent_dirs(&self, mut dir: Option<&Path>) {
1584 while let Some(path) = dir {
1585 if path == self.work_tree {
1586 break;
1587 }
1588 if std::fs::remove_dir(path).is_err() {
1589 break;
1590 }
1591 dir = path.parent();
1592 }
1593 }
1594
1595 /// List up to `limit` most-recent snapshots, newest first.
1596 pub fn list(&self, limit: usize) -> io::Result<Vec<Snapshot>> {
1597 // `git log -<n>` is the short form of `--max-count=<n>`; if `limit`
1598 // is `usize::MAX` (caller asked for "everything") we pass an empty
1599 // count so git defaults to no upper bound.
1600 let mut args: Vec<String> = vec!["log".to_string()];
1601 if limit < usize::MAX {
1602 args.push(format!("--max-count={limit}"));
1603 }
1604 args.push("--pretty=format:%H%x09%T%x09%at%x09%s".to_string());
1605 args.push("--no-color".to_string());
1606 let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect();
1607 let log = run_git(&self.git_dir, &self.work_tree, &arg_refs)?;
1608 if !log.status.success() {
1609 let head = run_git(
1610 &self.git_dir,
1611 &self.work_tree,
1612 &["symbolic-ref", "-q", "HEAD"],
1613 )?;
1614 if head.status.success() {
1615 let reference = String::from_utf8_lossy(&head.stdout);
1616 let exists = run_git(
1617 &self.git_dir,
1618 &self.work_tree,
1619 &["show-ref", "--verify", "--quiet", reference.trim()],
1620 )?;
1621 if exists.status.code() == Some(1) {
1622 return Ok(Vec::new());
1623 }
1624 }
1625 return Err(io_other(format!(
1626 "git log failed: {}",
1627 String::from_utf8_lossy(&log.stderr).trim()
1628 )));
1629 }
1630 let stdout = String::from_utf8_lossy(&log.stdout);
1631 let mut out = Vec::new();
1632 for line in stdout.lines() {
1633 let mut parts = line.splitn(4, '\t');
1634 let sha = parts.next().unwrap_or("").to_string();
1635 let tree = parts.next().unwrap_or("").to_string();
1636 let ts = parts
1637 .next()
1638 .and_then(|s| s.parse::<i64>().ok())
1639 .unwrap_or(0);
1640 let subject = parts.next().unwrap_or("").to_string();
1641 // `git log --pretty=format:%H` only emits full hex ids; skip anything
1642 // else rather than let it become a revision argument later.
1643 let (Ok(id), Ok(tree)) = (SnapshotId::parse(&sha), SnapshotId::parse(&tree)) else {
1644 continue;
1645 };
1646 let (session_id, label) = Self::decode_session_label(&subject);
1647 out.push(Snapshot {
1648 id,
1649 tree,
1650 label,
1651 timestamp: ts,
1652 session_id,
1653 });
1654 }
1655 Ok(out)
1656 }
1657
1658 /// Drop snapshots older than `max_age`, returning the count removed.
1659 ///
1660 /// Strategy: identify keepable commits (younger than the cutoff),
1661 /// reset HEAD to the oldest survivor, then `git reflog expire` +
1662 /// `git gc --prune=now` to actually reclaim space. Cheap and avoids
1663 /// rewriting history when nothing has aged out.
1664 pub fn prune_older_than(&self, max_age: Duration) -> io::Result<usize> {
1665 self.with_write_lock(|| self.prune_older_than_locked(max_age))
1666 }
1667
1668 fn prune_older_than_locked(&self, max_age: Duration) -> io::Result<usize> {
1669 let now = SystemTime::now()
1670 .duration_since(UNIX_EPOCH)
1671 .map_err(|e| io_other(format!("clock error: {e}")))?
1672 .as_secs() as i64;
1673 let cutoff = now - max_age.as_secs() as i64;
1674
1675 let snapshots = self.list(usize::MAX)?;
1676 if snapshots.is_empty() {
1677 return Ok(0);
1678 }
1679
1680 // Snapshots are newest-first. Find the index of the first one
1681 // at-or-older than the cutoff — every entry from that index
1682 // onward is a candidate for removal. We use `<=` so a 0-second
1683 // retention drops same-second commits (otherwise tests calling
1684 // `prune_older_than(Duration::ZERO)` immediately after creating
1685 // a snapshot would never prune anything).
1686 let cut_index = snapshots.iter().position(|s| s.timestamp <= cutoff);
1687 let Some(cut) = cut_index else {
1688 return Ok(0);
1689 };
1690 let removed = snapshots.len() - cut;
1691 if removed == 0 {
1692 return Ok(0);
1693 }
1694
1695 if cut == 0 {
1696 // Every snapshot is older than the cutoff: unset HEAD so the next
1697 // snapshot starts a fresh history and gc reclaims the old one.
1698 self.move_head(None, Some(snapshots[0].id.as_str()))?;
1699 } else {
1700 // Keep the newest `cut` snapshots (indices [0..cut], newest-first)
1701 // and drop the older tail. This MUST rebuild the survivors as a
1702 // fresh orphan chain, not `update-ref HEAD <oldest survivor>`:
1703 // the snapshots are a parent-linked commit chain with the newest
1704 // at HEAD, so pointing HEAD at the oldest survivor orphaned every
1705 // NEWER snapshot (gc then destroyed them) while keeping the very
1706 // snapshots we meant to remove as its ancestors — the exact
1707 // inverse of the intent (2026-08-04 review, reproduced).
1708 self.rebuild_survivor_chain(&snapshots[..cut], &snapshots[0].id)?;
1709 }
1710
1711 self.reclaim_unreachable();
1712 Ok(removed)
1713 }
1714
1715 /// Expire the reflog and gc every unreachable object now, reclaiming the
1716 /// space of the snapshots a prune dropped. Runs only under the snapshot
1717 /// write lock: an immediate prune is safe because no other writer that
1718 /// takes the lock can hold a new, not yet referenced object meanwhile.
1719 fn reclaim_unreachable(&self) {
1720 let _ = run_git(
1721 &self.git_dir,
1722 &self.work_tree,
1723 &["reflog", "expire", "--expire=now", "--all"],
1724 );
1725 let _ = run_git(
1726 &self.git_dir,
1727 &self.work_tree,
1728 &["gc", "--prune=now", "--quiet"],
1729 );
1730 }
1731
1732 /// Run `op` holding the side repo's cross-process write lock (see
1733 /// [`SNAPSHOT_LOCK_FILE`]). Reentrant on one thread, so a locked
1734 /// operation may call another.
1735 fn with_write_lock<T>(&self, op: impl FnOnce() -> io::Result<T>) -> io::Result<T> {
1736 self.with_write_lock_within(SNAPSHOT_LOCK_WAIT, op)
1737 }
1738
1739 /// [`Self::with_write_lock`] with an explicit bound on the wait. The
1740 /// snapshot runs on the turn and tool path, so a peer's long gc (or a
1741 /// stopped process holding the lock) must fail this snapshot, not stall
1742 /// the turn and pin a blocking thread indefinitely.
1743 fn with_write_lock_within<T>(
1744 &self,
1745 wait: Duration,
1746 op: impl FnOnce() -> io::Result<T>,
1747 ) -> io::Result<T> {
1748 let held = HELD_SNAPSHOT_LOCKS.with(|held| held.borrow().contains(&self.git_dir));
1749 if held {
1750 return op();
1751 }
1752 let file = open_snapshot_lock_file(&self.git_dir.join(SNAPSHOT_LOCK_FILE))?;
1753 let mut lock = fd_lock::RwLock::new(file);
1754 let deadline = std::time::Instant::now() + wait;
1755 let _guard = loop {
1756 match lock.try_write() {
1757 Ok(guard) => break guard,
1758 Err(err) if std::time::Instant::now() >= deadline => {
1759 return Err(io::Error::new(
1760 io::ErrorKind::TimedOut,
1761 format!(
1762 "snapshot side repo is busy: another session held its lock for over {}s ({err})",
1763 wait.as_secs()
1764 ),
1765 ));
1766 }
1767 Err(_) => std::thread::sleep(SNAPSHOT_LOCK_POLL),
1768 }
1769 };
1770 // Declared after the guard, so it is dropped (and the thread's claim
1771 // released) before the file lock is.
1772 let _held = HeldSnapshotLock::claim(&self.git_dir);
1773 op()
1774 }
1775
1776 /// Stage every tracked and untracked path the workspace exposes.
1777 /// `--all` means `add` + `update` + `remove`, the set `git status` shows.
1778 fn stage_work_tree(&self) -> io::Result<()> {
1779 let add = run_git(&self.git_dir, &self.work_tree, &["add", "-A"])?;
1780 if !add.status.success() {
1781 return Err(io_other(format!(
1782 "git add -A failed: {}",
1783 String::from_utf8_lossy(&add.stderr).trim()
1784 )));
1785 }
1786 Ok(())
1787 }
1788
1789 /// Rebuild `survivors` (newest-first) as a fresh orphan commit chain and
1790 /// point HEAD at its tip, so every snapshot NOT in `survivors` becomes
1791 /// unreachable for gc to reclaim. Each survivor's tree, label, session
1792 /// id, and author/committer timestamp are preserved, so ages do not lie
1793 /// after a prune (finding: `prune_keep_last_n` previously reset them to
1794 /// "now"). Assumes `survivors` is non-empty.
1795 ///
1796 /// `listed_head` is the HEAD the survivors were chosen from; if another
1797 /// writer moved HEAD since, the rebuild fails rather than orphan that
1798 /// writer's snapshot.
1799 fn rebuild_survivor_chain(
1800 &self,
1801 survivors: &[Snapshot],
1802 listed_head: &SnapshotId,
1803 ) -> io::Result<()> {
1804 let mut prev_sha: Option<String> = None;
1805 for s in survivors.iter().rev() {
1806 let tree = run_git(
1807 &self.git_dir,
1808 &self.work_tree,
1809 &["rev-parse", &format!("{}^{{tree}}", s.id.as_str())],
1810 )?;
1811 if !tree.status.success() {
1812 return Err(io_other(format!(
1813 "rev-parse {}^{{tree}} failed: {}",
1814 s.id.as_str(),
1815 String::from_utf8_lossy(&tree.stderr).trim()
1816 )));
1817 }
1818 let tree_hash = String::from_utf8_lossy(&tree.stdout).trim().to_string();
1819
1820 let mut args = vec![
1821 "commit-tree".to_string(),
1822 "-m".to_string(),
1823 Self::encode_session_label(&s.label, s.session_id.as_deref()),
1824 tree_hash,
1825 ];
1826 if let Some(ref p) = prev_sha {
1827 args.push("-p".to_string());
1828 args.push(p.clone());
1829 }
1830 let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect();
1831 let new_sha = self.commit_tree_preserving_date(&arg_refs, s.timestamp)?;
1832 prev_sha = Some(new_sha);
1833 }
1834
1835 if let Some(final_sha) = prev_sha {
1836 self.move_head(Some(&final_sha), Some(listed_head.as_str()))?;
1837 }
1838 Ok(())
1839 }
1840
1841 /// Point HEAD at `new` (or delete it when `None`), only if it still names
1842 /// `expected` (`None`: HEAD must not exist yet). The compare-and-swap
1843 /// means a writer that does not take the snapshot lock, such as an older
1844 /// build, is never silently orphaned: the move fails instead.
1845 fn move_head(&self, new: Option<&str>, expected: Option<&str>) -> io::Result<()> {
1846 let expected = expected.unwrap_or("");
1847 let args = match new {
1848 Some(new) => ["update-ref", "HEAD", new, expected],
1849 None => ["update-ref", "-d", "HEAD", expected],
1850 };
1851 let update = run_git(&self.git_dir, &self.work_tree, &args)?;
1852 if !update.status.success() {
1853 return Err(io_other(format!(
1854 "git update-ref HEAD failed: {}",
1855 String::from_utf8_lossy(&update.stderr).trim()
1856 )));
1857 }
1858 Ok(())
1859 }
1860
1861 /// Run a `commit-tree` invocation with the author/committer dates pinned
1862 /// to `timestamp` (Unix seconds), so a rebuilt survivor keeps its real
1863 /// age instead of stamping "now".
1864 fn commit_tree_preserving_date(&self, args: &[&str], timestamp: i64) -> io::Result<String> {
1865 let date = format!("{timestamp} +0000");
1866 let mut command = crate::dependencies::Git::command()
1867 .ok_or_else(|| io::Error::new(io::ErrorKind::NotFound, "git not found on PATH"))?;
1868 command
1869 .arg("--git-dir")
1870 .arg(&self.git_dir)
1871 .arg("--work-tree")
1872 .arg(&self.work_tree)
1873 .env("GIT_AUTHOR_DATE", &date)
1874 .env("GIT_COMMITTER_DATE", &date)
1875 .args(args);
1876 let out = run_bounded_git(&mut command, args.first().copied().unwrap_or("git"))?;
1877 if !out.status.success() {
1878 return Err(io_other(format!(
1879 "commit-tree failed: {}",
1880 String::from_utf8_lossy(&out.stderr).trim()
1881 )));
1882 }
1883 Ok(String::from_utf8_lossy(&out.stdout).trim().to_string())
1884 }
1885
1886 /// Prune by count: keep the newest `max_count` snapshots, plus the newest
1887 /// `max_count` turn boundaries (`pre-turn:` / `post-turn:`), and drop the
1888 /// rest.
1889 ///
1890 /// Turn boundaries are the restore points turn-scoped undo resolves, and
1891 /// every other kind (`tool:`, `post-tool:`, `pre-restore:`) can arrive in
1892 /// bursts: one turn with more file-modifying tool calls than `max_count`,
1893 /// or several threads sharing the workspace, would otherwise push out the
1894 /// running turn's own `pre-turn:` snapshot before its `post-turn:` one is
1895 /// taken. So the boundaries are retained by their own count, and at most
1896 /// `2 * max_count` snapshots survive.
1897 ///
1898 /// The survivors are rebuilt as a fresh orphan chain; each keeps its
1899 /// tree, label, session id and timestamp, and the dropped ones become
1900 /// unreachable for gc to reclaim.
1901 #[cfg(test)]
1902 pub fn prune_keep_last_n(&self, max_count: usize) -> io::Result<usize> {
1903 self.with_write_lock(|| self.prune_keep_last_n_locked(max_count, 1))
1904 }
1905
1906 /// [`Self::prune_keep_last_n`] for the prune that follows every snapshot:
1907 /// it waits until half a window of snapshots is due to go and drops them
1908 /// together. Rebuilding the survivor chain costs two git processes per
1909 /// survivor, and doing that after every snapshot past the cap put seconds
1910 /// in front of every later turn's provider request. The store then holds
1911 /// at most half a window more than [`Self::prune_keep_last_n`] would keep.
1912 pub fn prune_keep_last_n_batched(&self, max_count: usize) -> io::Result<usize> {
1913 self.with_write_lock(|| self.prune_keep_last_n_locked(max_count, (max_count / 2).max(1)))
1914 }
1915
1916 fn prune_keep_last_n_locked(&self, max_count: usize, min_removed: usize) -> io::Result<usize> {
1917 let snapshots = self.list(usize::MAX)?;
1918 if snapshots.len() <= max_count {
1919 return Ok(0);
1920 }
1921 // Newest first: keep the first `max_count` of every kind, and the
1922 // first `max_count` turn boundaries wherever they sit.
1923 let mut boundaries_kept = 0usize;
1924 let survivors: Vec<Snapshot> = snapshots
1925 .iter()
1926 .enumerate()
1927 .filter(|(index, snapshot)| {
1928 let boundary = is_turn_boundary_label(&snapshot.label);
1929 let keep = *index < max_count || (boundary && boundaries_kept < max_count);
1930 if keep && boundary {
1931 boundaries_kept += 1;
1932 }
1933 keep
1934 })
1935 .map(|(_, snapshot)| snapshot.clone())
1936 .collect();
1937 let removed = snapshots.len() - survivors.len();
1938 if removed < min_removed || survivors.is_empty() {
1939 return Ok(0);
1940 }
1941 self.rebuild_survivor_chain(&survivors, &snapshots[0].id)?;
1942 self.reclaim_unreachable();
1943 Ok(removed)
1944 }
1945
1946 /// Whether a snapshot of the workspace would leave `rel` out: it is
1947 /// excluded by the workspace's `.gitignore` files or the built-in
1948 /// snapshot exclusions and not already tracked. Such a path is never in
1949 /// any snapshot, so no restore can put it back.
1950 pub fn path_is_excluded(&self, rel: &Path) -> io::Result<bool> {
1951 let rel_str = rel
1952 .to_str()
1953 .ok_or_else(|| io_other("snapshot path must be UTF-8"))?;
1954 let out = run_git(
1955 &self.git_dir,
1956 &self.work_tree,
1957 &["check-ignore", "--quiet", "--", rel_str],
1958 )?;
1959 match out.status.code() {
1960 Some(0) => Ok(true),
1961 Some(1) => Ok(false),
1962 _ => Err(io_other(format!(
1963 "git check-ignore failed: {}",
1964 String::from_utf8_lossy(&out.stderr).trim()
1965 ))),
1966 }
1967 }
1968
1969 /// Drop unreachable loose objects left behind by interrupted or
1970 /// orphaned side-repo operations.
1971 pub fn prune_unreachable_objects(&self) -> io::Result<()> {
1972 self.with_write_lock(|| {
1973 let prune = run_git(&self.git_dir, &self.work_tree, &["prune", "--expire=now"])?;
1974 if !prune.status.success() {
1975 return Err(io_other(format!(
1976 "git prune failed: {}",
1977 String::from_utf8_lossy(&prune.stderr).trim()
1978 )));
1979 }
1980 Ok(())
1981 })
1982 }
1983
1984 /// Return the side-repo's `.git` directory.
1985 pub fn git_dir(&self) -> &Path {
1986 &self.git_dir
1987 }
1988
1989 /// Return the work tree path.
1990 pub fn work_tree(&self) -> &Path {
1991 &self.work_tree
1992 }
1993 }
1994
1995 /// Whether `label` marks a turn boundary (`pre-turn:` / `post-turn:`), the
1996 /// restore points [`SnapshotRepo::prune_keep_last_n`] retains by their own
1997 /// count.
1998 fn is_turn_boundary_label(label: &str) -> bool {
1999 label.starts_with("pre-turn:") || label.starts_with("post-turn:")
2000 }
2001
2002 /// Which snapshots a size-pressure prune keeps (newest first): the newest
2003 /// `keep`, plus the newest snapshot and, for every session, its newest
2004 /// `pre-turn:` and `post-turn:` boundaries wherever they sit. Sessions and
2005 /// sub-agents share the side repo, so protecting only the globally newest
2006 /// boundaries let one session's snapshots push out another's running turn;
2007 /// each session's current turn and the one before it stay restorable
2008 /// however hard the prune has to cut.
2009 fn size_pressure_survivors(snapshots: &[Snapshot], keep: usize) -> Vec<Snapshot> {
2010 let mut boundaries_seen: HashSet<(Option<&str>, bool)> = HashSet::new();
2011 snapshots
2012 .iter()
2013 .enumerate()
2014 .filter(|(index, snapshot)| {
2015 let kind = if snapshot.label.starts_with("pre-turn:") {
2016 Some(true)
2017 } else if snapshot.label.starts_with("post-turn:") {
2018 Some(false)
2019 } else {
2020 None
2021 };
2022 let newest_boundary = kind
2023 .is_some_and(|pre| boundaries_seen.insert((snapshot.session_id.as_deref(), pre)));
2024 *index == 0 || *index < keep || newest_boundary
2025 })
2026 .map(|(_, snapshot)| snapshot.clone())
2027 .collect()
2028 }
2029
2030 /// This thread's claim on a side repo's write lock, released on drop.
2031 struct HeldSnapshotLock(PathBuf);
2032
2033 impl HeldSnapshotLock {
2034 fn claim(git_dir: &Path) -> Self {
2035 HELD_SNAPSHOT_LOCKS.with(|held| held.borrow_mut().push(git_dir.to_path_buf()));
2036 Self(git_dir.to_path_buf())
2037 }
2038 }
2039
2040 impl Drop for HeldSnapshotLock {
2041 fn drop(&mut self) {
2042 HELD_SNAPSHOT_LOCKS.with(|held| {
2043 let mut held = held.borrow_mut();
2044 if let Some(index) = held.iter().rposition(|path| path == &self.0) {
2045 held.remove(index);
2046 }
2047 });
2048 }
2049 }
2050
2051 /// Open (creating if needed) the side repo's lock file without following a
2052 /// symlink planted in its place.
2053 fn open_snapshot_lock_file(path: &Path) -> io::Result<std::fs::File> {
2054 let mut options = std::fs::OpenOptions::new();
2055 options.create(true).truncate(false).read(true).write(true);
2056 #[cfg(unix)]
2057 {
2058 use std::os::unix::fs::OpenOptionsExt;
2059 options
2060 .mode(0o600)
2061 .custom_flags(libc::O_NOFOLLOW | libc::O_CLOEXEC);
2062 }
2063 options.open(path)
2064 }
2065
2066 /// Keep the side repo's `info/exclude` at [`BUILTIN_EXCLUDES`]. Every open
2067 /// runs this without the write lock, while another session may be inside
2068 /// `git add -A` reading the file, so it is left alone when already current
2069 /// and otherwise replaced by rename: a truncate-then-write let that reader
2070 /// see an empty exclude list and stage `node_modules/`, `target/` and the
2071 /// like into the shared side repo.
2072 fn write_builtin_excludes(git_dir: &Path) -> io::Result<()> {
2073 let info_dir = git_dir.join("info");
2074 let exclude = info_dir.join("exclude");
2075 if std::fs::read(&exclude).is_ok_and(|current| current == BUILTIN_EXCLUDES.as_bytes()) {
2076 return Ok(());
2077 }
2078 std::fs::create_dir_all(&info_dir)?;
2079 crate::utils::write_atomic(&exclude, BUILTIN_EXCLUDES.as_bytes())
2080 }
2081
2082 /// Recursively compute the total size of a directory in bytes.
2083 fn dir_size_bytes(root: &Path) -> io::Result<u64> {
2084 fn walk(dir: &Path, total: &mut u64) -> io::Result<()> {
2085 if !dir.is_dir() {
2086 return Ok(());
2087 }
2088 for entry in std::fs::read_dir(dir)? {
2089 let entry = entry?;
2090 let path = entry.path();
2091 let ft = entry.file_type()?;
2092 if ft.is_symlink() {
2093 continue;
2094 }
2095 if ft.is_dir() {
2096 walk(&path, total)?;
2097 } else if ft.is_file() {
2098 *total = total.saturating_add(entry.metadata().map(|m| m.len()).unwrap_or(0));
2099 }
2100 }
2101 Ok(())
2102 }
2103 let mut total: u64 = 0;
2104 walk(root, &mut total)?;
2105 Ok(total)
2106 }
2107
2108 /// One prominent notice per workspace per process when the size-pressure
2109 /// prune destroys restore points — silent loss of undo history is the S5
2110 /// failure mode (2026-08-04 snapshot hunt). The stderr print is deliberate:
2111 /// headless/CLI stderr is the user surface for once-per-workspace snapshot
2112 /// warnings, matching `maybe_notify_snapshots_disabled_once` in
2113 /// `core/turn.rs`.
2114 #[allow(clippy::print_stderr)]
2115 fn notify_snapshot_history_pruned_once(workspace: &Path, removed: usize) {
2116 use std::collections::HashSet;
2117 use std::sync::{Mutex, OnceLock};
2118 static NOTIFIED: OnceLock<Mutex<HashSet<String>>> = OnceLock::new();
2119 let key = workspace.to_string_lossy().into_owned();
2120 let set = NOTIFIED.get_or_init(|| Mutex::new(HashSet::new()));
2121 let Ok(mut guard) = set.lock() else {
2122 return;
2123 };
2124 if !guard.insert(key) {
2125 return;
2126 }
2127 drop(guard);
2128 eprint!("{}", snapshot_history_pruned_message(workspace, removed));
2129 }
2130
2131 /// Build the user-visible notice for a size-pressure prune. Kept pure and
2132 /// separate from the emit/dedup shell so the content is unit-testable.
2133 fn snapshot_history_pruned_message(workspace: &Path, removed: usize) -> String {
2134 format!(
2135 "warning: snapshot/undo history for {} was pruned to stay under the {} MB snapshot storage cap.
2136 {} snapshot(s) were removed and can no longer be restored.
2137 The cap bounds the undo side-repo's disk use; high-churn or large workspaces hit it sooner.
2138 ",
2139 workspace.display(),
2140 MAX_SNAPSHOT_SIZE_MB,
2141 removed
2142 )
2143 }
2144
2145 fn cleanup_stale_pack_temps(git_dir: &Path, stale_age: Duration) -> io::Result<usize> {
2146 let pack_dir = git_dir.join("objects").join("pack");
2147 if !pack_dir.exists() {
2148 return Ok(0);
2149 }
2150 cleanup_stale_pack_temps_in(&pack_dir, stale_age, SystemTime::now())
2151 }
2152
2153 fn cleanup_stale_pack_temps_in(
2154 pack_dir: &Path,
2155 stale_age: Duration,
2156 now: SystemTime,
2157 ) -> io::Result<usize> {
2158 let mut removed = 0;
2159 for entry in std::fs::read_dir(pack_dir)? {
2160 let entry = entry?;
2161 let name = entry.file_name();
2162 let Some(name) = name.to_str() else {
2163 continue;
2164 };
2165 if !name.starts_with("tmp_pack_") {
2166 continue;
2167 }
2168 if !entry.file_type()?.is_file() {
2169 continue;
2170 }
2171
2172 let metadata = entry.metadata()?;
2173 let Ok(modified) = metadata.modified() else {
2174 continue;
2175 };
2176 let Ok(age) = now.duration_since(modified) else {
2177 continue;
2178 };
2179 if age < stale_age {
2180 continue;
2181 }
2182
2183 match std::fs::remove_file(entry.path()) {
2184 Ok(()) => removed += 1,
2185 Err(err) if err.kind() == io::ErrorKind::NotFound => {}
2186 Err(err) => return Err(err),
2187 }
2188 }
2189 Ok(removed)
2190 }
2191
2192 // Generous budget: `git add -A` on a large workspace is legitimately slow,
2193 // but a wedged git (stalled NFS/FUSE, hung hook) must not block the turn
2194 // pipeline forever — every caller treats a snapshot error as
2195 // snapshot-disabled-with-warning and proceeds without the git data. Tests
2196 // use a tighter budget so a regression that deadlocks a child on its own
2197 // output fails in seconds instead of hanging for the full window.
2198 #[cfg(not(test))]
2199 const GIT_COMMAND_TIMEOUT: Duration = Duration::from_secs(300);
2200 #[cfg(test)]
2201 const GIT_COMMAND_TIMEOUT: Duration = Duration::from_secs(30);
2202
2203 /// Grace granted to the pipe readers after git has exited. A clean git
2204 /// closes its own write ends, so EOF is already waiting; only a grandchild
2205 /// that inherited the pipes (a post-checkout hook, a `git gc` pack worker)
2206 /// can hold them past exit, and it must not hold the turn pipeline either.
2207 const GIT_PIPE_DRAIN_GRACE: Duration = Duration::from_secs(2);
2208
2209 /// Run a pre-configured git command under [`GIT_COMMAND_TIMEOUT`] with both
2210 /// pipes drained while the child runs (the same concurrent drain
2211 /// `Command::output` performs). Every git invocation in this module goes
2212 /// through here, so a wedged git — stalled NFS/FUSE, hung hook — degrades
2213 /// the snapshot with an error instead of hanging the turn pipeline.
2214 /// (`delta.rs`'s read-view git commands are not routed here yet; that is a
2215 /// known follow-up.)
2216 ///
2217 /// The timeout path kills the child and reaps it on a detached thread: on a
2218 /// hard-wedged mount git can sit in uninterruptible kernel I/O where even
2219 /// SIGKILL is deferred, and a blocking `wait()` would hang the pipeline
2220 /// exactly like the wedged git would. Killing without git's own cleanup can
2221 /// leave a fresh `index.lock` behind; later snapshots then fail fast on the
2222 /// lock with an error naming it.
2223 fn run_bounded_git(cmd: &mut std::process::Command, subcommand: &str) -> io::Result<Output> {
2224 run_bounded_git_with_timeout(cmd, subcommand, GIT_COMMAND_TIMEOUT)
2225 }
2226
2227 fn run_bounded_git_with_timeout(
2228 cmd: &mut std::process::Command,
2229 subcommand: &str,
2230 timeout: Duration,
2231 ) -> io::Result<Output> {
2232 cmd.stdin(std::process::Stdio::null())
2233 .stdout(std::process::Stdio::piped())
2234 .stderr(std::process::Stdio::piped());
2235 let mut child = cmd.spawn()?;
2236 // Drain both pipes while waiting: the restore path's `ls-tree -r` and
2237 // the diff commands emit output that grows with workspace size, and a
2238 // child blocked on a full pipe buffer never exits — it would turn every
2239 // such call into a guaranteed timeout. Readers stream into shared
2240 // buffers, and the completion channel lets the collect below bound how
2241 // long a grandchild-held pipe may hold the call after git has exited.
2242 let (done_tx, done_rx) = std::sync::mpsc::channel::<()>();
2243 let stdout_pipe = child.stdout.take();
2244 let stderr_pipe = child.stderr.take();
2245 let stdout_buf = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
2246 let stderr_buf = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
2247 {
2248 let buf = std::sync::Arc::clone(&stdout_buf);
2249 let done_tx = done_tx.clone();
2250 std::thread::spawn(move || {
2251 if let Some(mut reader) = stdout_pipe {
2252 read_pipe_to_buffer(&mut reader, &buf);
2253 }
2254 let _ = done_tx.send(());
2255 });
2256 }
2257 {
2258 let buf = std::sync::Arc::clone(&stderr_buf);
2259 let done_tx = done_tx.clone();
2260 std::thread::spawn(move || {
2261 if let Some(mut reader) = stderr_pipe {
2262 read_pipe_to_buffer(&mut reader, &buf);
2263 }
2264 let _ = done_tx.send(());
2265 });
2266 }
2267 drop(done_tx);
2268
2269 let Some(status) = child.wait_timeout(timeout)? else {
2270 let _ = child.kill();
2271 // Reap off the pipeline thread (see the doc comment): the kernel may
2272 // not deliver the kill until an uninterruptible syscall returns.
2273 std::thread::spawn(move || {
2274 let _ = child.wait();
2275 });
2276 return Err(io::Error::new(
2277 io::ErrorKind::TimedOut,
2278 format!("git {subcommand} timed out after {}s", timeout.as_secs()),
2279 ));
2280 };
2281 // Wait for the readers up to the grace, then return what was captured. A
2282 // clean git already closed its write ends, so both readers deliver
2283 // immediately; the grace only covers a pipe still held open by an
2284 // inherited copy (a daemonizing hook). On expiry the captured output is
2285 // returned with a note on stderr instead of silently truncating; the
2286 // reader itself ends whenever whatever holds the pipe exits, holding
2287 // only a pipe read — never the turn pipeline.
2288 let mut partial = false;
2289 let mut drained = 0;
2290 while drained < 2 {
2291 match done_rx.recv_timeout(GIT_PIPE_DRAIN_GRACE) {
2292 Ok(()) => drained += 1,
2293 Err(_) => {
2294 partial = true;
2295 break;
2296 }
2297 }
2298 }
2299 let stdout = stdout_buf.lock().map(|buf| buf.clone()).unwrap_or_default();
2300 let mut stderr = stderr_buf.lock().map(|buf| buf.clone()).unwrap_or_default();
2301 if partial {
2302 if !stderr.is_empty() && stderr.last() != Some(&b'\n') {
2303 stderr.push(b'\n');
2304 }
2305 stderr.extend_from_slice(
2306 b"[codewhale] git output pipes did not close after git exited \
2307 (kept open by a hook or subprocess?); captured output may be partial\n",
2308 );
2309 }
2310 Ok(Output {
2311 status,
2312 stdout,
2313 stderr,
2314 })
2315 }
2316
2317 /// Read a child's pipe to end into a shared buffer, tolerating interruption.
2318 fn read_pipe_to_buffer(reader: &mut impl io::Read, buf: &std::sync::Mutex<Vec<u8>>) {
2319 let mut chunk = [0_u8; 8192];
2320 loop {
2321 match reader.read(&mut chunk) {
2322 Ok(0) => return,
2323 Ok(read) => {
2324 if let Ok(mut buf) = buf.lock() {
2325 buf.extend_from_slice(&chunk[..read]);
2326 }
2327 }
2328 Err(err) if err.kind() == io::ErrorKind::Interrupted => continue,
2329 Err(_) => return,
2330 }
2331 }
2332 }
2333
2334 fn run_git(git_dir: &Path, work_tree: &Path, args: &[&str]) -> io::Result<Output> {
2335 let mut cmd = crate::dependencies::Git::command()
2336 .ok_or_else(|| io::Error::new(io::ErrorKind::NotFound, "git not found on PATH"))?;
2337 cmd.arg("--git-dir")
2338 .arg(git_dir)
2339 .arg("--work-tree")
2340 .arg(work_tree)
2341 .args(args);
2342 run_bounded_git(&mut cmd, args.first().copied().unwrap_or("git"))
2343 }
2344
2345 fn git_diff_matches(output: Output) -> io::Result<bool> {
2346 match output.status.code() {
2347 Some(0) => Ok(true),
2348 Some(1) => Ok(false),
2349 _ => Err(io_other(format!(
2350 "git diff failed: {}",
2351 String::from_utf8_lossy(&output.stderr).trim()
2352 ))),
2353 }
2354 }
2355
2356 fn io_other(msg: impl Into<String>) -> io::Error {
2357 io::Error::other(msg.into())
2358 }
2359
2360 /// Walk `workspace` and accumulate file sizes, returning `Ok(total)`
2361 /// when the workspace fits under `cap_bytes` and `Err(gate)` naming the
2362 /// bound that tripped. Honors `.gitignore` — whether or not the
2363 /// workspace is itself a git repo, matching the `git add -A` that the
2364 /// snapshot commit actually runs against this work tree — and the
2365 /// snapshot-specific skip list above, so the measured size reflects
2366 /// what would land in a snapshot commit rather than the raw `du -sh`
2367 /// total.
2368 ///
2369 /// The walk is bounded by both `cap_bytes` and `max_entries`, and the
2370 /// two bounds are reported separately because they have different
2371 /// recoveries. A `cap_bytes` of `0` disables the byte cap entirely (so
2372 /// config can opt out) but not the entry bound.
2373 ///
2374 /// Production passes [`SIZE_WALK_MAX_ENTRIES`] for `max_entries`; it is
2375 /// a parameter only so the entry bound is reachable in a test without
2376 /// creating 200,000 inodes. It must not be threaded up through
2377 /// [`SnapshotRepo::open_or_init_with_cap`]: [`WorkspaceGate::describe`]
2378 /// interpolates the constant into the user-facing message, so a weaker
2379 /// injected bound would report a number that did not trip.
2380 pub fn estimate_workspace_size_bounded(
2381 workspace: &Path,
2382 cap_bytes: u64,
2383 max_entries: usize,
2384 ) -> Result<u64, WorkspaceGate> {
2385 use ignore::WalkBuilder;
2386 let mut total: u64 = 0;
2387 let mut entries: usize = 0;
2388 let skip: HashSet<&'static str> = SIZE_WALK_SKIP_DIRS.iter().copied().collect();
2389 let walker = WalkBuilder::new(workspace)
2390 .hidden(false)
2391 // `ignore` defaults to `require_git(true)`, which silently disables
2392 // every gitignore rule when the workspace is not inside a git repo.
2393 // The snapshot's own `git add -A` honors `.gitignore` regardless, so
2394 // without this the estimator over-counts a non-git workspace and can
2395 // refuse it while offering a `.gitignore` remedy that cannot work.
2396 .require_git(false)
2397 .follow_links(false)
2398 .filter_entry(move |entry| {
2399 // Skip the well-known build-output directories at any depth.
2400 // The `ignore` crate calls `filter_entry` once per dir/file;
2401 // returning `false` here prunes the whole subtree.
2402 entry
2403 .file_name()
2404 .to_str()
2405 .is_none_or(|name| !skip.contains(name))
2406 })
2407 .build();
2408 for entry in walker.flatten() {
2409 entries += 1;
2410 if entries > max_entries {
2411 return Err(WorkspaceGate::TooManyEntries);
2412 }
2413 if let Ok(meta) = entry.metadata()
2414 && meta.is_file()
2415 {
2416 total = total.saturating_add(meta.len());
2417 if cap_bytes > 0 && total > cap_bytes {
2418 return Err(WorkspaceGate::TooLarge);
2419 }
2420 }
2421 }
2422 Ok(total)
2423 }
2424
2425 pub(crate) fn unsafe_workspace_snapshot_reason(
2426 workspace: &Path,
2427 home: Option<&Path>,
2428 ) -> Option<&'static str> {
2429 let workspace = normalize_path_for_safety(workspace);
2430 if is_filesystem_root(&workspace) {
2431 return Some("filesystem root");
2432 }
2433
2434 if is_home_directory(&workspace, home) {
2435 return Some("home directory");
2436 }
2437
2438 let home = home.map(normalize_path_for_safety)?;
2439 if workspace.parent() == Some(home.as_path()) {
2440 let name = workspace.file_name().and_then(|name| name.to_str());
2441 if matches!(
2442 name,
2443 Some(
2444 "Desktop" | "Documents" | "Downloads" | "Library" | "Movies" | "Music" | "Pictures"
2445 )
2446 ) {
2447 return Some("home collection directory");
2448 }
2449 }
2450
2451 None
2452 }
2453
2454 fn normalize_path_for_safety(path: &Path) -> PathBuf {
2455 path.canonicalize().unwrap_or_else(|_| path.to_path_buf())
2456 }
2457
2458 fn is_filesystem_root(path: &Path) -> bool {
2459 path.parent().is_none()
2460 }
2461
2462 fn is_home_directory(work_tree: &Path, home: Option<&Path>) -> bool {
2463 let Some(home) = home else {
2464 return false;
2465 };
2466
2467 let home_canonical = home.canonicalize().unwrap_or_else(|_| home.to_path_buf());
2468 work_tree == home_canonical
2469 }
2470
2471 fn parse_nul_paths(bytes: &[u8]) -> HashSet<PathBuf> {
2472 bytes
2473 .split(|b| *b == 0)
2474 .filter(|chunk| !chunk.is_empty())
2475 .map(|chunk| PathBuf::from(String::from_utf8_lossy(chunk).into_owned()))
2476 .collect()
2477 }
2478
2479 fn is_safe_relative_path(path: &Path) -> bool {
2480 !path.as_os_str().is_empty()
2481 && path
2482 .components()
2483 .all(|component| matches!(component, Component::Normal(_)))
2484 }
2485
2486 /// Whether one path component names the repository metadata directory as the
2487 /// filesystem resolves it, not just as spelled: `.git` in any letter case
2488 /// (macOS and Windows default to case-insensitive names) and, on Windows,
2489 /// with the trailing dots/spaces or `:stream` suffix it drops and the `GIT~N`
2490 /// short-name alias. The one `.git` rule for workspace file routes, file
2491 /// restore, displayed workspace paths, the write carve-out and sub-agent
2492 /// deliverables.
2493 pub fn is_git_metadata_name(name: &std::ffi::OsStr) -> bool {
2494 git_metadata_name(&name.to_string_lossy(), cfg!(windows))
2495 }
2496
2497 fn git_metadata_name(name: &str, windows: bool) -> bool {
2498 let name = if windows {
2499 name.split(':')
2500 .next()
2501 .unwrap_or_default()
2502 .trim_end_matches(['.', ' '])
2503 } else {
2504 name
2505 };
2506 name.eq_ignore_ascii_case(".git")
2507 || (windows
2508 && name.len() > 4
2509 && name
2510 .get(..4)
2511 .is_some_and(|prefix| prefix.eq_ignore_ascii_case("git~"))
2512 && name[4..].bytes().all(|byte| byte.is_ascii_digit()))
2513 }
2514
2515 /// Normalize a caller-supplied path into a safe workspace-relative path.
2516 ///
2517 /// Accepts either a workspace-relative path or an absolute path inside
2518 /// `workspace`. Returns `None` when the result is not a plain relative path —
2519 /// absolute, empty, containing `..`, or pointing outside the workspace. Every
2520 /// file-scoped restore path passes through here, so a caller never gets to
2521 /// name a path the snapshot repo would resolve outside the work tree.
2522 ///
2523 /// The name is literal: leading or trailing spaces, brackets and glob
2524 /// characters are filename bytes, never trimmed and never patterns. Git is
2525 /// invoked with `--literal-pathspecs` for every file-scoped operation.
2526 pub fn workspace_relative_path(workspace: &Path, raw: &str) -> Option<PathBuf> {
2527 if raw.is_empty() {
2528 return None;
2529 }
2530 let candidate = Path::new(raw);
2531 let rel = if candidate.is_absolute() {
2532 candidate.strip_prefix(workspace).ok()?.to_path_buf()
2533 } else {
2534 candidate.to_path_buf()
2535 };
2536 is_safe_relative_path(&rel).then_some(rel)
2537 }
2538
2539 /// Whether some existing ancestor of `rel` (within `root`) is not a
2540 /// directory: restoring `rel` would then turn a file into a directory.
2541 fn ancestor_is_not_a_directory(root: &Path, rel: &Path) -> bool {
2542 rel.ancestors()
2543 .skip(1)
2544 .filter(|ancestor| !ancestor.as_os_str().is_empty())
2545 .any(|ancestor| {
2546 std::fs::symlink_metadata(root.join(ancestor)).is_ok_and(|metadata| !metadata.is_dir())
2547 })
2548 }
2549
2550 #[cfg(test)]
2551 mod tests {
2552 use super::*;
2553 use crate::test_support::lock_test_env;
2554 use std::fs::{File, FileTimes};
2555 use tempfile::tempdir;
2556
2557 #[test]
2558 fn git_metadata_name_covers_case_and_windows_aliases() {
2559 for windows in [false, true] {
2560 for name in [".git", ".GIT", ".Git", ".gIt"] {
2561 assert!(git_metadata_name(name, windows), "{name} windows={windows}");
2562 }
2563 for name in [
2564 ".github",
2565 ".gitignore",
2566 "a.git",
2567 "git",
2568 "",
2569 "git~",
2570 "GIT~1a",
2571 ] {
2572 assert!(
2573 !git_metadata_name(name, windows),
2574 "{name} windows={windows}"
2575 );
2576 }
2577 }
2578 // Windows drops trailing dots/spaces and `:stream` suffixes, and
2579 // `GIT~N` is the 8.3 short name of `.git`; elsewhere these are
2580 // ordinary names.
2581 for name in [
2582 ".git.",
2583 ".git ",
2584 ".GIT. .",
2585 ".git::$INDEX_ALLOCATION",
2586 ".git:stream",
2587 "GIT~1",
2588 "git~12",
2589 ] {
2590 assert!(git_metadata_name(name, true), "{name}");
2591 assert!(!git_metadata_name(name, false), "{name}");
2592 }
2593 }
2594
2595 #[test]
2596 fn snapshot_id_parse_accepts_only_full_hex_object_ids() {
2597 let sha1 = "0123456789abcdefABCDEF0123456789abcdef01";
2598 let sha256 = "a".repeat(64);
2599 assert_eq!(SnapshotId::parse(sha1).expect("sha1").as_str(), sha1);
2600 assert!(SnapshotId::parse(&sha256).is_ok());
2601 for bad in [
2602 "",
2603 "HEAD",
2604 "abc123",
2605 "--output=/tmp/x",
2606 "-0123456789abcdef0123456789abcdef0123456",
2607 "0123456789abcdef0123456789abcdef0123456g",
2608 "0123456789abcdef0123456789abcdef01234567~1",
2609 "0123456789abcdef0123456789abcdef012345678",
2610 ] {
2611 let err = SnapshotId::parse(bad).expect_err(bad);
2612 assert_eq!(err.kind(), io::ErrorKind::InvalidInput, "{bad:?}");
2613 }
2614 }
2615
2616 /// Holds the home directory pinned to a tempdir for the lifetime of a test. Also
2617 /// owns the process-wide env-var mutex so tests across modules
2618 /// don't trample each other's home env vars.
2619 pub(super) struct ScopedHome {
2620 _vars: Vec<crate::test_support::EnvVarGuard>,
2621 _guard: crate::test_support::TestEnvLock,
2622 }
2623 pub(super) fn scoped_home(home: &Path) -> ScopedHome {
2624 use crate::test_support::EnvVarGuard;
2625 let guard = lock_test_env();
2626 ScopedHome {
2627 _vars: vec![
2628 EnvVarGuard::set("HOME", home),
2629 EnvVarGuard::set("USERPROFILE", home),
2630 EnvVarGuard::remove("HOMEDRIVE"),
2631 EnvVarGuard::remove("HOMEPATH"),
2632 EnvVarGuard::set("CODEWHALE_HOME", home.join(".codewhale")),
2633 ],
2634 _guard: guard,
2635 }
2636 }
2637
2638 /// Build a side-repo inside the test's selected profile. Return its
2639 /// environment guard so reads and writes stay isolated for the whole test.
2640 fn make_repo(tmp: &Path) -> (SnapshotRepo, ScopedHome) {
2641 let workspace = tmp.join("workspace");
2642 std::fs::create_dir_all(&workspace).unwrap();
2643 let guard = scoped_home(tmp);
2644 let repo = SnapshotRepo::open_or_init(&workspace).expect("open_or_init");
2645 (repo, guard)
2646 }
2647
2648 #[test]
2649 fn snapshot_creates_commit_in_side_repo_only() {
2650 let tmp = tempdir().unwrap();
2651 let (repo, _home) = make_repo(tmp.path());
2652 std::fs::write(repo.work_tree().join("a.txt"), b"alpha").unwrap();
2653
2654 let id = repo.snapshot("pre-turn:1").expect("snapshot");
2655 assert_eq!(id.as_str().len(), 40);
2656
2657 let list = repo.list(10).expect("list");
2658 assert_eq!(list.len(), 1);
2659 assert_eq!(list[0].label, "pre-turn:1");
2660
2661 // The user's workspace must NOT have a real `.git` because we
2662 // never created one in their workspace — only in the side dir.
2663 assert!(!repo.work_tree().join(".git").exists());
2664 }
2665
2666 /// B2: a side repo whose HEAD names a missing commit made every snapshot
2667 /// fail on `commit-tree -p` ("is not a valid object"), so /undo died
2668 /// silently. The broken ref is reset and snapshots resume.
2669 #[test]
2670 fn broken_head_is_repaired_and_snapshots_resume() {
2671 let tmp = tempdir().unwrap();
2672 let (repo, _home) = make_repo(tmp.path());
2673 std::fs::write(repo.work_tree().join("a.txt"), b"alpha").unwrap();
2674 repo.snapshot("pre-turn:1").expect("first snapshot");
2675 assert!(!repo.repair_broken_head().expect("healthy head"));
2676
2677 repo.point_head_at_missing_commit_for_test();
2678 assert!(
2679 repo.list(10).is_err() || repo.list(10).unwrap().is_empty(),
2680 "a broken head cannot list its history"
2681 );
2682
2683 // With a reflog, the last commit that still exists is restored and
2684 // the restore points before the break survive.
2685 assert!(
2686 !repo.repair_broken_head().expect("recover"),
2687 "recovered from the reflog, not restarted"
2688 );
2689 let list = repo.list(10).expect("list after recovery");
2690 assert_eq!(list.len(), 1, "{list:?}");
2691 assert_eq!(list[0].label, "pre-turn:1");
2692
2693 // Without one, the broken ref is deleted and history restarts.
2694 repo.point_head_at_missing_commit_for_test();
2695 std::fs::remove_dir_all(repo.git_dir.join("logs")).expect("drop reflogs");
2696 assert!(repo.repair_broken_head().expect("repair"), "repaired once");
2697 assert!(!repo.repair_broken_head().expect("idempotent"));
2698 std::fs::write(repo.work_tree().join("a.txt"), b"beta").unwrap();
2699 repo.snapshot("pre-turn:2").expect("snapshots resume");
2700 let list = repo.list(10).expect("list after repair");
2701 assert_eq!(list.len(), 1, "history restarts at the repair: {list:?}");
2702 assert_eq!(list[0].label, "pre-turn:2");
2703
2704 // The snapshot path repairs on its own too, for callers that never
2705 // asked.
2706 repo.point_head_at_missing_commit_for_test();
2707 repo.snapshot("pre-turn:3")
2708 .expect("snapshot repairs on its own");
2709 }
2710
2711 #[test]
2712 fn open_existing_is_read_only_and_does_not_initialize() {
2713 let tmp = tempdir().unwrap();
2714 let workspace = tmp.path().join("workspace");
2715 std::fs::create_dir_all(&workspace).unwrap();
2716 let _home = scoped_home(tmp.path());
2717
2718 let before = SnapshotRepo::open_existing(&workspace).expect("open existing");
2719 assert!(before.is_none());
2720 assert!(
2721 !snapshot_git_dir(&workspace)
2722 .expect("snapshot path")
2723 .exists(),
2724 "read-only open must not create the side repo"
2725 );
2726
2727 let repo = SnapshotRepo::open_or_init(&workspace).expect("open_or_init");
2728 std::fs::write(repo.work_tree().join("a.txt"), b"alpha").unwrap();
2729 repo.snapshot("pre-turn:1").expect("snapshot");
2730
2731 let after = SnapshotRepo::open_existing(&workspace).expect("open existing");
2732 assert!(after.is_some());
2733 }
2734
2735 #[test]
2736 fn restore_reverts_workspace_files() {
2737 let tmp = tempdir().unwrap();
2738 let (repo, _home) = make_repo(tmp.path());
2739 let f = repo.work_tree().join("file.txt");
2740
2741 std::fs::write(&f, b"original").unwrap();
2742 let id = repo.snapshot("pre-turn:1").expect("snapshot");
2743
2744 std::fs::write(&f, b"clobbered").unwrap();
2745 repo.snapshot("post-turn:1").expect("snapshot 2");
2746
2747 repo.restore(&id).expect("restore");
2748 let after = std::fs::read_to_string(&f).unwrap();
2749 assert_eq!(after, "original");
2750 }
2751
2752 #[test]
2753 fn restore_removes_files_added_after_target_snapshot() {
2754 let tmp = tempdir().unwrap();
2755 let (repo, _home) = make_repo(tmp.path());
2756 let original = repo.work_tree().join("original.txt");
2757 let added = repo.work_tree().join("added.txt");
2758
2759 std::fs::write(&original, b"original").unwrap();
2760 let id = repo.snapshot("pre-turn:1").expect("snapshot");
2761
2762 std::fs::write(&added, b"new file").unwrap();
2763 repo.snapshot("post-turn:1").expect("snapshot 2");
2764
2765 repo.restore(&id).expect("restore");
2766 assert!(original.exists());
2767 assert!(!added.exists(), "restore must remove tracked added files");
2768 }
2769
2770 /// The first snapshot of an empty project holds the empty tree. Restoring
2771 /// it used to fail (`pathspec ':/' did not match`) before removing
2772 /// anything, so every file created since stayed.
2773 #[test]
2774 fn restore_to_an_empty_snapshot_removes_the_files_created_since() {
2775 let tmp = tempdir().unwrap();
2776 let (repo, _home) = make_repo(tmp.path());
2777 let empty = repo.snapshot("pre-turn:1").expect("empty snapshot");
2778 let created = repo.work_tree().join("src").join("main.rs");
2779 std::fs::create_dir_all(created.parent().unwrap()).unwrap();
2780 std::fs::write(&created, b"fn main() {}").unwrap();
2781 std::fs::write(repo.work_tree().join("notes.txt"), b"notes").unwrap();
2782 repo.snapshot("post-turn:1").expect("snapshot 2");
2783
2784 repo.restore(&empty).expect("restore to the empty tree");
2785 assert!(!created.exists(), "restore must remove files created since");
2786 assert!(!repo.work_tree().join("notes.txt").exists());
2787 assert!(
2788 !repo.work_tree().join("src").exists(),
2789 "directories the removal emptied go too"
2790 );
2791 }
2792
2793 #[test]
2794 fn restore_keeps_current_files_when_safety_snapshot_fails() {
2795 let tmp = tempdir().unwrap();
2796 let (repo, _home) = make_repo(tmp.path());
2797 let empty = repo.snapshot("pre-turn:1").unwrap();
2798 let file = repo.work_tree().join("new-work.txt");
2799 std::fs::write(&file, b"only copy of new work").unwrap();
2800 repo.snapshot("post-turn:1").unwrap();
2801 std::fs::write(repo.git_dir().join("index.lock"), b"").unwrap();
2802
2803 let error = repo.restore(&empty).expect_err("backup is mandatory");
2804 assert!(
2805 error.to_string().contains("safety snapshot failed"),
2806 "{error}"
2807 );
2808 assert_eq!(std::fs::read(&file).unwrap(), b"only copy of new work");
2809 }
2810
2811 #[test]
2812 fn restore_rolls_back_files_after_a_partial_checkout_error() {
2813 let tmp = tempdir().unwrap();
2814 let (repo, _home) = make_repo(tmp.path());
2815 let first = repo.work_tree().join("a.txt");
2816 let last_dir = repo.work_tree().join("z");
2817 let last = last_dir.join("last.txt");
2818 std::fs::create_dir(&last_dir).unwrap();
2819 std::fs::write(&first, b"old first").unwrap();
2820 std::fs::write(&last, b"old last").unwrap();
2821 let target = repo.snapshot("pre-turn:1").unwrap();
2822 let object = run_git(
2823 repo.git_dir(),
2824 repo.work_tree(),
2825 &["rev-parse", &format!("{}:z/last.txt", target.as_str())],
2826 )
2827 .unwrap();
2828 assert!(object.status.success());
2829 let object = String::from_utf8(object.stdout).unwrap();
2830 let object = object.trim();
2831 std::fs::write(&first, b"new first").unwrap();
2832 std::fs::remove_file(&last).unwrap();
2833 let before = repo.snapshot("before-restore-proof").unwrap();
2834 // A missing target-only blob makes Git fail after it writes a.txt.
2835 // The backup needs neither that blob nor z/last.txt for recovery.
2836 std::fs::remove_file(
2837 repo.git_dir()
2838 .join("objects")
2839 .join(&object[..2])
2840 .join(&object[2..]),
2841 )
2842 .unwrap();
2843
2844 // Prove the failure fixture really permits a partial overwrite.
2845 let partial = run_git(
2846 repo.git_dir(),
2847 repo.work_tree(),
2848 &["checkout", target.as_str(), "--", ":/"],
2849 )
2850 .unwrap();
2851 assert!(!partial.status.success());
2852 assert_eq!(std::fs::read(&first).unwrap(), b"old first");
2853 let reset = run_git(
2854 repo.git_dir(),
2855 repo.work_tree(),
2856 &["checkout", before.as_str(), "--", ":/"],
2857 )
2858 .unwrap();
2859 assert!(reset.status.success());
2860 assert_eq!(std::fs::read(&first).unwrap(), b"new first");
2861
2862 let error = repo.restore(&target).expect_err("target blob is missing");
2863 assert!(error.to_string().contains("unable to read"), "{error}");
2864 assert!(
2865 error
2866 .to_string()
2867 .contains("previous snapshot files were restored"),
2868 "{error}"
2869 );
2870 assert!(error.to_string().contains("safety snapshot"), "{error}");
2871 assert_eq!(std::fs::read(&first).unwrap(), b"new first");
2872 assert!(!last.exists());
2873 }
2874
2875 #[test]
2876 fn restore_refuses_file_directory_transitions_without_changing_files() {
2877 for target_is_dir in [false, true] {
2878 let tmp = tempdir().unwrap();
2879 let (repo, _home) = make_repo(tmp.path());
2880 let path = repo.work_tree().join("a");
2881 let child = path.join("child");
2882 if target_is_dir {
2883 std::fs::create_dir(&path).unwrap();
2884 std::fs::write(&child, b"old child").unwrap();
2885 } else {
2886 std::fs::write(&path, b"old file").unwrap();
2887 }
2888 let target = repo.snapshot("pre-turn:1").unwrap();
2889 if target_is_dir {
2890 std::fs::remove_file(&child).unwrap();
2891 std::fs::remove_dir(&path).unwrap();
2892 std::fs::write(&path, b"new file").unwrap();
2893 } else {
2894 std::fs::remove_file(&path).unwrap();
2895 std::fs::create_dir(&path).unwrap();
2896 std::fs::write(&child, b"new child").unwrap();
2897 }
2898
2899 let error = repo
2900 .restore(&target)
2901 .expect_err("type transition is refused");
2902 assert!(
2903 error.to_string().contains("file/directory transition"),
2904 "{error}"
2905 );
2906 if target_is_dir {
2907 assert_eq!(std::fs::read(&path).unwrap(), b"new file");
2908 } else {
2909 assert_eq!(std::fs::read(&child).unwrap(), b"new child");
2910 }
2911 }
2912 }
2913
2914 #[test]
2915 fn a_path_under_a_live_file_needs_a_transition() {
2916 let tmp = tempdir().unwrap();
2917 std::fs::write(tmp.path().join("a"), b"file").unwrap();
2918 std::fs::create_dir(tmp.path().join("d")).unwrap();
2919 assert!(super::ancestor_is_not_a_directory(
2920 tmp.path(),
2921 Path::new("a/child")
2922 ));
2923 assert!(!super::ancestor_is_not_a_directory(
2924 tmp.path(),
2925 Path::new("d/child")
2926 ));
2927 assert!(!super::ancestor_is_not_a_directory(
2928 tmp.path(),
2929 Path::new("missing/child")
2930 ));
2931 assert!(!super::ancestor_is_not_a_directory(
2932 tmp.path(),
2933 Path::new("a")
2934 ));
2935 }
2936
2937 /// A failed safety snapshot refuses before a symlinked parent could be
2938 /// followed, even when restoring to an empty target skips checkout.
2939 #[cfg(unix)]
2940 #[test]
2941 fn restore_refuses_to_remove_through_a_symlinked_parent() {
2942 let tmp = tempdir().unwrap();
2943 let (repo, _home) = make_repo(tmp.path());
2944 let empty = repo.snapshot("pre-turn:1").expect("empty snapshot");
2945 let src = repo.work_tree().join("src");
2946 std::fs::create_dir_all(&src).unwrap();
2947 std::fs::write(src.join("victim"), b"inside").unwrap();
2948 repo.snapshot("post-turn:1")
2949 .expect("snapshot holding src/victim");
2950
2951 let outside = tempdir().unwrap();
2952 let sentinel = outside.path().join("victim");
2953 std::fs::write(&sentinel, b"outside").unwrap();
2954 std::fs::remove_dir_all(&src).unwrap();
2955 std::os::unix::fs::symlink(outside.path(), &src).unwrap();
2956 // A stale index lock makes the mandatory safety snapshot fail.
2957 std::fs::write(repo.git_dir().join("index.lock"), b"").unwrap();
2958
2959 let err = repo
2960 .restore(&empty)
2961 .expect_err("a removal through a symlinked parent is refused");
2962 assert!(err.to_string().contains("safety snapshot failed"), "{err}");
2963 assert_eq!(
2964 std::fs::read(&sentinel).unwrap(),
2965 b"outside",
2966 "the file outside the workspace survives"
2967 );
2968 assert!(
2969 std::fs::symlink_metadata(&src)
2970 .unwrap()
2971 .file_type()
2972 .is_symlink()
2973 );
2974 }
2975
2976 /// Every session, turn and sub-agent in a workspace shares its side repo.
2977 /// A snapshot must wait while another writer holds the repo's write lock
2978 /// instead of racing its index, HEAD and gc.
2979 #[test]
2980 fn snapshot_waits_for_the_side_repo_write_lock() {
2981 let tmp = tempdir().unwrap();
2982 let (repo, _home) = make_repo(tmp.path());
2983 std::fs::write(repo.work_tree().join("f.txt"), b"v0").unwrap();
2984 repo.snapshot("pre-turn:1").expect("first snapshot");
2985
2986 let lock_path = repo.git_dir().join(SNAPSHOT_LOCK_FILE);
2987 let mut peer = fd_lock::RwLock::new(open_snapshot_lock_file(&lock_path).unwrap());
2988 let held = peer.write().expect("peer holds the write lock");
2989
2990 let (done_tx, done_rx) = std::sync::mpsc::channel();
2991 let git_dir = repo.git_dir().to_path_buf();
2992 let work_tree = repo.work_tree().to_path_buf();
2993 let writer = std::thread::spawn(move || {
2994 let repo = SnapshotRepo { git_dir, work_tree };
2995 std::fs::write(repo.work_tree().join("f.txt"), b"v1").unwrap();
2996 let taken = repo.snapshot("post-turn:1");
2997 let _ = done_tx.send(());
2998 taken
2999 });
3000 assert!(
3001 done_rx.recv_timeout(Duration::from_millis(400)).is_err(),
3002 "a snapshot must not run while another writer holds the lock"
3003 );
3004 drop(held);
3005 done_rx
3006 .recv_timeout(Duration::from_secs(30))
3007 .expect("snapshot proceeds once the lock is released");
3008 writer
3009 .join()
3010 .expect("writer thread")
3011 .expect("snapshot after release");
3012 assert_eq!(repo.list(usize::MAX).unwrap().len(), 2);
3013 }
3014
3015 /// Two writers sharing one side repo, each snapshotting and pruning, must
3016 /// all succeed and leave a history whose every snapshot still restores.
3017 #[test]
3018 fn concurrent_snapshot_and_prune_keep_the_side_repo_consistent() {
3019 let tmp = tempdir().unwrap();
3020 let (repo, _home) = make_repo(tmp.path());
3021 std::fs::write(repo.work_tree().join("seed.txt"), b"seed").unwrap();
3022 repo.snapshot("pre-turn:0").expect("seed snapshot");
3023 let git_dir = repo.git_dir().to_path_buf();
3024 let work_tree = repo.work_tree().to_path_buf();
3025 let writers: Vec<_> = (0..2)
3026 .map(|writer| {
3027 let git_dir = git_dir.clone();
3028 let work_tree = work_tree.clone();
3029 std::thread::spawn(move || -> io::Result<()> {
3030 let repo = SnapshotRepo { git_dir, work_tree };
3031 for turn in 0..6 {
3032 std::fs::write(
3033 repo.work_tree().join(format!("w{writer}.txt")),
3034 format!("{writer}-{turn}"),
3035 )?;
3036 repo.snapshot(&format!("post-turn:{writer}-{turn}"))?;
3037 repo.prune_keep_last_n(2)?;
3038 }
3039 Ok(())
3040 })
3041 })
3042 .collect();
3043 for writer in writers {
3044 writer
3045 .join()
3046 .expect("writer thread")
3047 .expect("writer ran clean");
3048 }
3049 assert!(!repo.repair_broken_head().expect("head check"));
3050 let history = repo.list(usize::MAX).expect("history lists");
3051 assert!(!history.is_empty());
3052 for snapshot in &history {
3053 assert!(
3054 repo.is_commit(snapshot.id.as_str()).unwrap(),
3055 "listed snapshot {} must exist",
3056 snapshot.id.as_str()
3057 );
3058 }
3059 }
3060
3061 /// A peer stuck holding the lock (a long gc, a stopped process) used to
3062 /// block the snapshot, and the turn waiting on it, with no end. The wait
3063 /// is bounded and ends in an error the turn reports.
3064 #[test]
3065 fn side_repo_lock_wait_is_bounded() {
3066 let tmp = tempdir().unwrap();
3067 let (repo, _home) = make_repo(tmp.path());
3068 let lock_path = repo.git_dir().join(SNAPSHOT_LOCK_FILE);
3069 let mut peer = fd_lock::RwLock::new(open_snapshot_lock_file(&lock_path).unwrap());
3070 let _held = peer.write().expect("peer holds the write lock");
3071
3072 let started = std::time::Instant::now();
3073 let err = repo
3074 .with_write_lock_within(Duration::from_millis(200), || Ok(()))
3075 .expect_err("a held lock times out");
3076 assert_eq!(err.kind(), io::ErrorKind::TimedOut, "{err}");
3077 assert!(err.to_string().contains("busy"), "{err}");
3078 assert!(
3079 started.elapsed() < Duration::from_secs(10),
3080 "the wait ends near its bound"
3081 );
3082 }
3083
3084 /// First-time init (git init plus the pinned identity) runs under the
3085 /// write lock, so two sessions opening a new workspace do not race it; a
3086 /// `.git` holding only a peer's lock file is initialized, not trusted.
3087 #[test]
3088 fn first_init_waits_for_the_side_repo_write_lock() {
3089 let tmp = tempdir().unwrap();
3090 let (repo, _home) = make_repo(tmp.path());
3091 let git_dir = repo.git_dir().to_path_buf();
3092 let workspace = repo.work_tree().to_path_buf();
3093 drop(repo);
3094 std::fs::remove_dir_all(&git_dir).unwrap();
3095 std::fs::create_dir_all(&git_dir).unwrap();
3096 let mut peer = fd_lock::RwLock::new(
3097 open_snapshot_lock_file(&git_dir.join(SNAPSHOT_LOCK_FILE)).unwrap(),
3098 );
3099 let held = peer.write().expect("peer holds the write lock");
3100
3101 let (done_tx, done_rx) = std::sync::mpsc::channel();
3102 let env_ticket = crate::test_support::env_scope_ticket();
3103 let opener = std::thread::spawn(move || {
3104 let _membership = crate::test_support::join_env_scope(env_ticket);
3105 let opened = SnapshotRepo::open_or_init(&workspace);
3106 let _ = done_tx.send(());
3107 opened
3108 });
3109 assert!(
3110 done_rx.recv_timeout(Duration::from_millis(400)).is_err(),
3111 "init must not run while another writer holds the lock"
3112 );
3113 drop(held);
3114 done_rx
3115 .recv_timeout(Duration::from_secs(30))
3116 .expect("init proceeds once the lock is released");
3117 let repo = opener.join().expect("opener thread").expect("open_or_init");
3118 assert!(
3119 repo.git_dir().join("HEAD").exists(),
3120 "the repo was initialized"
3121 );
3122 std::fs::write(repo.work_tree().join("f.txt"), b"v").unwrap();
3123 repo.snapshot("pre-turn:1").expect("the new repo snapshots");
3124 }
3125
3126 /// Opening runs on every snapshot without the write lock while a peer may
3127 /// be staging. It used to truncate and rewrite `info/exclude` each time,
3128 /// so the peer's `git add -A` could read an empty exclude list.
3129 #[test]
3130 fn opening_leaves_a_current_exclude_file_untouched() {
3131 let tmp = tempdir().unwrap();
3132 let (repo, _home) = make_repo(tmp.path());
3133 let exclude = repo.git_dir().join("info").join("exclude");
3134 let old = SystemTime::UNIX_EPOCH + Duration::from_secs(1_000_000);
3135 File::options()
3136 .write(true)
3137 .open(&exclude)
3138 .unwrap()
3139 .set_times(FileTimes::new().set_modified(old))
3140 .unwrap();
3141
3142 SnapshotRepo::open_or_init(repo.work_tree()).expect("reopen");
3143 assert_eq!(
3144 std::fs::metadata(&exclude).unwrap().modified().unwrap(),
3145 old,
3146 "a current exclude file is not rewritten"
3147 );
3148
3149 std::fs::write(&exclude, b"stale\n").unwrap();
3150 SnapshotRepo::open_or_init(repo.work_tree()).expect("reopen");
3151 assert_eq!(
3152 std::fs::read_to_string(&exclude).unwrap(),
3153 BUILTIN_EXCLUDES,
3154 "a stale exclude file is replaced"
3155 );
3156 }
3157
3158 /// HEAD moves only from the value the writer read. A writer that does not
3159 /// take the lock (an older build) must not be orphaned by a later move.
3160 #[test]
3161 fn head_moves_only_from_the_expected_commit() {
3162 let tmp = tempdir().unwrap();
3163 let (repo, _home) = make_repo(tmp.path());
3164 std::fs::write(repo.work_tree().join("f.txt"), b"v1").unwrap();
3165 let first = repo.snapshot("pre-turn:1").expect("snapshot 1");
3166 std::fs::write(repo.work_tree().join("f.txt"), b"v2").unwrap();
3167 let second = repo.snapshot("post-turn:1").expect("snapshot 2");
3168
3169 repo.move_head(Some(first.as_str()), Some(first.as_str()))
3170 .expect_err("HEAD no longer names the expected commit");
3171 repo.move_head(Some(first.as_str()), None)
3172 .expect_err("HEAD exists, so an unborn expectation fails");
3173 repo.move_head(None, Some(first.as_str()))
3174 .expect_err("deleting HEAD also checks the expected commit");
3175 assert_eq!(repo.list(1).unwrap()[0].id, second, "HEAD is unchanged");
3176
3177 repo.move_head(Some(first.as_str()), Some(second.as_str()))
3178 .expect("the expected commit moves");
3179 assert_eq!(repo.list(1).unwrap()[0].id, first);
3180 }
3181
3182 /// HEAD repair looks up a missing commit, then moves HEAD off it. A peer
3183 /// that published a snapshot in between used to be overwritten by the
3184 /// reflog reset or deleted by the fresh-history fallback; both moves now
3185 /// require HEAD to still name the missing commit.
3186 #[test]
3187 fn head_repair_leaves_a_peers_new_snapshot_alone() {
3188 let tmp = tempdir().unwrap();
3189 let (repo, _home) = make_repo(tmp.path());
3190 std::fs::write(repo.work_tree().join("f.txt"), b"v1").unwrap();
3191 repo.snapshot("pre-turn:1").expect("snapshot 1");
3192 std::fs::write(repo.work_tree().join("f.txt"), b"v2").unwrap();
3193 let peer = repo.snapshot("post-turn:peer").expect("peer snapshot");
3194 let missing = "1111111111111111111111111111111111111111";
3195
3196 repo.reset_missing_head(missing)
3197 .expect_err("the reflog reset checks HEAD still names the missing commit");
3198 assert_eq!(repo.list(1).unwrap()[0].id, peer, "HEAD is unchanged");
3199
3200 std::fs::remove_dir_all(repo.git_dir().join("logs")).unwrap();
3201 repo.reset_missing_head(missing)
3202 .expect_err("the fresh-history delete checks it too");
3203 assert_eq!(
3204 repo.list(1).unwrap()[0].id,
3205 peer,
3206 "the peer's snapshot stays"
3207 );
3208 }
3209
3210 /// A prune rebuilds the survivors it listed. If another writer committed
3211 /// after that listing, the rebuild used to overwrite HEAD and orphan the
3212 /// new snapshot for gc; now it fails and the snapshot stays.
3213 #[test]
3214 fn survivor_rebuild_refuses_when_another_writer_moved_head() {
3215 let tmp = tempdir().unwrap();
3216 let (repo, _home) = make_repo(tmp.path());
3217 for i in 0..3 {
3218 std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap();
3219 repo.snapshot(&format!("tool:{i}")).expect("snapshot");
3220 }
3221 let listed = repo.list(usize::MAX).unwrap();
3222 std::fs::write(repo.work_tree().join("f.txt"), b"peer").unwrap();
3223 let peer = repo.snapshot("post-turn:peer").expect("peer snapshot");
3224
3225 repo.rebuild_survivor_chain(&listed[..1], &listed[0].id)
3226 .expect_err("HEAD moved since the listing");
3227 let history = repo.list(usize::MAX).unwrap();
3228 assert_eq!(history.len(), 4, "nothing was dropped");
3229 assert_eq!(history[0].id, peer, "the peer's snapshot is still HEAD");
3230 }
3231
3232 /// An index naming objects that no longer exist (an interrupted or racing
3233 /// gc) failed every later `write-tree`, because `add -A` does not rehash
3234 /// files whose stat data is unchanged. The index is rebuilt once instead.
3235 #[test]
3236 fn snapshot_rebuilds_an_index_naming_missing_objects() {
3237 let tmp = tempdir().unwrap();
3238 let (repo, _home) = make_repo(tmp.path());
3239 let file = repo.work_tree().join("f.txt");
3240 std::fs::write(&file, b"content").unwrap();
3241 // An mtime well before the index keeps git from treating the entry as
3242 // racily clean and rehashing it, which would hide the broken index.
3243 File::options()
3244 .write(true)
3245 .open(&file)
3246 .unwrap()
3247 .set_times(FileTimes::new().set_modified(SystemTime::now() - Duration::from_secs(3600)))
3248 .unwrap();
3249 let taken = repo
3250 .take_snapshot("pre-turn:1", None)
3251 .expect("first snapshot");
3252 let blob = git_output(&repo, &["rev-parse", "HEAD:f.txt"]);
3253 for oid in [blob.as_str(), taken.tree.as_str()] {
3254 let object = repo
3255 .git_dir()
3256 .join("objects")
3257 .join(&oid[..2])
3258 .join(&oid[2..]);
3259 std::fs::remove_file(&object).expect("loose object removed");
3260 }
3261
3262 let again = repo
3263 .take_snapshot("post-turn:1", None)
3264 .expect("the snapshot rebuilds the index and succeeds");
3265 assert_eq!(
3266 git_output(
3267 &repo,
3268 &["cat-file", "-p", &format!("{}:f.txt", again.id.as_str())]
3269 ),
3270 "content"
3271 );
3272 }
3273
3274 fn git_output(repo: &SnapshotRepo, args: &[&str]) -> String {
3275 let out = run_git(repo.git_dir(), repo.work_tree(), args).unwrap();
3276 assert!(
3277 out.status.success(),
3278 "git {args:?}: {}",
3279 String::from_utf8_lossy(&out.stderr)
3280 );
3281 String::from_utf8_lossy(&out.stdout).trim().to_string()
3282 }
3283
3284 #[test]
3285 fn restore_paths_leaves_unrelated_files_alone() {
3286 let tmp = tempdir().unwrap();
3287 let (repo, _home) = make_repo(tmp.path());
3288 let wanted = repo.work_tree().join("wanted.txt");
3289 let unrelated = repo.work_tree().join("unrelated.txt");
3290
3291 std::fs::write(&wanted, b"original").unwrap();
3292 std::fs::write(&unrelated, b"original").unwrap();
3293 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3294
3295 std::fs::write(&wanted, b"clobbered").unwrap();
3296 std::fs::write(&unrelated, b"also clobbered").unwrap();
3297 repo.snapshot("post-turn:1").expect("snapshot 2");
3298
3299 let outcomes = repo
3300 .restore_paths(&id, &[PathBuf::from("wanted.txt")])
3301 .expect("scoped restore");
3302
3303 assert_eq!(std::fs::read_to_string(&wanted).unwrap(), "original");
3304 assert_eq!(
3305 std::fs::read_to_string(&unrelated).unwrap(),
3306 "also clobbered",
3307 "a file-scoped restore must not touch a path it was not given"
3308 );
3309 assert_eq!(outcomes.len(), 1);
3310 assert_eq!(outcomes[0].path, PathBuf::from("wanted.txt"));
3311 assert_eq!(outcomes[0].action, PathRestoreAction::Modified);
3312 }
3313
3314 #[test]
3315 fn only_restore_paths_is_safe_for_a_single_file_action() {
3316 // Characterizes the difference the per-file Revert control depends on.
3317 // `restore()` is what the TUI's `patch_undo()` and the runtime's
3318 // `patch-undo` endpoint both call. It is scoped in *snapshot selection*
3319 // (it picks a recent `tool:` snapshot) but not in *effect*: it checks
3320 // out the whole tree, so it also rolls back a working-tree path that no
3321 // tool touched. That is the data loss #2 removed the control over, and
3322 // it is why a per-file action cannot be built on top of it.
3323 let tmp = tempdir().unwrap();
3324 let (repo, _home) = make_repo(tmp.path());
3325 let touched = repo.work_tree().join("touched.txt");
3326 let unrelated = repo.work_tree().join("unrelated.txt");
3327
3328 std::fs::write(&touched, b"v1").unwrap();
3329 std::fs::write(&unrelated, b"snapshot-time").unwrap();
3330 let id = repo.snapshot("tool:call-1").expect("snapshot");
3331
3332 // The tool edits one file; something else — the user, another editor —
3333 // changes the other one after the snapshot was taken.
3334 std::fs::write(&touched, b"v2").unwrap();
3335 std::fs::write(&unrelated, b"user-work-in-progress").unwrap();
3336
3337 repo.restore(&id).expect("whole-tree restore");
3338 assert_eq!(std::fs::read_to_string(&touched).unwrap(), "v1");
3339 assert_eq!(
3340 std::fs::read_to_string(&unrelated).unwrap(),
3341 "snapshot-time",
3342 "whole-tree restore rolls back a file no tool touched"
3343 );
3344
3345 // Same situation again, but through the file-scoped path the
3346 // `file-revert` endpoint uses.
3347 std::fs::write(&touched, b"v1").unwrap();
3348 std::fs::write(&unrelated, b"snapshot-time").unwrap();
3349 let id2 = repo.snapshot("tool:call-2").expect("snapshot 2");
3350 std::fs::write(&touched, b"v2").unwrap();
3351 std::fs::write(&unrelated, b"user-work-in-progress").unwrap();
3352
3353 repo.restore_paths(&id2, &[PathBuf::from("touched.txt")])
3354 .expect("scoped restore");
3355 assert_eq!(std::fs::read_to_string(&touched).unwrap(), "v1");
3356 assert_eq!(
3357 std::fs::read_to_string(&unrelated).unwrap(),
3358 "user-work-in-progress",
3359 "the scoped restore must leave the unrelated edit alone"
3360 );
3361 }
3362
3363 #[test]
3364 fn restore_paths_removes_a_file_created_after_the_snapshot() {
3365 let tmp = tempdir().unwrap();
3366 let (repo, _home) = make_repo(tmp.path());
3367 let kept = repo.work_tree().join("kept.txt");
3368 let created = repo.work_tree().join("created.txt");
3369
3370 std::fs::write(&kept, b"kept").unwrap();
3371 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3372
3373 std::fs::write(&created, b"new file").unwrap();
3374 repo.snapshot("post-turn:1").expect("snapshot 2");
3375
3376 let outcomes = repo
3377 .restore_paths(&id, &[PathBuf::from("created.txt")])
3378 .expect("scoped restore");
3379
3380 assert!(
3381 !created.exists(),
3382 "a created file must be removed by revert"
3383 );
3384 assert!(kept.exists(), "the untouched file must survive");
3385 assert_eq!(outcomes[0].action, PathRestoreAction::Removed);
3386 }
3387
3388 #[test]
3389 fn restore_paths_recreates_a_file_deleted_after_the_snapshot() {
3390 let tmp = tempdir().unwrap();
3391 let (repo, _home) = make_repo(tmp.path());
3392 let deleted = repo.work_tree().join("deleted.txt");
3393
3394 std::fs::write(&deleted, b"content").unwrap();
3395 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3396
3397 std::fs::remove_file(&deleted).unwrap();
3398 repo.snapshot("post-turn:1").expect("snapshot 2");
3399
3400 let outcomes = repo
3401 .restore_paths(&id, &[PathBuf::from("deleted.txt")])
3402 .expect("scoped restore");
3403
3404 assert_eq!(std::fs::read_to_string(&deleted).unwrap(), "content");
3405 assert_eq!(outcomes[0].action, PathRestoreAction::Recreated);
3406 }
3407
3408 #[test]
3409 fn restore_paths_rejects_parent_traversal() {
3410 let tmp = tempdir().unwrap();
3411 let (repo, _home) = make_repo(tmp.path());
3412 std::fs::write(repo.work_tree().join("a.txt"), b"a").unwrap();
3413 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3414
3415 let err = repo
3416 .restore_paths(&id, &[PathBuf::from("../escape.txt")])
3417 .expect_err("traversal must be refused");
3418 assert!(err.to_string().contains("unsafe path"), "got: {err}");
3419 }
3420
3421 #[test]
3422 fn path_differs_from_snapshot_is_scoped_to_the_named_path() {
3423 let tmp = tempdir().unwrap();
3424 let (repo, _home) = make_repo(tmp.path());
3425 let touched = repo.work_tree().join("touched.txt");
3426 let untouched = repo.work_tree().join("untouched.txt");
3427
3428 std::fs::write(&touched, b"v1").unwrap();
3429 std::fs::write(&untouched, b"v1").unwrap();
3430 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3431
3432 std::fs::write(&touched, b"v2").unwrap();
3433
3434 assert!(
3435 repo.path_differs_from_snapshot(&id, Path::new("touched.txt"))
3436 .expect("differs")
3437 );
3438 assert!(
3439 !repo
3440 .path_differs_from_snapshot(&id, Path::new("untouched.txt"))
3441 .expect("differs")
3442 );
3443 }
3444
3445 #[test]
3446 fn workspace_relative_path_accepts_inside_paths_and_refuses_outside_ones() {
3447 // A real absolute temp path so the fixture is absolute on Windows too
3448 // (`/tmp/ws` is a relative path with a root-dir component there).
3449 let temp = std::env::temp_dir();
3450 let workspace = temp.join("ws");
3451 let other = temp.join("other");
3452
3453 assert_eq!(
3454 workspace_relative_path(&workspace, "src/lib.rs"),
3455 Some(PathBuf::from("src/lib.rs"))
3456 );
3457 let inside = workspace.join("src").join("lib.rs");
3458 assert_eq!(
3459 workspace_relative_path(&workspace, &inside.to_string_lossy()),
3460 Some(PathBuf::from("src").join("lib.rs"))
3461 );
3462 let outside = other.join("lib.rs");
3463 assert_eq!(
3464 workspace_relative_path(&workspace, &outside.to_string_lossy()),
3465 None
3466 );
3467 assert_eq!(workspace_relative_path(&workspace, "../escape"), None);
3468 assert_eq!(
3469 workspace_relative_path(&workspace, "src/../../escape"),
3470 None
3471 );
3472 assert_eq!(workspace_relative_path(&workspace, ""), None);
3473 // Whitespace and glob characters are literal filename bytes.
3474 assert_eq!(
3475 workspace_relative_path(&workspace, " padded.txt "),
3476 Some(PathBuf::from(" padded.txt "))
3477 );
3478 assert_eq!(
3479 workspace_relative_path(&workspace, "file[12].txt"),
3480 Some(PathBuf::from("file[12].txt"))
3481 );
3482 }
3483
3484 fn sha256_hash(path: &Path) -> String {
3485 format!(
3486 "sha256:{}",
3487 crate::hashing::sha256_hex(std::fs::read(path).unwrap())
3488 )
3489 }
3490
3491 /// The Git primitive treats `[12]` as a pattern even after `--`; the
3492 /// file-scoped restore must not. `file[12].txt` and `file1.txt` both exist
3493 /// in the snapshot, so only literal pathspecs keep the second one intact.
3494 #[test]
3495 fn restore_file_if_unchanged_treats_glob_characters_literally() {
3496 let tmp = tempdir().unwrap();
3497 let (repo, _home) = make_repo(tmp.path());
3498 let literal = repo.work_tree().join("file[12].txt");
3499 let sibling = repo.work_tree().join("file1.txt");
3500 std::fs::write(&literal, b"literal-before").unwrap();
3501 std::fs::write(&sibling, b"sibling-before").unwrap();
3502 let id = repo.snapshot("tool:call-1").expect("snapshot");
3503 std::fs::write(&literal, b"literal-after").unwrap();
3504 std::fs::write(&sibling, b"sibling-after").unwrap();
3505
3506 assert!(
3507 repo.path_differs_from_snapshot(&id, Path::new("file[12].txt"))
3508 .unwrap()
3509 );
3510 let outcomes = repo
3511 .restore_file_if_unchanged(&id, Path::new("file[12].txt"), &sha256_hash(&literal))
3512 .expect("literal restore");
3513 assert_eq!(outcomes.len(), 1);
3514 assert_eq!(outcomes[0].action, PathRestoreAction::Modified);
3515 assert_eq!(std::fs::read_to_string(&literal).unwrap(), "literal-before");
3516 assert_eq!(
3517 std::fs::read_to_string(&sibling).unwrap(),
3518 "sibling-after",
3519 "a bracketed filename must never restore its glob siblings"
3520 );
3521 }
3522
3523 #[test]
3524 fn restore_file_if_unchanged_refuses_when_the_reviewed_bytes_changed() {
3525 let tmp = tempdir().unwrap();
3526 let (repo, _home) = make_repo(tmp.path());
3527 let file = repo.work_tree().join("a.txt");
3528 std::fs::write(&file, b"v1").unwrap();
3529 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3530 std::fs::write(&file, b"v2").unwrap();
3531 let reviewed = sha256_hash(&file);
3532 // The user edits again after the client captured its change record.
3533 std::fs::write(&file, b"v3-user-edit").unwrap();
3534
3535 let err = repo
3536 .restore_file_if_unchanged(&id, Path::new("a.txt"), &reviewed)
3537 .expect_err("stale hash must refuse");
3538 assert_eq!(err.kind(), io::ErrorKind::WouldBlock);
3539 assert_eq!(std::fs::read_to_string(&file).unwrap(), "v3-user-edit");
3540 // `absent` is only valid for a file the client saw as deleted.
3541 let err = repo
3542 .restore_file_if_unchanged(&id, Path::new("a.txt"), "absent")
3543 .expect_err("absent must not match an existing file");
3544 assert_eq!(err.kind(), io::ErrorKind::WouldBlock);
3545 // The exact current bytes restore.
3546 let outcomes = repo
3547 .restore_file_if_unchanged(&id, Path::new("a.txt"), &sha256_hash(&file))
3548 .expect("current hash restores");
3549 assert_eq!(outcomes[0].action, PathRestoreAction::Modified);
3550 assert_eq!(std::fs::read_to_string(&file).unwrap(), "v1");
3551 }
3552
3553 #[test]
3554 fn restore_file_if_unchanged_handles_deleted_and_created_files() {
3555 let tmp = tempdir().unwrap();
3556 let (repo, _home) = make_repo(tmp.path());
3557 let deleted = repo.work_tree().join("deleted.txt");
3558 std::fs::write(&deleted, b"content").unwrap();
3559 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3560 std::fs::remove_file(&deleted).unwrap();
3561 let created = repo.work_tree().join("created.txt");
3562 std::fs::write(&created, b"new").unwrap();
3563
3564 let outcomes = repo
3565 .restore_file_if_unchanged(&id, Path::new("deleted.txt"), "absent")
3566 .expect("recreate");
3567 assert_eq!(outcomes[0].action, PathRestoreAction::Recreated);
3568 assert_eq!(std::fs::read_to_string(&deleted).unwrap(), "content");
3569
3570 let outcomes = repo
3571 .restore_file_if_unchanged(&id, Path::new("created.txt"), &sha256_hash(&created))
3572 .expect("remove");
3573 assert_eq!(outcomes[0].action, PathRestoreAction::Removed);
3574 assert!(!created.exists());
3575 // A created file inside a new directory is removed alone; the
3576 // directory the user made stays.
3577 let nested_dir = repo.work_tree().join("newdir");
3578 std::fs::create_dir_all(&nested_dir).unwrap();
3579 let nested = nested_dir.join("only.txt");
3580 std::fs::write(&nested, b"n").unwrap();
3581 let outcomes = repo
3582 .restore_file_if_unchanged(&id, Path::new("newdir/only.txt"), &sha256_hash(&nested))
3583 .expect("remove nested");
3584 assert_eq!(outcomes[0].action, PathRestoreAction::Removed);
3585 assert!(!nested.exists());
3586 assert!(nested_dir.is_dir(), "the parent directory is not pruned");
3587 // A path missing on both sides is not a change and reports nothing.
3588 assert!(
3589 !repo
3590 .path_differs_from_snapshot(&id, Path::new("never.txt"))
3591 .unwrap()
3592 );
3593 }
3594
3595 #[test]
3596 fn restore_file_if_unchanged_refuses_directories_git_metadata_and_ignored_files() {
3597 let tmp = tempdir().unwrap();
3598 let (repo, _home) = make_repo(tmp.path());
3599 let dir = repo.work_tree().join("src");
3600 std::fs::create_dir_all(&dir).unwrap();
3601 std::fs::write(dir.join("lib.rs"), b"fn a() {}").unwrap();
3602 std::fs::write(
3603 repo.work_tree().join(".gitignore"),
3604 "ignored.txt
3605 ",
3606 )
3607 .unwrap();
3608 std::fs::write(repo.work_tree().join("ignored.txt"), b"secret").unwrap();
3609 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3610 std::fs::write(dir.join("lib.rs"), b"fn b() {}").unwrap();
3611
3612 let mut refused = vec!["src", ".git/config", "src/.GIT/x", ".git"];
3613 if cfg!(windows) {
3614 refused.extend([".git./config", ".git /config", "GIT~1/config"]);
3615 }
3616 for rel in refused {
3617 let err = repo.validate_restore_file(Path::new(rel)).expect_err(rel);
3618 assert_eq!(err.kind(), io::ErrorKind::InvalidInput, "{rel}");
3619 }
3620 let err = repo
3621 .restore_file_if_unchanged(&id, Path::new("src"), "absent")
3622 .expect_err("directories are refused");
3623 assert_eq!(err.kind(), io::ErrorKind::InvalidInput);
3624 assert_eq!(
3625 std::fs::read_to_string(dir.join("lib.rs")).unwrap(),
3626 "fn b() {}"
3627 );
3628
3629 // A gitignored file is excluded from the safety backup, so removing
3630 // it would be unrecoverable: refuse and leave it in place.
3631 let ignored = repo.work_tree().join("ignored.txt");
3632 let err = repo
3633 .restore_file_if_unchanged(&id, Path::new("ignored.txt"), &sha256_hash(&ignored))
3634 .expect_err("ignored files are refused");
3635 assert!(err.to_string().contains("safety snapshot"), "got: {err}");
3636 assert_eq!(std::fs::read_to_string(&ignored).unwrap(), "secret");
3637 }
3638
3639 #[cfg(unix)]
3640 #[test]
3641 fn restore_file_if_unchanged_refuses_symlinks_anywhere_in_the_path() {
3642 let tmp = tempdir().unwrap();
3643 let (repo, _home) = make_repo(tmp.path());
3644 let outside = tmp.path().join("outside");
3645 std::fs::create_dir_all(&outside).unwrap();
3646 std::fs::write(outside.join("target.txt"), b"outside").unwrap();
3647 std::fs::write(repo.work_tree().join("real.txt"), b"real").unwrap();
3648 std::os::unix::fs::symlink(&outside, repo.work_tree().join("linkdir")).unwrap();
3649 std::os::unix::fs::symlink(
3650 outside.join("target.txt"),
3651 repo.work_tree().join("link.txt"),
3652 )
3653 .unwrap();
3654 let id = repo.snapshot("pre-turn:1").expect("snapshot");
3655
3656 for rel in ["link.txt", "linkdir/target.txt"] {
3657 let err = repo
3658 .restore_file_if_unchanged(&id, Path::new(rel), "absent")
3659 .expect_err(rel);
3660 assert_eq!(err.kind(), io::ErrorKind::InvalidInput, "{rel}");
3661 }
3662 assert_eq!(
3663 std::fs::read_to_string(outside.join("target.txt")).unwrap(),
3664 "outside"
3665 );
3666 // The snapshot side is checked too: a symlink entry in the tree is
3667 // not a regular file even when the work tree copy is gone.
3668 std::fs::remove_file(repo.work_tree().join("link.txt")).unwrap();
3669 let err = repo
3670 .restore_file_if_unchanged(&id, Path::new("link.txt"), "absent")
3671 .expect_err("snapshot symlink entry");
3672 assert_eq!(err.kind(), io::ErrorKind::InvalidInput);
3673 assert!(!repo.work_tree().join("link.txt").exists());
3674 }
3675
3676 #[test]
3677 fn list_distinguishes_an_unborn_head_from_broken_history() {
3678 let tmp = tempdir().unwrap();
3679 let (repo, _home) = make_repo(tmp.path());
3680 assert!(repo.list(10).expect("unborn HEAD lists nothing").is_empty());
3681
3682 std::fs::write(repo.work_tree().join("a.txt"), b"a").unwrap();
3683 repo.snapshot("pre-turn:1").expect("snapshot");
3684 assert_eq!(repo.list(10).unwrap().len(), 1);
3685
3686 // Point the branch at an object that does not exist: the history is
3687 // now broken, which must surface as an error rather than "no
3688 // snapshots" (an empty list would let patch-undo drop a turn).
3689 let head = String::from_utf8(
3690 run_git(repo.git_dir(), repo.work_tree(), &["symbolic-ref", "HEAD"])
3691 .unwrap()
3692 .stdout,
3693 )
3694 .unwrap();
3695 std::fs::write(
3696 repo.git_dir().join(head.trim()),
3697 "0123456789abcdef0123456789abcdef01234567\n",
3698 )
3699 .unwrap();
3700 let err = repo.list(10).expect_err("broken history must error");
3701 assert!(err.to_string().contains("git log failed"), "got: {err}");
3702 }
3703
3704 #[test]
3705 fn restore_takes_a_pre_restore_safety_snapshot_that_round_trips() {
3706 let tmp = tempdir().unwrap();
3707 let (repo, _home) = make_repo(tmp.path());
3708 let f = repo.work_tree().join("file.txt");
3709
3710 std::fs::write(&f, b"v1").unwrap();
3711 let id1 = repo.snapshot("pre-turn:1").expect("snapshot v1");
3712
3713 std::fs::write(&f, b"v2").unwrap();
3714 repo.snapshot("post-turn:1").expect("snapshot v2");
3715
3716 repo.restore(&id1).expect("restore to v1");
3717 assert_eq!(std::fs::read_to_string(&f).unwrap(), "v1");
3718
3719 // The restore must have captured the pre-restore state (v2) under a
3720 // `pre-restore:` label naming its target, so the destructive op is
3721 // itself reversible (2026-08-04 snapshot hunt).
3722 let snapshots = repo.list(usize::MAX).expect("list");
3723 let safety = snapshots
3724 .iter()
3725 .find(|s| s.label.starts_with("pre-restore:"))
3726 .expect("a pre-restore safety snapshot must exist");
3727 assert!(
3728 safety.label.ends_with(&id1.as_str()[..12]),
3729 "safety label should name the restore target: {}",
3730 safety.label
3731 );
3732
3733 repo.restore(&safety.id)
3734 .expect("restore the safety snapshot");
3735 assert_eq!(
3736 std::fs::read_to_string(&f).unwrap(),
3737 "v2",
3738 "the safety snapshot must bring back the pre-restore state"
3739 );
3740 }
3741
3742 #[test]
3743 fn snapshot_and_restore_do_not_move_user_git_head() {
3744 let tmp = tempdir().unwrap();
3745 let workspace = tmp.path().join("workspace");
3746 std::fs::create_dir_all(&workspace).unwrap();
3747 crate::dependencies::Git::command()
3748 .expect("git not found")
3749 .arg("-C")
3750 .arg(&workspace)
3751 .arg("init")
3752 .arg("--quiet")
3753 .status()
3754 .unwrap();
3755 std::fs::write(workspace.join("tracked.txt"), b"committed").unwrap();
3756 crate::dependencies::Git::command()
3757 .expect("git not found")
3758 .arg("-C")
3759 .arg(&workspace)
3760 .arg("add")
3761 .arg("tracked.txt")
3762 .status()
3763 .unwrap();
3764 crate::dependencies::Git::command()
3765 .expect("git not found")
3766 .arg("-C")
3767 .arg(&workspace)
3768 .arg("-c")
3769 .arg("user.name=user")
3770 .arg("-c")
3771 .arg("user.email=user@example.test")
3772 .arg("commit")
3773 .arg("--quiet")
3774 .arg("-m")
3775 .arg("init")
3776 .status()
3777 .unwrap();
3778 let user_head_before = crate::dependencies::Git::command()
3779 .expect("git not found")
3780 .arg("-C")
3781 .arg(&workspace)
3782 .args(["rev-parse", "HEAD"])
3783 .output()
3784 .unwrap()
3785 .stdout;
3786
3787 let _home = scoped_home(tmp.path());
3788 let repo = SnapshotRepo::open_or_init(&workspace).unwrap();
3789 std::fs::write(workspace.join("tracked.txt"), b"dirty-before").unwrap();
3790 let id = repo.snapshot("pre-turn:1").unwrap();
3791 std::fs::write(workspace.join("tracked.txt"), b"dirty-after").unwrap();
3792 repo.snapshot("post-turn:1").unwrap();
3793 repo.restore(&id).unwrap();
3794
3795 let user_head_after = crate::dependencies::Git::command()
3796 .expect("git not found")
3797 .arg("-C")
3798 .arg(&workspace)
3799 .args(["rev-parse", "HEAD"])
3800 .output()
3801 .unwrap()
3802 .stdout;
3803 assert_eq!(user_head_after, user_head_before);
3804 assert_eq!(
3805 std::fs::read_to_string(workspace.join("tracked.txt")).unwrap(),
3806 "dirty-before"
3807 );
3808 }
3809
3810 #[test]
3811 fn list_respects_limit() {
3812 let tmp = tempdir().unwrap();
3813 let (repo, _home) = make_repo(tmp.path());
3814 for i in 0..5 {
3815 std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap();
3816 repo.snapshot(&format!("turn:{i}")).unwrap();
3817 }
3818 let three = repo.list(3).unwrap();
3819 assert_eq!(three.len(), 3);
3820 // Newest first.
3821 assert_eq!(three[0].label, "turn:4");
3822 }
3823
3824 #[test]
3825 fn prune_drops_snapshots_older_than_threshold() {
3826 let tmp = tempdir().unwrap();
3827 let (repo, _home) = make_repo(tmp.path());
3828 std::fs::write(repo.work_tree().join("f.txt"), "v0").unwrap();
3829 repo.snapshot("turn:0").unwrap();
3830
3831 // Wait one second so the snapshot's commit timestamp is strictly
3832 // in the past relative to the prune call's "now" — otherwise
3833 // same-second comparisons make the assertion flaky.
3834 std::thread::sleep(Duration::from_millis(1100));
3835
3836 let removed = repo.prune_older_than(Duration::from_secs(0)).unwrap();
3837 assert!(removed >= 1, "expected at least 1 pruned, got {removed}");
3838
3839 // After pruning everything, the next snapshot should start a
3840 // fresh history.
3841 std::fs::write(repo.work_tree().join("f.txt"), "v1").unwrap();
3842 repo.snapshot("turn:1").unwrap();
3843 let list = repo.list(10).unwrap();
3844 assert_eq!(list.len(), 1);
3845 assert_eq!(list[0].label, "turn:1");
3846 }
3847
3848 /// The 2026-08-04 regression: with a cut in the MIDDLE of history,
3849 /// `prune_older_than` used to `update-ref HEAD <oldest survivor>`, which
3850 /// orphaned (and gc destroyed) the NEWEST snapshots while keeping the
3851 /// old ones as ancestors — the inverse of the intent, firing on every
3852 /// boot. This pins the correct partial-cut behavior.
3853 #[test]
3854 fn prune_older_than_keeps_the_newest_and_drops_only_the_old_tail() {
3855 let tmp = tempdir().unwrap();
3856 let (repo, _home) = make_repo(tmp.path());
3857
3858 // Two "old" snapshots, then a pause, then two "new" ones.
3859 for i in 0..2 {
3860 std::fs::write(repo.work_tree().join("f.txt"), format!("old{i}")).unwrap();
3861 repo.snapshot(&format!("old:{i}")).unwrap();
3862 std::thread::sleep(Duration::from_millis(1100));
3863 }
3864 // A wide gap so git's whole-second commit timestamps land the cut
3865 // unambiguously between the old and new pairs. The margins are
3866 // deliberately generous: this test runs under full-suite parallelism
3867 // where a sleep can overrun, and the cut is wall-clock. At prune time
3868 // the newest pair is ~0-1.2s old against a 6s cutoff, and the old
3869 // pair is ~9s old — ~5s of slack in both directions.
3870 std::thread::sleep(Duration::from_secs(8));
3871 for i in 0..2 {
3872 std::fs::write(repo.work_tree().join("f.txt"), format!("new{i}")).unwrap();
3873 repo.snapshot(&format!("new:{i}")).unwrap();
3874 if i == 0 {
3875 std::thread::sleep(Duration::from_millis(1100));
3876 }
3877 }
3878 let before = repo.list(usize::MAX).unwrap();
3879 assert_eq!(before.len(), 4);
3880 // Derive the cut from the timestamps actually recorded rather than a
3881 // fixed 6s. A fixed cut assumes `repo.snapshot()` is fast: `new:0` is
3882 // only ~1.2s plus one git subprocess older than prune time, so on a
3883 // loaded Windows runner that subprocess alone pushed it past 6s and
3884 // three snapshots were pruned instead of two. (The old fixture guard
3885 // could not catch it either — it checked `before[0]` and `before[2]`,
3886 // and `before[1]` is the entry that drifts.)
3887 let now = std::time::SystemTime::now()
3888 .duration_since(std::time::UNIX_EPOCH)
3889 .unwrap()
3890 .as_secs() as i64;
3891 // Newest-first: [new:1, new:0, old:1, old:0]. The cut must land
3892 // strictly between the pairs, so aim at the midpoint of the 8s gap —
3893 // that leaves ~4s of slack against clock drift and a slow runner in
3894 // both directions.
3895 let survivor = before[1].timestamp;
3896 let victim = before[2].timestamp;
3897 assert!(
3898 survivor - victim >= 8,
3899 "fixture needs an 8s gap between the pairs (survivor {survivor}, victim {victim})"
3900 );
3901 let midpoint = victim + (survivor - victim) / 2;
3902 assert!(
3903 now > midpoint,
3904 "fixture cutoff is not before the current time"
3905 );
3906 let max_age = Duration::from_secs((now - midpoint) as u64);
3907
3908 // The two old snapshots drop, the two new ones survive.
3909 let removed = repo.prune_older_than(max_age).unwrap();
3910 assert_eq!(removed, 2, "only the old tail should be removed");
3911
3912 let remaining = repo.list(usize::MAX).unwrap();
3913 assert_eq!(remaining.len(), 2, "the two newest must survive");
3914 assert_eq!(
3915 remaining[0].label, "new:1",
3916 "newest survives (was destroyed before)"
3917 );
3918 assert_eq!(remaining[1].label, "new:0");
3919 assert!(
3920 !remaining.iter().any(|s| s.label.starts_with("old:")),
3921 "old snapshots must be gone, not kept as ancestors: {:?}",
3922 remaining.iter().map(|s| &s.label).collect::<Vec<_>>()
3923 );
3924
3925 // The survivors' contents are intact and restorable.
3926 repo.restore(&remaining[0].id).unwrap();
3927 assert_eq!(
3928 std::fs::read_to_string(repo.work_tree().join("f.txt")).unwrap(),
3929 "new1"
3930 );
3931 }
3932
3933 #[test]
3934 fn prune_keep_last_n_keeps_latest_and_gc_reclaims_rest() {
3935 let tmp = tempdir().unwrap();
3936 let (repo, _home) = make_repo(tmp.path());
3937
3938 for i in 0..3 {
3939 std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap();
3940 repo.snapshot(&format!("turn:{i}")).unwrap();
3941 std::thread::sleep(Duration::from_millis(1100));
3942 }
3943
3944 assert_eq!(repo.list(usize::MAX).unwrap().len(), 3);
3945
3946 let removed = repo.prune_keep_last_n(1).unwrap();
3947 assert_eq!(removed, 2);
3948
3949 let remaining = repo.list(usize::MAX).unwrap();
3950 assert_eq!(remaining.len(), 1);
3951 assert_eq!(remaining[0].label, "turn:2");
3952
3953 // New snapshot starts a clean chain (not appending to old).
3954 std::fs::write(repo.work_tree().join("f.txt"), "fresh").unwrap();
3955 repo.snapshot("turn:new").unwrap();
3956 assert_eq!(repo.list(usize::MAX).unwrap().len(), 2);
3957 }
3958
3959 /// The per-snapshot prune drops half a window at once instead of
3960 /// rebuilding the chain for every snapshot past the cap.
3961 #[test]
3962 fn batched_prune_waits_for_half_a_window_then_drops_it_together() {
3963 let tmp = tempdir().unwrap();
3964 let (repo, _home) = make_repo(tmp.path());
3965 let max = 4;
3966 for n in 0..max + 1 {
3967 std::fs::write(repo.work_tree().join("f.txt"), format!("{n}")).unwrap();
3968 repo.snapshot(&format!("tool:{n}")).unwrap();
3969 }
3970 // One over the cap: the plain prune would rebuild now, the batched
3971 // one waits.
3972 assert_eq!(repo.prune_keep_last_n_batched(max).unwrap(), 0);
3973 assert_eq!(repo.list(usize::MAX).unwrap().len(), max + 1);
3974
3975 std::fs::write(repo.work_tree().join("f.txt"), "last").unwrap();
3976 repo.snapshot("tool:last").unwrap();
3977 assert_eq!(repo.prune_keep_last_n_batched(max).unwrap(), 2);
3978 let kept = repo.list(usize::MAX).unwrap();
3979 assert_eq!(kept.len(), max);
3980 assert_eq!(kept[0].label, "tool:last", "the newest snapshots survive");
3981 }
3982
3983 #[test]
3984 fn prune_keep_last_n_preserves_multiple_snapshots_in_order() {
3985 let tmp = tempdir().unwrap();
3986 let (repo, _home) = make_repo(tmp.path());
3987
3988 for i in 0..4 {
3989 std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap();
3990 repo.snapshot(&format!("turn:{i}")).unwrap();
3991 std::thread::sleep(Duration::from_millis(1100));
3992 }
3993
3994 assert_eq!(repo.list(usize::MAX).unwrap().len(), 4);
3995
3996 let removed = repo.prune_keep_last_n(2).unwrap();
3997 assert_eq!(removed, 2);
3998
3999 let remaining = repo.list(usize::MAX).unwrap();
4000 assert_eq!(remaining.len(), 2);
4001 // Should be newest-first: turn:3 (newest), turn:2 (second newest)
4002 assert_eq!(remaining[0].label, "turn:3");
4003 assert_eq!(remaining[1].label, "turn:2");
4004
4005 // New snapshot continues the chain.
4006 std::fs::write(repo.work_tree().join("f.txt"), "fresh").unwrap();
4007 repo.snapshot("turn:new").unwrap();
4008 let after = repo.list(usize::MAX).unwrap();
4009 assert_eq!(after.len(), 3);
4010 assert_eq!(after[0].label, "turn:new");
4011 }
4012
4013 #[test]
4014 fn open_or_init_removes_stale_tmp_pack_files_only() {
4015 let tmp = tempdir().unwrap();
4016 let (repo, _home) = make_repo(tmp.path());
4017 let workspace = repo.work_tree().to_path_buf();
4018 let pack_dir = repo.git_dir().join("objects").join("pack");
4019 std::fs::create_dir_all(&pack_dir).unwrap();
4020
4021 let stale = pack_dir.join("tmp_pack_stale");
4022 let fresh = pack_dir.join("tmp_pack_fresh");
4023 let ordinary_pack = pack_dir.join("pack-kept.pack");
4024 std::fs::write(&stale, b"stale").unwrap();
4025 std::fs::write(&fresh, b"fresh").unwrap();
4026 std::fs::write(&ordinary_pack, b"pack").unwrap();
4027
4028 let old_time = SystemTime::now() - STALE_TMP_PACK_AGE - Duration::from_secs(60);
4029 {
4030 let file = File::options().write(true).open(&stale).unwrap();
4031 file.set_times(FileTimes::new().set_modified(old_time))
4032 .unwrap();
4033 }
4034
4035 SnapshotRepo::open_or_init(&workspace).unwrap();
4036
4037 assert!(!stale.exists(), "stale tmp_pack file should be removed");
4038 assert!(fresh.exists(), "fresh tmp_pack file should be kept");
4039 assert!(ordinary_pack.exists(), "non-temp pack file should be kept");
4040 }
4041
4042 #[test]
4043 fn snapshot_respects_workspace_gitignore() {
4044 let tmp = tempdir().unwrap();
4045 let (repo, _home) = make_repo(tmp.path());
4046 std::fs::write(repo.work_tree().join(".gitignore"), "ignored.txt\n").unwrap();
4047 std::fs::write(repo.work_tree().join("ignored.txt"), b"secret").unwrap();
4048 std::fs::write(repo.work_tree().join("kept.txt"), b"public").unwrap();
4049
4050 let id = repo.snapshot("pre-turn:1").expect("snapshot");
4051
4052 // `git ls-tree` against the snapshot's commit shouldn't list ignored.txt.
4053 let ls = run_git(
4054 repo.git_dir(),
4055 repo.work_tree(),
4056 &["ls-tree", "-r", "--name-only", id.as_str()],
4057 )
4058 .expect("ls-tree");
4059 let names = String::from_utf8_lossy(&ls.stdout);
4060 assert!(names.contains("kept.txt"), "kept.txt missing: {names}");
4061 assert!(
4062 !names.contains("ignored.txt"),
4063 "ignored.txt should not be in snapshot: {names}",
4064 );
4065 }
4066
4067 #[test]
4068 fn unsafe_workspace_rejects_home_directory_workspace() {
4069 let tmp = tempdir().unwrap();
4070 let home = tmp.path();
4071
4072 assert_eq!(
4073 unsafe_workspace_snapshot_reason(home, Some(home)),
4074 Some("home directory")
4075 );
4076 }
4077
4078 #[test]
4079 fn unsafe_workspace_rejects_home_collection_directories() {
4080 let tmp = tempdir().unwrap();
4081 let home = tmp.path();
4082 let desktop = tmp.path().join("Desktop");
4083 std::fs::create_dir_all(&desktop).unwrap();
4084
4085 assert_eq!(
4086 unsafe_workspace_snapshot_reason(&desktop, Some(home)),
4087 Some("home collection directory")
4088 );
4089 }
4090
4091 #[test]
4092 fn unsafe_workspace_allows_project_directories_under_home() {
4093 let tmp = tempdir().unwrap();
4094 let home = tmp.path();
4095 let workspace = tmp.path().join("code").join("project");
4096 std::fs::create_dir_all(&workspace).unwrap();
4097
4098 assert_eq!(
4099 unsafe_workspace_snapshot_reason(&workspace, Some(home)),
4100 None
4101 );
4102 }
4103
4104 #[test]
4105 fn snapshot_respects_builtin_excludes() {
4106 let tmp = tempdir().unwrap();
4107 let (repo, _home) = make_repo(tmp.path());
4108 std::fs::create_dir_all(repo.work_tree().join("node_modules/pkg")).unwrap();
4109 std::fs::create_dir_all(repo.work_tree().join(".next/cache")).unwrap();
4110 std::fs::create_dir_all(repo.work_tree().join("src")).unwrap();
4111 std::fs::write(
4112 repo.work_tree().join("node_modules/pkg/index.js"),
4113 b"generated",
4114 )
4115 .unwrap();
4116 std::fs::write(repo.work_tree().join(".next/cache/chunk.bin"), b"generated").unwrap();
4117 std::fs::write(repo.work_tree().join("debug.wasm"), b"binary").unwrap();
4118 std::fs::write(repo.work_tree().join("src/main.rs"), b"fn main() {}").unwrap();
4119
4120 let excludes = std::fs::read_to_string(repo.git_dir().join("info/exclude")).unwrap();
4121 assert!(excludes.contains("node_modules/"));
4122 assert!(excludes.contains(".next/"));
4123 assert!(excludes.contains("*.wasm"));
4124
4125 let id = repo.snapshot("pre-turn:1").expect("snapshot");
4126 let ls = run_git(
4127 repo.git_dir(),
4128 repo.work_tree(),
4129 &["ls-tree", "-r", "--name-only", id.as_str()],
4130 )
4131 .expect("ls-tree");
4132 let names = String::from_utf8_lossy(&ls.stdout);
4133 assert!(
4134 names.contains("src/main.rs"),
4135 "src/main.rs missing: {names}"
4136 );
4137 assert!(
4138 !names.contains("node_modules"),
4139 "node_modules should not be in snapshot: {names}",
4140 );
4141 assert!(
4142 !names.contains(".next"),
4143 ".next should not be in snapshot: {names}",
4144 );
4145 assert!(
4146 !names.contains("debug.wasm"),
4147 "binary artifacts should not be in snapshot: {names}",
4148 );
4149 }
4150
4151 #[test]
4152 fn open_or_init_is_idempotent() {
4153 let tmp = tempdir().unwrap();
4154 let (_r, _h) = make_repo(tmp.path());
4155 // Second open should not panic and should reuse the existing
4156 // `.git`. We re-open via the public API rather than make_repo to
4157 // avoid double-acquiring HOME (the guard would deadlock).
4158 drop((_r, _h));
4159 let (_r2, _h2) = make_repo(tmp.path());
4160 }
4161
4162 #[test]
4163 fn home_directory_guard_matches_canonical_paths() {
4164 let tmp = tempdir().unwrap();
4165 let home = tmp.path();
4166 let home_canonical = home.canonicalize().unwrap();
4167 let workspace = home.join("workspace");
4168 std::fs::create_dir_all(&workspace).unwrap();
4169 let workspace_canonical = workspace.canonicalize().unwrap();
4170
4171 assert!(is_home_directory(&home_canonical, Some(home)));
4172 assert!(!is_home_directory(&workspace_canonical, Some(home)));
4173 assert!(!is_home_directory(&home_canonical, None));
4174 }
4175
4176 #[test]
4177 fn dir_size_bytes_measures_directory_bytes() {
4178 let tmp = tempdir().unwrap();
4179 let dir = tmp.path().join("sizedir");
4180 std::fs::create_dir_all(dir.join("sub")).unwrap();
4181 // 3 bytes per file.
4182 std::fs::write(dir.join("a.txt"), b"abc").unwrap();
4183 std::fs::write(dir.join("sub/b.txt"), b"xyz").unwrap();
4184
4185 let size = dir_size_bytes(&dir).expect("dir_size_bytes");
4186 assert_eq!(size, 6, "two 3-byte files should measure 6 bytes");
4187
4188 // Write 2 MB of data.
4189 let big = dir.join("big.bin");
4190 std::fs::write(&big, vec![0u8; 2 * 1024 * 1024]).unwrap();
4191 let size = dir_size_bytes(&dir).expect("dir_size_bytes after big write");
4192 assert_eq!(
4193 size,
4194 2 * 1024 * 1024 + 6,
4195 "expected 2 MB + 6 bytes after writing a 2 MB file"
4196 );
4197 }
4198
4199 /// Regression: snapshot size cap (#1112). When the snapshot dir grows,
4200 /// `snapshot()` must prune old snapshots to stay under the limit.
4201 /// This test uses the real size constants, which are 500/400 MB —
4202 /// we can't easily blow up a temp dir to 500 MB in a unit test.
4203 /// Instead we verify the guard logic doesn't panic or error on a
4204 /// small repo (well under the cap), and that `snapshot()` still works.
4205 #[test]
4206 fn snapshot_succeeds_when_under_size_cap() {
4207 let tmp = tempdir().unwrap();
4208 let (repo, _home) = make_repo(tmp.path());
4209 // The side repo is tiny — well under 500 MB. Snapshot should work.
4210 std::fs::write(repo.work_tree().join("f.txt"), b"hello").unwrap();
4211 let id = repo.snapshot("pre-turn:1").expect("snapshot under cap");
4212 assert_eq!(id.as_str().len(), 40);
4213 }
4214
4215 /// Sessions and sub-agents share one side repo. The size-pressure prune
4216 /// used to protect only the globally newest turn boundaries, so another
4217 /// session's snapshots pushed a running turn's `pre-turn:` out and its
4218 /// undo had nothing to restore.
4219 #[test]
4220 fn prune_size_pressure_keeps_every_sessions_turn_boundaries() {
4221 let tmp = tempdir().unwrap();
4222 let (repo, _home) = make_repo(tmp.path());
4223 for (i, (label, sid)) in [
4224 ("pre-turn:5", "A"),
4225 ("tool:a", "A"),
4226 ("pre-turn:1", "B"),
4227 ("tool:b", "B"),
4228 ("post-turn:1", "B"),
4229 ("pre-turn:2", "B"),
4230 ("tool:c", "B"),
4231 ]
4232 .into_iter()
4233 .enumerate()
4234 {
4235 std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap();
4236 repo.snapshot_with_session(label, Some(sid))
4237 .expect("snapshot");
4238 }
4239 repo.prune_size_pressure(0, 0).expect("prune_size_pressure");
4240 let kept: Vec<(String, Option<String>)> = repo
4241 .list(usize::MAX)
4242 .unwrap()
4243 .into_iter()
4244 .map(|s| (s.label, s.session_id))
4245 .collect();
4246 let kept_a_pre = kept
4247 .iter()
4248 .any(|(label, sid)| label == "pre-turn:5" && sid.as_deref() == Some("A"));
4249 assert!(
4250 kept_a_pre,
4251 "session A's running turn stays restorable: {kept:?}"
4252 );
4253 assert_eq!(
4254 kept.iter()
4255 .map(|(label, _)| label.as_str())
4256 .collect::<Vec<_>>(),
4257 ["tool:c", "pre-turn:2", "post-turn:1", "pre-turn:5"],
4258 "only boundaries and the newest snapshot survive a full cut"
4259 );
4260 }
4261
4262 /// The size-pressure prune drops the oldest snapshots first and keeps the
4263 /// newest one plus the newest turn boundaries. It used to prune by age
4264 /// from one second down, which wiped every restore point, the running
4265 /// turn's own `pre-turn:` included, on each snapshot of a side repo over
4266 /// the cap.
4267 #[test]
4268 fn prune_size_pressure_drops_oldest_first_and_keeps_turn_boundaries() {
4269 let tmp = tempdir().unwrap();
4270 let (repo, _home) = make_repo(tmp.path());
4271 for (i, label) in [
4272 "pre-turn:1",
4273 "tool:a",
4274 "post-turn:1",
4275 "pre-turn:2",
4276 "tool:b",
4277 "tool:c",
4278 ]
4279 .into_iter()
4280 .enumerate()
4281 {
4282 std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap();
4283 repo.snapshot(label).expect("snapshot");
4284 }
4285 // A zero byte limit makes any non-empty side repo "over limit", so the
4286 // prune cuts as far as it may and reports exactly what it destroyed;
4287 // the count is what the user-visible notice is built from.
4288 let removed = repo.prune_size_pressure(0, 0).expect("prune_size_pressure");
4289 let labels: Vec<String> = repo
4290 .list(usize::MAX)
4291 .unwrap()
4292 .into_iter()
4293 .map(|s| s.label)
4294 .collect();
4295 assert_eq!(
4296 labels,
4297 ["tool:c", "pre-turn:2", "post-turn:1"],
4298 "the newest snapshot and the newest turn boundaries survive"
4299 );
4300 assert_eq!(removed, 3, "every dropped snapshot is reported");
4301 // The survivors still restore: the running turn can be undone.
4302 let pre = repo.list(usize::MAX).unwrap()[1].id.clone();
4303 repo.restore(&pre)
4304 .expect("restore the running turn's boundary");
4305 assert_eq!(
4306 std::fs::read_to_string(repo.work_tree().join("f.txt")).unwrap(),
4307 "v3"
4308 );
4309 }
4310
4311 #[test]
4312 fn prune_size_pressure_is_a_noop_under_the_limit() {
4313 let tmp = tempdir().unwrap();
4314 let (repo, _home) = make_repo(tmp.path());
4315 std::fs::write(repo.work_tree().join("f.txt"), b"v0").unwrap();
4316 repo.snapshot("pre-turn:0").expect("snapshot");
4317 let removed = repo
4318 .prune_size_pressure(u64::MAX, u64::MAX)
4319 .expect("prune_size_pressure");
4320 assert_eq!(removed, 0, "under the limit nothing may be removed");
4321 assert_eq!(repo.list(usize::MAX).unwrap().len(), 1);
4322 }
4323
4324 #[test]
4325 fn snapshot_history_pruned_message_names_workspace_count_and_cap() {
4326 let msg = snapshot_history_pruned_message(Path::new("/tmp/ws"), 7);
4327 assert!(msg.contains("/tmp/ws"), "message must name the workspace");
4328 assert!(msg.contains("7"), "message must state the removed count");
4329 assert!(
4330 msg.contains(&MAX_SNAPSHOT_SIZE_MB.to_string()),
4331 "message must state the storage cap"
4332 );
4333 }
4334
4335 #[test]
4336 fn estimate_workspace_size_bounded_returns_total_when_under_cap() {
4337 let tmp = tempdir().unwrap();
4338 let workspace = tmp.path().join("workspace");
4339 std::fs::create_dir_all(&workspace).unwrap();
4340 std::fs::write(workspace.join("a.txt"), vec![b'a'; 100]).unwrap();
4341 std::fs::write(workspace.join("b.txt"), vec![b'b'; 50]).unwrap();
4342 let total = estimate_workspace_size_bounded(&workspace, 10_000, SIZE_WALK_MAX_ENTRIES)
4343 .expect("under-cap walk must return a total");
4344 assert!(
4345 total >= 150,
4346 "total ({total}) must include both files (≥150 bytes)"
4347 );
4348 }
4349
4350 #[test]
4351 fn estimate_workspace_size_bounded_reports_the_size_gate_when_over_cap() {
4352 let tmp = tempdir().unwrap();
4353 let workspace = tmp.path().join("workspace");
4354 std::fs::create_dir_all(&workspace).unwrap();
4355 // Two 1 KB files, cap at 1 KB — second file should trip the cap.
4356 std::fs::write(workspace.join("a.bin"), vec![b'a'; 1024]).unwrap();
4357 std::fs::write(workspace.join("b.bin"), vec![b'b'; 1024]).unwrap();
4358 assert_eq!(
4359 estimate_workspace_size_bounded(&workspace, 1024, SIZE_WALK_MAX_ENTRIES),
4360 Err(WorkspaceGate::TooLarge),
4361 "over-cap walk must name the size gate for early bailout"
4362 );
4363 }
4364
4365 #[test]
4366 fn oversize_gate_message_states_the_byte_cap_without_a_remedy() {
4367 // The remedy is localized by the notice surfaces; repeating it here is
4368 // what produced the doubled warning.
4369 let message =
4370 WorkspaceGate::TooLarge.describe(2 * 1024 * 1024 * 1024, Path::new("/tmp/ws"));
4371 assert!(message.starts_with(GATE_TOO_LARGE_MARKER));
4372 assert!(message.contains("/tmp/ws"));
4373 assert!(!message.contains("max_workspace_gb"));
4374 assert_eq!(message.lines().count(), 1, "the gate message is one line");
4375 }
4376
4377 #[test]
4378 fn entry_gate_message_is_distinct_and_never_blames_the_size_cap() {
4379 let message = WorkspaceGate::TooManyEntries.describe(0, Path::new("/tmp/ws"));
4380 assert!(message.starts_with(GATE_TOO_MANY_ENTRIES_MARKER));
4381 assert!(
4382 !message.contains(GATE_TOO_LARGE_MARKER),
4383 "the entry gate must not be reported as a size trip"
4384 );
4385 assert!(message.contains(&SIZE_WALK_MAX_ENTRIES.to_string()));
4386 assert!(!message.contains("max_workspace_gb"));
4387 }
4388
4389 #[test]
4390 fn estimate_workspace_size_bounded_skips_builtin_excluded_dirs() {
4391 let tmp = tempdir().unwrap();
4392 let workspace = tmp.path().join("workspace");
4393 std::fs::create_dir_all(workspace.join("node_modules")).unwrap();
4394 std::fs::create_dir_all(workspace.join("target")).unwrap();
4395 std::fs::create_dir_all(workspace.join("src")).unwrap();
4396 // 2 MB of "build output" in excluded dirs — must not count toward
4397 // the cap.
4398 std::fs::write(workspace.join("node_modules/big.bin"), vec![0u8; 1_000_000]).unwrap();
4399 std::fs::write(workspace.join("target/big.bin"), vec![0u8; 1_000_000]).unwrap();
4400 std::fs::write(workspace.join("src/lib.rs"), b"// real source").unwrap();
4401 let total = estimate_workspace_size_bounded(&workspace, 500_000, SIZE_WALK_MAX_ENTRIES)
4402 .expect("walk must succeed since real source is tiny");
4403 assert!(
4404 total < 1_000,
4405 "total ({total}) must reflect only src/, not node_modules/ or target/"
4406 );
4407 }
4408
4409 #[test]
4410 fn estimate_workspace_size_bounded_cap_zero_disables_cap() {
4411 let tmp = tempdir().unwrap();
4412 let workspace = tmp.path().join("workspace");
4413 std::fs::create_dir_all(&workspace).unwrap();
4414 // 10 KB file — would trip a 1 KB cap, but cap=0 means no cap.
4415 std::fs::write(workspace.join("big.bin"), vec![0u8; 10 * 1024]).unwrap();
4416 let total = estimate_workspace_size_bounded(&workspace, 0, SIZE_WALK_MAX_ENTRIES)
4417 .expect("cap=0 must always return a total");
4418 assert!(
4419 total >= 10 * 1024,
4420 "total ({total}) must include the 10 KB file when cap is disabled"
4421 );
4422 }
4423
4424 /// The entry ceiling is the bound that no test could reach before
4425 /// `max_entries` became a parameter: 200,000 inodes per run is not a
4426 /// price a unit test should pay. A byte-cheap workspace must still be
4427 /// refused, and refused as the *entry* gate — reporting `TooLarge` here
4428 /// would offer `max_workspace_gb` as a remedy that cannot lift it.
4429 #[test]
4430 fn entry_ceiling_refuses_a_byte_cheap_workspace_with_too_many_entries() {
4431 let tmp = tempdir().unwrap();
4432 let workspace = tmp.path().join("workspace");
4433 std::fs::create_dir_all(&workspace).unwrap();
4434 for i in 0..10 {
4435 std::fs::write(workspace.join(format!("f{i}.txt")), b"x").unwrap();
4436 }
4437 assert_eq!(
4438 estimate_workspace_size_bounded(&workspace, 10_000_000, 3),
4439 Err(WorkspaceGate::TooManyEntries),
4440 "ten tiny files under a 10 MB cap must trip the entry bound, not the byte cap"
4441 );
4442 }
4443
4444 /// The invariant documented on `WorkspaceGate::TooManyEntries` and on the
4445 /// estimator: `max_workspace_gb = 0` opts out of the byte cap only. A
4446 /// future "if `cap_bytes == 0`, skip the walk" shortcut would satisfy
4447 /// every other test here and silently delete the ceiling that exists to
4448 /// stop a multi-minute `git add -A`.
4449 #[test]
4450 fn cap_zero_does_not_lift_the_entry_ceiling() {
4451 let tmp = tempdir().unwrap();
4452 let workspace = tmp.path().join("workspace");
4453 std::fs::create_dir_all(&workspace).unwrap();
4454 for i in 0..10 {
4455 std::fs::write(workspace.join(format!("f{i}.txt")), b"x").unwrap();
4456 }
4457 assert_eq!(
4458 estimate_workspace_size_bounded(&workspace, 0, 3),
4459 Err(WorkspaceGate::TooManyEntries),
4460 "cap_bytes = 0 disables the byte cap, never the entry ceiling"
4461 );
4462 }
4463
4464 /// `ignore` disables gitignore matching entirely when no ancestor holds a
4465 /// `.git` (`require_git` defaults to true), but the snapshot's own
4466 /// `git add -A --work-tree <workspace>` reads `.gitignore` either way. A
4467 /// non-git workspace was therefore measured on content that would never
4468 /// be staged — and then told to fix it by editing `.gitignore`.
4469 ///
4470 /// Deliberately creates no `.git`: the point is the non-repo case.
4471 #[test]
4472 fn gitignored_content_is_excluded_outside_a_git_repo() {
4473 let tmp = tempdir().unwrap();
4474 let workspace = tmp.path().join("workspace");
4475 std::fs::create_dir_all(workspace.join("src")).unwrap();
4476 std::fs::write(workspace.join(".gitignore"), "big.bin\n").unwrap();
4477 std::fs::write(workspace.join("big.bin"), vec![0u8; 1_000_000]).unwrap();
4478 std::fs::write(workspace.join("src/lib.rs"), b"// real source").unwrap();
4479 assert!(
4480 !workspace.join(".git").exists(),
4481 "this test is only meaningful outside a git repo"
4482 );
4483 let total = estimate_workspace_size_bounded(&workspace, 500_000, SIZE_WALK_MAX_ENTRIES)
4484 .expect("the only large file is gitignored, so the walk must fit under the cap");
4485 assert!(
4486 total < 1_000,
4487 "total ({total}) must exclude the gitignored 1 MB file"
4488 );
4489 }
4490
4491 #[test]
4492 fn open_or_init_with_cap_rejects_oversized_workspace() {
4493 let tmp = tempdir().unwrap();
4494 let workspace = tmp.path().join("workspace");
4495 std::fs::create_dir_all(&workspace).unwrap();
4496 let _home = scoped_home(tmp.path());
4497 // Drop a 4 KB file under a 1 KB cap.
4498 std::fs::write(workspace.join("big.bin"), vec![0u8; 4096]).unwrap();
4499 let outcome = SnapshotRepo::open_or_init_with_cap(&workspace, 1024);
4500 let err = match outcome {
4501 Ok(_) => panic!("oversized workspace must fail open_or_init_with_cap"),
4502 Err(e) => e,
4503 };
4504 let msg = err.to_string();
4505 assert!(
4506 msg.contains(GATE_TOO_LARGE_MARKER),
4507 "error must call out the size cap; got: {msg}"
4508 );
4509 let named_owned = workspace.display().to_string();
4510 let named = named_owned
4511 .strip_prefix(r"\\?\")
4512 .or_else(|| named_owned.strip_prefix("//?/"))
4513 .unwrap_or(named_owned.as_str());
4514 assert!(
4515 msg.contains(named),
4516 "error must name the workspace it refused; got: {msg}"
4517 );
4518 // The remedy belongs to the localized notice. Repeating it here is
4519 // what produced the doubled, three-line warning users saw.
4520 assert!(
4521 !msg.contains("max_workspace_gb"),
4522 "gate error must not carry its own remedy copy; got: {msg}"
4523 );
4524 }
4525
4526 #[test]
4527 fn open_or_init_with_cap_zero_disables_size_check() {
4528 let tmp = tempdir().unwrap();
4529 let workspace = tmp.path().join("workspace");
4530 std::fs::create_dir_all(&workspace).unwrap();
4531 let _home = scoped_home(tmp.path());
4532 // 4 KB file but cap=0 → should still succeed.
4533 std::fs::write(workspace.join("big.bin"), vec![0u8; 4096]).unwrap();
4534 let repo = SnapshotRepo::open_or_init_with_cap(&workspace, 0)
4535 .expect("cap=0 must skip the size check");
4536 let id = repo
4537 .snapshot("pre-turn:1")
4538 .expect("snapshot under disabled cap");
4539 assert_eq!(id.as_str().len(), 40);
4540 }
4541
4542 #[test]
4543 fn session_tagged_snapshot_round_trips_through_list() {
4544 let tmp = tempdir().unwrap();
4545 let (repo, _home) = make_repo(tmp.path());
4546 std::fs::write(repo.work_tree().join("a.txt"), b"x").unwrap();
4547
4548 repo.snapshot_with_session("pre-turn:1", Some("sess-42"))
4549 .expect("snapshot with session");
4550
4551 let list = repo.list(10).expect("list");
4552 assert_eq!(list.len(), 1);
4553 // The visible label stays clean; the session id is decoded separately.
4554 assert_eq!(list[0].label, "pre-turn:1");
4555 assert_eq!(list[0].session_id.as_deref(), Some("sess-42"));
4556 }
4557
4558 #[test]
4559 fn untagged_snapshot_decodes_without_session() {
4560 let tmp = tempdir().unwrap();
4561 let (repo, _home) = make_repo(tmp.path());
4562 std::fs::write(repo.work_tree().join("a.txt"), b"x").unwrap();
4563
4564 repo.snapshot("pre-turn:1").expect("snapshot");
4565
4566 let list = repo.list(10).expect("list");
4567 assert_eq!(list.len(), 1);
4568 assert_eq!(list[0].label, "pre-turn:1");
4569 assert_eq!(list[0].session_id, None);
4570 }
4571
4572 /// A burst of per-tool snapshots larger than the count cap never pushes
4573 /// out the turn boundaries: the running turn's `pre-turn:` restore point
4574 /// survives its own 50-write turn and another thread's burst.
4575 #[test]
4576 fn prune_keep_last_n_retains_turn_boundaries_through_a_tool_burst() {
4577 let tmp = tempdir().unwrap();
4578 let (repo, _home) = make_repo(tmp.path());
4579 let file = repo.work_tree().join("a.txt");
4580 std::fs::write(&file, "v0").unwrap();
4581 let older_post = repo
4582 .take_snapshot("post-turn:0", Some("thr_a"))
4583 .expect("snapshot");
4584 let pre = repo
4585 .take_snapshot("pre-turn:1", Some("thr_a"))
4586 .expect("snapshot");
4587 for i in 0..6 {
4588 std::fs::write(&file, format!("v{}", i + 1)).unwrap();
4589 repo.take_snapshot(&format!("tool:call-{i}"), Some("thr_a"))
4590 .expect("snapshot");
4591 repo.take_snapshot(&format!("post-tool:call-{i}"), Some("thr_a"))
4592 .expect("snapshot");
4593 }
4594 // 14 snapshots, cap 3: the newest three plus the (two) boundaries.
4595 let removed = repo.prune_keep_last_n(3).expect("prune");
4596 assert_eq!(removed, 9);
4597 let labels: Vec<String> = repo
4598 .list(usize::MAX)
4599 .unwrap()
4600 .into_iter()
4601 .map(|snapshot| snapshot.label)
4602 .collect();
4603 assert_eq!(
4604 labels,
4605 [
4606 "post-tool:call-5",
4607 "tool:call-5",
4608 "post-tool:call-4",
4609 "pre-turn:1",
4610 "post-turn:0",
4611 ]
4612 );
4613 let trees: Vec<SnapshotId> = repo
4614 .list(usize::MAX)
4615 .unwrap()
4616 .into_iter()
4617 .map(|snapshot| snapshot.tree)
4618 .collect();
4619 assert!(trees.contains(&pre.tree) && trees.contains(&older_post.tree));
4620
4621 // Boundaries are themselves capped at the same count.
4622 for i in 2..6 {
4623 repo.take_snapshot(&format!("pre-turn:{i}"), Some("thr_a"))
4624 .expect("snapshot");
4625 }
4626 repo.prune_keep_last_n(3).expect("prune");
4627 let boundaries = repo
4628 .list(usize::MAX)
4629 .unwrap()
4630 .into_iter()
4631 .filter(|snapshot| is_turn_boundary_label(&snapshot.label))
4632 .count();
4633 assert_eq!(boundaries, 3);
4634 }
4635
4636 #[test]
4637 fn prune_keep_last_n_preserves_session_tags() {
4638 let tmp = tempdir().unwrap();
4639 let (repo, _home) = make_repo(tmp.path());
4640 let file = repo.work_tree().join("a.txt");
4641
4642 // More snapshots than DEFAULT_MAX_SNAPSHOTS (50) so the survivor
4643 // chain is rebuilt as orphan commits — the path that previously
4644 // dropped the [sid=...] label prefix and turned every surviving
4645 // snapshot into a "legacy" (untagged) one.
4646 for i in 0..55 {
4647 std::fs::write(&file, format!("v{i}")).unwrap();
4648 repo.snapshot_with_session(&format!("pre-turn:{i}"), Some("sess-p"))
4649 .expect("tagged snapshot");
4650 }
4651
4652 let removed = repo.prune_keep_last_n(50).expect("prune");
4653 assert!(removed > 0, "expected prune to drop older snapshots");
4654
4655 let list = repo.list(usize::MAX).expect("list");
4656 assert_eq!(list.len(), 50);
4657 assert!(
4658 list.iter()
4659 .all(|s| s.session_id.as_deref() == Some("sess-p")),
4660 "prune must preserve [sid=...] prefixes; got untagged survivors"
4661 );
4662 }
4663
4664 #[test]
4665 fn tagged_and_untagged_snapshots_coexist_in_one_chain() {
4666 let tmp = tempdir().unwrap();
4667 let (repo, _home) = make_repo(tmp.path());
4668 std::fs::write(repo.work_tree().join("a.txt"), b"v1").unwrap();
4669
4670 // Legacy untagged snapshot, then a session-tagged one.
4671 repo.snapshot("pre-turn:1").expect("legacy snapshot");
4672 std::fs::write(repo.work_tree().join("a.txt"), b"v2").unwrap();
4673 repo.snapshot_with_session("pre-turn:1", Some("sess-a"))
4674 .expect("tagged snapshot");
4675
4676 let list = repo.list(10).expect("list");
4677 assert_eq!(list.len(), 2);
4678 // Newest first.
4679 assert_eq!(list[0].session_id.as_deref(), Some("sess-a"));
4680 assert_eq!(list[1].session_id, None);
4681 assert_eq!(list[1].label, "pre-turn:1");
4682 }
4683
4684 #[test]
4685 fn run_git_drains_output_larger_than_the_pipe_buffer() {
4686 let tmp = tempdir().unwrap();
4687 let (repo, _home) = make_repo(tmp.path());
4688 // ~5000 paths is well past the 64 KiB OS pipe buffer: output of this
4689 // size is only reachable when the pipes are drained while the child
4690 // runs. A bounded read that only starts after the child exits would
4691 // deadlock the child on its own output and die at the command
4692 // timeout instead of returning the tree listing.
4693 for i in 0..5000 {
4694 std::fs::write(repo.work_tree().join(format!("file_{i:05}.txt")), b"x").unwrap();
4695 }
4696 let id = repo.snapshot("large-output").expect("snapshot");
4697 let paths = repo
4698 .tree_paths(id.as_str())
4699 .expect("tree_paths must drain output instead of timing out");
4700 assert!(
4701 paths.len() >= 5000,
4702 "expected every file in the tree listing, got {}",
4703 paths.len()
4704 );
4705 }
4706 }
4707
4708 /// The bounded-git tests drive the core with real children, so they need a
4709 /// POSIX shell; the timeout pin also depends on wall-clock behavior.
4710 #[cfg(all(test, unix))]
4711 mod bounded_git_tests {
4712 use super::*;
4713
4714 #[test]
4715 fn bounded_git_times_out_a_wedged_child_and_reports_timed_out() {
4716 // A child that never exits stands in for a wedged git (stalled
4717 // mount, hung hook): the call must come back with a TimedOut error
4718 // promptly instead of blocking the turn pipeline forever.
4719 let started = std::time::Instant::now();
4720 let mut wedged = std::process::Command::new("sh");
4721 wedged.arg("-c").arg("sleep 30");
4722 let err = run_bounded_git_with_timeout(&mut wedged, "sleep", Duration::from_millis(250))
4723 .expect_err("a wedged child must hit the bound");
4724 assert_eq!(err.kind(), io::ErrorKind::TimedOut);
4725 assert!(
4726 err.to_string().contains("timed out after"),
4727 "the error must report the bound; got {err}"
4728 );
4729 let elapsed = started.elapsed();
4730 assert!(
4731 elapsed < Duration::from_secs(15),
4732 "the bound must be enforced promptly; took {elapsed:?}"
4733 );
4734 }
4735
4736 #[test]
4737 fn bounded_git_returns_promptly_when_a_grandchild_holds_the_pipes() {
4738 // The direct child (sh) exits after the echo; the backgrounded
4739 // sleep inherits both pipes and holds them for 30s. The call must
4740 // still return promptly with the output captured before the grace,
4741 // annotated on stderr — not wait out the grandchild.
4742 let started = std::time::Instant::now();
4743 let mut sh = std::process::Command::new("sh");
4744 sh.arg("-c").arg("echo bounded-git-grandchild; sleep 30 &");
4745 let output = run_bounded_git(&mut sh, "sh").expect("sh must succeed");
4746 let elapsed = started.elapsed();
4747 assert!(output.status.success());
4748 assert!(
4749 String::from_utf8_lossy(&output.stdout).contains("bounded-git-grandchild"),
4750 "output captured before the grace must survive: {:?}",
4751 output.stdout
4752 );
4753 assert!(
4754 String::from_utf8_lossy(&output.stderr)
4755 .contains("git output pipes did not close after git exited"),
4756 "the partial-output note must explain the early return: {:?}",
4757 output.stderr
4758 );
4759 assert!(
4760 elapsed < Duration::from_secs(15),
4761 "a pipe-holding grandchild must not hold the call past the grace; took {elapsed:?}"
4762 );
4763 }
4764 }
4765
4765 lines RUST