| 1 | //! Side-git repository wrapper for workspace snapshots. |
| 2 | //! |
| 3 | //! `SnapshotRepo` shells out to the system `git` binary (we deliberately |
| 4 | //! avoid `git2` to dodge its LGPL surface). The two paths that matter: |
| 5 | //! |
| 6 | //! - `git_dir` → `<snapshot state dir>/<project_hash>/<worktree_hash>/.git` |
| 7 | //! - `work_tree` → the user's actual workspace |
| 8 | //! |
| 9 | //! Every git invocation passes both `--git-dir` AND `--work-tree`. That is |
| 10 | //! the single biggest safety mechanism: it guarantees we never accidentally |
| 11 | //! mutate the user's own `.git` directory. If git can't find the side |
| 12 | //! repo, the command fails fast instead of falling back to "current |
| 13 | //! directory". |
| 14 | |
| 15 | use std::cell::RefCell; |
| 16 | use std::collections::{HashMap, HashSet}; |
| 17 | use std::io; |
| 18 | use std::path::{Component, Path, PathBuf}; |
| 19 | use std::process::Output; |
| 20 | use std::time::{Duration, SystemTime, UNIX_EPOCH}; |
| 21 | |
| 22 | use wait_timeout::ChildExt as _; |
| 23 | |
| 24 | use crate::dependencies::ExternalTool; |
| 25 | |
| 26 | use super::paths::{ensure_snapshot_dir, snapshot_git_dir}; |
| 27 | |
| 28 | /// Identifier for a snapshot — the underlying git commit id. |
| 29 | /// |
| 30 | /// The field is private: [`SnapshotId::parse`] is the only way to build one, |
| 31 | /// so every value handed to `git` as a revision is a full SHA-1 or SHA-256 |
| 32 | /// hex object id and can never be read as an option or a revision expression. |
| 33 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 34 | pub struct SnapshotId(String); |
| 35 | |
| 36 | impl SnapshotId { |
| 37 | /// Accept exactly a full hex object id: 40 (SHA-1) or 64 (SHA-256) |
| 38 | /// ASCII hex digits. Anything else is `InvalidInput`. |
| 39 | pub fn parse(id: &str) -> io::Result<Self> { |
| 40 | if Self::is_well_formed(id) { |
| 41 | Ok(Self(id.to_string())) |
| 42 | } else { |
| 43 | Err(io::Error::new( |
| 44 | io::ErrorKind::InvalidInput, |
| 45 | "snapshot id must be a full hexadecimal commit id", |
| 46 | )) |
| 47 | } |
| 48 | } |
| 49 | |
| 50 | /// Whether `id` would be accepted by [`SnapshotId::parse`]. |
| 51 | pub fn is_well_formed(id: &str) -> bool { |
| 52 | matches!(id.len(), 40 | 64) && id.bytes().all(|b| b.is_ascii_hexdigit()) |
| 53 | } |
| 54 | |
| 55 | /// Take the id string out. |
| 56 | pub fn into_string(self) -> String { |
| 57 | self.0 |
| 58 | } |
| 59 | |
| 60 | /// Borrow the SHA as a string slice. |
| 61 | pub fn as_str(&self) -> &str { |
| 62 | &self.0 |
| 63 | } |
| 64 | } |
| 65 | |
| 66 | /// A single snapshot record (one row in `git log`). |
| 67 | #[derive(Debug, Clone)] |
| 68 | pub struct Snapshot { |
| 69 | /// Commit SHA inside the side repo. |
| 70 | pub id: SnapshotId, |
| 71 | /// Subject line — the label passed to [`SnapshotRepo::snapshot`]. |
| 72 | pub label: String, |
| 73 | /// Root tree of the snapshot commit. Unlike [`Self::id`] it survives the |
| 74 | /// survivor-chain rebuild every prune performs (the rebuild re-commits |
| 75 | /// the same tree under a new commit id), so it is the durable identity a |
| 76 | /// recorded restore point is resolved by. |
| 77 | pub tree: SnapshotId, |
| 78 | /// Author timestamp (Unix seconds). |
| 79 | pub timestamp: i64, |
| 80 | /// Session this snapshot belongs to, when recorded (encoded as a |
| 81 | /// `[sid=...] ` label prefix). `None` for legacy snapshots taken |
| 82 | /// before session tagging existed. |
| 83 | pub session_id: Option<String>, |
| 84 | } |
| 85 | |
| 86 | /// One path that differs between two snapshots |
| 87 | /// ([`SnapshotRepo::path_changes_between`]). |
| 88 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 89 | pub struct SnapshotPathChange { |
| 90 | /// Workspace-relative path, as git names it. |
| 91 | pub path: String, |
| 92 | /// git's status letter: `A` added, `D` deleted, `M` modified, `T` type |
| 93 | /// changed. |
| 94 | pub status: char, |
| 95 | /// Lines added and removed; `None` for a binary file. |
| 96 | pub added: Option<u64>, |
| 97 | pub removed: Option<u64>, |
| 98 | } |
| 99 | |
| 100 | /// What a file-scoped restore did to one path, relative to the working tree |
| 101 | /// it was applied to. |
| 102 | #[derive(Debug, Clone, Copy, PartialEq, Eq)] |
| 103 | pub enum PathRestoreAction { |
| 104 | /// Both the snapshot and the working tree had the path; its content came |
| 105 | /// back from the snapshot. |
| 106 | Modified, |
| 107 | /// The snapshot had the path and the working tree no longer did, so the |
| 108 | /// restore recreated the file. |
| 109 | Recreated, |
| 110 | /// The working tree had the path and the snapshot did not, so the restore |
| 111 | /// removed the file. |
| 112 | Removed, |
| 113 | } |
| 114 | |
| 115 | impl PathRestoreAction { |
| 116 | /// Stable wire name, also used by the runtime API response. |
| 117 | pub fn as_str(self) -> &'static str { |
| 118 | match self { |
| 119 | Self::Modified => "modified", |
| 120 | Self::Recreated => "recreated", |
| 121 | Self::Removed => "removed", |
| 122 | } |
| 123 | } |
| 124 | } |
| 125 | |
| 126 | /// The safety snapshot a path restore writes first, or one already taken. |
| 127 | enum RestoreBackup<'a> { |
| 128 | Take(&'a str), |
| 129 | Existing(&'a SnapshotId), |
| 130 | } |
| 131 | |
| 132 | /// Report of what [`SnapshotRepo::restore_paths`] did to one path. |
| 133 | #[derive(Debug, Clone)] |
| 134 | pub struct PathRestoreOutcome { |
| 135 | /// Workspace-relative path that was restored. |
| 136 | pub path: PathBuf, |
| 137 | /// How the working tree changed. |
| 138 | pub action: PathRestoreAction, |
| 139 | } |
| 140 | |
| 141 | /// A snapshot as it was just written: its commit and the root tree it holds. |
| 142 | /// |
| 143 | /// The commit id can be rewritten by the prune that follows every snapshot |
| 144 | /// once the repo holds more than [`crate::snapshot::DEFAULT_MAX_SNAPSHOTS`] |
| 145 | /// (see `rebuild_survivor_chain`); the tree id cannot, because a tree is |
| 146 | /// content-addressed and the rebuild re-commits the same tree. |
| 147 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 148 | pub struct TakenSnapshot { |
| 149 | pub id: SnapshotId, |
| 150 | pub tree: SnapshotId, |
| 151 | } |
| 152 | |
| 153 | /// Wrapper around the per-workspace side-git repo. |
| 154 | pub struct SnapshotRepo { |
| 155 | git_dir: PathBuf, |
| 156 | work_tree: PathBuf, |
| 157 | } |
| 158 | |
| 159 | const STALE_TMP_PACK_AGE: Duration = Duration::from_secs(60 * 60); |
| 160 | |
| 161 | /// Advisory lock file inside the side repo's `.git`. Every session, turn and |
| 162 | /// sub-agent working in one workspace shares one side repo, and its index, |
| 163 | /// HEAD and object store are only consistent when snapshot, restore, prune |
| 164 | /// and gc run one at a time: a gc in one session otherwise deleted another |
| 165 | /// session's new, not yet referenced commit and left HEAD on a missing |
| 166 | /// object. |
| 167 | const SNAPSHOT_LOCK_FILE: &str = "codewhale-snapshot.lock"; |
| 168 | |
| 169 | /// Longest a snapshot, restore or prune waits for another writer's lock |
| 170 | /// before failing (the turn then shows the snapshot-failure notice). |
| 171 | const SNAPSHOT_LOCK_WAIT: Duration = Duration::from_secs(30); |
| 172 | |
| 173 | /// How often a waiting writer retries the lock. |
| 174 | const SNAPSHOT_LOCK_POLL: Duration = Duration::from_millis(25); |
| 175 | |
| 176 | thread_local! { |
| 177 | /// Side repos whose write lock this thread already holds, so a locked |
| 178 | /// operation that calls another (restore takes a safety snapshot, a |
| 179 | /// snapshot runs the size prune) does not wait on its own lock. |
| 180 | static HELD_SNAPSHOT_LOCKS: RefCell<Vec<PathBuf>> = const { RefCell::new(Vec::new()) }; |
| 181 | } |
| 182 | |
| 183 | /// Maximum total snapshot storage in megabytes before pruning kicks in at |
| 184 | /// snapshot time. Keeps the side repo from blowing up the user's disk during |
| 185 | /// long-running or high-churn sessions (#1112). |
| 186 | const MAX_SNAPSHOT_SIZE_MB: u64 = 500; |
| 187 | |
| 188 | const BYTES_PER_MB: u64 = 1024 * 1024; |
| 189 | |
| 190 | /// Grace margin below `MAX_SNAPSHOT_SIZE_MB` used as the prune target |
| 191 | /// so the repo doesn't hit the limit again one snapshot later. |
| 192 | const PRUNE_TARGET_MB: u64 = 400; |
| 193 | |
| 194 | /// Default workspace-size ceiling above which snapshots self-disable |
| 195 | /// on first use (2 GB of non-excluded content). Reports from users with |
| 196 | /// multi-hundred-GB project directories — datasets, model weights, |
| 197 | /// docker image dumps that fall outside the built-in excludes — |
| 198 | /// surfaced that `git add -A` on first init would hang the TUI for |
| 199 | /// minutes-to-hours while indexing the workspace. Snapshots are a |
| 200 | /// rollback safety net, not a backup tool; bailing out on workspaces |
| 201 | /// that big is the right tradeoff. Users with legitimate large |
| 202 | /// monorepos can raise `[snapshots] max_workspace_gb` (or set it to |
| 203 | /// `0` to disable the cap entirely). |
| 204 | pub const DEFAULT_MAX_WORKSPACE_BYTES_FOR_SNAPSHOT: u64 = 2 * 1024 * 1024 * 1024; |
| 205 | |
| 206 | /// Hard cap on the number of file entries the bounded size estimator |
| 207 | /// will inspect before declaring the workspace "too large". Protects |
| 208 | /// against a workspace with millions of tiny files (no individual |
| 209 | /// file is large, but `git add -A` would still take forever). |
| 210 | pub const SIZE_WALK_MAX_ENTRIES: usize = 200_000; |
| 211 | |
| 212 | /// Which snapshot gate refused a workspace. The recovery differs per gate — |
| 213 | /// raising `[snapshots] max_workspace_gb` lifts only [`WorkspaceGate::TooLarge`] |
| 214 | /// — so callers must not offer one gate's remedy for another's failure. |
| 215 | #[derive(Debug, Clone, Copy, PartialEq, Eq)] |
| 216 | pub enum WorkspaceGate { |
| 217 | /// Snapshot-eligible content exceeds the configured byte cap. |
| 218 | TooLarge, |
| 219 | /// The bounded walk hit [`SIZE_WALK_MAX_ENTRIES`]. This bound is |
| 220 | /// independent of the byte cap: `max_workspace_gb = 0` does not lift it. |
| 221 | TooManyEntries, |
| 222 | } |
| 223 | |
| 224 | /// Leading text of the `io::Error` each gate produces. `core::turn` matches on |
| 225 | /// these to pick the right consequence/recovery notice, so they are one |
| 226 | /// declaration shared by producer and matcher rather than two literals. |
| 227 | pub const GATE_TOO_LARGE_MARKER: &str = "workspace too large for snapshots"; |
| 228 | pub const GATE_TOO_MANY_ENTRIES_MARKER: &str = "workspace has too many files for snapshots"; |
| 229 | pub const GATE_UNSAFE_LOCATION_MARKER: &str = "workspace snapshots are disabled"; |
| 230 | |
| 231 | /// Display a workspace path in gate diagnostics. The diagnostic names the |
| 232 | /// path the caller passed in — canonicalization is for filesystem and |
| 233 | /// security logic, not for the message. On Windows `Path::canonicalize` |
| 234 | /// rewrites more than the verbatim (`\\?\`) prefix (case, 8.3 names), so a |
| 235 | /// canonical spelling can never be relied on to match what users name; the |
| 236 | /// verbatim prefix is still stripped when present for readability. |
| 237 | fn display_workspace_for_gate(workspace: &Path) -> String { |
| 238 | let raw = workspace.display().to_string(); |
| 239 | raw.strip_prefix(r"\\?\") |
| 240 | .or_else(|| raw.strip_prefix("//?/")) |
| 241 | .unwrap_or(&raw) |
| 242 | .to_string() |
| 243 | } |
| 244 | |
| 245 | impl WorkspaceGate { |
| 246 | /// One-line English diagnostic for logs, `/undo`, and the gate matcher. |
| 247 | /// The user-facing consequence and recovery are localized by the notice |
| 248 | /// surfaces; this string must not restate them. |
| 249 | fn describe(self, cap_bytes: u64, workspace: &Path) -> String { |
| 250 | let workspace = display_workspace_for_gate(workspace); |
| 251 | match self { |
| 252 | Self::TooLarge => format!( |
| 253 | "{GATE_TOO_LARGE_MARKER}: over {} bytes of snapshot-eligible content in {workspace}", |
| 254 | cap_bytes, |
| 255 | ), |
| 256 | Self::TooManyEntries => format!( |
| 257 | "{GATE_TOO_MANY_ENTRIES_MARKER}: over {SIZE_WALK_MAX_ENTRIES} snapshot-eligible entries in {workspace}" |
| 258 | ), |
| 259 | } |
| 260 | } |
| 261 | } |
| 262 | |
| 263 | /// Top-level directory and extension patterns that the snapshot path |
| 264 | /// already excludes via `BUILTIN_EXCLUDES`. The estimator skips these |
| 265 | /// up front so the size walk reflects what would actually land in the |
| 266 | /// snapshot commit. Kept narrow to common build-output dirs — anything |
| 267 | /// else falls back to the `.gitignore` filter. |
| 268 | const SIZE_WALK_SKIP_DIRS: &[&str] = &[ |
| 269 | "node_modules", |
| 270 | "target", |
| 271 | "dist", |
| 272 | "build", |
| 273 | ".build", |
| 274 | ".next", |
| 275 | ".nuxt", |
| 276 | ".svelte-kit", |
| 277 | ".turbo", |
| 278 | ".parcel-cache", |
| 279 | "vendor", |
| 280 | ".cargo", |
| 281 | ".rustup", |
| 282 | ".npm", |
| 283 | ".bun", |
| 284 | ".yarn", |
| 285 | ".pnpm-store", |
| 286 | ".cache", |
| 287 | ".venv", |
| 288 | "venv", |
| 289 | ".tox", |
| 290 | "__pycache__", |
| 291 | ".mypy_cache", |
| 292 | ".pytest_cache", |
| 293 | ".ruff_cache", |
| 294 | ".gradle", |
| 295 | ".m2", |
| 296 | ".local", |
| 297 | ".git", |
| 298 | ]; |
| 299 | |
| 300 | const BUILTIN_EXCLUDES: &str = "\ |
| 301 | # CodeWhale built-in snapshot exclusions |
| 302 | node_modules/ |
| 303 | target/ |
| 304 | dist/ |
| 305 | build/ |
| 306 | .build/ |
| 307 | .next/ |
| 308 | .nuxt/ |
| 309 | .svelte-kit/ |
| 310 | .turbo/ |
| 311 | .parcel-cache/ |
| 312 | vendor/ |
| 313 | .cargo/ |
| 314 | .rustup/ |
| 315 | .npm/ |
| 316 | .bun/ |
| 317 | .yarn/ |
| 318 | .pnpm-store/ |
| 319 | .cache/ |
| 320 | .venv/ |
| 321 | venv/ |
| 322 | .tox/ |
| 323 | __pycache__/ |
| 324 | *.pyc |
| 325 | .mypy_cache/ |
| 326 | .pytest_cache/ |
| 327 | .ruff_cache/ |
| 328 | .gradle/ |
| 329 | .m2/ |
| 330 | .local/ |
| 331 | .DS_Store |
| 332 | |
| 333 | # Binary and generated artifacts. Snapshots are source rollback checkpoints, |
| 334 | # not a full binary backup; keeping these out avoids side-repo bloat. |
| 335 | *.exe |
| 336 | *.dll |
| 337 | *.so |
| 338 | *.dylib |
| 339 | *.wasm |
| 340 | *.o |
| 341 | *.obj |
| 342 | *.class |
| 343 | *.pdb |
| 344 | *.dSYM |
| 345 | *.zip |
| 346 | *.tar |
| 347 | *.tar.gz |
| 348 | *.tgz |
| 349 | *.tar.bz2 |
| 350 | *.tar.xz |
| 351 | *.7z |
| 352 | *.rar |
| 353 | *.iso |
| 354 | *.dmg |
| 355 | *.bin |
| 356 | *.mp4 |
| 357 | *.mov |
| 358 | *.mkv |
| 359 | *.avi |
| 360 | *.webm |
| 361 | *.mp3 |
| 362 | *.wav |
| 363 | *.flac |
| 364 | *.aac |
| 365 | "; |
| 366 | |
| 367 | impl SnapshotRepo { |
| 368 | /// Open an existing snapshot repo for `workspace` without creating or |
| 369 | /// initializing anything on disk. |
| 370 | /// |
| 371 | /// This is useful for read-only UI surfaces that want to report checkpoint |
| 372 | /// availability without paying the first-init size walk or surprising the |
| 373 | /// user by creating a side repo from a view action. |
| 374 | pub fn open_existing(workspace: &Path) -> io::Result<Option<Self>> { |
| 375 | let work_tree = workspace |
| 376 | .canonicalize() |
| 377 | .unwrap_or_else(|_| workspace.to_path_buf()); |
| 378 | let git_dir = snapshot_git_dir(&work_tree)?; |
| 379 | if !git_dir.exists() || !git_dir.join("HEAD").exists() { |
| 380 | return Ok(None); |
| 381 | } |
| 382 | Ok(Some(Self { git_dir, work_tree })) |
| 383 | } |
| 384 | |
| 385 | /// Open or initialize the snapshot repo for `workspace`. |
| 386 | /// |
| 387 | /// On first use this: |
| 388 | /// 1. Creates the `.git` dir under the resolved snapshot store. |
| 389 | /// 2. Runs `git init --bare=false --quiet`. |
| 390 | /// 3. Sets a fixed `user.name` / `user.email` so commits don't pick up |
| 391 | /// the user's global git identity (we don't want our snapshots to |
| 392 | /// look like they came from the user). |
| 393 | pub fn open_or_init(workspace: &Path) -> io::Result<Self> { |
| 394 | Self::open_or_init_with_cap(workspace, DEFAULT_MAX_WORKSPACE_BYTES_FOR_SNAPSHOT) |
| 395 | } |
| 396 | |
| 397 | /// Variant of [`Self::open_or_init`] that accepts an explicit |
| 398 | /// workspace-size cap. `cap_bytes = 0` disables the cap entirely |
| 399 | /// (always snapshot, regardless of size). |
| 400 | /// |
| 401 | /// When the workspace exceeds the cap and the side repo hasn't |
| 402 | /// been initialized yet, returns `Err(InvalidInput)` with a |
| 403 | /// "workspace too large" reason. Subsequent calls (after the user |
| 404 | /// shrinks the workspace or raises the cap via config) succeed. |
| 405 | pub fn open_or_init_with_cap(workspace: &Path, cap_bytes: u64) -> io::Result<Self> { |
| 406 | let work_tree = workspace |
| 407 | .canonicalize() |
| 408 | .unwrap_or_else(|_| workspace.to_path_buf()); |
| 409 | if let Some(reason) = unsafe_workspace_snapshot_reason( |
| 410 | &work_tree, |
| 411 | crate::config::effective_home_dir().as_deref(), |
| 412 | ) { |
| 413 | return Err(io::Error::new( |
| 414 | io::ErrorKind::InvalidInput, |
| 415 | format!( |
| 416 | "{GATE_UNSAFE_LOCATION_MARKER} for {reason}: {}", |
| 417 | display_workspace_for_gate(workspace) |
| 418 | ), |
| 419 | )); |
| 420 | } |
| 421 | |
| 422 | let git_dir = ensure_snapshot_dir(&work_tree)?.join(".git"); |
| 423 | |
| 424 | // A `.git` without HEAD is an init that never finished (or only a |
| 425 | // peer's lock file): initialize it rather than use it half-made. |
| 426 | let needs_init = !git_dir.join("HEAD").exists(); |
| 427 | if needs_init { |
| 428 | // First-init size guard. Skipping this on subsequent opens |
| 429 | // is intentional: paying a workspace walk on every snapshot |
| 430 | // would defeat the purpose of the cap, and a workspace |
| 431 | // that fit on first init is allowed to grow within the |
| 432 | // existing repo's `MAX_SNAPSHOT_SIZE_MB` budget. Users on |
| 433 | // workspaces that grew past the cap mid-session get the |
| 434 | // existing aggressive-pruning path in `snapshot()`. |
| 435 | if let Err(gate) = |
| 436 | estimate_workspace_size_bounded(&work_tree, cap_bytes, SIZE_WALK_MAX_ENTRIES) |
| 437 | { |
| 438 | return Err(io::Error::new( |
| 439 | io::ErrorKind::InvalidInput, |
| 440 | gate.describe(cap_bytes, workspace), |
| 441 | )); |
| 442 | } |
| 443 | // The lock file lives in `.git`, so create it first; `git init` |
| 444 | // accepts an existing `.git` directory. |
| 445 | std::fs::create_dir_all(&git_dir)?; |
| 446 | } |
| 447 | let repo = Self { git_dir, work_tree }; |
| 448 | if needs_init { |
| 449 | // Two sessions opening a new workspace at once must not both run |
| 450 | // `git init` and the identity config: a config write that loses |
| 451 | // git's own config lock is ignored, and a snapshot committed |
| 452 | // before the identity lands would carry the user's. |
| 453 | repo.with_write_lock(|| { |
| 454 | if repo.git_dir.join("HEAD").exists() { |
| 455 | return Ok(()); |
| 456 | } |
| 457 | repo.init_side_repo() |
| 458 | })?; |
| 459 | } |
| 460 | |
| 461 | write_builtin_excludes(&repo.git_dir)?; |
| 462 | if let Err(err) = cleanup_stale_pack_temps(&repo.git_dir, STALE_TMP_PACK_AGE) { |
| 463 | tracing::debug!( |
| 464 | target: "snapshot", |
| 465 | "failed to clean stale snapshot tmp_pack files: {err}" |
| 466 | ); |
| 467 | } |
| 468 | Ok(repo) |
| 469 | } |
| 470 | |
| 471 | /// `git init` the side repo and pin its config. Runs under the write lock. |
| 472 | fn init_side_repo(&self) -> io::Result<()> { |
| 473 | let (git_dir, work_tree) = (&self.git_dir, &self.work_tree); |
| 474 | let parent = git_dir.parent().ok_or_else(|| { |
| 475 | io::Error::new(io::ErrorKind::InvalidInput, "snapshot dir has no parent") |
| 476 | })?; |
| 477 | // `git init` here uses the parent directory as the work tree |
| 478 | // and stores metadata in `.git`. We then continue to use |
| 479 | // explicit `--git-dir` / `--work-tree` flags for every other |
| 480 | // command so behaviour is invariant of cwd. |
| 481 | let mut git = |
| 482 | crate::dependencies::Git::command().ok_or_else(|| io_other("git not found on PATH"))?; |
| 483 | let init = git.arg("init").arg("--quiet").arg(parent); |
| 484 | let init = run_bounded_git(init, "init") |
| 485 | .map_err(|e| io_other(format!("failed to run git init: {e}")))?; |
| 486 | if !init.status.success() { |
| 487 | return Err(io_other(format!( |
| 488 | "git init failed: {}", |
| 489 | String::from_utf8_lossy(&init.stderr).trim() |
| 490 | ))); |
| 491 | } |
| 492 | |
| 493 | // Pin a stable identity so snapshot commits are recognisable |
| 494 | // and don't bleed into the user's git config. |
| 495 | let _ = run_git( |
| 496 | git_dir, |
| 497 | work_tree, |
| 498 | &["config", "user.name", "deepseek-snapshots"], |
| 499 | ); |
| 500 | let _ = run_git( |
| 501 | git_dir, |
| 502 | work_tree, |
| 503 | &["config", "user.email", "snapshots@codewhale.local"], |
| 504 | ); |
| 505 | // Don't auto-gc on every commit; we manage pruning ourselves. |
| 506 | let _ = run_git(git_dir, work_tree, &["config", "gc.auto", "0"]); |
| 507 | // Ignore CRLF rewriting — we want byte-for-byte fidelity. |
| 508 | let _ = run_git(git_dir, work_tree, &["config", "core.autocrlf", "false"]); |
| 509 | Ok(()) |
| 510 | } |
| 511 | |
| 512 | /// Take a snapshot of the current working tree. |
| 513 | /// |
| 514 | /// Internally: `git add -A`, `git write-tree`, `git commit-tree`, then |
| 515 | /// `git update-ref HEAD <commit>`. |
| 516 | /// `git add -A` honours the user's workspace ignore rules while staging |
| 517 | /// into the side repo's index. |
| 518 | /// |
| 519 | /// Before committing, checks whether the snapshot directory exceeds |
| 520 | /// [`MAX_SNAPSHOT_SIZE_MB`] and prunes the oldest snapshots if it does. |
| 521 | /// |
| 522 | /// Returns the snapshot's commit SHA. |
| 523 | #[allow(dead_code)] // convenience entry kept for tests and legacy callers; production writes go through snapshot_with_session |
| 524 | pub fn snapshot(&self, label: &str) -> io::Result<SnapshotId> { |
| 525 | self.snapshot_with_session(label, None) |
| 526 | } |
| 527 | |
| 528 | /// Take a snapshot, tagging it with the owning session id. |
| 529 | /// |
| 530 | /// The session id is encoded into the commit message as a `[sid=...] ` |
| 531 | /// label prefix. [`Self::list`] decodes it back into |
| 532 | /// [`Snapshot::session_id`] and strips the prefix from the visible |
| 533 | /// label, so existing listing surfaces keep showing the plain label. |
| 534 | /// Legacy snapshots taken through [`Self::snapshot`] carry no prefix |
| 535 | /// and decode with `session_id == None`. |
| 536 | pub fn snapshot_with_session( |
| 537 | &self, |
| 538 | label: &str, |
| 539 | session_id: Option<&str>, |
| 540 | ) -> io::Result<SnapshotId> { |
| 541 | self.take_snapshot(label, session_id).map(|taken| taken.id) |
| 542 | } |
| 543 | |
| 544 | /// [`Self::snapshot_with_session`], also reporting the root tree the |
| 545 | /// snapshot holds — the identity a recorded restore point keeps. |
| 546 | pub fn take_snapshot( |
| 547 | &self, |
| 548 | label: &str, |
| 549 | session_id: Option<&str>, |
| 550 | ) -> io::Result<TakenSnapshot> { |
| 551 | self.with_write_lock(|| self.take_snapshot_locked(label, session_id)) |
| 552 | } |
| 553 | |
| 554 | fn take_snapshot_locked( |
| 555 | &self, |
| 556 | label: &str, |
| 557 | session_id: Option<&str>, |
| 558 | ) -> io::Result<TakenSnapshot> { |
| 559 | // Guard against disk blowup (#1112): if the snapshot directory has |
| 560 | // grown beyond the limit, prune aggressively before adding more. |
| 561 | // When the prune actually destroys restore points the user is told |
| 562 | // once per workspace — losing undo history to a log line is the S5 |
| 563 | // failure mode (2026-08-04 snapshot hunt). |
| 564 | if let Ok(removed) = self.prune_size_pressure( |
| 565 | MAX_SNAPSHOT_SIZE_MB * BYTES_PER_MB, |
| 566 | PRUNE_TARGET_MB * BYTES_PER_MB, |
| 567 | ) && removed > 0 |
| 568 | { |
| 569 | notify_snapshot_history_pruned_once(&self.work_tree, removed); |
| 570 | } |
| 571 | // A HEAD that names a missing commit would make every `commit-tree |
| 572 | // -p` below fail ("is not a valid object"), silently ending undo. |
| 573 | self.repair_broken_head()?; |
| 574 | let parent = run_git( |
| 575 | &self.git_dir, |
| 576 | &self.work_tree, |
| 577 | &["rev-parse", "--verify", "--quiet", "HEAD^{commit}"], |
| 578 | )?; |
| 579 | let parent = parent |
| 580 | .status |
| 581 | .success() |
| 582 | .then(|| String::from_utf8_lossy(&parent.stdout).trim().to_string()) |
| 583 | .filter(|s| !s.is_empty()); |
| 584 | |
| 585 | self.stage_work_tree()?; |
| 586 | let (tree, sha) = match self.commit_staged_tree(parent.as_deref(), label, session_id) { |
| 587 | Ok(done) => done, |
| 588 | Err(first) => { |
| 589 | // An index naming objects that no longer exist (left by an |
| 590 | // interrupted or racing gc) breaks every later snapshot: |
| 591 | // `add -A` does not rehash files whose stat data is |
| 592 | // unchanged, and `write-tree` hands back the index's cached |
| 593 | // tree id even when that tree is gone, so `commit-tree` then |
| 594 | // fails. Rebuild the index from the work tree once. |
| 595 | let reset = run_git(&self.git_dir, &self.work_tree, &["read-tree", "--empty"])?; |
| 596 | if !reset.status.success() { |
| 597 | return Err(first); |
| 598 | } |
| 599 | self.stage_work_tree()?; |
| 600 | self.commit_staged_tree(parent.as_deref(), label, session_id)? |
| 601 | } |
| 602 | }; |
| 603 | |
| 604 | self.move_head(Some(&sha), parent.as_deref())?; |
| 605 | |
| 606 | let id = SnapshotId::parse(&sha).map_err(|_| { |
| 607 | io_other(format!( |
| 608 | "git commit-tree returned a malformed commit id: {sha:?}" |
| 609 | )) |
| 610 | })?; |
| 611 | let tree = SnapshotId::parse(&tree).map_err(|_| { |
| 612 | io_other(format!( |
| 613 | "git write-tree returned a malformed tree id: {tree:?}" |
| 614 | )) |
| 615 | })?; |
| 616 | Ok(TakenSnapshot { id, tree }) |
| 617 | } |
| 618 | |
| 619 | /// `write-tree` the staged index and `commit-tree` it onto `parent`, |
| 620 | /// returning the tree and commit ids. |
| 621 | fn commit_staged_tree( |
| 622 | &self, |
| 623 | parent: Option<&str>, |
| 624 | label: &str, |
| 625 | session_id: Option<&str>, |
| 626 | ) -> io::Result<(String, String)> { |
| 627 | let tree = run_git(&self.git_dir, &self.work_tree, &["write-tree"])?; |
| 628 | if !tree.status.success() { |
| 629 | return Err(io_other(format!( |
| 630 | "git write-tree failed: {}", |
| 631 | String::from_utf8_lossy(&tree.stderr).trim() |
| 632 | ))); |
| 633 | } |
| 634 | let tree = String::from_utf8_lossy(&tree.stdout).trim().to_string(); |
| 635 | |
| 636 | let mut args = vec!["commit-tree".to_string(), tree.clone()]; |
| 637 | if let Some(parent) = parent { |
| 638 | args.push("-p".to_string()); |
| 639 | args.push(parent.to_string()); |
| 640 | } |
| 641 | args.push("-m".to_string()); |
| 642 | args.push(Self::encode_session_label(label, session_id)); |
| 643 | let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect(); |
| 644 | |
| 645 | // `commit-tree` creates marker commits even when the tree matches its |
| 646 | // parent, and it does not run user/global commit hooks. |
| 647 | let commit = run_git(&self.git_dir, &self.work_tree, &arg_refs)?; |
| 648 | if !commit.status.success() { |
| 649 | return Err(io_other(format!( |
| 650 | "git commit-tree failed: {}", |
| 651 | String::from_utf8_lossy(&commit.stderr).trim() |
| 652 | ))); |
| 653 | } |
| 654 | let sha = String::from_utf8_lossy(&commit.stdout).trim().to_string(); |
| 655 | Ok((tree, sha)) |
| 656 | } |
| 657 | |
| 658 | /// Repair a side repo whose HEAD names a commit that no longer exists |
| 659 | /// (an interrupted gc or prune, a copied or partially deleted |
| 660 | /// `~/.codewhale/snapshots` directory). Left alone, every later snapshot |
| 661 | /// fails on `commit-tree -p <missing>` and /undo is dead without a word. |
| 662 | /// |
| 663 | /// The broken ref is deleted so the next snapshot starts a fresh history. |
| 664 | /// Restore points before the break cannot be recovered; the caller tells |
| 665 | /// the user. Returns `true` when a repair happened. |
| 666 | pub fn repair_broken_head(&self) -> io::Result<bool> { |
| 667 | self.with_write_lock(|| self.repair_broken_head_locked()) |
| 668 | } |
| 669 | |
| 670 | fn repair_broken_head_locked(&self) -> io::Result<bool> { |
| 671 | let commit = run_git( |
| 672 | &self.git_dir, |
| 673 | &self.work_tree, |
| 674 | &["rev-parse", "--verify", "--quiet", "HEAD^{commit}"], |
| 675 | )?; |
| 676 | if commit.status.success() { |
| 677 | return Ok(false); |
| 678 | } |
| 679 | // An unborn branch (fresh repo) names nothing: nothing to repair. |
| 680 | let named = run_git( |
| 681 | &self.git_dir, |
| 682 | &self.work_tree, |
| 683 | &["rev-parse", "--verify", "--quiet", "HEAD"], |
| 684 | )?; |
| 685 | if !named.status.success() { |
| 686 | return Ok(false); |
| 687 | } |
| 688 | let missing = String::from_utf8_lossy(&named.stdout).trim().to_string(); |
| 689 | // A lookup can fail for a moment while another session sharing this |
| 690 | // repo runs gc or repack. Only a commit that is really absent |
| 691 | // justifies touching HEAD: deleting it discards every restore point. |
| 692 | if self.is_commit(&missing)? { |
| 693 | return Ok(false); |
| 694 | } |
| 695 | self.reset_missing_head(&missing) |
| 696 | } |
| 697 | |
| 698 | /// Move HEAD off `missing`, a commit id it named when this repair looked. |
| 699 | /// Both moves are compare-and-swap against `missing`: a writer that does |
| 700 | /// not take the snapshot lock (an older build) may have published a valid |
| 701 | /// snapshot since, and neither the reflog reset nor the delete may then |
| 702 | /// overwrite it. A refused reset is reported, never followed by a delete. |
| 703 | fn reset_missing_head(&self, missing: &str) -> io::Result<bool> { |
| 704 | // The newest reflog entry that still names a commit keeps the |
| 705 | // restore points before the break. |
| 706 | if let Some(recovered) = self.newest_reflog_commit(missing)? { |
| 707 | self.move_head(Some(&recovered), Some(missing)) |
| 708 | .map_err(|error| { |
| 709 | io_other(format!( |
| 710 | "snapshot history HEAD pointed at missing commit {missing} and was not reset to {recovered}: {error}" |
| 711 | )) |
| 712 | })?; |
| 713 | tracing::warn!( |
| 714 | target: "snapshot", |
| 715 | "snapshot history HEAD pointed at missing commit {missing}; reset to {recovered} from the reflog" |
| 716 | ); |
| 717 | return Ok(false); |
| 718 | } |
| 719 | self.move_head(None, Some(missing)).map_err(|error| { |
| 720 | io_other(format!( |
| 721 | "snapshot history HEAD points at missing commit {missing} and could not be reset: {error}" |
| 722 | )) |
| 723 | })?; |
| 724 | tracing::warn!( |
| 725 | target: "snapshot", |
| 726 | "snapshot history HEAD pointed at missing commit {missing}; started a fresh history" |
| 727 | ); |
| 728 | Ok(true) |
| 729 | } |
| 730 | |
| 731 | fn is_commit(&self, oid: &str) -> io::Result<bool> { |
| 732 | let object = format!("{oid}^{{commit}}"); |
| 733 | Ok( |
| 734 | run_git(&self.git_dir, &self.work_tree, &["cat-file", "-e", &object])? |
| 735 | .status |
| 736 | .success(), |
| 737 | ) |
| 738 | } |
| 739 | |
| 740 | /// The newest commit recorded in HEAD's reflogs (its branch's, then |
| 741 | /// HEAD's own) that still exists, skipping `missing`. |
| 742 | fn newest_reflog_commit(&self, missing: &str) -> io::Result<Option<String>> { |
| 743 | let branch = run_git(&self.git_dir, &self.work_tree, &["symbolic-ref", "HEAD"])?; |
| 744 | let mut logs = Vec::new(); |
| 745 | if branch.status.success() { |
| 746 | let name = String::from_utf8_lossy(&branch.stdout).trim().to_string(); |
| 747 | logs.push(self.git_dir.join("logs").join(name)); |
| 748 | } |
| 749 | logs.push(self.git_dir.join("logs").join("HEAD")); |
| 750 | for log in logs { |
| 751 | let Ok(text) = std::fs::read_to_string(&log) else { |
| 752 | continue; |
| 753 | }; |
| 754 | // Each line is `<old> <new> <who> <when>\t<message>`. |
| 755 | for line in text.lines().rev() { |
| 756 | let mut fields = line.split(' '); |
| 757 | let (Some(old), Some(new)) = (fields.next(), fields.next()) else { |
| 758 | continue; |
| 759 | }; |
| 760 | for oid in [new, old] { |
| 761 | if oid.len() >= 40 |
| 762 | && oid != missing |
| 763 | && oid.bytes().any(|b| b != b'0') |
| 764 | && self.is_commit(oid)? |
| 765 | { |
| 766 | return Ok(Some(oid.to_string())); |
| 767 | } |
| 768 | } |
| 769 | } |
| 770 | } |
| 771 | Ok(None) |
| 772 | } |
| 773 | |
| 774 | /// Point the side repo's HEAD branch at a commit id that does not exist. |
| 775 | #[cfg(test)] |
| 776 | pub(crate) fn point_head_at_missing_commit_for_test(&self) { |
| 777 | let branch = run_git(&self.git_dir, &self.work_tree, &["symbolic-ref", "HEAD"]) |
| 778 | .expect("symbolic-ref"); |
| 779 | let branch = String::from_utf8_lossy(&branch.stdout).trim().to_string(); |
| 780 | let path = self.git_dir.join(&branch); |
| 781 | std::fs::create_dir_all(path.parent().expect("ref parent")).expect("ref dir"); |
| 782 | std::fs::write(path, "1111111111111111111111111111111111111111\n").expect("write ref"); |
| 783 | } |
| 784 | |
| 785 | /// Prefix a snapshot label with its owning session id, if any. |
| 786 | fn encode_session_label(label: &str, session_id: Option<&str>) -> String { |
| 787 | match session_id { |
| 788 | Some(sid) if !sid.is_empty() => format!("[sid={sid}] {label}"), |
| 789 | _ => label.to_string(), |
| 790 | } |
| 791 | } |
| 792 | |
| 793 | /// Split a possibly session-tagged label back into `(session_id, label)`. |
| 794 | /// |
| 795 | /// Returns `(None, label)` for untagged labels. The decoded label is |
| 796 | /// the original one without the `[sid=...] ` prefix, so consumers that |
| 797 | /// match on `pre-turn:`/`tool:`/`redo:` prefixes keep working unchanged. |
| 798 | fn decode_session_label(label: &str) -> (Option<String>, String) { |
| 799 | let Some(rest) = label.strip_prefix("[sid=") else { |
| 800 | return (None, label.to_string()); |
| 801 | }; |
| 802 | let Some(end) = rest.find("] ") else { |
| 803 | return (None, label.to_string()); |
| 804 | }; |
| 805 | let sid = &rest[..end]; |
| 806 | let plain = &rest[end + 2..]; |
| 807 | if sid.is_empty() || plain.is_empty() { |
| 808 | return (None, label.to_string()); |
| 809 | } |
| 810 | (Some(sid.to_string()), plain.to_string()) |
| 811 | } |
| 812 | /// Size-pressure prune (#1112): if the side repo exceeds `max_bytes`, |
| 813 | /// drop the oldest half of the snapshots, repeatedly, until the store is |
| 814 | /// at or under `target_bytes` or only the protected restore points are |
| 815 | /// left. Returns the number of snapshots destroyed, so the caller can |
| 816 | /// tell the user their undo history shrank (S5). |
| 817 | /// |
| 818 | /// The prune goes oldest first by count and always keeps the newest |
| 819 | /// snapshot plus each session's newest `pre-turn:` and `post-turn:` |
| 820 | /// boundaries, so every session's running turn (and the one before it) |
| 821 | /// stays restorable. It used to |
| 822 | /// prune by age starting at one second, which on the first pass dropped |
| 823 | /// every snapshot older than a second and then wiped the rest: a workspace |
| 824 | /// whose side repo sat above the cap lost all undo history on every |
| 825 | /// snapshot. |
| 826 | fn prune_size_pressure(&self, max_bytes: u64, target_bytes: u64) -> io::Result<usize> { |
| 827 | self.with_write_lock(|| { |
| 828 | let current_bytes = dir_size_bytes(&self.git_dir)?; |
| 829 | if current_bytes <= max_bytes { |
| 830 | return Ok(0); |
| 831 | } |
| 832 | tracing::warn!( |
| 833 | target: "snapshot", |
| 834 | current_mb = current_bytes / BYTES_PER_MB, |
| 835 | limit_mb = max_bytes / BYTES_PER_MB, |
| 836 | "snapshot storage over limit — pruning the oldest snapshots" |
| 837 | ); |
| 838 | let mut removed_total: usize = 0; |
| 839 | loop { |
| 840 | let snapshots = self.list(usize::MAX)?; |
| 841 | let mut survivors = size_pressure_survivors(&snapshots, snapshots.len() / 2); |
| 842 | if survivors.len() >= snapshots.len() { |
| 843 | // Halving kept only protected points: cut to those alone. |
| 844 | survivors = size_pressure_survivors(&snapshots, 0); |
| 845 | } |
| 846 | if survivors.is_empty() || survivors.len() >= snapshots.len() { |
| 847 | break; |
| 848 | } |
| 849 | self.rebuild_survivor_chain(&survivors, &snapshots[0].id)?; |
| 850 | self.reclaim_unreachable(); |
| 851 | removed_total = removed_total.saturating_add(snapshots.len() - survivors.len()); |
| 852 | let new_size = dir_size_bytes(&self.git_dir)?; |
| 853 | if new_size <= target_bytes { |
| 854 | tracing::info!( |
| 855 | target: "snapshot", |
| 856 | new_size_mb = new_size / BYTES_PER_MB, |
| 857 | "pruned snapshot storage back under limit" |
| 858 | ); |
| 859 | break; |
| 860 | } |
| 861 | } |
| 862 | Ok(removed_total) |
| 863 | }) |
| 864 | } |
| 865 | |
| 866 | /// Restore the workspace to the state at `id`. |
| 867 | /// |
| 868 | /// Requires a durable safety snapshot before changing files. A failed |
| 869 | /// restore attempts to put those files back; if that also fails, the |
| 870 | /// error identifies the retained safety snapshot for recovery. This is |
| 871 | /// not atomic against an external editor changing the live workspace. |
| 872 | /// File/directory transitions are refused before checkout because the |
| 873 | /// replaced directory may contain files excluded from the backup. |
| 874 | /// We never touch the user's own `.git`. |
| 875 | pub fn restore(&self, id: &SnapshotId) -> io::Result<()> { |
| 876 | self.with_write_lock(|| self.restore_locked(id)) |
| 877 | } |
| 878 | |
| 879 | fn restore_locked(&self, id: &SnapshotId) -> io::Result<()> { |
| 880 | // The backup label is deliberately not an undo/revert-turn candidate. |
| 881 | let target_short = &id.as_str()[..id.as_str().len().min(12)]; |
| 882 | let backup = self |
| 883 | .snapshot_with_session(&format!("pre-restore:{target_short}"), None) |
| 884 | .map_err(|error| { |
| 885 | io_other(format!( |
| 886 | "pre-restore safety snapshot failed; no workspace files were changed: {error}" |
| 887 | )) |
| 888 | })?; |
| 889 | let current_paths = self.tree_paths(backup.as_str())?; |
| 890 | let target_paths = self.tree_paths(id.as_str())?; |
| 891 | for rel in &target_paths { |
| 892 | match std::fs::symlink_metadata(self.work_tree.join(rel)) { |
| 893 | Ok(metadata) if metadata.is_dir() => { |
| 894 | return Err(io_other(format!( |
| 895 | "'{}' requires a file/directory transition; nothing was restored", |
| 896 | rel.display() |
| 897 | ))); |
| 898 | } |
| 899 | Ok(_) if !current_paths.contains(rel) => { |
| 900 | return Err(io_other(format!( |
| 901 | "'{}' was excluded from the safety snapshot; nothing was restored", |
| 902 | rel.display() |
| 903 | ))); |
| 904 | } |
| 905 | // Unix reports a path under a live file as NotADirectory; |
| 906 | // Windows reports NotFound, so look for the file itself. |
| 907 | Err(error) |
| 908 | if error.kind() == io::ErrorKind::NotADirectory |
| 909 | || (error.kind() == io::ErrorKind::NotFound |
| 910 | && ancestor_is_not_a_directory(&self.work_tree, rel)) => |
| 911 | { |
| 912 | return Err(io_other(format!( |
| 913 | "'{}' requires a file/directory transition; nothing was restored", |
| 914 | rel.display() |
| 915 | ))); |
| 916 | } |
| 917 | Err(error) if error.kind() != io::ErrorKind::NotFound => return Err(error), |
| 918 | _ => {} |
| 919 | } |
| 920 | } |
| 921 | if let Err(error) = self.restore_tree(id, ¤t_paths, &target_paths) { |
| 922 | let recovery = match self.restore_tree(&backup, &target_paths, ¤t_paths) { |
| 923 | Ok(()) => "previous snapshot files were restored".to_string(), |
| 924 | Err(rollback) => format!("rollback also failed: {rollback}"), |
| 925 | }; |
| 926 | return Err(io_other(format!( |
| 927 | "restore failed: {error}; {recovery}; safety snapshot {} retains the previous files", |
| 928 | backup.as_str() |
| 929 | ))); |
| 930 | } |
| 931 | Ok(()) |
| 932 | } |
| 933 | |
| 934 | fn restore_tree( |
| 935 | &self, |
| 936 | id: &SnapshotId, |
| 937 | current_paths: &HashSet<PathBuf>, |
| 938 | target_paths: &HashSet<PathBuf>, |
| 939 | ) -> io::Result<()> { |
| 940 | // An empty target (the first snapshot of an empty directory) has no |
| 941 | // path for `:/` to match, and git refuses the checkout outright; there |
| 942 | // is nothing to write back, only the later files to remove. |
| 943 | if !target_paths.is_empty() { |
| 944 | let checkout = run_git( |
| 945 | &self.git_dir, |
| 946 | &self.work_tree, |
| 947 | &["checkout", "--end-of-options", id.as_str(), "--", ":/"], |
| 948 | )?; |
| 949 | if !checkout.status.success() { |
| 950 | return Err(io_other(format!( |
| 951 | "git checkout failed: {}", |
| 952 | String::from_utf8_lossy(&checkout.stderr).trim() |
| 953 | ))); |
| 954 | } |
| 955 | } |
| 956 | self.remove_paths_missing_from_target(current_paths, target_paths) |
| 957 | } |
| 958 | |
| 959 | /// File restore never traverses symlinks, directories, or Git metadata. |
| 960 | /// Validate every existing component before reading, backing up or writing. |
| 961 | pub fn validate_restore_file(&self, rel: &Path) -> io::Result<bool> { |
| 962 | if !is_safe_relative_path(rel) |
| 963 | || rel |
| 964 | .components() |
| 965 | .any(|part| is_git_metadata_name(part.as_os_str())) |
| 966 | { |
| 967 | return Err(io::Error::new( |
| 968 | io::ErrorKind::InvalidInput, |
| 969 | format!( |
| 970 | "refusing to restore unsafe path '{}': restore requires a regular workspace file", |
| 971 | rel.display() |
| 972 | ), |
| 973 | )); |
| 974 | } |
| 975 | let mut path = self.work_tree.clone(); |
| 976 | for part in rel.components() { |
| 977 | path.push(part); |
| 978 | match std::fs::symlink_metadata(&path) { |
| 979 | Ok(meta) |
| 980 | if meta.file_type().is_symlink() |
| 981 | || (path == self.work_tree.join(rel) && !meta.is_file()) |
| 982 | || (path != self.work_tree.join(rel) && !meta.is_dir()) => |
| 983 | { |
| 984 | return Err(io::Error::new( |
| 985 | io::ErrorKind::InvalidInput, |
| 986 | "restore refuses directories, symlinks and non-regular files", |
| 987 | )); |
| 988 | } |
| 989 | Ok(_) => {} |
| 990 | Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false), |
| 991 | Err(error) => return Err(error), |
| 992 | } |
| 993 | } |
| 994 | Ok(true) |
| 995 | } |
| 996 | |
| 997 | fn snapshot_contains_regular_file(&self, id: &SnapshotId, rel: &Path) -> io::Result<bool> { |
| 998 | self.snapshot_file_blob(id, rel).map(|blob| blob.is_some()) |
| 999 | } |
| 1000 | |
| 1001 | /// The blob id `rel` holds in snapshot (commit or tree) `id`, or `None` |
| 1002 | /// when the snapshot does not contain it. Anything other than a regular |
| 1003 | /// file is `InvalidInput`: file-scoped restores never write symlinks or |
| 1004 | /// directories. |
| 1005 | fn snapshot_file_blob(&self, id: &SnapshotId, rel: &Path) -> io::Result<Option<String>> { |
| 1006 | let entry = run_git( |
| 1007 | &self.git_dir, |
| 1008 | &self.work_tree, |
| 1009 | &[ |
| 1010 | "--literal-pathspecs", |
| 1011 | "ls-tree", |
| 1012 | "-z", |
| 1013 | "--end-of-options", |
| 1014 | id.as_str(), |
| 1015 | "--", |
| 1016 | rel.to_str() |
| 1017 | .ok_or_else(|| io_other("restore path must be UTF-8"))?, |
| 1018 | ], |
| 1019 | )?; |
| 1020 | if !entry.status.success() { |
| 1021 | return Err(io_other(format!( |
| 1022 | "Failed to inspect snapshot file: {}", |
| 1023 | String::from_utf8_lossy(&entry.stderr).trim() |
| 1024 | ))); |
| 1025 | } |
| 1026 | if entry.stdout.is_empty() { |
| 1027 | return Ok(None); |
| 1028 | } |
| 1029 | let line = String::from_utf8_lossy(&entry.stdout); |
| 1030 | let blob = line |
| 1031 | .strip_prefix("100644 blob ") |
| 1032 | .or_else(|| line.strip_prefix("100755 blob ")) |
| 1033 | .and_then(|rest| rest.split('\t').next()) |
| 1034 | .filter(|sha| SnapshotId::is_well_formed(sha)) |
| 1035 | .ok_or_else(|| { |
| 1036 | io::Error::new( |
| 1037 | io::ErrorKind::InvalidInput, |
| 1038 | "snapshot path is not a regular file", |
| 1039 | ) |
| 1040 | })?; |
| 1041 | Ok(Some(blob.to_string())) |
| 1042 | } |
| 1043 | |
| 1044 | /// Whether the working-tree bytes of `rel` are exactly what snapshot `id` |
| 1045 | /// holds for it (both absent counts as a match). |
| 1046 | /// |
| 1047 | /// Compares content hashes directly instead of `git diff <id>`: the side |
| 1048 | /// repo's index only lists paths present at the last snapshot, and a diff |
| 1049 | /// against a commit reports a path the index lost as deleted even when |
| 1050 | /// the file is on disk. |
| 1051 | pub fn path_matches_snapshot(&self, id: &SnapshotId, rel: &Path) -> io::Result<bool> { |
| 1052 | let in_work = self.validate_restore_file(rel)?; |
| 1053 | match (self.snapshot_file_blob(id, rel)?, in_work) { |
| 1054 | (None, false) => Ok(true), |
| 1055 | (None, true) | (Some(_), false) => Ok(false), |
| 1056 | (Some(blob), true) => { |
| 1057 | let rel_str = rel |
| 1058 | .to_str() |
| 1059 | .ok_or_else(|| io_other("restore path must be UTF-8"))?; |
| 1060 | let abs = self.work_tree.join(rel); |
| 1061 | let abs = abs |
| 1062 | .to_str() |
| 1063 | .ok_or_else(|| io_other("restore path must be UTF-8"))?; |
| 1064 | // Git for Windows cannot open the `\\?\` verbatim form that |
| 1065 | // `canonicalize` gives the work tree. |
| 1066 | let abs = match abs.strip_prefix(r"\\?\") { |
| 1067 | Some(rest) => match rest.strip_prefix(r"UNC\") { |
| 1068 | Some(unc) => format!(r"\\{unc}"), |
| 1069 | None => rest.to_string(), |
| 1070 | }, |
| 1071 | None => abs.to_string(), |
| 1072 | }; |
| 1073 | let hashed = run_git( |
| 1074 | &self.git_dir, |
| 1075 | &self.work_tree, |
| 1076 | &["hash-object", &format!("--path={rel_str}"), "--", &abs], |
| 1077 | )?; |
| 1078 | if !hashed.status.success() { |
| 1079 | return Err(io_other(format!( |
| 1080 | "git hash-object failed: {}", |
| 1081 | String::from_utf8_lossy(&hashed.stderr).trim() |
| 1082 | ))); |
| 1083 | } |
| 1084 | Ok(String::from_utf8_lossy(&hashed.stdout).trim() == blob) |
| 1085 | } |
| 1086 | } |
| 1087 | } |
| 1088 | |
| 1089 | /// Paths whose content differs between snapshots `from` and `to` (commit |
| 1090 | /// or tree ids), with renames reported as a delete plus an add so each |
| 1091 | /// side's path is restorable on its own. |
| 1092 | pub fn changed_paths_between( |
| 1093 | &self, |
| 1094 | from: &SnapshotId, |
| 1095 | to: &SnapshotId, |
| 1096 | ) -> io::Result<Vec<PathBuf>> { |
| 1097 | let diff = run_git( |
| 1098 | &self.git_dir, |
| 1099 | &self.work_tree, |
| 1100 | &[ |
| 1101 | "diff", |
| 1102 | "--no-renames", |
| 1103 | "--name-only", |
| 1104 | "-z", |
| 1105 | "--end-of-options", |
| 1106 | from.as_str(), |
| 1107 | to.as_str(), |
| 1108 | "--", |
| 1109 | ], |
| 1110 | )?; |
| 1111 | if !diff.status.success() { |
| 1112 | return Err(io_other(format!( |
| 1113 | "git diff --name-only failed: {}", |
| 1114 | String::from_utf8_lossy(&diff.stderr).trim() |
| 1115 | ))); |
| 1116 | } |
| 1117 | let mut paths: Vec<PathBuf> = parse_nul_paths(&diff.stdout).into_iter().collect(); |
| 1118 | paths.sort(); |
| 1119 | Ok(paths) |
| 1120 | } |
| 1121 | |
| 1122 | /// Whether `rel` has the same content (or is absent) in both snapshots. |
| 1123 | pub fn path_same_in_snapshots( |
| 1124 | &self, |
| 1125 | a: &SnapshotId, |
| 1126 | b: &SnapshotId, |
| 1127 | rel: &Path, |
| 1128 | ) -> io::Result<bool> { |
| 1129 | Ok(self.snapshot_file_blob(a, rel)? == self.snapshot_file_blob(b, rel)?) |
| 1130 | } |
| 1131 | |
| 1132 | /// Return whether `rel` differs between snapshot `id` and the current |
| 1133 | /// working tree. |
| 1134 | /// |
| 1135 | /// This is the single-path counterpart of |
| 1136 | /// [`Self::work_tree_matches_snapshot`]: it answers "would restoring just |
| 1137 | /// this file change anything?", which is what file-scoped revert |
| 1138 | /// cursoring needs. A path that exists in neither the snapshot nor the |
| 1139 | /// working tree does not differ. |
| 1140 | pub fn path_differs_from_snapshot(&self, id: &SnapshotId, rel: &Path) -> io::Result<bool> { |
| 1141 | let in_work = self.validate_restore_file(rel)?; |
| 1142 | let in_target = self.snapshot_contains_regular_file(id, rel)?; |
| 1143 | match (in_target, in_work) { |
| 1144 | // Neither side has it: nothing to restore and nothing to remove. |
| 1145 | (false, false) => Ok(false), |
| 1146 | // The snapshot has it and the working tree lost it. |
| 1147 | (true, false) => Ok(true), |
| 1148 | // The path was created after the snapshot. |
| 1149 | (false, true) => Ok(true), |
| 1150 | (true, true) => { |
| 1151 | let rel = rel.to_string_lossy().into_owned(); |
| 1152 | let diff = run_git( |
| 1153 | &self.git_dir, |
| 1154 | &self.work_tree, |
| 1155 | &[ |
| 1156 | "--literal-pathspecs", |
| 1157 | "diff", |
| 1158 | "--quiet", |
| 1159 | "--end-of-options", |
| 1160 | id.as_str(), |
| 1161 | "--", |
| 1162 | rel.as_str(), |
| 1163 | ], |
| 1164 | )?; |
| 1165 | git_diff_matches(diff).map(|matches| !matches) |
| 1166 | } |
| 1167 | } |
| 1168 | } |
| 1169 | |
| 1170 | /// Restore only `rel_paths` from snapshot `id`. |
| 1171 | /// |
| 1172 | /// This is the file-scoped counterpart of [`Self::restore`]. The |
| 1173 | /// difference that matters: the whole-tree `git checkout <sha> -- :/` is |
| 1174 | /// replaced by a pathspec-limited checkout, so a working-tree path outside |
| 1175 | /// `rel_paths` is never written or deleted. The safety backup reads the workspace. |
| 1176 | /// |
| 1177 | /// A path the snapshot does not track is removed from the working tree |
| 1178 | /// (that is how a file created after the snapshot is reverted), and a path |
| 1179 | /// the snapshot tracks but the working tree lost is recreated. A path that |
| 1180 | /// exists in neither side produces no outcome at all, rather than a |
| 1181 | /// report claiming a change that did not happen. |
| 1182 | #[cfg(test)] |
| 1183 | pub fn restore_paths( |
| 1184 | &self, |
| 1185 | id: &SnapshotId, |
| 1186 | rel_paths: &[PathBuf], |
| 1187 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1188 | self.restore_paths_checked(id, rel_paths, || Ok(())) |
| 1189 | } |
| 1190 | |
| 1191 | pub fn restore_file_if_unchanged( |
| 1192 | &self, |
| 1193 | id: &SnapshotId, |
| 1194 | rel: &Path, |
| 1195 | expected_hash: &str, |
| 1196 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1197 | let verify = || { |
| 1198 | let actual = if self.validate_restore_file(rel)? { |
| 1199 | let bytes = std::fs::read(self.work_tree.join(rel))?; |
| 1200 | format!("sha256:{}", crate::hashing::sha256_hex(bytes)) |
| 1201 | } else { |
| 1202 | "absent".to_string() |
| 1203 | }; |
| 1204 | if actual != expected_hash { |
| 1205 | return Err(io::Error::new( |
| 1206 | io::ErrorKind::WouldBlock, |
| 1207 | "The file changed after the selected change record. Refresh and review it before restoring; nothing was changed.", |
| 1208 | )); |
| 1209 | } |
| 1210 | Ok(()) |
| 1211 | }; |
| 1212 | verify()?; |
| 1213 | self.restore_paths_checked(id, &[rel.to_path_buf()], verify) |
| 1214 | } |
| 1215 | |
| 1216 | fn restore_paths_checked( |
| 1217 | &self, |
| 1218 | id: &SnapshotId, |
| 1219 | rel_paths: &[PathBuf], |
| 1220 | preflight: impl FnOnce() -> io::Result<()>, |
| 1221 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1222 | let target_short = &id.as_str()[..id.as_str().len().min(12)]; |
| 1223 | let plan: Vec<(PathBuf, SnapshotId)> = rel_paths |
| 1224 | .iter() |
| 1225 | .map(|rel| (rel.clone(), id.clone())) |
| 1226 | .collect(); |
| 1227 | self.restore_path_plan( |
| 1228 | &plan, |
| 1229 | &format!("pre-restore:{target_short}"), |
| 1230 | false, |
| 1231 | preflight, |
| 1232 | ) |
| 1233 | } |
| 1234 | |
| 1235 | /// Restore each `(path, source)` pair of `plan` from its own snapshot |
| 1236 | /// (commit or tree id), and nothing else. |
| 1237 | /// |
| 1238 | /// The whole plan is validated, then one mandatory `backup_label` safety |
| 1239 | /// snapshot is written, then `preflight` runs immediately before the |
| 1240 | /// first mutation. A path its source does not hold is removed; with |
| 1241 | /// `prune_emptied_dirs` the directories that removal leaves empty go too |
| 1242 | /// (a turn-scoped undo removes the directories a dropped turn created), |
| 1243 | /// otherwise they stay (a file-scoped revert names a file, not a tree). |
| 1244 | pub fn restore_path_plan( |
| 1245 | &self, |
| 1246 | plan: &[(PathBuf, SnapshotId)], |
| 1247 | backup_label: &str, |
| 1248 | prune_emptied_dirs: bool, |
| 1249 | preflight: impl FnOnce() -> io::Result<()>, |
| 1250 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1251 | self.restore_path_plan_backed_up( |
| 1252 | plan, |
| 1253 | RestoreBackup::Take(backup_label), |
| 1254 | prune_emptied_dirs, |
| 1255 | preflight, |
| 1256 | ) |
| 1257 | } |
| 1258 | |
| 1259 | /// [`Self::restore_path_plan`] with a safety snapshot the caller already |
| 1260 | /// took (`backup`, a commit id) instead of a new one. `preflight` must |
| 1261 | /// prove every planned path is still as `backup` holds it, so the backup |
| 1262 | /// is as good as one taken now; reusing it keeps a second snapshot, and |
| 1263 | /// the prune that comes with it, out of the window between planning and |
| 1264 | /// the first write. |
| 1265 | pub fn restore_path_plan_with_backup( |
| 1266 | &self, |
| 1267 | plan: &[(PathBuf, SnapshotId)], |
| 1268 | backup: &SnapshotId, |
| 1269 | prune_emptied_dirs: bool, |
| 1270 | preflight: impl FnOnce() -> io::Result<()>, |
| 1271 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1272 | self.restore_path_plan_backed_up( |
| 1273 | plan, |
| 1274 | RestoreBackup::Existing(backup), |
| 1275 | prune_emptied_dirs, |
| 1276 | preflight, |
| 1277 | ) |
| 1278 | } |
| 1279 | |
| 1280 | fn restore_path_plan_backed_up( |
| 1281 | &self, |
| 1282 | plan: &[(PathBuf, SnapshotId)], |
| 1283 | backup: RestoreBackup<'_>, |
| 1284 | prune_emptied_dirs: bool, |
| 1285 | preflight: impl FnOnce() -> io::Result<()>, |
| 1286 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1287 | self.with_write_lock(|| { |
| 1288 | self.restore_path_plan_locked(plan, backup, prune_emptied_dirs, preflight) |
| 1289 | }) |
| 1290 | } |
| 1291 | |
| 1292 | fn restore_path_plan_locked( |
| 1293 | &self, |
| 1294 | plan: &[(PathBuf, SnapshotId)], |
| 1295 | backup: RestoreBackup<'_>, |
| 1296 | prune_emptied_dirs: bool, |
| 1297 | preflight: impl FnOnce() -> io::Result<()>, |
| 1298 | ) -> io::Result<Vec<PathRestoreOutcome>> { |
| 1299 | if plan.is_empty() { |
| 1300 | return Ok(Vec::new()); |
| 1301 | } |
| 1302 | // Validate the entire request before any mutation or backup. A snapshot |
| 1303 | // directory entry must not turn a file action into recursive checkout. |
| 1304 | let mut pre_state = Vec::with_capacity(plan.len()); |
| 1305 | for (rel, id) in plan { |
| 1306 | let in_work = self.validate_restore_file(rel)?; |
| 1307 | let in_target = self.snapshot_contains_regular_file(id, rel)?; |
| 1308 | pre_state.push((rel.clone(), id.clone(), in_target, in_work)); |
| 1309 | } |
| 1310 | |
| 1311 | // A durable backup is required for this destructive API. Ignored |
| 1312 | // files cannot be removed/overwritten if the snapshot cannot retain them. |
| 1313 | let backup = match backup { |
| 1314 | RestoreBackup::Take(label) => self.snapshot_with_session(label, None)?, |
| 1315 | RestoreBackup::Existing(id) => id.clone(), |
| 1316 | }; |
| 1317 | for (rel, _, _, in_work) in &pre_state { |
| 1318 | if *in_work && !self.snapshot_contains_regular_file(&backup, rel)? { |
| 1319 | return Err(io_other( |
| 1320 | "File was excluded from the safety snapshot; nothing was restored", |
| 1321 | )); |
| 1322 | } |
| 1323 | self.validate_restore_file(rel)?; |
| 1324 | } |
| 1325 | |
| 1326 | // Recheck after the potentially slow safety snapshot, immediately |
| 1327 | // before checkout/removal. New editor work is retained in the backup. |
| 1328 | preflight()?; |
| 1329 | |
| 1330 | // One pathspec-limited checkout per source snapshot, in plan order. |
| 1331 | let mut sources: Vec<&SnapshotId> = Vec::new(); |
| 1332 | for (_, id, in_target, _) in &pre_state { |
| 1333 | if *in_target && !sources.contains(&id) { |
| 1334 | sources.push(id); |
| 1335 | } |
| 1336 | } |
| 1337 | for source in sources { |
| 1338 | let tracked: Vec<String> = pre_state |
| 1339 | .iter() |
| 1340 | .filter(|(_, id, in_target, _)| *in_target && id == source) |
| 1341 | .map(|(rel, _, _, _)| rel.to_string_lossy().into_owned()) |
| 1342 | .collect(); |
| 1343 | let mut args: Vec<String> = vec![ |
| 1344 | "--literal-pathspecs".to_string(), |
| 1345 | "checkout".to_string(), |
| 1346 | "--end-of-options".to_string(), |
| 1347 | source.as_str().to_string(), |
| 1348 | "--".to_string(), |
| 1349 | ]; |
| 1350 | args.extend(tracked); |
| 1351 | let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect(); |
| 1352 | let checkout = run_git(&self.git_dir, &self.work_tree, &arg_refs)?; |
| 1353 | if !checkout.status.success() { |
| 1354 | return Err(io_other(format!( |
| 1355 | "git checkout failed: {} (safety snapshot {} holds the previous files)", |
| 1356 | String::from_utf8_lossy(&checkout.stderr).trim(), |
| 1357 | backup.as_str() |
| 1358 | ))); |
| 1359 | } |
| 1360 | } |
| 1361 | |
| 1362 | let mut outcomes = Vec::new(); |
| 1363 | for (rel, _, in_target, was_in_work) in pre_state { |
| 1364 | match (in_target, was_in_work) { |
| 1365 | (true, true) => outcomes.push(PathRestoreOutcome { |
| 1366 | path: rel, |
| 1367 | action: PathRestoreAction::Modified, |
| 1368 | }), |
| 1369 | (true, false) => outcomes.push(PathRestoreOutcome { |
| 1370 | path: rel, |
| 1371 | action: PathRestoreAction::Recreated, |
| 1372 | }), |
| 1373 | (false, true) => { |
| 1374 | let path = self.work_tree.join(&rel); |
| 1375 | self.validate_restore_file(&rel)?; |
| 1376 | std::fs::remove_file(&path).map_err(|error| { |
| 1377 | io_other(format!( |
| 1378 | "removing '{}' failed: {error} (safety snapshot {} holds the previous files)", |
| 1379 | rel.display(), |
| 1380 | backup.as_str() |
| 1381 | )) |
| 1382 | })?; |
| 1383 | if prune_emptied_dirs { |
| 1384 | self.prune_empty_parent_dirs(path.parent()); |
| 1385 | } |
| 1386 | outcomes.push(PathRestoreOutcome { |
| 1387 | path: rel, |
| 1388 | action: PathRestoreAction::Removed, |
| 1389 | }); |
| 1390 | } |
| 1391 | // Already in the snapshot's state. |
| 1392 | (false, false) => {} |
| 1393 | } |
| 1394 | } |
| 1395 | Ok(outcomes) |
| 1396 | } |
| 1397 | |
| 1398 | /// Return whether the current workspace matches the given snapshot's |
| 1399 | /// tracked file content. |
| 1400 | /// |
| 1401 | /// This is intentionally narrower than a full "workspace identical" |
| 1402 | /// claim: it compares the current working tree against the snapshot's |
| 1403 | /// tracked paths via git's diff machinery. That is sufficient for |
| 1404 | /// `/undo` cursoring — if the diff is empty, restoring this snapshot |
| 1405 | /// again would be a no-op, so the caller should continue scanning |
| 1406 | /// older snapshots. |
| 1407 | pub fn work_tree_matches_snapshot(&self, id: &SnapshotId) -> io::Result<bool> { |
| 1408 | let diff = run_git( |
| 1409 | &self.git_dir, |
| 1410 | &self.work_tree, |
| 1411 | &[ |
| 1412 | "diff", |
| 1413 | "--quiet", |
| 1414 | "--end-of-options", |
| 1415 | id.as_str(), |
| 1416 | "--", |
| 1417 | ":/", |
| 1418 | ], |
| 1419 | )?; |
| 1420 | git_diff_matches(diff) |
| 1421 | } |
| 1422 | |
| 1423 | /// Paths that differ between snapshots `from` and `to`, in git's order, |
| 1424 | /// one [`SnapshotPathChange`] each: its `status` is git's `A`/`M`/`D`/`T` |
| 1425 | /// letter and its line counts are `None` for a binary file. Paths come |
| 1426 | /// back as git stores them (`-z`), control characters included, so a |
| 1427 | /// caller that prints one must escape it. Both trees are read from the side repo; neither the |
| 1428 | /// work tree nor the index is touched. At most `limit` paths are |
| 1429 | /// returned; the flag says whether more differed. |
| 1430 | pub fn path_changes_between( |
| 1431 | &self, |
| 1432 | from: &SnapshotId, |
| 1433 | to: &SnapshotId, |
| 1434 | limit: usize, |
| 1435 | ) -> io::Result<(Vec<SnapshotPathChange>, bool)> { |
| 1436 | let run = |format: &str| -> io::Result<String> { |
| 1437 | let output = run_git( |
| 1438 | &self.git_dir, |
| 1439 | &self.work_tree, |
| 1440 | &[ |
| 1441 | "diff", |
| 1442 | "--no-renames", |
| 1443 | "--no-ext-diff", |
| 1444 | "--no-textconv", |
| 1445 | format, |
| 1446 | "-z", |
| 1447 | "--end-of-options", |
| 1448 | from.as_str(), |
| 1449 | to.as_str(), |
| 1450 | ], |
| 1451 | )?; |
| 1452 | if !output.status.success() { |
| 1453 | return Err(io_other(format!( |
| 1454 | "git diff {format} failed: {}", |
| 1455 | String::from_utf8_lossy(&output.stderr).trim() |
| 1456 | ))); |
| 1457 | } |
| 1458 | Ok(String::from_utf8_lossy(&output.stdout).into_owned()) |
| 1459 | }; |
| 1460 | // `--numstat -z`: `added\tremoved\tpath\0`, `-` for a binary side. |
| 1461 | let numstat = run("--numstat")?; |
| 1462 | let mut counts: HashMap<String, (Option<u64>, Option<u64>)> = HashMap::new(); |
| 1463 | for record in numstat.split('\0').filter(|record| !record.is_empty()) { |
| 1464 | let mut fields = record.splitn(3, '\t'); |
| 1465 | let (Some(added), Some(removed), Some(path)) = |
| 1466 | (fields.next(), fields.next(), fields.next()) |
| 1467 | else { |
| 1468 | continue; |
| 1469 | }; |
| 1470 | counts.insert(path.to_string(), (added.parse().ok(), removed.parse().ok())); |
| 1471 | } |
| 1472 | // `--name-status -z`: `status\0path\0` pairs. |
| 1473 | let name_status = run("--name-status")?; |
| 1474 | let mut fields = name_status.split('\0').filter(|field| !field.is_empty()); |
| 1475 | let mut changes = Vec::new(); |
| 1476 | let mut truncated = false; |
| 1477 | while let (Some(status), Some(path)) = (fields.next(), fields.next()) { |
| 1478 | if changes.len() == limit { |
| 1479 | truncated = true; |
| 1480 | break; |
| 1481 | } |
| 1482 | let (added, removed) = counts.get(path).copied().unwrap_or((None, None)); |
| 1483 | changes.push(SnapshotPathChange { |
| 1484 | path: path.to_string(), |
| 1485 | status: status.chars().next().unwrap_or('M'), |
| 1486 | added, |
| 1487 | removed, |
| 1488 | }); |
| 1489 | } |
| 1490 | Ok((changes, truncated)) |
| 1491 | } |
| 1492 | |
| 1493 | fn tree_paths(&self, treeish: &str) -> io::Result<HashSet<PathBuf>> { |
| 1494 | let ls = run_git( |
| 1495 | &self.git_dir, |
| 1496 | &self.work_tree, |
| 1497 | &[ |
| 1498 | "ls-tree", |
| 1499 | "-r", |
| 1500 | "-z", |
| 1501 | "--name-only", |
| 1502 | "--end-of-options", |
| 1503 | treeish, |
| 1504 | ], |
| 1505 | )?; |
| 1506 | if !ls.status.success() { |
| 1507 | return Err(io_other(format!( |
| 1508 | "git ls-tree failed: {}", |
| 1509 | String::from_utf8_lossy(&ls.stderr).trim() |
| 1510 | ))); |
| 1511 | } |
| 1512 | Ok(parse_nul_paths(&ls.stdout)) |
| 1513 | } |
| 1514 | |
| 1515 | fn remove_paths_missing_from_target( |
| 1516 | &self, |
| 1517 | current_paths: &HashSet<PathBuf>, |
| 1518 | target_paths: &HashSet<PathBuf>, |
| 1519 | ) -> io::Result<()> { |
| 1520 | let removals: Vec<&PathBuf> = current_paths |
| 1521 | .difference(target_paths) |
| 1522 | .filter(|rel| is_safe_relative_path(rel)) |
| 1523 | .collect(); |
| 1524 | // The removal list comes from the side repo, not the live tree. A |
| 1525 | // directory may have been replaced by a symlink since the backup, and |
| 1526 | // `remove_file` would follow it and delete outside the workspace. |
| 1527 | // Refuse the whole removal before deleting anything. |
| 1528 | for rel in &removals { |
| 1529 | self.removal_parent_is_real(rel)?; |
| 1530 | } |
| 1531 | for rel in removals { |
| 1532 | // Checked again next to the delete: the tree is live. |
| 1533 | if !self.removal_parent_is_real(rel)? { |
| 1534 | continue; |
| 1535 | } |
| 1536 | let path = self.work_tree.join(rel); |
| 1537 | let metadata = match std::fs::symlink_metadata(&path) { |
| 1538 | Ok(metadata) => metadata, |
| 1539 | Err(error) if error.kind() == io::ErrorKind::NotFound => continue, |
| 1540 | Err(error) => return Err(error), |
| 1541 | }; |
| 1542 | if metadata.file_type().is_dir() { |
| 1543 | // A file-to-directory transition can make this path a |
| 1544 | // required parent of files just restored from the target. |
| 1545 | if target_paths.iter().any(|target| target.starts_with(rel)) { |
| 1546 | continue; |
| 1547 | } |
| 1548 | std::fs::remove_dir(&path)?; |
| 1549 | } else { |
| 1550 | std::fs::remove_file(&path)?; |
| 1551 | } |
| 1552 | self.prune_empty_parent_dirs(path.parent()); |
| 1553 | } |
| 1554 | Ok(()) |
| 1555 | } |
| 1556 | |
| 1557 | /// Whether every directory between the work tree and `rel` is a real |
| 1558 | /// directory. `Ok(false)`: one is missing, so `rel` is gone too. An |
| 1559 | /// ancestor that is a symlink or a file is an `InvalidInput` refusal. |
| 1560 | fn removal_parent_is_real(&self, rel: &Path) -> io::Result<bool> { |
| 1561 | let mut dir = self.work_tree.clone(); |
| 1562 | for part in rel.parent().into_iter().flat_map(Path::components) { |
| 1563 | dir.push(part); |
| 1564 | match std::fs::symlink_metadata(&dir) { |
| 1565 | Ok(meta) if meta.file_type().is_dir() => {} |
| 1566 | Ok(_) => { |
| 1567 | return Err(io::Error::new( |
| 1568 | io::ErrorKind::InvalidInput, |
| 1569 | format!( |
| 1570 | "restore refuses to remove '{}': '{}' is no longer a directory (a symlink could lead outside the workspace)", |
| 1571 | rel.display(), |
| 1572 | dir.strip_prefix(&self.work_tree).unwrap_or(&dir).display() |
| 1573 | ), |
| 1574 | )); |
| 1575 | } |
| 1576 | Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false), |
| 1577 | Err(error) => return Err(error), |
| 1578 | } |
| 1579 | } |
| 1580 | Ok(true) |
| 1581 | } |
| 1582 | |
| 1583 | fn prune_empty_parent_dirs(&self, mut dir: Option<&Path>) { |
| 1584 | while let Some(path) = dir { |
| 1585 | if path == self.work_tree { |
| 1586 | break; |
| 1587 | } |
| 1588 | if std::fs::remove_dir(path).is_err() { |
| 1589 | break; |
| 1590 | } |
| 1591 | dir = path.parent(); |
| 1592 | } |
| 1593 | } |
| 1594 | |
| 1595 | /// List up to `limit` most-recent snapshots, newest first. |
| 1596 | pub fn list(&self, limit: usize) -> io::Result<Vec<Snapshot>> { |
| 1597 | // `git log -<n>` is the short form of `--max-count=<n>`; if `limit` |
| 1598 | // is `usize::MAX` (caller asked for "everything") we pass an empty |
| 1599 | // count so git defaults to no upper bound. |
| 1600 | let mut args: Vec<String> = vec!["log".to_string()]; |
| 1601 | if limit < usize::MAX { |
| 1602 | args.push(format!("--max-count={limit}")); |
| 1603 | } |
| 1604 | args.push("--pretty=format:%H%x09%T%x09%at%x09%s".to_string()); |
| 1605 | args.push("--no-color".to_string()); |
| 1606 | let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect(); |
| 1607 | let log = run_git(&self.git_dir, &self.work_tree, &arg_refs)?; |
| 1608 | if !log.status.success() { |
| 1609 | let head = run_git( |
| 1610 | &self.git_dir, |
| 1611 | &self.work_tree, |
| 1612 | &["symbolic-ref", "-q", "HEAD"], |
| 1613 | )?; |
| 1614 | if head.status.success() { |
| 1615 | let reference = String::from_utf8_lossy(&head.stdout); |
| 1616 | let exists = run_git( |
| 1617 | &self.git_dir, |
| 1618 | &self.work_tree, |
| 1619 | &["show-ref", "--verify", "--quiet", reference.trim()], |
| 1620 | )?; |
| 1621 | if exists.status.code() == Some(1) { |
| 1622 | return Ok(Vec::new()); |
| 1623 | } |
| 1624 | } |
| 1625 | return Err(io_other(format!( |
| 1626 | "git log failed: {}", |
| 1627 | String::from_utf8_lossy(&log.stderr).trim() |
| 1628 | ))); |
| 1629 | } |
| 1630 | let stdout = String::from_utf8_lossy(&log.stdout); |
| 1631 | let mut out = Vec::new(); |
| 1632 | for line in stdout.lines() { |
| 1633 | let mut parts = line.splitn(4, '\t'); |
| 1634 | let sha = parts.next().unwrap_or("").to_string(); |
| 1635 | let tree = parts.next().unwrap_or("").to_string(); |
| 1636 | let ts = parts |
| 1637 | .next() |
| 1638 | .and_then(|s| s.parse::<i64>().ok()) |
| 1639 | .unwrap_or(0); |
| 1640 | let subject = parts.next().unwrap_or("").to_string(); |
| 1641 | // `git log --pretty=format:%H` only emits full hex ids; skip anything |
| 1642 | // else rather than let it become a revision argument later. |
| 1643 | let (Ok(id), Ok(tree)) = (SnapshotId::parse(&sha), SnapshotId::parse(&tree)) else { |
| 1644 | continue; |
| 1645 | }; |
| 1646 | let (session_id, label) = Self::decode_session_label(&subject); |
| 1647 | out.push(Snapshot { |
| 1648 | id, |
| 1649 | tree, |
| 1650 | label, |
| 1651 | timestamp: ts, |
| 1652 | session_id, |
| 1653 | }); |
| 1654 | } |
| 1655 | Ok(out) |
| 1656 | } |
| 1657 | |
| 1658 | /// Drop snapshots older than `max_age`, returning the count removed. |
| 1659 | /// |
| 1660 | /// Strategy: identify keepable commits (younger than the cutoff), |
| 1661 | /// reset HEAD to the oldest survivor, then `git reflog expire` + |
| 1662 | /// `git gc --prune=now` to actually reclaim space. Cheap and avoids |
| 1663 | /// rewriting history when nothing has aged out. |
| 1664 | pub fn prune_older_than(&self, max_age: Duration) -> io::Result<usize> { |
| 1665 | self.with_write_lock(|| self.prune_older_than_locked(max_age)) |
| 1666 | } |
| 1667 | |
| 1668 | fn prune_older_than_locked(&self, max_age: Duration) -> io::Result<usize> { |
| 1669 | let now = SystemTime::now() |
| 1670 | .duration_since(UNIX_EPOCH) |
| 1671 | .map_err(|e| io_other(format!("clock error: {e}")))? |
| 1672 | .as_secs() as i64; |
| 1673 | let cutoff = now - max_age.as_secs() as i64; |
| 1674 | |
| 1675 | let snapshots = self.list(usize::MAX)?; |
| 1676 | if snapshots.is_empty() { |
| 1677 | return Ok(0); |
| 1678 | } |
| 1679 | |
| 1680 | // Snapshots are newest-first. Find the index of the first one |
| 1681 | // at-or-older than the cutoff — every entry from that index |
| 1682 | // onward is a candidate for removal. We use `<=` so a 0-second |
| 1683 | // retention drops same-second commits (otherwise tests calling |
| 1684 | // `prune_older_than(Duration::ZERO)` immediately after creating |
| 1685 | // a snapshot would never prune anything). |
| 1686 | let cut_index = snapshots.iter().position(|s| s.timestamp <= cutoff); |
| 1687 | let Some(cut) = cut_index else { |
| 1688 | return Ok(0); |
| 1689 | }; |
| 1690 | let removed = snapshots.len() - cut; |
| 1691 | if removed == 0 { |
| 1692 | return Ok(0); |
| 1693 | } |
| 1694 | |
| 1695 | if cut == 0 { |
| 1696 | // Every snapshot is older than the cutoff: unset HEAD so the next |
| 1697 | // snapshot starts a fresh history and gc reclaims the old one. |
| 1698 | self.move_head(None, Some(snapshots[0].id.as_str()))?; |
| 1699 | } else { |
| 1700 | // Keep the newest `cut` snapshots (indices [0..cut], newest-first) |
| 1701 | // and drop the older tail. This MUST rebuild the survivors as a |
| 1702 | // fresh orphan chain, not `update-ref HEAD <oldest survivor>`: |
| 1703 | // the snapshots are a parent-linked commit chain with the newest |
| 1704 | // at HEAD, so pointing HEAD at the oldest survivor orphaned every |
| 1705 | // NEWER snapshot (gc then destroyed them) while keeping the very |
| 1706 | // snapshots we meant to remove as its ancestors — the exact |
| 1707 | // inverse of the intent (2026-08-04 review, reproduced). |
| 1708 | self.rebuild_survivor_chain(&snapshots[..cut], &snapshots[0].id)?; |
| 1709 | } |
| 1710 | |
| 1711 | self.reclaim_unreachable(); |
| 1712 | Ok(removed) |
| 1713 | } |
| 1714 | |
| 1715 | /// Expire the reflog and gc every unreachable object now, reclaiming the |
| 1716 | /// space of the snapshots a prune dropped. Runs only under the snapshot |
| 1717 | /// write lock: an immediate prune is safe because no other writer that |
| 1718 | /// takes the lock can hold a new, not yet referenced object meanwhile. |
| 1719 | fn reclaim_unreachable(&self) { |
| 1720 | let _ = run_git( |
| 1721 | &self.git_dir, |
| 1722 | &self.work_tree, |
| 1723 | &["reflog", "expire", "--expire=now", "--all"], |
| 1724 | ); |
| 1725 | let _ = run_git( |
| 1726 | &self.git_dir, |
| 1727 | &self.work_tree, |
| 1728 | &["gc", "--prune=now", "--quiet"], |
| 1729 | ); |
| 1730 | } |
| 1731 | |
| 1732 | /// Run `op` holding the side repo's cross-process write lock (see |
| 1733 | /// [`SNAPSHOT_LOCK_FILE`]). Reentrant on one thread, so a locked |
| 1734 | /// operation may call another. |
| 1735 | fn with_write_lock<T>(&self, op: impl FnOnce() -> io::Result<T>) -> io::Result<T> { |
| 1736 | self.with_write_lock_within(SNAPSHOT_LOCK_WAIT, op) |
| 1737 | } |
| 1738 | |
| 1739 | /// [`Self::with_write_lock`] with an explicit bound on the wait. The |
| 1740 | /// snapshot runs on the turn and tool path, so a peer's long gc (or a |
| 1741 | /// stopped process holding the lock) must fail this snapshot, not stall |
| 1742 | /// the turn and pin a blocking thread indefinitely. |
| 1743 | fn with_write_lock_within<T>( |
| 1744 | &self, |
| 1745 | wait: Duration, |
| 1746 | op: impl FnOnce() -> io::Result<T>, |
| 1747 | ) -> io::Result<T> { |
| 1748 | let held = HELD_SNAPSHOT_LOCKS.with(|held| held.borrow().contains(&self.git_dir)); |
| 1749 | if held { |
| 1750 | return op(); |
| 1751 | } |
| 1752 | let file = open_snapshot_lock_file(&self.git_dir.join(SNAPSHOT_LOCK_FILE))?; |
| 1753 | let mut lock = fd_lock::RwLock::new(file); |
| 1754 | let deadline = std::time::Instant::now() + wait; |
| 1755 | let _guard = loop { |
| 1756 | match lock.try_write() { |
| 1757 | Ok(guard) => break guard, |
| 1758 | Err(err) if std::time::Instant::now() >= deadline => { |
| 1759 | return Err(io::Error::new( |
| 1760 | io::ErrorKind::TimedOut, |
| 1761 | format!( |
| 1762 | "snapshot side repo is busy: another session held its lock for over {}s ({err})", |
| 1763 | wait.as_secs() |
| 1764 | ), |
| 1765 | )); |
| 1766 | } |
| 1767 | Err(_) => std::thread::sleep(SNAPSHOT_LOCK_POLL), |
| 1768 | } |
| 1769 | }; |
| 1770 | // Declared after the guard, so it is dropped (and the thread's claim |
| 1771 | // released) before the file lock is. |
| 1772 | let _held = HeldSnapshotLock::claim(&self.git_dir); |
| 1773 | op() |
| 1774 | } |
| 1775 | |
| 1776 | /// Stage every tracked and untracked path the workspace exposes. |
| 1777 | /// `--all` means `add` + `update` + `remove`, the set `git status` shows. |
| 1778 | fn stage_work_tree(&self) -> io::Result<()> { |
| 1779 | let add = run_git(&self.git_dir, &self.work_tree, &["add", "-A"])?; |
| 1780 | if !add.status.success() { |
| 1781 | return Err(io_other(format!( |
| 1782 | "git add -A failed: {}", |
| 1783 | String::from_utf8_lossy(&add.stderr).trim() |
| 1784 | ))); |
| 1785 | } |
| 1786 | Ok(()) |
| 1787 | } |
| 1788 | |
| 1789 | /// Rebuild `survivors` (newest-first) as a fresh orphan commit chain and |
| 1790 | /// point HEAD at its tip, so every snapshot NOT in `survivors` becomes |
| 1791 | /// unreachable for gc to reclaim. Each survivor's tree, label, session |
| 1792 | /// id, and author/committer timestamp are preserved, so ages do not lie |
| 1793 | /// after a prune (finding: `prune_keep_last_n` previously reset them to |
| 1794 | /// "now"). Assumes `survivors` is non-empty. |
| 1795 | /// |
| 1796 | /// `listed_head` is the HEAD the survivors were chosen from; if another |
| 1797 | /// writer moved HEAD since, the rebuild fails rather than orphan that |
| 1798 | /// writer's snapshot. |
| 1799 | fn rebuild_survivor_chain( |
| 1800 | &self, |
| 1801 | survivors: &[Snapshot], |
| 1802 | listed_head: &SnapshotId, |
| 1803 | ) -> io::Result<()> { |
| 1804 | let mut prev_sha: Option<String> = None; |
| 1805 | for s in survivors.iter().rev() { |
| 1806 | let tree = run_git( |
| 1807 | &self.git_dir, |
| 1808 | &self.work_tree, |
| 1809 | &["rev-parse", &format!("{}^{{tree}}", s.id.as_str())], |
| 1810 | )?; |
| 1811 | if !tree.status.success() { |
| 1812 | return Err(io_other(format!( |
| 1813 | "rev-parse {}^{{tree}} failed: {}", |
| 1814 | s.id.as_str(), |
| 1815 | String::from_utf8_lossy(&tree.stderr).trim() |
| 1816 | ))); |
| 1817 | } |
| 1818 | let tree_hash = String::from_utf8_lossy(&tree.stdout).trim().to_string(); |
| 1819 | |
| 1820 | let mut args = vec![ |
| 1821 | "commit-tree".to_string(), |
| 1822 | "-m".to_string(), |
| 1823 | Self::encode_session_label(&s.label, s.session_id.as_deref()), |
| 1824 | tree_hash, |
| 1825 | ]; |
| 1826 | if let Some(ref p) = prev_sha { |
| 1827 | args.push("-p".to_string()); |
| 1828 | args.push(p.clone()); |
| 1829 | } |
| 1830 | let arg_refs: Vec<&str> = args.iter().map(String::as_str).collect(); |
| 1831 | let new_sha = self.commit_tree_preserving_date(&arg_refs, s.timestamp)?; |
| 1832 | prev_sha = Some(new_sha); |
| 1833 | } |
| 1834 | |
| 1835 | if let Some(final_sha) = prev_sha { |
| 1836 | self.move_head(Some(&final_sha), Some(listed_head.as_str()))?; |
| 1837 | } |
| 1838 | Ok(()) |
| 1839 | } |
| 1840 | |
| 1841 | /// Point HEAD at `new` (or delete it when `None`), only if it still names |
| 1842 | /// `expected` (`None`: HEAD must not exist yet). The compare-and-swap |
| 1843 | /// means a writer that does not take the snapshot lock, such as an older |
| 1844 | /// build, is never silently orphaned: the move fails instead. |
| 1845 | fn move_head(&self, new: Option<&str>, expected: Option<&str>) -> io::Result<()> { |
| 1846 | let expected = expected.unwrap_or(""); |
| 1847 | let args = match new { |
| 1848 | Some(new) => ["update-ref", "HEAD", new, expected], |
| 1849 | None => ["update-ref", "-d", "HEAD", expected], |
| 1850 | }; |
| 1851 | let update = run_git(&self.git_dir, &self.work_tree, &args)?; |
| 1852 | if !update.status.success() { |
| 1853 | return Err(io_other(format!( |
| 1854 | "git update-ref HEAD failed: {}", |
| 1855 | String::from_utf8_lossy(&update.stderr).trim() |
| 1856 | ))); |
| 1857 | } |
| 1858 | Ok(()) |
| 1859 | } |
| 1860 | |
| 1861 | /// Run a `commit-tree` invocation with the author/committer dates pinned |
| 1862 | /// to `timestamp` (Unix seconds), so a rebuilt survivor keeps its real |
| 1863 | /// age instead of stamping "now". |
| 1864 | fn commit_tree_preserving_date(&self, args: &[&str], timestamp: i64) -> io::Result<String> { |
| 1865 | let date = format!("{timestamp} +0000"); |
| 1866 | let mut command = crate::dependencies::Git::command() |
| 1867 | .ok_or_else(|| io::Error::new(io::ErrorKind::NotFound, "git not found on PATH"))?; |
| 1868 | command |
| 1869 | .arg("--git-dir") |
| 1870 | .arg(&self.git_dir) |
| 1871 | .arg("--work-tree") |
| 1872 | .arg(&self.work_tree) |
| 1873 | .env("GIT_AUTHOR_DATE", &date) |
| 1874 | .env("GIT_COMMITTER_DATE", &date) |
| 1875 | .args(args); |
| 1876 | let out = run_bounded_git(&mut command, args.first().copied().unwrap_or("git"))?; |
| 1877 | if !out.status.success() { |
| 1878 | return Err(io_other(format!( |
| 1879 | "commit-tree failed: {}", |
| 1880 | String::from_utf8_lossy(&out.stderr).trim() |
| 1881 | ))); |
| 1882 | } |
| 1883 | Ok(String::from_utf8_lossy(&out.stdout).trim().to_string()) |
| 1884 | } |
| 1885 | |
| 1886 | /// Prune by count: keep the newest `max_count` snapshots, plus the newest |
| 1887 | /// `max_count` turn boundaries (`pre-turn:` / `post-turn:`), and drop the |
| 1888 | /// rest. |
| 1889 | /// |
| 1890 | /// Turn boundaries are the restore points turn-scoped undo resolves, and |
| 1891 | /// every other kind (`tool:`, `post-tool:`, `pre-restore:`) can arrive in |
| 1892 | /// bursts: one turn with more file-modifying tool calls than `max_count`, |
| 1893 | /// or several threads sharing the workspace, would otherwise push out the |
| 1894 | /// running turn's own `pre-turn:` snapshot before its `post-turn:` one is |
| 1895 | /// taken. So the boundaries are retained by their own count, and at most |
| 1896 | /// `2 * max_count` snapshots survive. |
| 1897 | /// |
| 1898 | /// The survivors are rebuilt as a fresh orphan chain; each keeps its |
| 1899 | /// tree, label, session id and timestamp, and the dropped ones become |
| 1900 | /// unreachable for gc to reclaim. |
| 1901 | #[cfg(test)] |
| 1902 | pub fn prune_keep_last_n(&self, max_count: usize) -> io::Result<usize> { |
| 1903 | self.with_write_lock(|| self.prune_keep_last_n_locked(max_count, 1)) |
| 1904 | } |
| 1905 | |
| 1906 | /// [`Self::prune_keep_last_n`] for the prune that follows every snapshot: |
| 1907 | /// it waits until half a window of snapshots is due to go and drops them |
| 1908 | /// together. Rebuilding the survivor chain costs two git processes per |
| 1909 | /// survivor, and doing that after every snapshot past the cap put seconds |
| 1910 | /// in front of every later turn's provider request. The store then holds |
| 1911 | /// at most half a window more than [`Self::prune_keep_last_n`] would keep. |
| 1912 | pub fn prune_keep_last_n_batched(&self, max_count: usize) -> io::Result<usize> { |
| 1913 | self.with_write_lock(|| self.prune_keep_last_n_locked(max_count, (max_count / 2).max(1))) |
| 1914 | } |
| 1915 | |
| 1916 | fn prune_keep_last_n_locked(&self, max_count: usize, min_removed: usize) -> io::Result<usize> { |
| 1917 | let snapshots = self.list(usize::MAX)?; |
| 1918 | if snapshots.len() <= max_count { |
| 1919 | return Ok(0); |
| 1920 | } |
| 1921 | // Newest first: keep the first `max_count` of every kind, and the |
| 1922 | // first `max_count` turn boundaries wherever they sit. |
| 1923 | let mut boundaries_kept = 0usize; |
| 1924 | let survivors: Vec<Snapshot> = snapshots |
| 1925 | .iter() |
| 1926 | .enumerate() |
| 1927 | .filter(|(index, snapshot)| { |
| 1928 | let boundary = is_turn_boundary_label(&snapshot.label); |
| 1929 | let keep = *index < max_count || (boundary && boundaries_kept < max_count); |
| 1930 | if keep && boundary { |
| 1931 | boundaries_kept += 1; |
| 1932 | } |
| 1933 | keep |
| 1934 | }) |
| 1935 | .map(|(_, snapshot)| snapshot.clone()) |
| 1936 | .collect(); |
| 1937 | let removed = snapshots.len() - survivors.len(); |
| 1938 | if removed < min_removed || survivors.is_empty() { |
| 1939 | return Ok(0); |
| 1940 | } |
| 1941 | self.rebuild_survivor_chain(&survivors, &snapshots[0].id)?; |
| 1942 | self.reclaim_unreachable(); |
| 1943 | Ok(removed) |
| 1944 | } |
| 1945 | |
| 1946 | /// Whether a snapshot of the workspace would leave `rel` out: it is |
| 1947 | /// excluded by the workspace's `.gitignore` files or the built-in |
| 1948 | /// snapshot exclusions and not already tracked. Such a path is never in |
| 1949 | /// any snapshot, so no restore can put it back. |
| 1950 | pub fn path_is_excluded(&self, rel: &Path) -> io::Result<bool> { |
| 1951 | let rel_str = rel |
| 1952 | .to_str() |
| 1953 | .ok_or_else(|| io_other("snapshot path must be UTF-8"))?; |
| 1954 | let out = run_git( |
| 1955 | &self.git_dir, |
| 1956 | &self.work_tree, |
| 1957 | &["check-ignore", "--quiet", "--", rel_str], |
| 1958 | )?; |
| 1959 | match out.status.code() { |
| 1960 | Some(0) => Ok(true), |
| 1961 | Some(1) => Ok(false), |
| 1962 | _ => Err(io_other(format!( |
| 1963 | "git check-ignore failed: {}", |
| 1964 | String::from_utf8_lossy(&out.stderr).trim() |
| 1965 | ))), |
| 1966 | } |
| 1967 | } |
| 1968 | |
| 1969 | /// Drop unreachable loose objects left behind by interrupted or |
| 1970 | /// orphaned side-repo operations. |
| 1971 | pub fn prune_unreachable_objects(&self) -> io::Result<()> { |
| 1972 | self.with_write_lock(|| { |
| 1973 | let prune = run_git(&self.git_dir, &self.work_tree, &["prune", "--expire=now"])?; |
| 1974 | if !prune.status.success() { |
| 1975 | return Err(io_other(format!( |
| 1976 | "git prune failed: {}", |
| 1977 | String::from_utf8_lossy(&prune.stderr).trim() |
| 1978 | ))); |
| 1979 | } |
| 1980 | Ok(()) |
| 1981 | }) |
| 1982 | } |
| 1983 | |
| 1984 | /// Return the side-repo's `.git` directory. |
| 1985 | pub fn git_dir(&self) -> &Path { |
| 1986 | &self.git_dir |
| 1987 | } |
| 1988 | |
| 1989 | /// Return the work tree path. |
| 1990 | pub fn work_tree(&self) -> &Path { |
| 1991 | &self.work_tree |
| 1992 | } |
| 1993 | } |
| 1994 | |
| 1995 | /// Whether `label` marks a turn boundary (`pre-turn:` / `post-turn:`), the |
| 1996 | /// restore points [`SnapshotRepo::prune_keep_last_n`] retains by their own |
| 1997 | /// count. |
| 1998 | fn is_turn_boundary_label(label: &str) -> bool { |
| 1999 | label.starts_with("pre-turn:") || label.starts_with("post-turn:") |
| 2000 | } |
| 2001 | |
| 2002 | /// Which snapshots a size-pressure prune keeps (newest first): the newest |
| 2003 | /// `keep`, plus the newest snapshot and, for every session, its newest |
| 2004 | /// `pre-turn:` and `post-turn:` boundaries wherever they sit. Sessions and |
| 2005 | /// sub-agents share the side repo, so protecting only the globally newest |
| 2006 | /// boundaries let one session's snapshots push out another's running turn; |
| 2007 | /// each session's current turn and the one before it stay restorable |
| 2008 | /// however hard the prune has to cut. |
| 2009 | fn size_pressure_survivors(snapshots: &[Snapshot], keep: usize) -> Vec<Snapshot> { |
| 2010 | let mut boundaries_seen: HashSet<(Option<&str>, bool)> = HashSet::new(); |
| 2011 | snapshots |
| 2012 | .iter() |
| 2013 | .enumerate() |
| 2014 | .filter(|(index, snapshot)| { |
| 2015 | let kind = if snapshot.label.starts_with("pre-turn:") { |
| 2016 | Some(true) |
| 2017 | } else if snapshot.label.starts_with("post-turn:") { |
| 2018 | Some(false) |
| 2019 | } else { |
| 2020 | None |
| 2021 | }; |
| 2022 | let newest_boundary = kind |
| 2023 | .is_some_and(|pre| boundaries_seen.insert((snapshot.session_id.as_deref(), pre))); |
| 2024 | *index == 0 || *index < keep || newest_boundary |
| 2025 | }) |
| 2026 | .map(|(_, snapshot)| snapshot.clone()) |
| 2027 | .collect() |
| 2028 | } |
| 2029 | |
| 2030 | /// This thread's claim on a side repo's write lock, released on drop. |
| 2031 | struct HeldSnapshotLock(PathBuf); |
| 2032 | |
| 2033 | impl HeldSnapshotLock { |
| 2034 | fn claim(git_dir: &Path) -> Self { |
| 2035 | HELD_SNAPSHOT_LOCKS.with(|held| held.borrow_mut().push(git_dir.to_path_buf())); |
| 2036 | Self(git_dir.to_path_buf()) |
| 2037 | } |
| 2038 | } |
| 2039 | |
| 2040 | impl Drop for HeldSnapshotLock { |
| 2041 | fn drop(&mut self) { |
| 2042 | HELD_SNAPSHOT_LOCKS.with(|held| { |
| 2043 | let mut held = held.borrow_mut(); |
| 2044 | if let Some(index) = held.iter().rposition(|path| path == &self.0) { |
| 2045 | held.remove(index); |
| 2046 | } |
| 2047 | }); |
| 2048 | } |
| 2049 | } |
| 2050 | |
| 2051 | /// Open (creating if needed) the side repo's lock file without following a |
| 2052 | /// symlink planted in its place. |
| 2053 | fn open_snapshot_lock_file(path: &Path) -> io::Result<std::fs::File> { |
| 2054 | let mut options = std::fs::OpenOptions::new(); |
| 2055 | options.create(true).truncate(false).read(true).write(true); |
| 2056 | #[cfg(unix)] |
| 2057 | { |
| 2058 | use std::os::unix::fs::OpenOptionsExt; |
| 2059 | options |
| 2060 | .mode(0o600) |
| 2061 | .custom_flags(libc::O_NOFOLLOW | libc::O_CLOEXEC); |
| 2062 | } |
| 2063 | options.open(path) |
| 2064 | } |
| 2065 | |
| 2066 | /// Keep the side repo's `info/exclude` at [`BUILTIN_EXCLUDES`]. Every open |
| 2067 | /// runs this without the write lock, while another session may be inside |
| 2068 | /// `git add -A` reading the file, so it is left alone when already current |
| 2069 | /// and otherwise replaced by rename: a truncate-then-write let that reader |
| 2070 | /// see an empty exclude list and stage `node_modules/`, `target/` and the |
| 2071 | /// like into the shared side repo. |
| 2072 | fn write_builtin_excludes(git_dir: &Path) -> io::Result<()> { |
| 2073 | let info_dir = git_dir.join("info"); |
| 2074 | let exclude = info_dir.join("exclude"); |
| 2075 | if std::fs::read(&exclude).is_ok_and(|current| current == BUILTIN_EXCLUDES.as_bytes()) { |
| 2076 | return Ok(()); |
| 2077 | } |
| 2078 | std::fs::create_dir_all(&info_dir)?; |
| 2079 | crate::utils::write_atomic(&exclude, BUILTIN_EXCLUDES.as_bytes()) |
| 2080 | } |
| 2081 | |
| 2082 | /// Recursively compute the total size of a directory in bytes. |
| 2083 | fn dir_size_bytes(root: &Path) -> io::Result<u64> { |
| 2084 | fn walk(dir: &Path, total: &mut u64) -> io::Result<()> { |
| 2085 | if !dir.is_dir() { |
| 2086 | return Ok(()); |
| 2087 | } |
| 2088 | for entry in std::fs::read_dir(dir)? { |
| 2089 | let entry = entry?; |
| 2090 | let path = entry.path(); |
| 2091 | let ft = entry.file_type()?; |
| 2092 | if ft.is_symlink() { |
| 2093 | continue; |
| 2094 | } |
| 2095 | if ft.is_dir() { |
| 2096 | walk(&path, total)?; |
| 2097 | } else if ft.is_file() { |
| 2098 | *total = total.saturating_add(entry.metadata().map(|m| m.len()).unwrap_or(0)); |
| 2099 | } |
| 2100 | } |
| 2101 | Ok(()) |
| 2102 | } |
| 2103 | let mut total: u64 = 0; |
| 2104 | walk(root, &mut total)?; |
| 2105 | Ok(total) |
| 2106 | } |
| 2107 | |
| 2108 | /// One prominent notice per workspace per process when the size-pressure |
| 2109 | /// prune destroys restore points — silent loss of undo history is the S5 |
| 2110 | /// failure mode (2026-08-04 snapshot hunt). The stderr print is deliberate: |
| 2111 | /// headless/CLI stderr is the user surface for once-per-workspace snapshot |
| 2112 | /// warnings, matching `maybe_notify_snapshots_disabled_once` in |
| 2113 | /// `core/turn.rs`. |
| 2114 | #[allow(clippy::print_stderr)] |
| 2115 | fn notify_snapshot_history_pruned_once(workspace: &Path, removed: usize) { |
| 2116 | use std::collections::HashSet; |
| 2117 | use std::sync::{Mutex, OnceLock}; |
| 2118 | static NOTIFIED: OnceLock<Mutex<HashSet<String>>> = OnceLock::new(); |
| 2119 | let key = workspace.to_string_lossy().into_owned(); |
| 2120 | let set = NOTIFIED.get_or_init(|| Mutex::new(HashSet::new())); |
| 2121 | let Ok(mut guard) = set.lock() else { |
| 2122 | return; |
| 2123 | }; |
| 2124 | if !guard.insert(key) { |
| 2125 | return; |
| 2126 | } |
| 2127 | drop(guard); |
| 2128 | eprint!("{}", snapshot_history_pruned_message(workspace, removed)); |
| 2129 | } |
| 2130 | |
| 2131 | /// Build the user-visible notice for a size-pressure prune. Kept pure and |
| 2132 | /// separate from the emit/dedup shell so the content is unit-testable. |
| 2133 | fn snapshot_history_pruned_message(workspace: &Path, removed: usize) -> String { |
| 2134 | format!( |
| 2135 | "warning: snapshot/undo history for {} was pruned to stay under the {} MB snapshot storage cap. |
| 2136 | {} snapshot(s) were removed and can no longer be restored. |
| 2137 | The cap bounds the undo side-repo's disk use; high-churn or large workspaces hit it sooner. |
| 2138 | ", |
| 2139 | workspace.display(), |
| 2140 | MAX_SNAPSHOT_SIZE_MB, |
| 2141 | removed |
| 2142 | ) |
| 2143 | } |
| 2144 | |
| 2145 | fn cleanup_stale_pack_temps(git_dir: &Path, stale_age: Duration) -> io::Result<usize> { |
| 2146 | let pack_dir = git_dir.join("objects").join("pack"); |
| 2147 | if !pack_dir.exists() { |
| 2148 | return Ok(0); |
| 2149 | } |
| 2150 | cleanup_stale_pack_temps_in(&pack_dir, stale_age, SystemTime::now()) |
| 2151 | } |
| 2152 | |
| 2153 | fn cleanup_stale_pack_temps_in( |
| 2154 | pack_dir: &Path, |
| 2155 | stale_age: Duration, |
| 2156 | now: SystemTime, |
| 2157 | ) -> io::Result<usize> { |
| 2158 | let mut removed = 0; |
| 2159 | for entry in std::fs::read_dir(pack_dir)? { |
| 2160 | let entry = entry?; |
| 2161 | let name = entry.file_name(); |
| 2162 | let Some(name) = name.to_str() else { |
| 2163 | continue; |
| 2164 | }; |
| 2165 | if !name.starts_with("tmp_pack_") { |
| 2166 | continue; |
| 2167 | } |
| 2168 | if !entry.file_type()?.is_file() { |
| 2169 | continue; |
| 2170 | } |
| 2171 | |
| 2172 | let metadata = entry.metadata()?; |
| 2173 | let Ok(modified) = metadata.modified() else { |
| 2174 | continue; |
| 2175 | }; |
| 2176 | let Ok(age) = now.duration_since(modified) else { |
| 2177 | continue; |
| 2178 | }; |
| 2179 | if age < stale_age { |
| 2180 | continue; |
| 2181 | } |
| 2182 | |
| 2183 | match std::fs::remove_file(entry.path()) { |
| 2184 | Ok(()) => removed += 1, |
| 2185 | Err(err) if err.kind() == io::ErrorKind::NotFound => {} |
| 2186 | Err(err) => return Err(err), |
| 2187 | } |
| 2188 | } |
| 2189 | Ok(removed) |
| 2190 | } |
| 2191 | |
| 2192 | // Generous budget: `git add -A` on a large workspace is legitimately slow, |
| 2193 | // but a wedged git (stalled NFS/FUSE, hung hook) must not block the turn |
| 2194 | // pipeline forever — every caller treats a snapshot error as |
| 2195 | // snapshot-disabled-with-warning and proceeds without the git data. Tests |
| 2196 | // use a tighter budget so a regression that deadlocks a child on its own |
| 2197 | // output fails in seconds instead of hanging for the full window. |
| 2198 | #[cfg(not(test))] |
| 2199 | const GIT_COMMAND_TIMEOUT: Duration = Duration::from_secs(300); |
| 2200 | #[cfg(test)] |
| 2201 | const GIT_COMMAND_TIMEOUT: Duration = Duration::from_secs(30); |
| 2202 | |
| 2203 | /// Grace granted to the pipe readers after git has exited. A clean git |
| 2204 | /// closes its own write ends, so EOF is already waiting; only a grandchild |
| 2205 | /// that inherited the pipes (a post-checkout hook, a `git gc` pack worker) |
| 2206 | /// can hold them past exit, and it must not hold the turn pipeline either. |
| 2207 | const GIT_PIPE_DRAIN_GRACE: Duration = Duration::from_secs(2); |
| 2208 | |
| 2209 | /// Run a pre-configured git command under [`GIT_COMMAND_TIMEOUT`] with both |
| 2210 | /// pipes drained while the child runs (the same concurrent drain |
| 2211 | /// `Command::output` performs). Every git invocation in this module goes |
| 2212 | /// through here, so a wedged git — stalled NFS/FUSE, hung hook — degrades |
| 2213 | /// the snapshot with an error instead of hanging the turn pipeline. |
| 2214 | /// (`delta.rs`'s read-view git commands are not routed here yet; that is a |
| 2215 | /// known follow-up.) |
| 2216 | /// |
| 2217 | /// The timeout path kills the child and reaps it on a detached thread: on a |
| 2218 | /// hard-wedged mount git can sit in uninterruptible kernel I/O where even |
| 2219 | /// SIGKILL is deferred, and a blocking `wait()` would hang the pipeline |
| 2220 | /// exactly like the wedged git would. Killing without git's own cleanup can |
| 2221 | /// leave a fresh `index.lock` behind; later snapshots then fail fast on the |
| 2222 | /// lock with an error naming it. |
| 2223 | fn run_bounded_git(cmd: &mut std::process::Command, subcommand: &str) -> io::Result<Output> { |
| 2224 | run_bounded_git_with_timeout(cmd, subcommand, GIT_COMMAND_TIMEOUT) |
| 2225 | } |
| 2226 | |
| 2227 | fn run_bounded_git_with_timeout( |
| 2228 | cmd: &mut std::process::Command, |
| 2229 | subcommand: &str, |
| 2230 | timeout: Duration, |
| 2231 | ) -> io::Result<Output> { |
| 2232 | cmd.stdin(std::process::Stdio::null()) |
| 2233 | .stdout(std::process::Stdio::piped()) |
| 2234 | .stderr(std::process::Stdio::piped()); |
| 2235 | let mut child = cmd.spawn()?; |
| 2236 | // Drain both pipes while waiting: the restore path's `ls-tree -r` and |
| 2237 | // the diff commands emit output that grows with workspace size, and a |
| 2238 | // child blocked on a full pipe buffer never exits — it would turn every |
| 2239 | // such call into a guaranteed timeout. Readers stream into shared |
| 2240 | // buffers, and the completion channel lets the collect below bound how |
| 2241 | // long a grandchild-held pipe may hold the call after git has exited. |
| 2242 | let (done_tx, done_rx) = std::sync::mpsc::channel::<()>(); |
| 2243 | let stdout_pipe = child.stdout.take(); |
| 2244 | let stderr_pipe = child.stderr.take(); |
| 2245 | let stdout_buf = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); |
| 2246 | let stderr_buf = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); |
| 2247 | { |
| 2248 | let buf = std::sync::Arc::clone(&stdout_buf); |
| 2249 | let done_tx = done_tx.clone(); |
| 2250 | std::thread::spawn(move || { |
| 2251 | if let Some(mut reader) = stdout_pipe { |
| 2252 | read_pipe_to_buffer(&mut reader, &buf); |
| 2253 | } |
| 2254 | let _ = done_tx.send(()); |
| 2255 | }); |
| 2256 | } |
| 2257 | { |
| 2258 | let buf = std::sync::Arc::clone(&stderr_buf); |
| 2259 | let done_tx = done_tx.clone(); |
| 2260 | std::thread::spawn(move || { |
| 2261 | if let Some(mut reader) = stderr_pipe { |
| 2262 | read_pipe_to_buffer(&mut reader, &buf); |
| 2263 | } |
| 2264 | let _ = done_tx.send(()); |
| 2265 | }); |
| 2266 | } |
| 2267 | drop(done_tx); |
| 2268 | |
| 2269 | let Some(status) = child.wait_timeout(timeout)? else { |
| 2270 | let _ = child.kill(); |
| 2271 | // Reap off the pipeline thread (see the doc comment): the kernel may |
| 2272 | // not deliver the kill until an uninterruptible syscall returns. |
| 2273 | std::thread::spawn(move || { |
| 2274 | let _ = child.wait(); |
| 2275 | }); |
| 2276 | return Err(io::Error::new( |
| 2277 | io::ErrorKind::TimedOut, |
| 2278 | format!("git {subcommand} timed out after {}s", timeout.as_secs()), |
| 2279 | )); |
| 2280 | }; |
| 2281 | // Wait for the readers up to the grace, then return what was captured. A |
| 2282 | // clean git already closed its write ends, so both readers deliver |
| 2283 | // immediately; the grace only covers a pipe still held open by an |
| 2284 | // inherited copy (a daemonizing hook). On expiry the captured output is |
| 2285 | // returned with a note on stderr instead of silently truncating; the |
| 2286 | // reader itself ends whenever whatever holds the pipe exits, holding |
| 2287 | // only a pipe read — never the turn pipeline. |
| 2288 | let mut partial = false; |
| 2289 | let mut drained = 0; |
| 2290 | while drained < 2 { |
| 2291 | match done_rx.recv_timeout(GIT_PIPE_DRAIN_GRACE) { |
| 2292 | Ok(()) => drained += 1, |
| 2293 | Err(_) => { |
| 2294 | partial = true; |
| 2295 | break; |
| 2296 | } |
| 2297 | } |
| 2298 | } |
| 2299 | let stdout = stdout_buf.lock().map(|buf| buf.clone()).unwrap_or_default(); |
| 2300 | let mut stderr = stderr_buf.lock().map(|buf| buf.clone()).unwrap_or_default(); |
| 2301 | if partial { |
| 2302 | if !stderr.is_empty() && stderr.last() != Some(&b'\n') { |
| 2303 | stderr.push(b'\n'); |
| 2304 | } |
| 2305 | stderr.extend_from_slice( |
| 2306 | b"[codewhale] git output pipes did not close after git exited \ |
| 2307 | (kept open by a hook or subprocess?); captured output may be partial\n", |
| 2308 | ); |
| 2309 | } |
| 2310 | Ok(Output { |
| 2311 | status, |
| 2312 | stdout, |
| 2313 | stderr, |
| 2314 | }) |
| 2315 | } |
| 2316 | |
| 2317 | /// Read a child's pipe to end into a shared buffer, tolerating interruption. |
| 2318 | fn read_pipe_to_buffer(reader: &mut impl io::Read, buf: &std::sync::Mutex<Vec<u8>>) { |
| 2319 | let mut chunk = [0_u8; 8192]; |
| 2320 | loop { |
| 2321 | match reader.read(&mut chunk) { |
| 2322 | Ok(0) => return, |
| 2323 | Ok(read) => { |
| 2324 | if let Ok(mut buf) = buf.lock() { |
| 2325 | buf.extend_from_slice(&chunk[..read]); |
| 2326 | } |
| 2327 | } |
| 2328 | Err(err) if err.kind() == io::ErrorKind::Interrupted => continue, |
| 2329 | Err(_) => return, |
| 2330 | } |
| 2331 | } |
| 2332 | } |
| 2333 | |
| 2334 | fn run_git(git_dir: &Path, work_tree: &Path, args: &[&str]) -> io::Result<Output> { |
| 2335 | let mut cmd = crate::dependencies::Git::command() |
| 2336 | .ok_or_else(|| io::Error::new(io::ErrorKind::NotFound, "git not found on PATH"))?; |
| 2337 | cmd.arg("--git-dir") |
| 2338 | .arg(git_dir) |
| 2339 | .arg("--work-tree") |
| 2340 | .arg(work_tree) |
| 2341 | .args(args); |
| 2342 | run_bounded_git(&mut cmd, args.first().copied().unwrap_or("git")) |
| 2343 | } |
| 2344 | |
| 2345 | fn git_diff_matches(output: Output) -> io::Result<bool> { |
| 2346 | match output.status.code() { |
| 2347 | Some(0) => Ok(true), |
| 2348 | Some(1) => Ok(false), |
| 2349 | _ => Err(io_other(format!( |
| 2350 | "git diff failed: {}", |
| 2351 | String::from_utf8_lossy(&output.stderr).trim() |
| 2352 | ))), |
| 2353 | } |
| 2354 | } |
| 2355 | |
| 2356 | fn io_other(msg: impl Into<String>) -> io::Error { |
| 2357 | io::Error::other(msg.into()) |
| 2358 | } |
| 2359 | |
| 2360 | /// Walk `workspace` and accumulate file sizes, returning `Ok(total)` |
| 2361 | /// when the workspace fits under `cap_bytes` and `Err(gate)` naming the |
| 2362 | /// bound that tripped. Honors `.gitignore` — whether or not the |
| 2363 | /// workspace is itself a git repo, matching the `git add -A` that the |
| 2364 | /// snapshot commit actually runs against this work tree — and the |
| 2365 | /// snapshot-specific skip list above, so the measured size reflects |
| 2366 | /// what would land in a snapshot commit rather than the raw `du -sh` |
| 2367 | /// total. |
| 2368 | /// |
| 2369 | /// The walk is bounded by both `cap_bytes` and `max_entries`, and the |
| 2370 | /// two bounds are reported separately because they have different |
| 2371 | /// recoveries. A `cap_bytes` of `0` disables the byte cap entirely (so |
| 2372 | /// config can opt out) but not the entry bound. |
| 2373 | /// |
| 2374 | /// Production passes [`SIZE_WALK_MAX_ENTRIES`] for `max_entries`; it is |
| 2375 | /// a parameter only so the entry bound is reachable in a test without |
| 2376 | /// creating 200,000 inodes. It must not be threaded up through |
| 2377 | /// [`SnapshotRepo::open_or_init_with_cap`]: [`WorkspaceGate::describe`] |
| 2378 | /// interpolates the constant into the user-facing message, so a weaker |
| 2379 | /// injected bound would report a number that did not trip. |
| 2380 | pub fn estimate_workspace_size_bounded( |
| 2381 | workspace: &Path, |
| 2382 | cap_bytes: u64, |
| 2383 | max_entries: usize, |
| 2384 | ) -> Result<u64, WorkspaceGate> { |
| 2385 | use ignore::WalkBuilder; |
| 2386 | let mut total: u64 = 0; |
| 2387 | let mut entries: usize = 0; |
| 2388 | let skip: HashSet<&'static str> = SIZE_WALK_SKIP_DIRS.iter().copied().collect(); |
| 2389 | let walker = WalkBuilder::new(workspace) |
| 2390 | .hidden(false) |
| 2391 | // `ignore` defaults to `require_git(true)`, which silently disables |
| 2392 | // every gitignore rule when the workspace is not inside a git repo. |
| 2393 | // The snapshot's own `git add -A` honors `.gitignore` regardless, so |
| 2394 | // without this the estimator over-counts a non-git workspace and can |
| 2395 | // refuse it while offering a `.gitignore` remedy that cannot work. |
| 2396 | .require_git(false) |
| 2397 | .follow_links(false) |
| 2398 | .filter_entry(move |entry| { |
| 2399 | // Skip the well-known build-output directories at any depth. |
| 2400 | // The `ignore` crate calls `filter_entry` once per dir/file; |
| 2401 | // returning `false` here prunes the whole subtree. |
| 2402 | entry |
| 2403 | .file_name() |
| 2404 | .to_str() |
| 2405 | .is_none_or(|name| !skip.contains(name)) |
| 2406 | }) |
| 2407 | .build(); |
| 2408 | for entry in walker.flatten() { |
| 2409 | entries += 1; |
| 2410 | if entries > max_entries { |
| 2411 | return Err(WorkspaceGate::TooManyEntries); |
| 2412 | } |
| 2413 | if let Ok(meta) = entry.metadata() |
| 2414 | && meta.is_file() |
| 2415 | { |
| 2416 | total = total.saturating_add(meta.len()); |
| 2417 | if cap_bytes > 0 && total > cap_bytes { |
| 2418 | return Err(WorkspaceGate::TooLarge); |
| 2419 | } |
| 2420 | } |
| 2421 | } |
| 2422 | Ok(total) |
| 2423 | } |
| 2424 | |
| 2425 | pub(crate) fn unsafe_workspace_snapshot_reason( |
| 2426 | workspace: &Path, |
| 2427 | home: Option<&Path>, |
| 2428 | ) -> Option<&'static str> { |
| 2429 | let workspace = normalize_path_for_safety(workspace); |
| 2430 | if is_filesystem_root(&workspace) { |
| 2431 | return Some("filesystem root"); |
| 2432 | } |
| 2433 | |
| 2434 | if is_home_directory(&workspace, home) { |
| 2435 | return Some("home directory"); |
| 2436 | } |
| 2437 | |
| 2438 | let home = home.map(normalize_path_for_safety)?; |
| 2439 | if workspace.parent() == Some(home.as_path()) { |
| 2440 | let name = workspace.file_name().and_then(|name| name.to_str()); |
| 2441 | if matches!( |
| 2442 | name, |
| 2443 | Some( |
| 2444 | "Desktop" | "Documents" | "Downloads" | "Library" | "Movies" | "Music" | "Pictures" |
| 2445 | ) |
| 2446 | ) { |
| 2447 | return Some("home collection directory"); |
| 2448 | } |
| 2449 | } |
| 2450 | |
| 2451 | None |
| 2452 | } |
| 2453 | |
| 2454 | fn normalize_path_for_safety(path: &Path) -> PathBuf { |
| 2455 | path.canonicalize().unwrap_or_else(|_| path.to_path_buf()) |
| 2456 | } |
| 2457 | |
| 2458 | fn is_filesystem_root(path: &Path) -> bool { |
| 2459 | path.parent().is_none() |
| 2460 | } |
| 2461 | |
| 2462 | fn is_home_directory(work_tree: &Path, home: Option<&Path>) -> bool { |
| 2463 | let Some(home) = home else { |
| 2464 | return false; |
| 2465 | }; |
| 2466 | |
| 2467 | let home_canonical = home.canonicalize().unwrap_or_else(|_| home.to_path_buf()); |
| 2468 | work_tree == home_canonical |
| 2469 | } |
| 2470 | |
| 2471 | fn parse_nul_paths(bytes: &[u8]) -> HashSet<PathBuf> { |
| 2472 | bytes |
| 2473 | .split(|b| *b == 0) |
| 2474 | .filter(|chunk| !chunk.is_empty()) |
| 2475 | .map(|chunk| PathBuf::from(String::from_utf8_lossy(chunk).into_owned())) |
| 2476 | .collect() |
| 2477 | } |
| 2478 | |
| 2479 | fn is_safe_relative_path(path: &Path) -> bool { |
| 2480 | !path.as_os_str().is_empty() |
| 2481 | && path |
| 2482 | .components() |
| 2483 | .all(|component| matches!(component, Component::Normal(_))) |
| 2484 | } |
| 2485 | |
| 2486 | /// Whether one path component names the repository metadata directory as the |
| 2487 | /// filesystem resolves it, not just as spelled: `.git` in any letter case |
| 2488 | /// (macOS and Windows default to case-insensitive names) and, on Windows, |
| 2489 | /// with the trailing dots/spaces or `:stream` suffix it drops and the `GIT~N` |
| 2490 | /// short-name alias. The one `.git` rule for workspace file routes, file |
| 2491 | /// restore, displayed workspace paths, the write carve-out and sub-agent |
| 2492 | /// deliverables. |
| 2493 | pub fn is_git_metadata_name(name: &std::ffi::OsStr) -> bool { |
| 2494 | git_metadata_name(&name.to_string_lossy(), cfg!(windows)) |
| 2495 | } |
| 2496 | |
| 2497 | fn git_metadata_name(name: &str, windows: bool) -> bool { |
| 2498 | let name = if windows { |
| 2499 | name.split(':') |
| 2500 | .next() |
| 2501 | .unwrap_or_default() |
| 2502 | .trim_end_matches(['.', ' ']) |
| 2503 | } else { |
| 2504 | name |
| 2505 | }; |
| 2506 | name.eq_ignore_ascii_case(".git") |
| 2507 | || (windows |
| 2508 | && name.len() > 4 |
| 2509 | && name |
| 2510 | .get(..4) |
| 2511 | .is_some_and(|prefix| prefix.eq_ignore_ascii_case("git~")) |
| 2512 | && name[4..].bytes().all(|byte| byte.is_ascii_digit())) |
| 2513 | } |
| 2514 | |
| 2515 | /// Normalize a caller-supplied path into a safe workspace-relative path. |
| 2516 | /// |
| 2517 | /// Accepts either a workspace-relative path or an absolute path inside |
| 2518 | /// `workspace`. Returns `None` when the result is not a plain relative path — |
| 2519 | /// absolute, empty, containing `..`, or pointing outside the workspace. Every |
| 2520 | /// file-scoped restore path passes through here, so a caller never gets to |
| 2521 | /// name a path the snapshot repo would resolve outside the work tree. |
| 2522 | /// |
| 2523 | /// The name is literal: leading or trailing spaces, brackets and glob |
| 2524 | /// characters are filename bytes, never trimmed and never patterns. Git is |
| 2525 | /// invoked with `--literal-pathspecs` for every file-scoped operation. |
| 2526 | pub fn workspace_relative_path(workspace: &Path, raw: &str) -> Option<PathBuf> { |
| 2527 | if raw.is_empty() { |
| 2528 | return None; |
| 2529 | } |
| 2530 | let candidate = Path::new(raw); |
| 2531 | let rel = if candidate.is_absolute() { |
| 2532 | candidate.strip_prefix(workspace).ok()?.to_path_buf() |
| 2533 | } else { |
| 2534 | candidate.to_path_buf() |
| 2535 | }; |
| 2536 | is_safe_relative_path(&rel).then_some(rel) |
| 2537 | } |
| 2538 | |
| 2539 | /// Whether some existing ancestor of `rel` (within `root`) is not a |
| 2540 | /// directory: restoring `rel` would then turn a file into a directory. |
| 2541 | fn ancestor_is_not_a_directory(root: &Path, rel: &Path) -> bool { |
| 2542 | rel.ancestors() |
| 2543 | .skip(1) |
| 2544 | .filter(|ancestor| !ancestor.as_os_str().is_empty()) |
| 2545 | .any(|ancestor| { |
| 2546 | std::fs::symlink_metadata(root.join(ancestor)).is_ok_and(|metadata| !metadata.is_dir()) |
| 2547 | }) |
| 2548 | } |
| 2549 | |
| 2550 | #[cfg(test)] |
| 2551 | mod tests { |
| 2552 | use super::*; |
| 2553 | use crate::test_support::lock_test_env; |
| 2554 | use std::fs::{File, FileTimes}; |
| 2555 | use tempfile::tempdir; |
| 2556 | |
| 2557 | #[test] |
| 2558 | fn git_metadata_name_covers_case_and_windows_aliases() { |
| 2559 | for windows in [false, true] { |
| 2560 | for name in [".git", ".GIT", ".Git", ".gIt"] { |
| 2561 | assert!(git_metadata_name(name, windows), "{name} windows={windows}"); |
| 2562 | } |
| 2563 | for name in [ |
| 2564 | ".github", |
| 2565 | ".gitignore", |
| 2566 | "a.git", |
| 2567 | "git", |
| 2568 | "", |
| 2569 | "git~", |
| 2570 | "GIT~1a", |
| 2571 | ] { |
| 2572 | assert!( |
| 2573 | !git_metadata_name(name, windows), |
| 2574 | "{name} windows={windows}" |
| 2575 | ); |
| 2576 | } |
| 2577 | } |
| 2578 | // Windows drops trailing dots/spaces and `:stream` suffixes, and |
| 2579 | // `GIT~N` is the 8.3 short name of `.git`; elsewhere these are |
| 2580 | // ordinary names. |
| 2581 | for name in [ |
| 2582 | ".git.", |
| 2583 | ".git ", |
| 2584 | ".GIT. .", |
| 2585 | ".git::$INDEX_ALLOCATION", |
| 2586 | ".git:stream", |
| 2587 | "GIT~1", |
| 2588 | "git~12", |
| 2589 | ] { |
| 2590 | assert!(git_metadata_name(name, true), "{name}"); |
| 2591 | assert!(!git_metadata_name(name, false), "{name}"); |
| 2592 | } |
| 2593 | } |
| 2594 | |
| 2595 | #[test] |
| 2596 | fn snapshot_id_parse_accepts_only_full_hex_object_ids() { |
| 2597 | let sha1 = "0123456789abcdefABCDEF0123456789abcdef01"; |
| 2598 | let sha256 = "a".repeat(64); |
| 2599 | assert_eq!(SnapshotId::parse(sha1).expect("sha1").as_str(), sha1); |
| 2600 | assert!(SnapshotId::parse(&sha256).is_ok()); |
| 2601 | for bad in [ |
| 2602 | "", |
| 2603 | "HEAD", |
| 2604 | "abc123", |
| 2605 | "--output=/tmp/x", |
| 2606 | "-0123456789abcdef0123456789abcdef0123456", |
| 2607 | "0123456789abcdef0123456789abcdef0123456g", |
| 2608 | "0123456789abcdef0123456789abcdef01234567~1", |
| 2609 | "0123456789abcdef0123456789abcdef012345678", |
| 2610 | ] { |
| 2611 | let err = SnapshotId::parse(bad).expect_err(bad); |
| 2612 | assert_eq!(err.kind(), io::ErrorKind::InvalidInput, "{bad:?}"); |
| 2613 | } |
| 2614 | } |
| 2615 | |
| 2616 | /// Holds the home directory pinned to a tempdir for the lifetime of a test. Also |
| 2617 | /// owns the process-wide env-var mutex so tests across modules |
| 2618 | /// don't trample each other's home env vars. |
| 2619 | pub(super) struct ScopedHome { |
| 2620 | _vars: Vec<crate::test_support::EnvVarGuard>, |
| 2621 | _guard: crate::test_support::TestEnvLock, |
| 2622 | } |
| 2623 | pub(super) fn scoped_home(home: &Path) -> ScopedHome { |
| 2624 | use crate::test_support::EnvVarGuard; |
| 2625 | let guard = lock_test_env(); |
| 2626 | ScopedHome { |
| 2627 | _vars: vec![ |
| 2628 | EnvVarGuard::set("HOME", home), |
| 2629 | EnvVarGuard::set("USERPROFILE", home), |
| 2630 | EnvVarGuard::remove("HOMEDRIVE"), |
| 2631 | EnvVarGuard::remove("HOMEPATH"), |
| 2632 | EnvVarGuard::set("CODEWHALE_HOME", home.join(".codewhale")), |
| 2633 | ], |
| 2634 | _guard: guard, |
| 2635 | } |
| 2636 | } |
| 2637 | |
| 2638 | /// Build a side-repo inside the test's selected profile. Return its |
| 2639 | /// environment guard so reads and writes stay isolated for the whole test. |
| 2640 | fn make_repo(tmp: &Path) -> (SnapshotRepo, ScopedHome) { |
| 2641 | let workspace = tmp.join("workspace"); |
| 2642 | std::fs::create_dir_all(&workspace).unwrap(); |
| 2643 | let guard = scoped_home(tmp); |
| 2644 | let repo = SnapshotRepo::open_or_init(&workspace).expect("open_or_init"); |
| 2645 | (repo, guard) |
| 2646 | } |
| 2647 | |
| 2648 | #[test] |
| 2649 | fn snapshot_creates_commit_in_side_repo_only() { |
| 2650 | let tmp = tempdir().unwrap(); |
| 2651 | let (repo, _home) = make_repo(tmp.path()); |
| 2652 | std::fs::write(repo.work_tree().join("a.txt"), b"alpha").unwrap(); |
| 2653 | |
| 2654 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 2655 | assert_eq!(id.as_str().len(), 40); |
| 2656 | |
| 2657 | let list = repo.list(10).expect("list"); |
| 2658 | assert_eq!(list.len(), 1); |
| 2659 | assert_eq!(list[0].label, "pre-turn:1"); |
| 2660 | |
| 2661 | // The user's workspace must NOT have a real `.git` because we |
| 2662 | // never created one in their workspace — only in the side dir. |
| 2663 | assert!(!repo.work_tree().join(".git").exists()); |
| 2664 | } |
| 2665 | |
| 2666 | /// B2: a side repo whose HEAD names a missing commit made every snapshot |
| 2667 | /// fail on `commit-tree -p` ("is not a valid object"), so /undo died |
| 2668 | /// silently. The broken ref is reset and snapshots resume. |
| 2669 | #[test] |
| 2670 | fn broken_head_is_repaired_and_snapshots_resume() { |
| 2671 | let tmp = tempdir().unwrap(); |
| 2672 | let (repo, _home) = make_repo(tmp.path()); |
| 2673 | std::fs::write(repo.work_tree().join("a.txt"), b"alpha").unwrap(); |
| 2674 | repo.snapshot("pre-turn:1").expect("first snapshot"); |
| 2675 | assert!(!repo.repair_broken_head().expect("healthy head")); |
| 2676 | |
| 2677 | repo.point_head_at_missing_commit_for_test(); |
| 2678 | assert!( |
| 2679 | repo.list(10).is_err() || repo.list(10).unwrap().is_empty(), |
| 2680 | "a broken head cannot list its history" |
| 2681 | ); |
| 2682 | |
| 2683 | // With a reflog, the last commit that still exists is restored and |
| 2684 | // the restore points before the break survive. |
| 2685 | assert!( |
| 2686 | !repo.repair_broken_head().expect("recover"), |
| 2687 | "recovered from the reflog, not restarted" |
| 2688 | ); |
| 2689 | let list = repo.list(10).expect("list after recovery"); |
| 2690 | assert_eq!(list.len(), 1, "{list:?}"); |
| 2691 | assert_eq!(list[0].label, "pre-turn:1"); |
| 2692 | |
| 2693 | // Without one, the broken ref is deleted and history restarts. |
| 2694 | repo.point_head_at_missing_commit_for_test(); |
| 2695 | std::fs::remove_dir_all(repo.git_dir.join("logs")).expect("drop reflogs"); |
| 2696 | assert!(repo.repair_broken_head().expect("repair"), "repaired once"); |
| 2697 | assert!(!repo.repair_broken_head().expect("idempotent")); |
| 2698 | std::fs::write(repo.work_tree().join("a.txt"), b"beta").unwrap(); |
| 2699 | repo.snapshot("pre-turn:2").expect("snapshots resume"); |
| 2700 | let list = repo.list(10).expect("list after repair"); |
| 2701 | assert_eq!(list.len(), 1, "history restarts at the repair: {list:?}"); |
| 2702 | assert_eq!(list[0].label, "pre-turn:2"); |
| 2703 | |
| 2704 | // The snapshot path repairs on its own too, for callers that never |
| 2705 | // asked. |
| 2706 | repo.point_head_at_missing_commit_for_test(); |
| 2707 | repo.snapshot("pre-turn:3") |
| 2708 | .expect("snapshot repairs on its own"); |
| 2709 | } |
| 2710 | |
| 2711 | #[test] |
| 2712 | fn open_existing_is_read_only_and_does_not_initialize() { |
| 2713 | let tmp = tempdir().unwrap(); |
| 2714 | let workspace = tmp.path().join("workspace"); |
| 2715 | std::fs::create_dir_all(&workspace).unwrap(); |
| 2716 | let _home = scoped_home(tmp.path()); |
| 2717 | |
| 2718 | let before = SnapshotRepo::open_existing(&workspace).expect("open existing"); |
| 2719 | assert!(before.is_none()); |
| 2720 | assert!( |
| 2721 | !snapshot_git_dir(&workspace) |
| 2722 | .expect("snapshot path") |
| 2723 | .exists(), |
| 2724 | "read-only open must not create the side repo" |
| 2725 | ); |
| 2726 | |
| 2727 | let repo = SnapshotRepo::open_or_init(&workspace).expect("open_or_init"); |
| 2728 | std::fs::write(repo.work_tree().join("a.txt"), b"alpha").unwrap(); |
| 2729 | repo.snapshot("pre-turn:1").expect("snapshot"); |
| 2730 | |
| 2731 | let after = SnapshotRepo::open_existing(&workspace).expect("open existing"); |
| 2732 | assert!(after.is_some()); |
| 2733 | } |
| 2734 | |
| 2735 | #[test] |
| 2736 | fn restore_reverts_workspace_files() { |
| 2737 | let tmp = tempdir().unwrap(); |
| 2738 | let (repo, _home) = make_repo(tmp.path()); |
| 2739 | let f = repo.work_tree().join("file.txt"); |
| 2740 | |
| 2741 | std::fs::write(&f, b"original").unwrap(); |
| 2742 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 2743 | |
| 2744 | std::fs::write(&f, b"clobbered").unwrap(); |
| 2745 | repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 2746 | |
| 2747 | repo.restore(&id).expect("restore"); |
| 2748 | let after = std::fs::read_to_string(&f).unwrap(); |
| 2749 | assert_eq!(after, "original"); |
| 2750 | } |
| 2751 | |
| 2752 | #[test] |
| 2753 | fn restore_removes_files_added_after_target_snapshot() { |
| 2754 | let tmp = tempdir().unwrap(); |
| 2755 | let (repo, _home) = make_repo(tmp.path()); |
| 2756 | let original = repo.work_tree().join("original.txt"); |
| 2757 | let added = repo.work_tree().join("added.txt"); |
| 2758 | |
| 2759 | std::fs::write(&original, b"original").unwrap(); |
| 2760 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 2761 | |
| 2762 | std::fs::write(&added, b"new file").unwrap(); |
| 2763 | repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 2764 | |
| 2765 | repo.restore(&id).expect("restore"); |
| 2766 | assert!(original.exists()); |
| 2767 | assert!(!added.exists(), "restore must remove tracked added files"); |
| 2768 | } |
| 2769 | |
| 2770 | /// The first snapshot of an empty project holds the empty tree. Restoring |
| 2771 | /// it used to fail (`pathspec ':/' did not match`) before removing |
| 2772 | /// anything, so every file created since stayed. |
| 2773 | #[test] |
| 2774 | fn restore_to_an_empty_snapshot_removes_the_files_created_since() { |
| 2775 | let tmp = tempdir().unwrap(); |
| 2776 | let (repo, _home) = make_repo(tmp.path()); |
| 2777 | let empty = repo.snapshot("pre-turn:1").expect("empty snapshot"); |
| 2778 | let created = repo.work_tree().join("src").join("main.rs"); |
| 2779 | std::fs::create_dir_all(created.parent().unwrap()).unwrap(); |
| 2780 | std::fs::write(&created, b"fn main() {}").unwrap(); |
| 2781 | std::fs::write(repo.work_tree().join("notes.txt"), b"notes").unwrap(); |
| 2782 | repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 2783 | |
| 2784 | repo.restore(&empty).expect("restore to the empty tree"); |
| 2785 | assert!(!created.exists(), "restore must remove files created since"); |
| 2786 | assert!(!repo.work_tree().join("notes.txt").exists()); |
| 2787 | assert!( |
| 2788 | !repo.work_tree().join("src").exists(), |
| 2789 | "directories the removal emptied go too" |
| 2790 | ); |
| 2791 | } |
| 2792 | |
| 2793 | #[test] |
| 2794 | fn restore_keeps_current_files_when_safety_snapshot_fails() { |
| 2795 | let tmp = tempdir().unwrap(); |
| 2796 | let (repo, _home) = make_repo(tmp.path()); |
| 2797 | let empty = repo.snapshot("pre-turn:1").unwrap(); |
| 2798 | let file = repo.work_tree().join("new-work.txt"); |
| 2799 | std::fs::write(&file, b"only copy of new work").unwrap(); |
| 2800 | repo.snapshot("post-turn:1").unwrap(); |
| 2801 | std::fs::write(repo.git_dir().join("index.lock"), b"").unwrap(); |
| 2802 | |
| 2803 | let error = repo.restore(&empty).expect_err("backup is mandatory"); |
| 2804 | assert!( |
| 2805 | error.to_string().contains("safety snapshot failed"), |
| 2806 | "{error}" |
| 2807 | ); |
| 2808 | assert_eq!(std::fs::read(&file).unwrap(), b"only copy of new work"); |
| 2809 | } |
| 2810 | |
| 2811 | #[test] |
| 2812 | fn restore_rolls_back_files_after_a_partial_checkout_error() { |
| 2813 | let tmp = tempdir().unwrap(); |
| 2814 | let (repo, _home) = make_repo(tmp.path()); |
| 2815 | let first = repo.work_tree().join("a.txt"); |
| 2816 | let last_dir = repo.work_tree().join("z"); |
| 2817 | let last = last_dir.join("last.txt"); |
| 2818 | std::fs::create_dir(&last_dir).unwrap(); |
| 2819 | std::fs::write(&first, b"old first").unwrap(); |
| 2820 | std::fs::write(&last, b"old last").unwrap(); |
| 2821 | let target = repo.snapshot("pre-turn:1").unwrap(); |
| 2822 | let object = run_git( |
| 2823 | repo.git_dir(), |
| 2824 | repo.work_tree(), |
| 2825 | &["rev-parse", &format!("{}:z/last.txt", target.as_str())], |
| 2826 | ) |
| 2827 | .unwrap(); |
| 2828 | assert!(object.status.success()); |
| 2829 | let object = String::from_utf8(object.stdout).unwrap(); |
| 2830 | let object = object.trim(); |
| 2831 | std::fs::write(&first, b"new first").unwrap(); |
| 2832 | std::fs::remove_file(&last).unwrap(); |
| 2833 | let before = repo.snapshot("before-restore-proof").unwrap(); |
| 2834 | // A missing target-only blob makes Git fail after it writes a.txt. |
| 2835 | // The backup needs neither that blob nor z/last.txt for recovery. |
| 2836 | std::fs::remove_file( |
| 2837 | repo.git_dir() |
| 2838 | .join("objects") |
| 2839 | .join(&object[..2]) |
| 2840 | .join(&object[2..]), |
| 2841 | ) |
| 2842 | .unwrap(); |
| 2843 | |
| 2844 | // Prove the failure fixture really permits a partial overwrite. |
| 2845 | let partial = run_git( |
| 2846 | repo.git_dir(), |
| 2847 | repo.work_tree(), |
| 2848 | &["checkout", target.as_str(), "--", ":/"], |
| 2849 | ) |
| 2850 | .unwrap(); |
| 2851 | assert!(!partial.status.success()); |
| 2852 | assert_eq!(std::fs::read(&first).unwrap(), b"old first"); |
| 2853 | let reset = run_git( |
| 2854 | repo.git_dir(), |
| 2855 | repo.work_tree(), |
| 2856 | &["checkout", before.as_str(), "--", ":/"], |
| 2857 | ) |
| 2858 | .unwrap(); |
| 2859 | assert!(reset.status.success()); |
| 2860 | assert_eq!(std::fs::read(&first).unwrap(), b"new first"); |
| 2861 | |
| 2862 | let error = repo.restore(&target).expect_err("target blob is missing"); |
| 2863 | assert!(error.to_string().contains("unable to read"), "{error}"); |
| 2864 | assert!( |
| 2865 | error |
| 2866 | .to_string() |
| 2867 | .contains("previous snapshot files were restored"), |
| 2868 | "{error}" |
| 2869 | ); |
| 2870 | assert!(error.to_string().contains("safety snapshot"), "{error}"); |
| 2871 | assert_eq!(std::fs::read(&first).unwrap(), b"new first"); |
| 2872 | assert!(!last.exists()); |
| 2873 | } |
| 2874 | |
| 2875 | #[test] |
| 2876 | fn restore_refuses_file_directory_transitions_without_changing_files() { |
| 2877 | for target_is_dir in [false, true] { |
| 2878 | let tmp = tempdir().unwrap(); |
| 2879 | let (repo, _home) = make_repo(tmp.path()); |
| 2880 | let path = repo.work_tree().join("a"); |
| 2881 | let child = path.join("child"); |
| 2882 | if target_is_dir { |
| 2883 | std::fs::create_dir(&path).unwrap(); |
| 2884 | std::fs::write(&child, b"old child").unwrap(); |
| 2885 | } else { |
| 2886 | std::fs::write(&path, b"old file").unwrap(); |
| 2887 | } |
| 2888 | let target = repo.snapshot("pre-turn:1").unwrap(); |
| 2889 | if target_is_dir { |
| 2890 | std::fs::remove_file(&child).unwrap(); |
| 2891 | std::fs::remove_dir(&path).unwrap(); |
| 2892 | std::fs::write(&path, b"new file").unwrap(); |
| 2893 | } else { |
| 2894 | std::fs::remove_file(&path).unwrap(); |
| 2895 | std::fs::create_dir(&path).unwrap(); |
| 2896 | std::fs::write(&child, b"new child").unwrap(); |
| 2897 | } |
| 2898 | |
| 2899 | let error = repo |
| 2900 | .restore(&target) |
| 2901 | .expect_err("type transition is refused"); |
| 2902 | assert!( |
| 2903 | error.to_string().contains("file/directory transition"), |
| 2904 | "{error}" |
| 2905 | ); |
| 2906 | if target_is_dir { |
| 2907 | assert_eq!(std::fs::read(&path).unwrap(), b"new file"); |
| 2908 | } else { |
| 2909 | assert_eq!(std::fs::read(&child).unwrap(), b"new child"); |
| 2910 | } |
| 2911 | } |
| 2912 | } |
| 2913 | |
| 2914 | #[test] |
| 2915 | fn a_path_under_a_live_file_needs_a_transition() { |
| 2916 | let tmp = tempdir().unwrap(); |
| 2917 | std::fs::write(tmp.path().join("a"), b"file").unwrap(); |
| 2918 | std::fs::create_dir(tmp.path().join("d")).unwrap(); |
| 2919 | assert!(super::ancestor_is_not_a_directory( |
| 2920 | tmp.path(), |
| 2921 | Path::new("a/child") |
| 2922 | )); |
| 2923 | assert!(!super::ancestor_is_not_a_directory( |
| 2924 | tmp.path(), |
| 2925 | Path::new("d/child") |
| 2926 | )); |
| 2927 | assert!(!super::ancestor_is_not_a_directory( |
| 2928 | tmp.path(), |
| 2929 | Path::new("missing/child") |
| 2930 | )); |
| 2931 | assert!(!super::ancestor_is_not_a_directory( |
| 2932 | tmp.path(), |
| 2933 | Path::new("a") |
| 2934 | )); |
| 2935 | } |
| 2936 | |
| 2937 | /// A failed safety snapshot refuses before a symlinked parent could be |
| 2938 | /// followed, even when restoring to an empty target skips checkout. |
| 2939 | #[cfg(unix)] |
| 2940 | #[test] |
| 2941 | fn restore_refuses_to_remove_through_a_symlinked_parent() { |
| 2942 | let tmp = tempdir().unwrap(); |
| 2943 | let (repo, _home) = make_repo(tmp.path()); |
| 2944 | let empty = repo.snapshot("pre-turn:1").expect("empty snapshot"); |
| 2945 | let src = repo.work_tree().join("src"); |
| 2946 | std::fs::create_dir_all(&src).unwrap(); |
| 2947 | std::fs::write(src.join("victim"), b"inside").unwrap(); |
| 2948 | repo.snapshot("post-turn:1") |
| 2949 | .expect("snapshot holding src/victim"); |
| 2950 | |
| 2951 | let outside = tempdir().unwrap(); |
| 2952 | let sentinel = outside.path().join("victim"); |
| 2953 | std::fs::write(&sentinel, b"outside").unwrap(); |
| 2954 | std::fs::remove_dir_all(&src).unwrap(); |
| 2955 | std::os::unix::fs::symlink(outside.path(), &src).unwrap(); |
| 2956 | // A stale index lock makes the mandatory safety snapshot fail. |
| 2957 | std::fs::write(repo.git_dir().join("index.lock"), b"").unwrap(); |
| 2958 | |
| 2959 | let err = repo |
| 2960 | .restore(&empty) |
| 2961 | .expect_err("a removal through a symlinked parent is refused"); |
| 2962 | assert!(err.to_string().contains("safety snapshot failed"), "{err}"); |
| 2963 | assert_eq!( |
| 2964 | std::fs::read(&sentinel).unwrap(), |
| 2965 | b"outside", |
| 2966 | "the file outside the workspace survives" |
| 2967 | ); |
| 2968 | assert!( |
| 2969 | std::fs::symlink_metadata(&src) |
| 2970 | .unwrap() |
| 2971 | .file_type() |
| 2972 | .is_symlink() |
| 2973 | ); |
| 2974 | } |
| 2975 | |
| 2976 | /// Every session, turn and sub-agent in a workspace shares its side repo. |
| 2977 | /// A snapshot must wait while another writer holds the repo's write lock |
| 2978 | /// instead of racing its index, HEAD and gc. |
| 2979 | #[test] |
| 2980 | fn snapshot_waits_for_the_side_repo_write_lock() { |
| 2981 | let tmp = tempdir().unwrap(); |
| 2982 | let (repo, _home) = make_repo(tmp.path()); |
| 2983 | std::fs::write(repo.work_tree().join("f.txt"), b"v0").unwrap(); |
| 2984 | repo.snapshot("pre-turn:1").expect("first snapshot"); |
| 2985 | |
| 2986 | let lock_path = repo.git_dir().join(SNAPSHOT_LOCK_FILE); |
| 2987 | let mut peer = fd_lock::RwLock::new(open_snapshot_lock_file(&lock_path).unwrap()); |
| 2988 | let held = peer.write().expect("peer holds the write lock"); |
| 2989 | |
| 2990 | let (done_tx, done_rx) = std::sync::mpsc::channel(); |
| 2991 | let git_dir = repo.git_dir().to_path_buf(); |
| 2992 | let work_tree = repo.work_tree().to_path_buf(); |
| 2993 | let writer = std::thread::spawn(move || { |
| 2994 | let repo = SnapshotRepo { git_dir, work_tree }; |
| 2995 | std::fs::write(repo.work_tree().join("f.txt"), b"v1").unwrap(); |
| 2996 | let taken = repo.snapshot("post-turn:1"); |
| 2997 | let _ = done_tx.send(()); |
| 2998 | taken |
| 2999 | }); |
| 3000 | assert!( |
| 3001 | done_rx.recv_timeout(Duration::from_millis(400)).is_err(), |
| 3002 | "a snapshot must not run while another writer holds the lock" |
| 3003 | ); |
| 3004 | drop(held); |
| 3005 | done_rx |
| 3006 | .recv_timeout(Duration::from_secs(30)) |
| 3007 | .expect("snapshot proceeds once the lock is released"); |
| 3008 | writer |
| 3009 | .join() |
| 3010 | .expect("writer thread") |
| 3011 | .expect("snapshot after release"); |
| 3012 | assert_eq!(repo.list(usize::MAX).unwrap().len(), 2); |
| 3013 | } |
| 3014 | |
| 3015 | /// Two writers sharing one side repo, each snapshotting and pruning, must |
| 3016 | /// all succeed and leave a history whose every snapshot still restores. |
| 3017 | #[test] |
| 3018 | fn concurrent_snapshot_and_prune_keep_the_side_repo_consistent() { |
| 3019 | let tmp = tempdir().unwrap(); |
| 3020 | let (repo, _home) = make_repo(tmp.path()); |
| 3021 | std::fs::write(repo.work_tree().join("seed.txt"), b"seed").unwrap(); |
| 3022 | repo.snapshot("pre-turn:0").expect("seed snapshot"); |
| 3023 | let git_dir = repo.git_dir().to_path_buf(); |
| 3024 | let work_tree = repo.work_tree().to_path_buf(); |
| 3025 | let writers: Vec<_> = (0..2) |
| 3026 | .map(|writer| { |
| 3027 | let git_dir = git_dir.clone(); |
| 3028 | let work_tree = work_tree.clone(); |
| 3029 | std::thread::spawn(move || -> io::Result<()> { |
| 3030 | let repo = SnapshotRepo { git_dir, work_tree }; |
| 3031 | for turn in 0..6 { |
| 3032 | std::fs::write( |
| 3033 | repo.work_tree().join(format!("w{writer}.txt")), |
| 3034 | format!("{writer}-{turn}"), |
| 3035 | )?; |
| 3036 | repo.snapshot(&format!("post-turn:{writer}-{turn}"))?; |
| 3037 | repo.prune_keep_last_n(2)?; |
| 3038 | } |
| 3039 | Ok(()) |
| 3040 | }) |
| 3041 | }) |
| 3042 | .collect(); |
| 3043 | for writer in writers { |
| 3044 | writer |
| 3045 | .join() |
| 3046 | .expect("writer thread") |
| 3047 | .expect("writer ran clean"); |
| 3048 | } |
| 3049 | assert!(!repo.repair_broken_head().expect("head check")); |
| 3050 | let history = repo.list(usize::MAX).expect("history lists"); |
| 3051 | assert!(!history.is_empty()); |
| 3052 | for snapshot in &history { |
| 3053 | assert!( |
| 3054 | repo.is_commit(snapshot.id.as_str()).unwrap(), |
| 3055 | "listed snapshot {} must exist", |
| 3056 | snapshot.id.as_str() |
| 3057 | ); |
| 3058 | } |
| 3059 | } |
| 3060 | |
| 3061 | /// A peer stuck holding the lock (a long gc, a stopped process) used to |
| 3062 | /// block the snapshot, and the turn waiting on it, with no end. The wait |
| 3063 | /// is bounded and ends in an error the turn reports. |
| 3064 | #[test] |
| 3065 | fn side_repo_lock_wait_is_bounded() { |
| 3066 | let tmp = tempdir().unwrap(); |
| 3067 | let (repo, _home) = make_repo(tmp.path()); |
| 3068 | let lock_path = repo.git_dir().join(SNAPSHOT_LOCK_FILE); |
| 3069 | let mut peer = fd_lock::RwLock::new(open_snapshot_lock_file(&lock_path).unwrap()); |
| 3070 | let _held = peer.write().expect("peer holds the write lock"); |
| 3071 | |
| 3072 | let started = std::time::Instant::now(); |
| 3073 | let err = repo |
| 3074 | .with_write_lock_within(Duration::from_millis(200), || Ok(())) |
| 3075 | .expect_err("a held lock times out"); |
| 3076 | assert_eq!(err.kind(), io::ErrorKind::TimedOut, "{err}"); |
| 3077 | assert!(err.to_string().contains("busy"), "{err}"); |
| 3078 | assert!( |
| 3079 | started.elapsed() < Duration::from_secs(10), |
| 3080 | "the wait ends near its bound" |
| 3081 | ); |
| 3082 | } |
| 3083 | |
| 3084 | /// First-time init (git init plus the pinned identity) runs under the |
| 3085 | /// write lock, so two sessions opening a new workspace do not race it; a |
| 3086 | /// `.git` holding only a peer's lock file is initialized, not trusted. |
| 3087 | #[test] |
| 3088 | fn first_init_waits_for_the_side_repo_write_lock() { |
| 3089 | let tmp = tempdir().unwrap(); |
| 3090 | let (repo, _home) = make_repo(tmp.path()); |
| 3091 | let git_dir = repo.git_dir().to_path_buf(); |
| 3092 | let workspace = repo.work_tree().to_path_buf(); |
| 3093 | drop(repo); |
| 3094 | std::fs::remove_dir_all(&git_dir).unwrap(); |
| 3095 | std::fs::create_dir_all(&git_dir).unwrap(); |
| 3096 | let mut peer = fd_lock::RwLock::new( |
| 3097 | open_snapshot_lock_file(&git_dir.join(SNAPSHOT_LOCK_FILE)).unwrap(), |
| 3098 | ); |
| 3099 | let held = peer.write().expect("peer holds the write lock"); |
| 3100 | |
| 3101 | let (done_tx, done_rx) = std::sync::mpsc::channel(); |
| 3102 | let env_ticket = crate::test_support::env_scope_ticket(); |
| 3103 | let opener = std::thread::spawn(move || { |
| 3104 | let _membership = crate::test_support::join_env_scope(env_ticket); |
| 3105 | let opened = SnapshotRepo::open_or_init(&workspace); |
| 3106 | let _ = done_tx.send(()); |
| 3107 | opened |
| 3108 | }); |
| 3109 | assert!( |
| 3110 | done_rx.recv_timeout(Duration::from_millis(400)).is_err(), |
| 3111 | "init must not run while another writer holds the lock" |
| 3112 | ); |
| 3113 | drop(held); |
| 3114 | done_rx |
| 3115 | .recv_timeout(Duration::from_secs(30)) |
| 3116 | .expect("init proceeds once the lock is released"); |
| 3117 | let repo = opener.join().expect("opener thread").expect("open_or_init"); |
| 3118 | assert!( |
| 3119 | repo.git_dir().join("HEAD").exists(), |
| 3120 | "the repo was initialized" |
| 3121 | ); |
| 3122 | std::fs::write(repo.work_tree().join("f.txt"), b"v").unwrap(); |
| 3123 | repo.snapshot("pre-turn:1").expect("the new repo snapshots"); |
| 3124 | } |
| 3125 | |
| 3126 | /// Opening runs on every snapshot without the write lock while a peer may |
| 3127 | /// be staging. It used to truncate and rewrite `info/exclude` each time, |
| 3128 | /// so the peer's `git add -A` could read an empty exclude list. |
| 3129 | #[test] |
| 3130 | fn opening_leaves_a_current_exclude_file_untouched() { |
| 3131 | let tmp = tempdir().unwrap(); |
| 3132 | let (repo, _home) = make_repo(tmp.path()); |
| 3133 | let exclude = repo.git_dir().join("info").join("exclude"); |
| 3134 | let old = SystemTime::UNIX_EPOCH + Duration::from_secs(1_000_000); |
| 3135 | File::options() |
| 3136 | .write(true) |
| 3137 | .open(&exclude) |
| 3138 | .unwrap() |
| 3139 | .set_times(FileTimes::new().set_modified(old)) |
| 3140 | .unwrap(); |
| 3141 | |
| 3142 | SnapshotRepo::open_or_init(repo.work_tree()).expect("reopen"); |
| 3143 | assert_eq!( |
| 3144 | std::fs::metadata(&exclude).unwrap().modified().unwrap(), |
| 3145 | old, |
| 3146 | "a current exclude file is not rewritten" |
| 3147 | ); |
| 3148 | |
| 3149 | std::fs::write(&exclude, b"stale\n").unwrap(); |
| 3150 | SnapshotRepo::open_or_init(repo.work_tree()).expect("reopen"); |
| 3151 | assert_eq!( |
| 3152 | std::fs::read_to_string(&exclude).unwrap(), |
| 3153 | BUILTIN_EXCLUDES, |
| 3154 | "a stale exclude file is replaced" |
| 3155 | ); |
| 3156 | } |
| 3157 | |
| 3158 | /// HEAD moves only from the value the writer read. A writer that does not |
| 3159 | /// take the lock (an older build) must not be orphaned by a later move. |
| 3160 | #[test] |
| 3161 | fn head_moves_only_from_the_expected_commit() { |
| 3162 | let tmp = tempdir().unwrap(); |
| 3163 | let (repo, _home) = make_repo(tmp.path()); |
| 3164 | std::fs::write(repo.work_tree().join("f.txt"), b"v1").unwrap(); |
| 3165 | let first = repo.snapshot("pre-turn:1").expect("snapshot 1"); |
| 3166 | std::fs::write(repo.work_tree().join("f.txt"), b"v2").unwrap(); |
| 3167 | let second = repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 3168 | |
| 3169 | repo.move_head(Some(first.as_str()), Some(first.as_str())) |
| 3170 | .expect_err("HEAD no longer names the expected commit"); |
| 3171 | repo.move_head(Some(first.as_str()), None) |
| 3172 | .expect_err("HEAD exists, so an unborn expectation fails"); |
| 3173 | repo.move_head(None, Some(first.as_str())) |
| 3174 | .expect_err("deleting HEAD also checks the expected commit"); |
| 3175 | assert_eq!(repo.list(1).unwrap()[0].id, second, "HEAD is unchanged"); |
| 3176 | |
| 3177 | repo.move_head(Some(first.as_str()), Some(second.as_str())) |
| 3178 | .expect("the expected commit moves"); |
| 3179 | assert_eq!(repo.list(1).unwrap()[0].id, first); |
| 3180 | } |
| 3181 | |
| 3182 | /// HEAD repair looks up a missing commit, then moves HEAD off it. A peer |
| 3183 | /// that published a snapshot in between used to be overwritten by the |
| 3184 | /// reflog reset or deleted by the fresh-history fallback; both moves now |
| 3185 | /// require HEAD to still name the missing commit. |
| 3186 | #[test] |
| 3187 | fn head_repair_leaves_a_peers_new_snapshot_alone() { |
| 3188 | let tmp = tempdir().unwrap(); |
| 3189 | let (repo, _home) = make_repo(tmp.path()); |
| 3190 | std::fs::write(repo.work_tree().join("f.txt"), b"v1").unwrap(); |
| 3191 | repo.snapshot("pre-turn:1").expect("snapshot 1"); |
| 3192 | std::fs::write(repo.work_tree().join("f.txt"), b"v2").unwrap(); |
| 3193 | let peer = repo.snapshot("post-turn:peer").expect("peer snapshot"); |
| 3194 | let missing = "1111111111111111111111111111111111111111"; |
| 3195 | |
| 3196 | repo.reset_missing_head(missing) |
| 3197 | .expect_err("the reflog reset checks HEAD still names the missing commit"); |
| 3198 | assert_eq!(repo.list(1).unwrap()[0].id, peer, "HEAD is unchanged"); |
| 3199 | |
| 3200 | std::fs::remove_dir_all(repo.git_dir().join("logs")).unwrap(); |
| 3201 | repo.reset_missing_head(missing) |
| 3202 | .expect_err("the fresh-history delete checks it too"); |
| 3203 | assert_eq!( |
| 3204 | repo.list(1).unwrap()[0].id, |
| 3205 | peer, |
| 3206 | "the peer's snapshot stays" |
| 3207 | ); |
| 3208 | } |
| 3209 | |
| 3210 | /// A prune rebuilds the survivors it listed. If another writer committed |
| 3211 | /// after that listing, the rebuild used to overwrite HEAD and orphan the |
| 3212 | /// new snapshot for gc; now it fails and the snapshot stays. |
| 3213 | #[test] |
| 3214 | fn survivor_rebuild_refuses_when_another_writer_moved_head() { |
| 3215 | let tmp = tempdir().unwrap(); |
| 3216 | let (repo, _home) = make_repo(tmp.path()); |
| 3217 | for i in 0..3 { |
| 3218 | std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap(); |
| 3219 | repo.snapshot(&format!("tool:{i}")).expect("snapshot"); |
| 3220 | } |
| 3221 | let listed = repo.list(usize::MAX).unwrap(); |
| 3222 | std::fs::write(repo.work_tree().join("f.txt"), b"peer").unwrap(); |
| 3223 | let peer = repo.snapshot("post-turn:peer").expect("peer snapshot"); |
| 3224 | |
| 3225 | repo.rebuild_survivor_chain(&listed[..1], &listed[0].id) |
| 3226 | .expect_err("HEAD moved since the listing"); |
| 3227 | let history = repo.list(usize::MAX).unwrap(); |
| 3228 | assert_eq!(history.len(), 4, "nothing was dropped"); |
| 3229 | assert_eq!(history[0].id, peer, "the peer's snapshot is still HEAD"); |
| 3230 | } |
| 3231 | |
| 3232 | /// An index naming objects that no longer exist (an interrupted or racing |
| 3233 | /// gc) failed every later `write-tree`, because `add -A` does not rehash |
| 3234 | /// files whose stat data is unchanged. The index is rebuilt once instead. |
| 3235 | #[test] |
| 3236 | fn snapshot_rebuilds_an_index_naming_missing_objects() { |
| 3237 | let tmp = tempdir().unwrap(); |
| 3238 | let (repo, _home) = make_repo(tmp.path()); |
| 3239 | let file = repo.work_tree().join("f.txt"); |
| 3240 | std::fs::write(&file, b"content").unwrap(); |
| 3241 | // An mtime well before the index keeps git from treating the entry as |
| 3242 | // racily clean and rehashing it, which would hide the broken index. |
| 3243 | File::options() |
| 3244 | .write(true) |
| 3245 | .open(&file) |
| 3246 | .unwrap() |
| 3247 | .set_times(FileTimes::new().set_modified(SystemTime::now() - Duration::from_secs(3600))) |
| 3248 | .unwrap(); |
| 3249 | let taken = repo |
| 3250 | .take_snapshot("pre-turn:1", None) |
| 3251 | .expect("first snapshot"); |
| 3252 | let blob = git_output(&repo, &["rev-parse", "HEAD:f.txt"]); |
| 3253 | for oid in [blob.as_str(), taken.tree.as_str()] { |
| 3254 | let object = repo |
| 3255 | .git_dir() |
| 3256 | .join("objects") |
| 3257 | .join(&oid[..2]) |
| 3258 | .join(&oid[2..]); |
| 3259 | std::fs::remove_file(&object).expect("loose object removed"); |
| 3260 | } |
| 3261 | |
| 3262 | let again = repo |
| 3263 | .take_snapshot("post-turn:1", None) |
| 3264 | .expect("the snapshot rebuilds the index and succeeds"); |
| 3265 | assert_eq!( |
| 3266 | git_output( |
| 3267 | &repo, |
| 3268 | &["cat-file", "-p", &format!("{}:f.txt", again.id.as_str())] |
| 3269 | ), |
| 3270 | "content" |
| 3271 | ); |
| 3272 | } |
| 3273 | |
| 3274 | fn git_output(repo: &SnapshotRepo, args: &[&str]) -> String { |
| 3275 | let out = run_git(repo.git_dir(), repo.work_tree(), args).unwrap(); |
| 3276 | assert!( |
| 3277 | out.status.success(), |
| 3278 | "git {args:?}: {}", |
| 3279 | String::from_utf8_lossy(&out.stderr) |
| 3280 | ); |
| 3281 | String::from_utf8_lossy(&out.stdout).trim().to_string() |
| 3282 | } |
| 3283 | |
| 3284 | #[test] |
| 3285 | fn restore_paths_leaves_unrelated_files_alone() { |
| 3286 | let tmp = tempdir().unwrap(); |
| 3287 | let (repo, _home) = make_repo(tmp.path()); |
| 3288 | let wanted = repo.work_tree().join("wanted.txt"); |
| 3289 | let unrelated = repo.work_tree().join("unrelated.txt"); |
| 3290 | |
| 3291 | std::fs::write(&wanted, b"original").unwrap(); |
| 3292 | std::fs::write(&unrelated, b"original").unwrap(); |
| 3293 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3294 | |
| 3295 | std::fs::write(&wanted, b"clobbered").unwrap(); |
| 3296 | std::fs::write(&unrelated, b"also clobbered").unwrap(); |
| 3297 | repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 3298 | |
| 3299 | let outcomes = repo |
| 3300 | .restore_paths(&id, &[PathBuf::from("wanted.txt")]) |
| 3301 | .expect("scoped restore"); |
| 3302 | |
| 3303 | assert_eq!(std::fs::read_to_string(&wanted).unwrap(), "original"); |
| 3304 | assert_eq!( |
| 3305 | std::fs::read_to_string(&unrelated).unwrap(), |
| 3306 | "also clobbered", |
| 3307 | "a file-scoped restore must not touch a path it was not given" |
| 3308 | ); |
| 3309 | assert_eq!(outcomes.len(), 1); |
| 3310 | assert_eq!(outcomes[0].path, PathBuf::from("wanted.txt")); |
| 3311 | assert_eq!(outcomes[0].action, PathRestoreAction::Modified); |
| 3312 | } |
| 3313 | |
| 3314 | #[test] |
| 3315 | fn only_restore_paths_is_safe_for_a_single_file_action() { |
| 3316 | // Characterizes the difference the per-file Revert control depends on. |
| 3317 | // `restore()` is what the TUI's `patch_undo()` and the runtime's |
| 3318 | // `patch-undo` endpoint both call. It is scoped in *snapshot selection* |
| 3319 | // (it picks a recent `tool:` snapshot) but not in *effect*: it checks |
| 3320 | // out the whole tree, so it also rolls back a working-tree path that no |
| 3321 | // tool touched. That is the data loss #2 removed the control over, and |
| 3322 | // it is why a per-file action cannot be built on top of it. |
| 3323 | let tmp = tempdir().unwrap(); |
| 3324 | let (repo, _home) = make_repo(tmp.path()); |
| 3325 | let touched = repo.work_tree().join("touched.txt"); |
| 3326 | let unrelated = repo.work_tree().join("unrelated.txt"); |
| 3327 | |
| 3328 | std::fs::write(&touched, b"v1").unwrap(); |
| 3329 | std::fs::write(&unrelated, b"snapshot-time").unwrap(); |
| 3330 | let id = repo.snapshot("tool:call-1").expect("snapshot"); |
| 3331 | |
| 3332 | // The tool edits one file; something else — the user, another editor — |
| 3333 | // changes the other one after the snapshot was taken. |
| 3334 | std::fs::write(&touched, b"v2").unwrap(); |
| 3335 | std::fs::write(&unrelated, b"user-work-in-progress").unwrap(); |
| 3336 | |
| 3337 | repo.restore(&id).expect("whole-tree restore"); |
| 3338 | assert_eq!(std::fs::read_to_string(&touched).unwrap(), "v1"); |
| 3339 | assert_eq!( |
| 3340 | std::fs::read_to_string(&unrelated).unwrap(), |
| 3341 | "snapshot-time", |
| 3342 | "whole-tree restore rolls back a file no tool touched" |
| 3343 | ); |
| 3344 | |
| 3345 | // Same situation again, but through the file-scoped path the |
| 3346 | // `file-revert` endpoint uses. |
| 3347 | std::fs::write(&touched, b"v1").unwrap(); |
| 3348 | std::fs::write(&unrelated, b"snapshot-time").unwrap(); |
| 3349 | let id2 = repo.snapshot("tool:call-2").expect("snapshot 2"); |
| 3350 | std::fs::write(&touched, b"v2").unwrap(); |
| 3351 | std::fs::write(&unrelated, b"user-work-in-progress").unwrap(); |
| 3352 | |
| 3353 | repo.restore_paths(&id2, &[PathBuf::from("touched.txt")]) |
| 3354 | .expect("scoped restore"); |
| 3355 | assert_eq!(std::fs::read_to_string(&touched).unwrap(), "v1"); |
| 3356 | assert_eq!( |
| 3357 | std::fs::read_to_string(&unrelated).unwrap(), |
| 3358 | "user-work-in-progress", |
| 3359 | "the scoped restore must leave the unrelated edit alone" |
| 3360 | ); |
| 3361 | } |
| 3362 | |
| 3363 | #[test] |
| 3364 | fn restore_paths_removes_a_file_created_after_the_snapshot() { |
| 3365 | let tmp = tempdir().unwrap(); |
| 3366 | let (repo, _home) = make_repo(tmp.path()); |
| 3367 | let kept = repo.work_tree().join("kept.txt"); |
| 3368 | let created = repo.work_tree().join("created.txt"); |
| 3369 | |
| 3370 | std::fs::write(&kept, b"kept").unwrap(); |
| 3371 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3372 | |
| 3373 | std::fs::write(&created, b"new file").unwrap(); |
| 3374 | repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 3375 | |
| 3376 | let outcomes = repo |
| 3377 | .restore_paths(&id, &[PathBuf::from("created.txt")]) |
| 3378 | .expect("scoped restore"); |
| 3379 | |
| 3380 | assert!( |
| 3381 | !created.exists(), |
| 3382 | "a created file must be removed by revert" |
| 3383 | ); |
| 3384 | assert!(kept.exists(), "the untouched file must survive"); |
| 3385 | assert_eq!(outcomes[0].action, PathRestoreAction::Removed); |
| 3386 | } |
| 3387 | |
| 3388 | #[test] |
| 3389 | fn restore_paths_recreates_a_file_deleted_after_the_snapshot() { |
| 3390 | let tmp = tempdir().unwrap(); |
| 3391 | let (repo, _home) = make_repo(tmp.path()); |
| 3392 | let deleted = repo.work_tree().join("deleted.txt"); |
| 3393 | |
| 3394 | std::fs::write(&deleted, b"content").unwrap(); |
| 3395 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3396 | |
| 3397 | std::fs::remove_file(&deleted).unwrap(); |
| 3398 | repo.snapshot("post-turn:1").expect("snapshot 2"); |
| 3399 | |
| 3400 | let outcomes = repo |
| 3401 | .restore_paths(&id, &[PathBuf::from("deleted.txt")]) |
| 3402 | .expect("scoped restore"); |
| 3403 | |
| 3404 | assert_eq!(std::fs::read_to_string(&deleted).unwrap(), "content"); |
| 3405 | assert_eq!(outcomes[0].action, PathRestoreAction::Recreated); |
| 3406 | } |
| 3407 | |
| 3408 | #[test] |
| 3409 | fn restore_paths_rejects_parent_traversal() { |
| 3410 | let tmp = tempdir().unwrap(); |
| 3411 | let (repo, _home) = make_repo(tmp.path()); |
| 3412 | std::fs::write(repo.work_tree().join("a.txt"), b"a").unwrap(); |
| 3413 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3414 | |
| 3415 | let err = repo |
| 3416 | .restore_paths(&id, &[PathBuf::from("../escape.txt")]) |
| 3417 | .expect_err("traversal must be refused"); |
| 3418 | assert!(err.to_string().contains("unsafe path"), "got: {err}"); |
| 3419 | } |
| 3420 | |
| 3421 | #[test] |
| 3422 | fn path_differs_from_snapshot_is_scoped_to_the_named_path() { |
| 3423 | let tmp = tempdir().unwrap(); |
| 3424 | let (repo, _home) = make_repo(tmp.path()); |
| 3425 | let touched = repo.work_tree().join("touched.txt"); |
| 3426 | let untouched = repo.work_tree().join("untouched.txt"); |
| 3427 | |
| 3428 | std::fs::write(&touched, b"v1").unwrap(); |
| 3429 | std::fs::write(&untouched, b"v1").unwrap(); |
| 3430 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3431 | |
| 3432 | std::fs::write(&touched, b"v2").unwrap(); |
| 3433 | |
| 3434 | assert!( |
| 3435 | repo.path_differs_from_snapshot(&id, Path::new("touched.txt")) |
| 3436 | .expect("differs") |
| 3437 | ); |
| 3438 | assert!( |
| 3439 | !repo |
| 3440 | .path_differs_from_snapshot(&id, Path::new("untouched.txt")) |
| 3441 | .expect("differs") |
| 3442 | ); |
| 3443 | } |
| 3444 | |
| 3445 | #[test] |
| 3446 | fn workspace_relative_path_accepts_inside_paths_and_refuses_outside_ones() { |
| 3447 | // A real absolute temp path so the fixture is absolute on Windows too |
| 3448 | // (`/tmp/ws` is a relative path with a root-dir component there). |
| 3449 | let temp = std::env::temp_dir(); |
| 3450 | let workspace = temp.join("ws"); |
| 3451 | let other = temp.join("other"); |
| 3452 | |
| 3453 | assert_eq!( |
| 3454 | workspace_relative_path(&workspace, "src/lib.rs"), |
| 3455 | Some(PathBuf::from("src/lib.rs")) |
| 3456 | ); |
| 3457 | let inside = workspace.join("src").join("lib.rs"); |
| 3458 | assert_eq!( |
| 3459 | workspace_relative_path(&workspace, &inside.to_string_lossy()), |
| 3460 | Some(PathBuf::from("src").join("lib.rs")) |
| 3461 | ); |
| 3462 | let outside = other.join("lib.rs"); |
| 3463 | assert_eq!( |
| 3464 | workspace_relative_path(&workspace, &outside.to_string_lossy()), |
| 3465 | None |
| 3466 | ); |
| 3467 | assert_eq!(workspace_relative_path(&workspace, "../escape"), None); |
| 3468 | assert_eq!( |
| 3469 | workspace_relative_path(&workspace, "src/../../escape"), |
| 3470 | None |
| 3471 | ); |
| 3472 | assert_eq!(workspace_relative_path(&workspace, ""), None); |
| 3473 | // Whitespace and glob characters are literal filename bytes. |
| 3474 | assert_eq!( |
| 3475 | workspace_relative_path(&workspace, " padded.txt "), |
| 3476 | Some(PathBuf::from(" padded.txt ")) |
| 3477 | ); |
| 3478 | assert_eq!( |
| 3479 | workspace_relative_path(&workspace, "file[12].txt"), |
| 3480 | Some(PathBuf::from("file[12].txt")) |
| 3481 | ); |
| 3482 | } |
| 3483 | |
| 3484 | fn sha256_hash(path: &Path) -> String { |
| 3485 | format!( |
| 3486 | "sha256:{}", |
| 3487 | crate::hashing::sha256_hex(std::fs::read(path).unwrap()) |
| 3488 | ) |
| 3489 | } |
| 3490 | |
| 3491 | /// The Git primitive treats `[12]` as a pattern even after `--`; the |
| 3492 | /// file-scoped restore must not. `file[12].txt` and `file1.txt` both exist |
| 3493 | /// in the snapshot, so only literal pathspecs keep the second one intact. |
| 3494 | #[test] |
| 3495 | fn restore_file_if_unchanged_treats_glob_characters_literally() { |
| 3496 | let tmp = tempdir().unwrap(); |
| 3497 | let (repo, _home) = make_repo(tmp.path()); |
| 3498 | let literal = repo.work_tree().join("file[12].txt"); |
| 3499 | let sibling = repo.work_tree().join("file1.txt"); |
| 3500 | std::fs::write(&literal, b"literal-before").unwrap(); |
| 3501 | std::fs::write(&sibling, b"sibling-before").unwrap(); |
| 3502 | let id = repo.snapshot("tool:call-1").expect("snapshot"); |
| 3503 | std::fs::write(&literal, b"literal-after").unwrap(); |
| 3504 | std::fs::write(&sibling, b"sibling-after").unwrap(); |
| 3505 | |
| 3506 | assert!( |
| 3507 | repo.path_differs_from_snapshot(&id, Path::new("file[12].txt")) |
| 3508 | .unwrap() |
| 3509 | ); |
| 3510 | let outcomes = repo |
| 3511 | .restore_file_if_unchanged(&id, Path::new("file[12].txt"), &sha256_hash(&literal)) |
| 3512 | .expect("literal restore"); |
| 3513 | assert_eq!(outcomes.len(), 1); |
| 3514 | assert_eq!(outcomes[0].action, PathRestoreAction::Modified); |
| 3515 | assert_eq!(std::fs::read_to_string(&literal).unwrap(), "literal-before"); |
| 3516 | assert_eq!( |
| 3517 | std::fs::read_to_string(&sibling).unwrap(), |
| 3518 | "sibling-after", |
| 3519 | "a bracketed filename must never restore its glob siblings" |
| 3520 | ); |
| 3521 | } |
| 3522 | |
| 3523 | #[test] |
| 3524 | fn restore_file_if_unchanged_refuses_when_the_reviewed_bytes_changed() { |
| 3525 | let tmp = tempdir().unwrap(); |
| 3526 | let (repo, _home) = make_repo(tmp.path()); |
| 3527 | let file = repo.work_tree().join("a.txt"); |
| 3528 | std::fs::write(&file, b"v1").unwrap(); |
| 3529 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3530 | std::fs::write(&file, b"v2").unwrap(); |
| 3531 | let reviewed = sha256_hash(&file); |
| 3532 | // The user edits again after the client captured its change record. |
| 3533 | std::fs::write(&file, b"v3-user-edit").unwrap(); |
| 3534 | |
| 3535 | let err = repo |
| 3536 | .restore_file_if_unchanged(&id, Path::new("a.txt"), &reviewed) |
| 3537 | .expect_err("stale hash must refuse"); |
| 3538 | assert_eq!(err.kind(), io::ErrorKind::WouldBlock); |
| 3539 | assert_eq!(std::fs::read_to_string(&file).unwrap(), "v3-user-edit"); |
| 3540 | // `absent` is only valid for a file the client saw as deleted. |
| 3541 | let err = repo |
| 3542 | .restore_file_if_unchanged(&id, Path::new("a.txt"), "absent") |
| 3543 | .expect_err("absent must not match an existing file"); |
| 3544 | assert_eq!(err.kind(), io::ErrorKind::WouldBlock); |
| 3545 | // The exact current bytes restore. |
| 3546 | let outcomes = repo |
| 3547 | .restore_file_if_unchanged(&id, Path::new("a.txt"), &sha256_hash(&file)) |
| 3548 | .expect("current hash restores"); |
| 3549 | assert_eq!(outcomes[0].action, PathRestoreAction::Modified); |
| 3550 | assert_eq!(std::fs::read_to_string(&file).unwrap(), "v1"); |
| 3551 | } |
| 3552 | |
| 3553 | #[test] |
| 3554 | fn restore_file_if_unchanged_handles_deleted_and_created_files() { |
| 3555 | let tmp = tempdir().unwrap(); |
| 3556 | let (repo, _home) = make_repo(tmp.path()); |
| 3557 | let deleted = repo.work_tree().join("deleted.txt"); |
| 3558 | std::fs::write(&deleted, b"content").unwrap(); |
| 3559 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3560 | std::fs::remove_file(&deleted).unwrap(); |
| 3561 | let created = repo.work_tree().join("created.txt"); |
| 3562 | std::fs::write(&created, b"new").unwrap(); |
| 3563 | |
| 3564 | let outcomes = repo |
| 3565 | .restore_file_if_unchanged(&id, Path::new("deleted.txt"), "absent") |
| 3566 | .expect("recreate"); |
| 3567 | assert_eq!(outcomes[0].action, PathRestoreAction::Recreated); |
| 3568 | assert_eq!(std::fs::read_to_string(&deleted).unwrap(), "content"); |
| 3569 | |
| 3570 | let outcomes = repo |
| 3571 | .restore_file_if_unchanged(&id, Path::new("created.txt"), &sha256_hash(&created)) |
| 3572 | .expect("remove"); |
| 3573 | assert_eq!(outcomes[0].action, PathRestoreAction::Removed); |
| 3574 | assert!(!created.exists()); |
| 3575 | // A created file inside a new directory is removed alone; the |
| 3576 | // directory the user made stays. |
| 3577 | let nested_dir = repo.work_tree().join("newdir"); |
| 3578 | std::fs::create_dir_all(&nested_dir).unwrap(); |
| 3579 | let nested = nested_dir.join("only.txt"); |
| 3580 | std::fs::write(&nested, b"n").unwrap(); |
| 3581 | let outcomes = repo |
| 3582 | .restore_file_if_unchanged(&id, Path::new("newdir/only.txt"), &sha256_hash(&nested)) |
| 3583 | .expect("remove nested"); |
| 3584 | assert_eq!(outcomes[0].action, PathRestoreAction::Removed); |
| 3585 | assert!(!nested.exists()); |
| 3586 | assert!(nested_dir.is_dir(), "the parent directory is not pruned"); |
| 3587 | // A path missing on both sides is not a change and reports nothing. |
| 3588 | assert!( |
| 3589 | !repo |
| 3590 | .path_differs_from_snapshot(&id, Path::new("never.txt")) |
| 3591 | .unwrap() |
| 3592 | ); |
| 3593 | } |
| 3594 | |
| 3595 | #[test] |
| 3596 | fn restore_file_if_unchanged_refuses_directories_git_metadata_and_ignored_files() { |
| 3597 | let tmp = tempdir().unwrap(); |
| 3598 | let (repo, _home) = make_repo(tmp.path()); |
| 3599 | let dir = repo.work_tree().join("src"); |
| 3600 | std::fs::create_dir_all(&dir).unwrap(); |
| 3601 | std::fs::write(dir.join("lib.rs"), b"fn a() {}").unwrap(); |
| 3602 | std::fs::write( |
| 3603 | repo.work_tree().join(".gitignore"), |
| 3604 | "ignored.txt |
| 3605 | ", |
| 3606 | ) |
| 3607 | .unwrap(); |
| 3608 | std::fs::write(repo.work_tree().join("ignored.txt"), b"secret").unwrap(); |
| 3609 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3610 | std::fs::write(dir.join("lib.rs"), b"fn b() {}").unwrap(); |
| 3611 | |
| 3612 | let mut refused = vec!["src", ".git/config", "src/.GIT/x", ".git"]; |
| 3613 | if cfg!(windows) { |
| 3614 | refused.extend([".git./config", ".git /config", "GIT~1/config"]); |
| 3615 | } |
| 3616 | for rel in refused { |
| 3617 | let err = repo.validate_restore_file(Path::new(rel)).expect_err(rel); |
| 3618 | assert_eq!(err.kind(), io::ErrorKind::InvalidInput, "{rel}"); |
| 3619 | } |
| 3620 | let err = repo |
| 3621 | .restore_file_if_unchanged(&id, Path::new("src"), "absent") |
| 3622 | .expect_err("directories are refused"); |
| 3623 | assert_eq!(err.kind(), io::ErrorKind::InvalidInput); |
| 3624 | assert_eq!( |
| 3625 | std::fs::read_to_string(dir.join("lib.rs")).unwrap(), |
| 3626 | "fn b() {}" |
| 3627 | ); |
| 3628 | |
| 3629 | // A gitignored file is excluded from the safety backup, so removing |
| 3630 | // it would be unrecoverable: refuse and leave it in place. |
| 3631 | let ignored = repo.work_tree().join("ignored.txt"); |
| 3632 | let err = repo |
| 3633 | .restore_file_if_unchanged(&id, Path::new("ignored.txt"), &sha256_hash(&ignored)) |
| 3634 | .expect_err("ignored files are refused"); |
| 3635 | assert!(err.to_string().contains("safety snapshot"), "got: {err}"); |
| 3636 | assert_eq!(std::fs::read_to_string(&ignored).unwrap(), "secret"); |
| 3637 | } |
| 3638 | |
| 3639 | #[cfg(unix)] |
| 3640 | #[test] |
| 3641 | fn restore_file_if_unchanged_refuses_symlinks_anywhere_in_the_path() { |
| 3642 | let tmp = tempdir().unwrap(); |
| 3643 | let (repo, _home) = make_repo(tmp.path()); |
| 3644 | let outside = tmp.path().join("outside"); |
| 3645 | std::fs::create_dir_all(&outside).unwrap(); |
| 3646 | std::fs::write(outside.join("target.txt"), b"outside").unwrap(); |
| 3647 | std::fs::write(repo.work_tree().join("real.txt"), b"real").unwrap(); |
| 3648 | std::os::unix::fs::symlink(&outside, repo.work_tree().join("linkdir")).unwrap(); |
| 3649 | std::os::unix::fs::symlink( |
| 3650 | outside.join("target.txt"), |
| 3651 | repo.work_tree().join("link.txt"), |
| 3652 | ) |
| 3653 | .unwrap(); |
| 3654 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3655 | |
| 3656 | for rel in ["link.txt", "linkdir/target.txt"] { |
| 3657 | let err = repo |
| 3658 | .restore_file_if_unchanged(&id, Path::new(rel), "absent") |
| 3659 | .expect_err(rel); |
| 3660 | assert_eq!(err.kind(), io::ErrorKind::InvalidInput, "{rel}"); |
| 3661 | } |
| 3662 | assert_eq!( |
| 3663 | std::fs::read_to_string(outside.join("target.txt")).unwrap(), |
| 3664 | "outside" |
| 3665 | ); |
| 3666 | // The snapshot side is checked too: a symlink entry in the tree is |
| 3667 | // not a regular file even when the work tree copy is gone. |
| 3668 | std::fs::remove_file(repo.work_tree().join("link.txt")).unwrap(); |
| 3669 | let err = repo |
| 3670 | .restore_file_if_unchanged(&id, Path::new("link.txt"), "absent") |
| 3671 | .expect_err("snapshot symlink entry"); |
| 3672 | assert_eq!(err.kind(), io::ErrorKind::InvalidInput); |
| 3673 | assert!(!repo.work_tree().join("link.txt").exists()); |
| 3674 | } |
| 3675 | |
| 3676 | #[test] |
| 3677 | fn list_distinguishes_an_unborn_head_from_broken_history() { |
| 3678 | let tmp = tempdir().unwrap(); |
| 3679 | let (repo, _home) = make_repo(tmp.path()); |
| 3680 | assert!(repo.list(10).expect("unborn HEAD lists nothing").is_empty()); |
| 3681 | |
| 3682 | std::fs::write(repo.work_tree().join("a.txt"), b"a").unwrap(); |
| 3683 | repo.snapshot("pre-turn:1").expect("snapshot"); |
| 3684 | assert_eq!(repo.list(10).unwrap().len(), 1); |
| 3685 | |
| 3686 | // Point the branch at an object that does not exist: the history is |
| 3687 | // now broken, which must surface as an error rather than "no |
| 3688 | // snapshots" (an empty list would let patch-undo drop a turn). |
| 3689 | let head = String::from_utf8( |
| 3690 | run_git(repo.git_dir(), repo.work_tree(), &["symbolic-ref", "HEAD"]) |
| 3691 | .unwrap() |
| 3692 | .stdout, |
| 3693 | ) |
| 3694 | .unwrap(); |
| 3695 | std::fs::write( |
| 3696 | repo.git_dir().join(head.trim()), |
| 3697 | "0123456789abcdef0123456789abcdef01234567\n", |
| 3698 | ) |
| 3699 | .unwrap(); |
| 3700 | let err = repo.list(10).expect_err("broken history must error"); |
| 3701 | assert!(err.to_string().contains("git log failed"), "got: {err}"); |
| 3702 | } |
| 3703 | |
| 3704 | #[test] |
| 3705 | fn restore_takes_a_pre_restore_safety_snapshot_that_round_trips() { |
| 3706 | let tmp = tempdir().unwrap(); |
| 3707 | let (repo, _home) = make_repo(tmp.path()); |
| 3708 | let f = repo.work_tree().join("file.txt"); |
| 3709 | |
| 3710 | std::fs::write(&f, b"v1").unwrap(); |
| 3711 | let id1 = repo.snapshot("pre-turn:1").expect("snapshot v1"); |
| 3712 | |
| 3713 | std::fs::write(&f, b"v2").unwrap(); |
| 3714 | repo.snapshot("post-turn:1").expect("snapshot v2"); |
| 3715 | |
| 3716 | repo.restore(&id1).expect("restore to v1"); |
| 3717 | assert_eq!(std::fs::read_to_string(&f).unwrap(), "v1"); |
| 3718 | |
| 3719 | // The restore must have captured the pre-restore state (v2) under a |
| 3720 | // `pre-restore:` label naming its target, so the destructive op is |
| 3721 | // itself reversible (2026-08-04 snapshot hunt). |
| 3722 | let snapshots = repo.list(usize::MAX).expect("list"); |
| 3723 | let safety = snapshots |
| 3724 | .iter() |
| 3725 | .find(|s| s.label.starts_with("pre-restore:")) |
| 3726 | .expect("a pre-restore safety snapshot must exist"); |
| 3727 | assert!( |
| 3728 | safety.label.ends_with(&id1.as_str()[..12]), |
| 3729 | "safety label should name the restore target: {}", |
| 3730 | safety.label |
| 3731 | ); |
| 3732 | |
| 3733 | repo.restore(&safety.id) |
| 3734 | .expect("restore the safety snapshot"); |
| 3735 | assert_eq!( |
| 3736 | std::fs::read_to_string(&f).unwrap(), |
| 3737 | "v2", |
| 3738 | "the safety snapshot must bring back the pre-restore state" |
| 3739 | ); |
| 3740 | } |
| 3741 | |
| 3742 | #[test] |
| 3743 | fn snapshot_and_restore_do_not_move_user_git_head() { |
| 3744 | let tmp = tempdir().unwrap(); |
| 3745 | let workspace = tmp.path().join("workspace"); |
| 3746 | std::fs::create_dir_all(&workspace).unwrap(); |
| 3747 | crate::dependencies::Git::command() |
| 3748 | .expect("git not found") |
| 3749 | .arg("-C") |
| 3750 | .arg(&workspace) |
| 3751 | .arg("init") |
| 3752 | .arg("--quiet") |
| 3753 | .status() |
| 3754 | .unwrap(); |
| 3755 | std::fs::write(workspace.join("tracked.txt"), b"committed").unwrap(); |
| 3756 | crate::dependencies::Git::command() |
| 3757 | .expect("git not found") |
| 3758 | .arg("-C") |
| 3759 | .arg(&workspace) |
| 3760 | .arg("add") |
| 3761 | .arg("tracked.txt") |
| 3762 | .status() |
| 3763 | .unwrap(); |
| 3764 | crate::dependencies::Git::command() |
| 3765 | .expect("git not found") |
| 3766 | .arg("-C") |
| 3767 | .arg(&workspace) |
| 3768 | .arg("-c") |
| 3769 | .arg("user.name=user") |
| 3770 | .arg("-c") |
| 3771 | .arg("user.email=user@example.test") |
| 3772 | .arg("commit") |
| 3773 | .arg("--quiet") |
| 3774 | .arg("-m") |
| 3775 | .arg("init") |
| 3776 | .status() |
| 3777 | .unwrap(); |
| 3778 | let user_head_before = crate::dependencies::Git::command() |
| 3779 | .expect("git not found") |
| 3780 | .arg("-C") |
| 3781 | .arg(&workspace) |
| 3782 | .args(["rev-parse", "HEAD"]) |
| 3783 | .output() |
| 3784 | .unwrap() |
| 3785 | .stdout; |
| 3786 | |
| 3787 | let _home = scoped_home(tmp.path()); |
| 3788 | let repo = SnapshotRepo::open_or_init(&workspace).unwrap(); |
| 3789 | std::fs::write(workspace.join("tracked.txt"), b"dirty-before").unwrap(); |
| 3790 | let id = repo.snapshot("pre-turn:1").unwrap(); |
| 3791 | std::fs::write(workspace.join("tracked.txt"), b"dirty-after").unwrap(); |
| 3792 | repo.snapshot("post-turn:1").unwrap(); |
| 3793 | repo.restore(&id).unwrap(); |
| 3794 | |
| 3795 | let user_head_after = crate::dependencies::Git::command() |
| 3796 | .expect("git not found") |
| 3797 | .arg("-C") |
| 3798 | .arg(&workspace) |
| 3799 | .args(["rev-parse", "HEAD"]) |
| 3800 | .output() |
| 3801 | .unwrap() |
| 3802 | .stdout; |
| 3803 | assert_eq!(user_head_after, user_head_before); |
| 3804 | assert_eq!( |
| 3805 | std::fs::read_to_string(workspace.join("tracked.txt")).unwrap(), |
| 3806 | "dirty-before" |
| 3807 | ); |
| 3808 | } |
| 3809 | |
| 3810 | #[test] |
| 3811 | fn list_respects_limit() { |
| 3812 | let tmp = tempdir().unwrap(); |
| 3813 | let (repo, _home) = make_repo(tmp.path()); |
| 3814 | for i in 0..5 { |
| 3815 | std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap(); |
| 3816 | repo.snapshot(&format!("turn:{i}")).unwrap(); |
| 3817 | } |
| 3818 | let three = repo.list(3).unwrap(); |
| 3819 | assert_eq!(three.len(), 3); |
| 3820 | // Newest first. |
| 3821 | assert_eq!(three[0].label, "turn:4"); |
| 3822 | } |
| 3823 | |
| 3824 | #[test] |
| 3825 | fn prune_drops_snapshots_older_than_threshold() { |
| 3826 | let tmp = tempdir().unwrap(); |
| 3827 | let (repo, _home) = make_repo(tmp.path()); |
| 3828 | std::fs::write(repo.work_tree().join("f.txt"), "v0").unwrap(); |
| 3829 | repo.snapshot("turn:0").unwrap(); |
| 3830 | |
| 3831 | // Wait one second so the snapshot's commit timestamp is strictly |
| 3832 | // in the past relative to the prune call's "now" — otherwise |
| 3833 | // same-second comparisons make the assertion flaky. |
| 3834 | std::thread::sleep(Duration::from_millis(1100)); |
| 3835 | |
| 3836 | let removed = repo.prune_older_than(Duration::from_secs(0)).unwrap(); |
| 3837 | assert!(removed >= 1, "expected at least 1 pruned, got {removed}"); |
| 3838 | |
| 3839 | // After pruning everything, the next snapshot should start a |
| 3840 | // fresh history. |
| 3841 | std::fs::write(repo.work_tree().join("f.txt"), "v1").unwrap(); |
| 3842 | repo.snapshot("turn:1").unwrap(); |
| 3843 | let list = repo.list(10).unwrap(); |
| 3844 | assert_eq!(list.len(), 1); |
| 3845 | assert_eq!(list[0].label, "turn:1"); |
| 3846 | } |
| 3847 | |
| 3848 | /// The 2026-08-04 regression: with a cut in the MIDDLE of history, |
| 3849 | /// `prune_older_than` used to `update-ref HEAD <oldest survivor>`, which |
| 3850 | /// orphaned (and gc destroyed) the NEWEST snapshots while keeping the |
| 3851 | /// old ones as ancestors — the inverse of the intent, firing on every |
| 3852 | /// boot. This pins the correct partial-cut behavior. |
| 3853 | #[test] |
| 3854 | fn prune_older_than_keeps_the_newest_and_drops_only_the_old_tail() { |
| 3855 | let tmp = tempdir().unwrap(); |
| 3856 | let (repo, _home) = make_repo(tmp.path()); |
| 3857 | |
| 3858 | // Two "old" snapshots, then a pause, then two "new" ones. |
| 3859 | for i in 0..2 { |
| 3860 | std::fs::write(repo.work_tree().join("f.txt"), format!("old{i}")).unwrap(); |
| 3861 | repo.snapshot(&format!("old:{i}")).unwrap(); |
| 3862 | std::thread::sleep(Duration::from_millis(1100)); |
| 3863 | } |
| 3864 | // A wide gap so git's whole-second commit timestamps land the cut |
| 3865 | // unambiguously between the old and new pairs. The margins are |
| 3866 | // deliberately generous: this test runs under full-suite parallelism |
| 3867 | // where a sleep can overrun, and the cut is wall-clock. At prune time |
| 3868 | // the newest pair is ~0-1.2s old against a 6s cutoff, and the old |
| 3869 | // pair is ~9s old — ~5s of slack in both directions. |
| 3870 | std::thread::sleep(Duration::from_secs(8)); |
| 3871 | for i in 0..2 { |
| 3872 | std::fs::write(repo.work_tree().join("f.txt"), format!("new{i}")).unwrap(); |
| 3873 | repo.snapshot(&format!("new:{i}")).unwrap(); |
| 3874 | if i == 0 { |
| 3875 | std::thread::sleep(Duration::from_millis(1100)); |
| 3876 | } |
| 3877 | } |
| 3878 | let before = repo.list(usize::MAX).unwrap(); |
| 3879 | assert_eq!(before.len(), 4); |
| 3880 | // Derive the cut from the timestamps actually recorded rather than a |
| 3881 | // fixed 6s. A fixed cut assumes `repo.snapshot()` is fast: `new:0` is |
| 3882 | // only ~1.2s plus one git subprocess older than prune time, so on a |
| 3883 | // loaded Windows runner that subprocess alone pushed it past 6s and |
| 3884 | // three snapshots were pruned instead of two. (The old fixture guard |
| 3885 | // could not catch it either — it checked `before[0]` and `before[2]`, |
| 3886 | // and `before[1]` is the entry that drifts.) |
| 3887 | let now = std::time::SystemTime::now() |
| 3888 | .duration_since(std::time::UNIX_EPOCH) |
| 3889 | .unwrap() |
| 3890 | .as_secs() as i64; |
| 3891 | // Newest-first: [new:1, new:0, old:1, old:0]. The cut must land |
| 3892 | // strictly between the pairs, so aim at the midpoint of the 8s gap — |
| 3893 | // that leaves ~4s of slack against clock drift and a slow runner in |
| 3894 | // both directions. |
| 3895 | let survivor = before[1].timestamp; |
| 3896 | let victim = before[2].timestamp; |
| 3897 | assert!( |
| 3898 | survivor - victim >= 8, |
| 3899 | "fixture needs an 8s gap between the pairs (survivor {survivor}, victim {victim})" |
| 3900 | ); |
| 3901 | let midpoint = victim + (survivor - victim) / 2; |
| 3902 | assert!( |
| 3903 | now > midpoint, |
| 3904 | "fixture cutoff is not before the current time" |
| 3905 | ); |
| 3906 | let max_age = Duration::from_secs((now - midpoint) as u64); |
| 3907 | |
| 3908 | // The two old snapshots drop, the two new ones survive. |
| 3909 | let removed = repo.prune_older_than(max_age).unwrap(); |
| 3910 | assert_eq!(removed, 2, "only the old tail should be removed"); |
| 3911 | |
| 3912 | let remaining = repo.list(usize::MAX).unwrap(); |
| 3913 | assert_eq!(remaining.len(), 2, "the two newest must survive"); |
| 3914 | assert_eq!( |
| 3915 | remaining[0].label, "new:1", |
| 3916 | "newest survives (was destroyed before)" |
| 3917 | ); |
| 3918 | assert_eq!(remaining[1].label, "new:0"); |
| 3919 | assert!( |
| 3920 | !remaining.iter().any(|s| s.label.starts_with("old:")), |
| 3921 | "old snapshots must be gone, not kept as ancestors: {:?}", |
| 3922 | remaining.iter().map(|s| &s.label).collect::<Vec<_>>() |
| 3923 | ); |
| 3924 | |
| 3925 | // The survivors' contents are intact and restorable. |
| 3926 | repo.restore(&remaining[0].id).unwrap(); |
| 3927 | assert_eq!( |
| 3928 | std::fs::read_to_string(repo.work_tree().join("f.txt")).unwrap(), |
| 3929 | "new1" |
| 3930 | ); |
| 3931 | } |
| 3932 | |
| 3933 | #[test] |
| 3934 | fn prune_keep_last_n_keeps_latest_and_gc_reclaims_rest() { |
| 3935 | let tmp = tempdir().unwrap(); |
| 3936 | let (repo, _home) = make_repo(tmp.path()); |
| 3937 | |
| 3938 | for i in 0..3 { |
| 3939 | std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap(); |
| 3940 | repo.snapshot(&format!("turn:{i}")).unwrap(); |
| 3941 | std::thread::sleep(Duration::from_millis(1100)); |
| 3942 | } |
| 3943 | |
| 3944 | assert_eq!(repo.list(usize::MAX).unwrap().len(), 3); |
| 3945 | |
| 3946 | let removed = repo.prune_keep_last_n(1).unwrap(); |
| 3947 | assert_eq!(removed, 2); |
| 3948 | |
| 3949 | let remaining = repo.list(usize::MAX).unwrap(); |
| 3950 | assert_eq!(remaining.len(), 1); |
| 3951 | assert_eq!(remaining[0].label, "turn:2"); |
| 3952 | |
| 3953 | // New snapshot starts a clean chain (not appending to old). |
| 3954 | std::fs::write(repo.work_tree().join("f.txt"), "fresh").unwrap(); |
| 3955 | repo.snapshot("turn:new").unwrap(); |
| 3956 | assert_eq!(repo.list(usize::MAX).unwrap().len(), 2); |
| 3957 | } |
| 3958 | |
| 3959 | /// The per-snapshot prune drops half a window at once instead of |
| 3960 | /// rebuilding the chain for every snapshot past the cap. |
| 3961 | #[test] |
| 3962 | fn batched_prune_waits_for_half_a_window_then_drops_it_together() { |
| 3963 | let tmp = tempdir().unwrap(); |
| 3964 | let (repo, _home) = make_repo(tmp.path()); |
| 3965 | let max = 4; |
| 3966 | for n in 0..max + 1 { |
| 3967 | std::fs::write(repo.work_tree().join("f.txt"), format!("{n}")).unwrap(); |
| 3968 | repo.snapshot(&format!("tool:{n}")).unwrap(); |
| 3969 | } |
| 3970 | // One over the cap: the plain prune would rebuild now, the batched |
| 3971 | // one waits. |
| 3972 | assert_eq!(repo.prune_keep_last_n_batched(max).unwrap(), 0); |
| 3973 | assert_eq!(repo.list(usize::MAX).unwrap().len(), max + 1); |
| 3974 | |
| 3975 | std::fs::write(repo.work_tree().join("f.txt"), "last").unwrap(); |
| 3976 | repo.snapshot("tool:last").unwrap(); |
| 3977 | assert_eq!(repo.prune_keep_last_n_batched(max).unwrap(), 2); |
| 3978 | let kept = repo.list(usize::MAX).unwrap(); |
| 3979 | assert_eq!(kept.len(), max); |
| 3980 | assert_eq!(kept[0].label, "tool:last", "the newest snapshots survive"); |
| 3981 | } |
| 3982 | |
| 3983 | #[test] |
| 3984 | fn prune_keep_last_n_preserves_multiple_snapshots_in_order() { |
| 3985 | let tmp = tempdir().unwrap(); |
| 3986 | let (repo, _home) = make_repo(tmp.path()); |
| 3987 | |
| 3988 | for i in 0..4 { |
| 3989 | std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap(); |
| 3990 | repo.snapshot(&format!("turn:{i}")).unwrap(); |
| 3991 | std::thread::sleep(Duration::from_millis(1100)); |
| 3992 | } |
| 3993 | |
| 3994 | assert_eq!(repo.list(usize::MAX).unwrap().len(), 4); |
| 3995 | |
| 3996 | let removed = repo.prune_keep_last_n(2).unwrap(); |
| 3997 | assert_eq!(removed, 2); |
| 3998 | |
| 3999 | let remaining = repo.list(usize::MAX).unwrap(); |
| 4000 | assert_eq!(remaining.len(), 2); |
| 4001 | // Should be newest-first: turn:3 (newest), turn:2 (second newest) |
| 4002 | assert_eq!(remaining[0].label, "turn:3"); |
| 4003 | assert_eq!(remaining[1].label, "turn:2"); |
| 4004 | |
| 4005 | // New snapshot continues the chain. |
| 4006 | std::fs::write(repo.work_tree().join("f.txt"), "fresh").unwrap(); |
| 4007 | repo.snapshot("turn:new").unwrap(); |
| 4008 | let after = repo.list(usize::MAX).unwrap(); |
| 4009 | assert_eq!(after.len(), 3); |
| 4010 | assert_eq!(after[0].label, "turn:new"); |
| 4011 | } |
| 4012 | |
| 4013 | #[test] |
| 4014 | fn open_or_init_removes_stale_tmp_pack_files_only() { |
| 4015 | let tmp = tempdir().unwrap(); |
| 4016 | let (repo, _home) = make_repo(tmp.path()); |
| 4017 | let workspace = repo.work_tree().to_path_buf(); |
| 4018 | let pack_dir = repo.git_dir().join("objects").join("pack"); |
| 4019 | std::fs::create_dir_all(&pack_dir).unwrap(); |
| 4020 | |
| 4021 | let stale = pack_dir.join("tmp_pack_stale"); |
| 4022 | let fresh = pack_dir.join("tmp_pack_fresh"); |
| 4023 | let ordinary_pack = pack_dir.join("pack-kept.pack"); |
| 4024 | std::fs::write(&stale, b"stale").unwrap(); |
| 4025 | std::fs::write(&fresh, b"fresh").unwrap(); |
| 4026 | std::fs::write(&ordinary_pack, b"pack").unwrap(); |
| 4027 | |
| 4028 | let old_time = SystemTime::now() - STALE_TMP_PACK_AGE - Duration::from_secs(60); |
| 4029 | { |
| 4030 | let file = File::options().write(true).open(&stale).unwrap(); |
| 4031 | file.set_times(FileTimes::new().set_modified(old_time)) |
| 4032 | .unwrap(); |
| 4033 | } |
| 4034 | |
| 4035 | SnapshotRepo::open_or_init(&workspace).unwrap(); |
| 4036 | |
| 4037 | assert!(!stale.exists(), "stale tmp_pack file should be removed"); |
| 4038 | assert!(fresh.exists(), "fresh tmp_pack file should be kept"); |
| 4039 | assert!(ordinary_pack.exists(), "non-temp pack file should be kept"); |
| 4040 | } |
| 4041 | |
| 4042 | #[test] |
| 4043 | fn snapshot_respects_workspace_gitignore() { |
| 4044 | let tmp = tempdir().unwrap(); |
| 4045 | let (repo, _home) = make_repo(tmp.path()); |
| 4046 | std::fs::write(repo.work_tree().join(".gitignore"), "ignored.txt\n").unwrap(); |
| 4047 | std::fs::write(repo.work_tree().join("ignored.txt"), b"secret").unwrap(); |
| 4048 | std::fs::write(repo.work_tree().join("kept.txt"), b"public").unwrap(); |
| 4049 | |
| 4050 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 4051 | |
| 4052 | // `git ls-tree` against the snapshot's commit shouldn't list ignored.txt. |
| 4053 | let ls = run_git( |
| 4054 | repo.git_dir(), |
| 4055 | repo.work_tree(), |
| 4056 | &["ls-tree", "-r", "--name-only", id.as_str()], |
| 4057 | ) |
| 4058 | .expect("ls-tree"); |
| 4059 | let names = String::from_utf8_lossy(&ls.stdout); |
| 4060 | assert!(names.contains("kept.txt"), "kept.txt missing: {names}"); |
| 4061 | assert!( |
| 4062 | !names.contains("ignored.txt"), |
| 4063 | "ignored.txt should not be in snapshot: {names}", |
| 4064 | ); |
| 4065 | } |
| 4066 | |
| 4067 | #[test] |
| 4068 | fn unsafe_workspace_rejects_home_directory_workspace() { |
| 4069 | let tmp = tempdir().unwrap(); |
| 4070 | let home = tmp.path(); |
| 4071 | |
| 4072 | assert_eq!( |
| 4073 | unsafe_workspace_snapshot_reason(home, Some(home)), |
| 4074 | Some("home directory") |
| 4075 | ); |
| 4076 | } |
| 4077 | |
| 4078 | #[test] |
| 4079 | fn unsafe_workspace_rejects_home_collection_directories() { |
| 4080 | let tmp = tempdir().unwrap(); |
| 4081 | let home = tmp.path(); |
| 4082 | let desktop = tmp.path().join("Desktop"); |
| 4083 | std::fs::create_dir_all(&desktop).unwrap(); |
| 4084 | |
| 4085 | assert_eq!( |
| 4086 | unsafe_workspace_snapshot_reason(&desktop, Some(home)), |
| 4087 | Some("home collection directory") |
| 4088 | ); |
| 4089 | } |
| 4090 | |
| 4091 | #[test] |
| 4092 | fn unsafe_workspace_allows_project_directories_under_home() { |
| 4093 | let tmp = tempdir().unwrap(); |
| 4094 | let home = tmp.path(); |
| 4095 | let workspace = tmp.path().join("code").join("project"); |
| 4096 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4097 | |
| 4098 | assert_eq!( |
| 4099 | unsafe_workspace_snapshot_reason(&workspace, Some(home)), |
| 4100 | None |
| 4101 | ); |
| 4102 | } |
| 4103 | |
| 4104 | #[test] |
| 4105 | fn snapshot_respects_builtin_excludes() { |
| 4106 | let tmp = tempdir().unwrap(); |
| 4107 | let (repo, _home) = make_repo(tmp.path()); |
| 4108 | std::fs::create_dir_all(repo.work_tree().join("node_modules/pkg")).unwrap(); |
| 4109 | std::fs::create_dir_all(repo.work_tree().join(".next/cache")).unwrap(); |
| 4110 | std::fs::create_dir_all(repo.work_tree().join("src")).unwrap(); |
| 4111 | std::fs::write( |
| 4112 | repo.work_tree().join("node_modules/pkg/index.js"), |
| 4113 | b"generated", |
| 4114 | ) |
| 4115 | .unwrap(); |
| 4116 | std::fs::write(repo.work_tree().join(".next/cache/chunk.bin"), b"generated").unwrap(); |
| 4117 | std::fs::write(repo.work_tree().join("debug.wasm"), b"binary").unwrap(); |
| 4118 | std::fs::write(repo.work_tree().join("src/main.rs"), b"fn main() {}").unwrap(); |
| 4119 | |
| 4120 | let excludes = std::fs::read_to_string(repo.git_dir().join("info/exclude")).unwrap(); |
| 4121 | assert!(excludes.contains("node_modules/")); |
| 4122 | assert!(excludes.contains(".next/")); |
| 4123 | assert!(excludes.contains("*.wasm")); |
| 4124 | |
| 4125 | let id = repo.snapshot("pre-turn:1").expect("snapshot"); |
| 4126 | let ls = run_git( |
| 4127 | repo.git_dir(), |
| 4128 | repo.work_tree(), |
| 4129 | &["ls-tree", "-r", "--name-only", id.as_str()], |
| 4130 | ) |
| 4131 | .expect("ls-tree"); |
| 4132 | let names = String::from_utf8_lossy(&ls.stdout); |
| 4133 | assert!( |
| 4134 | names.contains("src/main.rs"), |
| 4135 | "src/main.rs missing: {names}" |
| 4136 | ); |
| 4137 | assert!( |
| 4138 | !names.contains("node_modules"), |
| 4139 | "node_modules should not be in snapshot: {names}", |
| 4140 | ); |
| 4141 | assert!( |
| 4142 | !names.contains(".next"), |
| 4143 | ".next should not be in snapshot: {names}", |
| 4144 | ); |
| 4145 | assert!( |
| 4146 | !names.contains("debug.wasm"), |
| 4147 | "binary artifacts should not be in snapshot: {names}", |
| 4148 | ); |
| 4149 | } |
| 4150 | |
| 4151 | #[test] |
| 4152 | fn open_or_init_is_idempotent() { |
| 4153 | let tmp = tempdir().unwrap(); |
| 4154 | let (_r, _h) = make_repo(tmp.path()); |
| 4155 | // Second open should not panic and should reuse the existing |
| 4156 | // `.git`. We re-open via the public API rather than make_repo to |
| 4157 | // avoid double-acquiring HOME (the guard would deadlock). |
| 4158 | drop((_r, _h)); |
| 4159 | let (_r2, _h2) = make_repo(tmp.path()); |
| 4160 | } |
| 4161 | |
| 4162 | #[test] |
| 4163 | fn home_directory_guard_matches_canonical_paths() { |
| 4164 | let tmp = tempdir().unwrap(); |
| 4165 | let home = tmp.path(); |
| 4166 | let home_canonical = home.canonicalize().unwrap(); |
| 4167 | let workspace = home.join("workspace"); |
| 4168 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4169 | let workspace_canonical = workspace.canonicalize().unwrap(); |
| 4170 | |
| 4171 | assert!(is_home_directory(&home_canonical, Some(home))); |
| 4172 | assert!(!is_home_directory(&workspace_canonical, Some(home))); |
| 4173 | assert!(!is_home_directory(&home_canonical, None)); |
| 4174 | } |
| 4175 | |
| 4176 | #[test] |
| 4177 | fn dir_size_bytes_measures_directory_bytes() { |
| 4178 | let tmp = tempdir().unwrap(); |
| 4179 | let dir = tmp.path().join("sizedir"); |
| 4180 | std::fs::create_dir_all(dir.join("sub")).unwrap(); |
| 4181 | // 3 bytes per file. |
| 4182 | std::fs::write(dir.join("a.txt"), b"abc").unwrap(); |
| 4183 | std::fs::write(dir.join("sub/b.txt"), b"xyz").unwrap(); |
| 4184 | |
| 4185 | let size = dir_size_bytes(&dir).expect("dir_size_bytes"); |
| 4186 | assert_eq!(size, 6, "two 3-byte files should measure 6 bytes"); |
| 4187 | |
| 4188 | // Write 2 MB of data. |
| 4189 | let big = dir.join("big.bin"); |
| 4190 | std::fs::write(&big, vec![0u8; 2 * 1024 * 1024]).unwrap(); |
| 4191 | let size = dir_size_bytes(&dir).expect("dir_size_bytes after big write"); |
| 4192 | assert_eq!( |
| 4193 | size, |
| 4194 | 2 * 1024 * 1024 + 6, |
| 4195 | "expected 2 MB + 6 bytes after writing a 2 MB file" |
| 4196 | ); |
| 4197 | } |
| 4198 | |
| 4199 | /// Regression: snapshot size cap (#1112). When the snapshot dir grows, |
| 4200 | /// `snapshot()` must prune old snapshots to stay under the limit. |
| 4201 | /// This test uses the real size constants, which are 500/400 MB — |
| 4202 | /// we can't easily blow up a temp dir to 500 MB in a unit test. |
| 4203 | /// Instead we verify the guard logic doesn't panic or error on a |
| 4204 | /// small repo (well under the cap), and that `snapshot()` still works. |
| 4205 | #[test] |
| 4206 | fn snapshot_succeeds_when_under_size_cap() { |
| 4207 | let tmp = tempdir().unwrap(); |
| 4208 | let (repo, _home) = make_repo(tmp.path()); |
| 4209 | // The side repo is tiny — well under 500 MB. Snapshot should work. |
| 4210 | std::fs::write(repo.work_tree().join("f.txt"), b"hello").unwrap(); |
| 4211 | let id = repo.snapshot("pre-turn:1").expect("snapshot under cap"); |
| 4212 | assert_eq!(id.as_str().len(), 40); |
| 4213 | } |
| 4214 | |
| 4215 | /// Sessions and sub-agents share one side repo. The size-pressure prune |
| 4216 | /// used to protect only the globally newest turn boundaries, so another |
| 4217 | /// session's snapshots pushed a running turn's `pre-turn:` out and its |
| 4218 | /// undo had nothing to restore. |
| 4219 | #[test] |
| 4220 | fn prune_size_pressure_keeps_every_sessions_turn_boundaries() { |
| 4221 | let tmp = tempdir().unwrap(); |
| 4222 | let (repo, _home) = make_repo(tmp.path()); |
| 4223 | for (i, (label, sid)) in [ |
| 4224 | ("pre-turn:5", "A"), |
| 4225 | ("tool:a", "A"), |
| 4226 | ("pre-turn:1", "B"), |
| 4227 | ("tool:b", "B"), |
| 4228 | ("post-turn:1", "B"), |
| 4229 | ("pre-turn:2", "B"), |
| 4230 | ("tool:c", "B"), |
| 4231 | ] |
| 4232 | .into_iter() |
| 4233 | .enumerate() |
| 4234 | { |
| 4235 | std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap(); |
| 4236 | repo.snapshot_with_session(label, Some(sid)) |
| 4237 | .expect("snapshot"); |
| 4238 | } |
| 4239 | repo.prune_size_pressure(0, 0).expect("prune_size_pressure"); |
| 4240 | let kept: Vec<(String, Option<String>)> = repo |
| 4241 | .list(usize::MAX) |
| 4242 | .unwrap() |
| 4243 | .into_iter() |
| 4244 | .map(|s| (s.label, s.session_id)) |
| 4245 | .collect(); |
| 4246 | let kept_a_pre = kept |
| 4247 | .iter() |
| 4248 | .any(|(label, sid)| label == "pre-turn:5" && sid.as_deref() == Some("A")); |
| 4249 | assert!( |
| 4250 | kept_a_pre, |
| 4251 | "session A's running turn stays restorable: {kept:?}" |
| 4252 | ); |
| 4253 | assert_eq!( |
| 4254 | kept.iter() |
| 4255 | .map(|(label, _)| label.as_str()) |
| 4256 | .collect::<Vec<_>>(), |
| 4257 | ["tool:c", "pre-turn:2", "post-turn:1", "pre-turn:5"], |
| 4258 | "only boundaries and the newest snapshot survive a full cut" |
| 4259 | ); |
| 4260 | } |
| 4261 | |
| 4262 | /// The size-pressure prune drops the oldest snapshots first and keeps the |
| 4263 | /// newest one plus the newest turn boundaries. It used to prune by age |
| 4264 | /// from one second down, which wiped every restore point, the running |
| 4265 | /// turn's own `pre-turn:` included, on each snapshot of a side repo over |
| 4266 | /// the cap. |
| 4267 | #[test] |
| 4268 | fn prune_size_pressure_drops_oldest_first_and_keeps_turn_boundaries() { |
| 4269 | let tmp = tempdir().unwrap(); |
| 4270 | let (repo, _home) = make_repo(tmp.path()); |
| 4271 | for (i, label) in [ |
| 4272 | "pre-turn:1", |
| 4273 | "tool:a", |
| 4274 | "post-turn:1", |
| 4275 | "pre-turn:2", |
| 4276 | "tool:b", |
| 4277 | "tool:c", |
| 4278 | ] |
| 4279 | .into_iter() |
| 4280 | .enumerate() |
| 4281 | { |
| 4282 | std::fs::write(repo.work_tree().join("f.txt"), format!("v{i}")).unwrap(); |
| 4283 | repo.snapshot(label).expect("snapshot"); |
| 4284 | } |
| 4285 | // A zero byte limit makes any non-empty side repo "over limit", so the |
| 4286 | // prune cuts as far as it may and reports exactly what it destroyed; |
| 4287 | // the count is what the user-visible notice is built from. |
| 4288 | let removed = repo.prune_size_pressure(0, 0).expect("prune_size_pressure"); |
| 4289 | let labels: Vec<String> = repo |
| 4290 | .list(usize::MAX) |
| 4291 | .unwrap() |
| 4292 | .into_iter() |
| 4293 | .map(|s| s.label) |
| 4294 | .collect(); |
| 4295 | assert_eq!( |
| 4296 | labels, |
| 4297 | ["tool:c", "pre-turn:2", "post-turn:1"], |
| 4298 | "the newest snapshot and the newest turn boundaries survive" |
| 4299 | ); |
| 4300 | assert_eq!(removed, 3, "every dropped snapshot is reported"); |
| 4301 | // The survivors still restore: the running turn can be undone. |
| 4302 | let pre = repo.list(usize::MAX).unwrap()[1].id.clone(); |
| 4303 | repo.restore(&pre) |
| 4304 | .expect("restore the running turn's boundary"); |
| 4305 | assert_eq!( |
| 4306 | std::fs::read_to_string(repo.work_tree().join("f.txt")).unwrap(), |
| 4307 | "v3" |
| 4308 | ); |
| 4309 | } |
| 4310 | |
| 4311 | #[test] |
| 4312 | fn prune_size_pressure_is_a_noop_under_the_limit() { |
| 4313 | let tmp = tempdir().unwrap(); |
| 4314 | let (repo, _home) = make_repo(tmp.path()); |
| 4315 | std::fs::write(repo.work_tree().join("f.txt"), b"v0").unwrap(); |
| 4316 | repo.snapshot("pre-turn:0").expect("snapshot"); |
| 4317 | let removed = repo |
| 4318 | .prune_size_pressure(u64::MAX, u64::MAX) |
| 4319 | .expect("prune_size_pressure"); |
| 4320 | assert_eq!(removed, 0, "under the limit nothing may be removed"); |
| 4321 | assert_eq!(repo.list(usize::MAX).unwrap().len(), 1); |
| 4322 | } |
| 4323 | |
| 4324 | #[test] |
| 4325 | fn snapshot_history_pruned_message_names_workspace_count_and_cap() { |
| 4326 | let msg = snapshot_history_pruned_message(Path::new("/tmp/ws"), 7); |
| 4327 | assert!(msg.contains("/tmp/ws"), "message must name the workspace"); |
| 4328 | assert!(msg.contains("7"), "message must state the removed count"); |
| 4329 | assert!( |
| 4330 | msg.contains(&MAX_SNAPSHOT_SIZE_MB.to_string()), |
| 4331 | "message must state the storage cap" |
| 4332 | ); |
| 4333 | } |
| 4334 | |
| 4335 | #[test] |
| 4336 | fn estimate_workspace_size_bounded_returns_total_when_under_cap() { |
| 4337 | let tmp = tempdir().unwrap(); |
| 4338 | let workspace = tmp.path().join("workspace"); |
| 4339 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4340 | std::fs::write(workspace.join("a.txt"), vec![b'a'; 100]).unwrap(); |
| 4341 | std::fs::write(workspace.join("b.txt"), vec![b'b'; 50]).unwrap(); |
| 4342 | let total = estimate_workspace_size_bounded(&workspace, 10_000, SIZE_WALK_MAX_ENTRIES) |
| 4343 | .expect("under-cap walk must return a total"); |
| 4344 | assert!( |
| 4345 | total >= 150, |
| 4346 | "total ({total}) must include both files (≥150 bytes)" |
| 4347 | ); |
| 4348 | } |
| 4349 | |
| 4350 | #[test] |
| 4351 | fn estimate_workspace_size_bounded_reports_the_size_gate_when_over_cap() { |
| 4352 | let tmp = tempdir().unwrap(); |
| 4353 | let workspace = tmp.path().join("workspace"); |
| 4354 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4355 | // Two 1 KB files, cap at 1 KB — second file should trip the cap. |
| 4356 | std::fs::write(workspace.join("a.bin"), vec![b'a'; 1024]).unwrap(); |
| 4357 | std::fs::write(workspace.join("b.bin"), vec![b'b'; 1024]).unwrap(); |
| 4358 | assert_eq!( |
| 4359 | estimate_workspace_size_bounded(&workspace, 1024, SIZE_WALK_MAX_ENTRIES), |
| 4360 | Err(WorkspaceGate::TooLarge), |
| 4361 | "over-cap walk must name the size gate for early bailout" |
| 4362 | ); |
| 4363 | } |
| 4364 | |
| 4365 | #[test] |
| 4366 | fn oversize_gate_message_states_the_byte_cap_without_a_remedy() { |
| 4367 | // The remedy is localized by the notice surfaces; repeating it here is |
| 4368 | // what produced the doubled warning. |
| 4369 | let message = |
| 4370 | WorkspaceGate::TooLarge.describe(2 * 1024 * 1024 * 1024, Path::new("/tmp/ws")); |
| 4371 | assert!(message.starts_with(GATE_TOO_LARGE_MARKER)); |
| 4372 | assert!(message.contains("/tmp/ws")); |
| 4373 | assert!(!message.contains("max_workspace_gb")); |
| 4374 | assert_eq!(message.lines().count(), 1, "the gate message is one line"); |
| 4375 | } |
| 4376 | |
| 4377 | #[test] |
| 4378 | fn entry_gate_message_is_distinct_and_never_blames_the_size_cap() { |
| 4379 | let message = WorkspaceGate::TooManyEntries.describe(0, Path::new("/tmp/ws")); |
| 4380 | assert!(message.starts_with(GATE_TOO_MANY_ENTRIES_MARKER)); |
| 4381 | assert!( |
| 4382 | !message.contains(GATE_TOO_LARGE_MARKER), |
| 4383 | "the entry gate must not be reported as a size trip" |
| 4384 | ); |
| 4385 | assert!(message.contains(&SIZE_WALK_MAX_ENTRIES.to_string())); |
| 4386 | assert!(!message.contains("max_workspace_gb")); |
| 4387 | } |
| 4388 | |
| 4389 | #[test] |
| 4390 | fn estimate_workspace_size_bounded_skips_builtin_excluded_dirs() { |
| 4391 | let tmp = tempdir().unwrap(); |
| 4392 | let workspace = tmp.path().join("workspace"); |
| 4393 | std::fs::create_dir_all(workspace.join("node_modules")).unwrap(); |
| 4394 | std::fs::create_dir_all(workspace.join("target")).unwrap(); |
| 4395 | std::fs::create_dir_all(workspace.join("src")).unwrap(); |
| 4396 | // 2 MB of "build output" in excluded dirs — must not count toward |
| 4397 | // the cap. |
| 4398 | std::fs::write(workspace.join("node_modules/big.bin"), vec![0u8; 1_000_000]).unwrap(); |
| 4399 | std::fs::write(workspace.join("target/big.bin"), vec![0u8; 1_000_000]).unwrap(); |
| 4400 | std::fs::write(workspace.join("src/lib.rs"), b"// real source").unwrap(); |
| 4401 | let total = estimate_workspace_size_bounded(&workspace, 500_000, SIZE_WALK_MAX_ENTRIES) |
| 4402 | .expect("walk must succeed since real source is tiny"); |
| 4403 | assert!( |
| 4404 | total < 1_000, |
| 4405 | "total ({total}) must reflect only src/, not node_modules/ or target/" |
| 4406 | ); |
| 4407 | } |
| 4408 | |
| 4409 | #[test] |
| 4410 | fn estimate_workspace_size_bounded_cap_zero_disables_cap() { |
| 4411 | let tmp = tempdir().unwrap(); |
| 4412 | let workspace = tmp.path().join("workspace"); |
| 4413 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4414 | // 10 KB file — would trip a 1 KB cap, but cap=0 means no cap. |
| 4415 | std::fs::write(workspace.join("big.bin"), vec![0u8; 10 * 1024]).unwrap(); |
| 4416 | let total = estimate_workspace_size_bounded(&workspace, 0, SIZE_WALK_MAX_ENTRIES) |
| 4417 | .expect("cap=0 must always return a total"); |
| 4418 | assert!( |
| 4419 | total >= 10 * 1024, |
| 4420 | "total ({total}) must include the 10 KB file when cap is disabled" |
| 4421 | ); |
| 4422 | } |
| 4423 | |
| 4424 | /// The entry ceiling is the bound that no test could reach before |
| 4425 | /// `max_entries` became a parameter: 200,000 inodes per run is not a |
| 4426 | /// price a unit test should pay. A byte-cheap workspace must still be |
| 4427 | /// refused, and refused as the *entry* gate — reporting `TooLarge` here |
| 4428 | /// would offer `max_workspace_gb` as a remedy that cannot lift it. |
| 4429 | #[test] |
| 4430 | fn entry_ceiling_refuses_a_byte_cheap_workspace_with_too_many_entries() { |
| 4431 | let tmp = tempdir().unwrap(); |
| 4432 | let workspace = tmp.path().join("workspace"); |
| 4433 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4434 | for i in 0..10 { |
| 4435 | std::fs::write(workspace.join(format!("f{i}.txt")), b"x").unwrap(); |
| 4436 | } |
| 4437 | assert_eq!( |
| 4438 | estimate_workspace_size_bounded(&workspace, 10_000_000, 3), |
| 4439 | Err(WorkspaceGate::TooManyEntries), |
| 4440 | "ten tiny files under a 10 MB cap must trip the entry bound, not the byte cap" |
| 4441 | ); |
| 4442 | } |
| 4443 | |
| 4444 | /// The invariant documented on `WorkspaceGate::TooManyEntries` and on the |
| 4445 | /// estimator: `max_workspace_gb = 0` opts out of the byte cap only. A |
| 4446 | /// future "if `cap_bytes == 0`, skip the walk" shortcut would satisfy |
| 4447 | /// every other test here and silently delete the ceiling that exists to |
| 4448 | /// stop a multi-minute `git add -A`. |
| 4449 | #[test] |
| 4450 | fn cap_zero_does_not_lift_the_entry_ceiling() { |
| 4451 | let tmp = tempdir().unwrap(); |
| 4452 | let workspace = tmp.path().join("workspace"); |
| 4453 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4454 | for i in 0..10 { |
| 4455 | std::fs::write(workspace.join(format!("f{i}.txt")), b"x").unwrap(); |
| 4456 | } |
| 4457 | assert_eq!( |
| 4458 | estimate_workspace_size_bounded(&workspace, 0, 3), |
| 4459 | Err(WorkspaceGate::TooManyEntries), |
| 4460 | "cap_bytes = 0 disables the byte cap, never the entry ceiling" |
| 4461 | ); |
| 4462 | } |
| 4463 | |
| 4464 | /// `ignore` disables gitignore matching entirely when no ancestor holds a |
| 4465 | /// `.git` (`require_git` defaults to true), but the snapshot's own |
| 4466 | /// `git add -A --work-tree <workspace>` reads `.gitignore` either way. A |
| 4467 | /// non-git workspace was therefore measured on content that would never |
| 4468 | /// be staged — and then told to fix it by editing `.gitignore`. |
| 4469 | /// |
| 4470 | /// Deliberately creates no `.git`: the point is the non-repo case. |
| 4471 | #[test] |
| 4472 | fn gitignored_content_is_excluded_outside_a_git_repo() { |
| 4473 | let tmp = tempdir().unwrap(); |
| 4474 | let workspace = tmp.path().join("workspace"); |
| 4475 | std::fs::create_dir_all(workspace.join("src")).unwrap(); |
| 4476 | std::fs::write(workspace.join(".gitignore"), "big.bin\n").unwrap(); |
| 4477 | std::fs::write(workspace.join("big.bin"), vec![0u8; 1_000_000]).unwrap(); |
| 4478 | std::fs::write(workspace.join("src/lib.rs"), b"// real source").unwrap(); |
| 4479 | assert!( |
| 4480 | !workspace.join(".git").exists(), |
| 4481 | "this test is only meaningful outside a git repo" |
| 4482 | ); |
| 4483 | let total = estimate_workspace_size_bounded(&workspace, 500_000, SIZE_WALK_MAX_ENTRIES) |
| 4484 | .expect("the only large file is gitignored, so the walk must fit under the cap"); |
| 4485 | assert!( |
| 4486 | total < 1_000, |
| 4487 | "total ({total}) must exclude the gitignored 1 MB file" |
| 4488 | ); |
| 4489 | } |
| 4490 | |
| 4491 | #[test] |
| 4492 | fn open_or_init_with_cap_rejects_oversized_workspace() { |
| 4493 | let tmp = tempdir().unwrap(); |
| 4494 | let workspace = tmp.path().join("workspace"); |
| 4495 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4496 | let _home = scoped_home(tmp.path()); |
| 4497 | // Drop a 4 KB file under a 1 KB cap. |
| 4498 | std::fs::write(workspace.join("big.bin"), vec![0u8; 4096]).unwrap(); |
| 4499 | let outcome = SnapshotRepo::open_or_init_with_cap(&workspace, 1024); |
| 4500 | let err = match outcome { |
| 4501 | Ok(_) => panic!("oversized workspace must fail open_or_init_with_cap"), |
| 4502 | Err(e) => e, |
| 4503 | }; |
| 4504 | let msg = err.to_string(); |
| 4505 | assert!( |
| 4506 | msg.contains(GATE_TOO_LARGE_MARKER), |
| 4507 | "error must call out the size cap; got: {msg}" |
| 4508 | ); |
| 4509 | let named_owned = workspace.display().to_string(); |
| 4510 | let named = named_owned |
| 4511 | .strip_prefix(r"\\?\") |
| 4512 | .or_else(|| named_owned.strip_prefix("//?/")) |
| 4513 | .unwrap_or(named_owned.as_str()); |
| 4514 | assert!( |
| 4515 | msg.contains(named), |
| 4516 | "error must name the workspace it refused; got: {msg}" |
| 4517 | ); |
| 4518 | // The remedy belongs to the localized notice. Repeating it here is |
| 4519 | // what produced the doubled, three-line warning users saw. |
| 4520 | assert!( |
| 4521 | !msg.contains("max_workspace_gb"), |
| 4522 | "gate error must not carry its own remedy copy; got: {msg}" |
| 4523 | ); |
| 4524 | } |
| 4525 | |
| 4526 | #[test] |
| 4527 | fn open_or_init_with_cap_zero_disables_size_check() { |
| 4528 | let tmp = tempdir().unwrap(); |
| 4529 | let workspace = tmp.path().join("workspace"); |
| 4530 | std::fs::create_dir_all(&workspace).unwrap(); |
| 4531 | let _home = scoped_home(tmp.path()); |
| 4532 | // 4 KB file but cap=0 → should still succeed. |
| 4533 | std::fs::write(workspace.join("big.bin"), vec![0u8; 4096]).unwrap(); |
| 4534 | let repo = SnapshotRepo::open_or_init_with_cap(&workspace, 0) |
| 4535 | .expect("cap=0 must skip the size check"); |
| 4536 | let id = repo |
| 4537 | .snapshot("pre-turn:1") |
| 4538 | .expect("snapshot under disabled cap"); |
| 4539 | assert_eq!(id.as_str().len(), 40); |
| 4540 | } |
| 4541 | |
| 4542 | #[test] |
| 4543 | fn session_tagged_snapshot_round_trips_through_list() { |
| 4544 | let tmp = tempdir().unwrap(); |
| 4545 | let (repo, _home) = make_repo(tmp.path()); |
| 4546 | std::fs::write(repo.work_tree().join("a.txt"), b"x").unwrap(); |
| 4547 | |
| 4548 | repo.snapshot_with_session("pre-turn:1", Some("sess-42")) |
| 4549 | .expect("snapshot with session"); |
| 4550 | |
| 4551 | let list = repo.list(10).expect("list"); |
| 4552 | assert_eq!(list.len(), 1); |
| 4553 | // The visible label stays clean; the session id is decoded separately. |
| 4554 | assert_eq!(list[0].label, "pre-turn:1"); |
| 4555 | assert_eq!(list[0].session_id.as_deref(), Some("sess-42")); |
| 4556 | } |
| 4557 | |
| 4558 | #[test] |
| 4559 | fn untagged_snapshot_decodes_without_session() { |
| 4560 | let tmp = tempdir().unwrap(); |
| 4561 | let (repo, _home) = make_repo(tmp.path()); |
| 4562 | std::fs::write(repo.work_tree().join("a.txt"), b"x").unwrap(); |
| 4563 | |
| 4564 | repo.snapshot("pre-turn:1").expect("snapshot"); |
| 4565 | |
| 4566 | let list = repo.list(10).expect("list"); |
| 4567 | assert_eq!(list.len(), 1); |
| 4568 | assert_eq!(list[0].label, "pre-turn:1"); |
| 4569 | assert_eq!(list[0].session_id, None); |
| 4570 | } |
| 4571 | |
| 4572 | /// A burst of per-tool snapshots larger than the count cap never pushes |
| 4573 | /// out the turn boundaries: the running turn's `pre-turn:` restore point |
| 4574 | /// survives its own 50-write turn and another thread's burst. |
| 4575 | #[test] |
| 4576 | fn prune_keep_last_n_retains_turn_boundaries_through_a_tool_burst() { |
| 4577 | let tmp = tempdir().unwrap(); |
| 4578 | let (repo, _home) = make_repo(tmp.path()); |
| 4579 | let file = repo.work_tree().join("a.txt"); |
| 4580 | std::fs::write(&file, "v0").unwrap(); |
| 4581 | let older_post = repo |
| 4582 | .take_snapshot("post-turn:0", Some("thr_a")) |
| 4583 | .expect("snapshot"); |
| 4584 | let pre = repo |
| 4585 | .take_snapshot("pre-turn:1", Some("thr_a")) |
| 4586 | .expect("snapshot"); |
| 4587 | for i in 0..6 { |
| 4588 | std::fs::write(&file, format!("v{}", i + 1)).unwrap(); |
| 4589 | repo.take_snapshot(&format!("tool:call-{i}"), Some("thr_a")) |
| 4590 | .expect("snapshot"); |
| 4591 | repo.take_snapshot(&format!("post-tool:call-{i}"), Some("thr_a")) |
| 4592 | .expect("snapshot"); |
| 4593 | } |
| 4594 | // 14 snapshots, cap 3: the newest three plus the (two) boundaries. |
| 4595 | let removed = repo.prune_keep_last_n(3).expect("prune"); |
| 4596 | assert_eq!(removed, 9); |
| 4597 | let labels: Vec<String> = repo |
| 4598 | .list(usize::MAX) |
| 4599 | .unwrap() |
| 4600 | .into_iter() |
| 4601 | .map(|snapshot| snapshot.label) |
| 4602 | .collect(); |
| 4603 | assert_eq!( |
| 4604 | labels, |
| 4605 | [ |
| 4606 | "post-tool:call-5", |
| 4607 | "tool:call-5", |
| 4608 | "post-tool:call-4", |
| 4609 | "pre-turn:1", |
| 4610 | "post-turn:0", |
| 4611 | ] |
| 4612 | ); |
| 4613 | let trees: Vec<SnapshotId> = repo |
| 4614 | .list(usize::MAX) |
| 4615 | .unwrap() |
| 4616 | .into_iter() |
| 4617 | .map(|snapshot| snapshot.tree) |
| 4618 | .collect(); |
| 4619 | assert!(trees.contains(&pre.tree) && trees.contains(&older_post.tree)); |
| 4620 | |
| 4621 | // Boundaries are themselves capped at the same count. |
| 4622 | for i in 2..6 { |
| 4623 | repo.take_snapshot(&format!("pre-turn:{i}"), Some("thr_a")) |
| 4624 | .expect("snapshot"); |
| 4625 | } |
| 4626 | repo.prune_keep_last_n(3).expect("prune"); |
| 4627 | let boundaries = repo |
| 4628 | .list(usize::MAX) |
| 4629 | .unwrap() |
| 4630 | .into_iter() |
| 4631 | .filter(|snapshot| is_turn_boundary_label(&snapshot.label)) |
| 4632 | .count(); |
| 4633 | assert_eq!(boundaries, 3); |
| 4634 | } |
| 4635 | |
| 4636 | #[test] |
| 4637 | fn prune_keep_last_n_preserves_session_tags() { |
| 4638 | let tmp = tempdir().unwrap(); |
| 4639 | let (repo, _home) = make_repo(tmp.path()); |
| 4640 | let file = repo.work_tree().join("a.txt"); |
| 4641 | |
| 4642 | // More snapshots than DEFAULT_MAX_SNAPSHOTS (50) so the survivor |
| 4643 | // chain is rebuilt as orphan commits — the path that previously |
| 4644 | // dropped the [sid=...] label prefix and turned every surviving |
| 4645 | // snapshot into a "legacy" (untagged) one. |
| 4646 | for i in 0..55 { |
| 4647 | std::fs::write(&file, format!("v{i}")).unwrap(); |
| 4648 | repo.snapshot_with_session(&format!("pre-turn:{i}"), Some("sess-p")) |
| 4649 | .expect("tagged snapshot"); |
| 4650 | } |
| 4651 | |
| 4652 | let removed = repo.prune_keep_last_n(50).expect("prune"); |
| 4653 | assert!(removed > 0, "expected prune to drop older snapshots"); |
| 4654 | |
| 4655 | let list = repo.list(usize::MAX).expect("list"); |
| 4656 | assert_eq!(list.len(), 50); |
| 4657 | assert!( |
| 4658 | list.iter() |
| 4659 | .all(|s| s.session_id.as_deref() == Some("sess-p")), |
| 4660 | "prune must preserve [sid=...] prefixes; got untagged survivors" |
| 4661 | ); |
| 4662 | } |
| 4663 | |
| 4664 | #[test] |
| 4665 | fn tagged_and_untagged_snapshots_coexist_in_one_chain() { |
| 4666 | let tmp = tempdir().unwrap(); |
| 4667 | let (repo, _home) = make_repo(tmp.path()); |
| 4668 | std::fs::write(repo.work_tree().join("a.txt"), b"v1").unwrap(); |
| 4669 | |
| 4670 | // Legacy untagged snapshot, then a session-tagged one. |
| 4671 | repo.snapshot("pre-turn:1").expect("legacy snapshot"); |
| 4672 | std::fs::write(repo.work_tree().join("a.txt"), b"v2").unwrap(); |
| 4673 | repo.snapshot_with_session("pre-turn:1", Some("sess-a")) |
| 4674 | .expect("tagged snapshot"); |
| 4675 | |
| 4676 | let list = repo.list(10).expect("list"); |
| 4677 | assert_eq!(list.len(), 2); |
| 4678 | // Newest first. |
| 4679 | assert_eq!(list[0].session_id.as_deref(), Some("sess-a")); |
| 4680 | assert_eq!(list[1].session_id, None); |
| 4681 | assert_eq!(list[1].label, "pre-turn:1"); |
| 4682 | } |
| 4683 | |
| 4684 | #[test] |
| 4685 | fn run_git_drains_output_larger_than_the_pipe_buffer() { |
| 4686 | let tmp = tempdir().unwrap(); |
| 4687 | let (repo, _home) = make_repo(tmp.path()); |
| 4688 | // ~5000 paths is well past the 64 KiB OS pipe buffer: output of this |
| 4689 | // size is only reachable when the pipes are drained while the child |
| 4690 | // runs. A bounded read that only starts after the child exits would |
| 4691 | // deadlock the child on its own output and die at the command |
| 4692 | // timeout instead of returning the tree listing. |
| 4693 | for i in 0..5000 { |
| 4694 | std::fs::write(repo.work_tree().join(format!("file_{i:05}.txt")), b"x").unwrap(); |
| 4695 | } |
| 4696 | let id = repo.snapshot("large-output").expect("snapshot"); |
| 4697 | let paths = repo |
| 4698 | .tree_paths(id.as_str()) |
| 4699 | .expect("tree_paths must drain output instead of timing out"); |
| 4700 | assert!( |
| 4701 | paths.len() >= 5000, |
| 4702 | "expected every file in the tree listing, got {}", |
| 4703 | paths.len() |
| 4704 | ); |
| 4705 | } |
| 4706 | } |
| 4707 | |
| 4708 | /// The bounded-git tests drive the core with real children, so they need a |
| 4709 | /// POSIX shell; the timeout pin also depends on wall-clock behavior. |
| 4710 | #[cfg(all(test, unix))] |
| 4711 | mod bounded_git_tests { |
| 4712 | use super::*; |
| 4713 | |
| 4714 | #[test] |
| 4715 | fn bounded_git_times_out_a_wedged_child_and_reports_timed_out() { |
| 4716 | // A child that never exits stands in for a wedged git (stalled |
| 4717 | // mount, hung hook): the call must come back with a TimedOut error |
| 4718 | // promptly instead of blocking the turn pipeline forever. |
| 4719 | let started = std::time::Instant::now(); |
| 4720 | let mut wedged = std::process::Command::new("sh"); |
| 4721 | wedged.arg("-c").arg("sleep 30"); |
| 4722 | let err = run_bounded_git_with_timeout(&mut wedged, "sleep", Duration::from_millis(250)) |
| 4723 | .expect_err("a wedged child must hit the bound"); |
| 4724 | assert_eq!(err.kind(), io::ErrorKind::TimedOut); |
| 4725 | assert!( |
| 4726 | err.to_string().contains("timed out after"), |
| 4727 | "the error must report the bound; got {err}" |
| 4728 | ); |
| 4729 | let elapsed = started.elapsed(); |
| 4730 | assert!( |
| 4731 | elapsed < Duration::from_secs(15), |
| 4732 | "the bound must be enforced promptly; took {elapsed:?}" |
| 4733 | ); |
| 4734 | } |
| 4735 | |
| 4736 | #[test] |
| 4737 | fn bounded_git_returns_promptly_when_a_grandchild_holds_the_pipes() { |
| 4738 | // The direct child (sh) exits after the echo; the backgrounded |
| 4739 | // sleep inherits both pipes and holds them for 30s. The call must |
| 4740 | // still return promptly with the output captured before the grace, |
| 4741 | // annotated on stderr — not wait out the grandchild. |
| 4742 | let started = std::time::Instant::now(); |
| 4743 | let mut sh = std::process::Command::new("sh"); |
| 4744 | sh.arg("-c").arg("echo bounded-git-grandchild; sleep 30 &"); |
| 4745 | let output = run_bounded_git(&mut sh, "sh").expect("sh must succeed"); |
| 4746 | let elapsed = started.elapsed(); |
| 4747 | assert!(output.status.success()); |
| 4748 | assert!( |
| 4749 | String::from_utf8_lossy(&output.stdout).contains("bounded-git-grandchild"), |
| 4750 | "output captured before the grace must survive: {:?}", |
| 4751 | output.stdout |
| 4752 | ); |
| 4753 | assert!( |
| 4754 | String::from_utf8_lossy(&output.stderr) |
| 4755 | .contains("git output pipes did not close after git exited"), |
| 4756 | "the partial-output note must explain the early return: {:?}", |
| 4757 | output.stderr |
| 4758 | ); |
| 4759 | assert!( |
| 4760 | elapsed < Duration::from_secs(15), |
| 4761 | "a pipe-holding grandchild must not hold the call past the grace; took {elapsed:?}" |
| 4762 | ); |
| 4763 | } |
| 4764 | } |
| 4765 |