git.lucas.co / cce-files
file manager
git clone https://git.lucas.co/cce-files.git

commit75ed1617657c762ec80128f8b0debf998b78b2f1
parentc52cd0224c
authorLucas Galante <lsgalante12@gmail.com>
date2026-10-06 12:53
fix(scan): enter mounts on the same disk, so a scan of / includes /home

The walk skipped every directory on another device than the scan
root's. On a btrfs layout each mounted subvolume is its own device: @
at / is 0:31, @home at /home 0:50, so a scan of / left out /home, most
of the disk, along with /var/log and the pacman cache. A separate /home
partition was dropped the same way.

The walk now enters a mount point when its topmost mount is on the same
physical disk as the mount the root lies in. Each source in
/proc/self/mountinfo is resolved through /sys/class/block, past a
partition to its disk and through device-mapper (LUKS, LVM) to what it
is built on. /proc, /sys, tmpfs, network shares and other drives have no
disk or another one and stay out, which is what the device check was
for. Unmounted nested subvolumes (snapshots, container layers) have a
device of their own and no mount, and stay out too, so a snapshot is not
counted as a second copy.

Crossing mounts makes a bind mount reachable, so every directory is now
entered once by (device, inode): a second sighting is skipped, counted
once, and cannot loop the walk. The root is canonicalised so its paths
match the mount table's.

On this machine a scan of / now crosses /home, /boot, /var/log and
/var/cache/pacman/pkg, and nothing else.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

 src/services/scan.rs | 192 +++++++++++++++++++++++++++++++++++++++++++++++----
 1 file changed, 180 insertions(+), 12 deletions(-)

diff --git a/src/services/scan.rs b/src/services/scan.rs
index 21ab615..e8c2d5e 100644
--- a/src/services/scan.rs
+++ b/src/services/scan.rs
@@ -6,7 +6,8 @@
 //! on the `FsService` thread like every other request; the app sees it only as
 //! progress messages followed by a completed tree.
 
-use std::path::Path;
+use std::collections::{HashMap, HashSet};
+use std::path::{Path, PathBuf};
 use std::os::unix::fs::MetadataExt;
 use std::sync::atomic::{AtomicBool, Ordering};
 use std::time::{Duration, Instant};
@@ -48,10 +49,15 @@ pub struct ScanResult {
 }
 
 struct Walker<'a> {
-    /// Device of the scan root. Entries on any other device are skipped, so
-    /// scanning `/` does not wander into `/proc`, `/sys`, or a mounted backup
-    /// drive — and cannot loop through a bind mount pointing back inside.
-    dev: u64,
+    /// Mount points the walk may descend into although they sit on another
+    /// device than their parent — see [`crossable`]. Any other change of
+    /// device is skipped, so scanning `/` does not wander into `/proc`, `/sys`,
+    /// a tmpfs, or a backup drive.
+    crossable: HashSet<PathBuf>,
+    /// Every directory entered, by (device, inode). Directories have no hard
+    /// links, so a second sighting is a bind mount: skipped, it is counted once
+    /// and cannot loop the walk back into itself.
+    seen: HashSet<(u64, u64)>,
     cancel: &'a AtomicBool,
     files: u64,
     dirs: u64,
@@ -61,7 +67,8 @@ struct Walker<'a> {
 }
 
 impl Walker<'_> {
-    fn walk(&mut self, dir: &Path, name: String, depth: u32) -> TreeNode {
+    /// `dev` is the device `dir` lies on; a child on any other is a mount.
+    fn walk(&mut self, dir: &Path, name: String, depth: u32, dev: u64) -> TreeNode {
         let mut node = TreeNode { name, size: 0, is_dir: true, children: Vec::new() };
         if depth >= MAX_DEPTH || self.cancel.load(Ordering::Relaxed) {
             return node;
@@ -81,17 +88,24 @@ impl Walker<'_> {
             // subtree's total, and cannot loop the walk back into itself.
             let Ok(meta) = entry.metadata() else { continue };
             let ft = meta.file_type();
-            if ft.is_symlink() || meta.dev() != self.dev {
+            if ft.is_symlink() {
                 continue;
             }
 
             let child_name = entry.file_name().to_string_lossy().into_owned();
             if ft.is_dir() {
+                let path = entry.path();
+                if meta.dev() != dev && !self.crossable.contains(&path) {
+                    continue;
+                }
+                if !self.seen.insert((meta.dev(), meta.ino())) {
+                    continue;
+                }
                 self.dirs += 1;
-                let child = self.walk(&entry.path(), child_name, depth + 1);
+                let child = self.walk(&path, child_name, depth + 1, meta.dev());
                 node.size += child.size;
                 node.children.push(child);
-            } else if ft.is_file() {
+            } else if ft.is_file() && meta.dev() == dev {
                 let size = meta.len();
                 self.files += 1;
                 self.bytes += size;
@@ -104,7 +118,7 @@ impl Walker<'_> {
                 });
             }
             // Sockets, fifos, and device nodes occupy no meaningful space and
-            // are dropped entirely.
+            // are dropped entirely, as is a file bind-mounted from elsewhere.
             self.maybe_report();
         }
 
@@ -120,6 +134,113 @@ impl Walker<'_> {
     }
 }
 
+/// One line of `/proc/self/mountinfo`, as much as the walk needs.
+#[derive(Debug, Clone, PartialEq)]
+struct Mount {
+    point: PathBuf,
+    /// What is mounted: a device path for a disk filesystem, a bare word
+    /// (`tmpfs`, `proc`) or `host:/path` for anything else.
+    source: String,
+}
+
+/// Parse mountinfo: `id parent maj:min root POINT opts [optional...] - fstype SOURCE superopts`.
+/// The optional fields vary in number, so the source is found after the ` - `.
+fn parse_mountinfo(text: &str) -> Vec<Mount> {
+    text.lines()
+        .filter_map(|line| {
+            let (left, right) = line.split_once(" - ")?;
+            let point = left.split(' ').nth(4)?;
+            let source = right.split(' ').nth(1)?;
+            Some(Mount { point: unescape_mount(point), source: unescape_mount(source).to_string_lossy().into_owned() })
+        })
+        .collect()
+}
+
+/// mountinfo writes a space, tab, newline or backslash in a path as a
+/// three-digit octal escape (`\040`).
+fn unescape_mount(s: &str) -> PathBuf {
+    use std::os::unix::ffi::OsStringExt;
+    let b = s.as_bytes();
+    let mut out = Vec::with_capacity(b.len());
+    let mut i = 0;
+    while i < b.len() {
+        if b[i] == b'\\' && i + 3 < b.len() && b[i + 1..i + 4].iter().all(|c| (b'0'..=b'7').contains(c)) {
+            out.push((b[i + 1] - b'0') * 64 + (b[i + 2] - b'0') * 8 + (b[i + 3] - b'0'));
+            i += 4;
+        } else {
+            out.push(b[i]);
+            i += 1;
+        }
+    }
+    PathBuf::from(std::ffi::OsString::from_vec(out))
+}
+
+/// The disk a mount source lives on: the device itself, a partition's
+/// parent, and through device-mapper (LUKS, LVM) the disk underneath.
+/// `None` for anything that is not a block device — tmpfs, proc, a share.
+fn disk_of(source: &str) -> Option<String> {
+    if !source.starts_with("/dev/") {
+        return None;
+    }
+    let dev = std::fs::canonicalize(source).ok()?;
+    disk_of_block(dev.file_name()?.to_str()?, 0)
+}
+
+fn disk_of_block(name: &str, depth: u32) -> Option<String> {
+    let sys = Path::new("/sys/class/block").join(name);
+    if !sys.exists() {
+        return None;
+    }
+    // A mapped device (dm-N) names what it is built on under `slaves/`.
+    if depth < 8 {
+        let slave = std::fs::read_dir(sys.join("slaves")).ok().and_then(|mut rd| rd.next()).and_then(|e| e.ok());
+        if let Some(slave) = slave {
+            return disk_of_block(&slave.file_name().to_string_lossy(), depth + 1);
+        }
+    }
+    // A partition's sysfs node sits inside its disk's.
+    if sys.join("partition").exists() {
+        let real = std::fs::canonicalize(&sys).ok()?;
+        return Some(real.parent()?.file_name()?.to_string_lossy().into_owned());
+    }
+    Some(name.to_string())
+}
+
+/// The mount points a scan of `root` may enter: every one whose topmost
+/// mount is on the same disk as the mount `root` lies in.
+///
+/// Comparing devices alone stopped at every mount, and on a btrfs layout
+/// (`@` at `/`, `@home` at `/home`) each subvolume is its own device: a scan
+/// of `/` left out `/home`, most of the disk, and a separate `/home`
+/// partition went the same way. Comparing disks keeps those and still keeps
+/// out what the device check was for — `/proc`, `/sys`, tmpfs, network
+/// shares, a backup drive. Nested btrfs subvolumes that are not mounted
+/// (snapshots, container layers) have a device of their own and no entry
+/// here, so they stay out, and a snapshot is not counted as a second copy.
+fn crossable(mounts: &[Mount], root: &Path, disk_of: impl Fn(&str) -> Option<String>) -> HashSet<PathBuf> {
+    // The last mount at a point is the one on top, the one the walk sees.
+    let mut top: HashMap<&Path, &str> = HashMap::new();
+    for m in mounts {
+        top.insert(&m.point, &m.source);
+    }
+    let Some((home, home_src)) = top
+        .iter()
+        .filter(|(p, _)| root.starts_with(p))
+        .max_by_key(|(p, _)| p.components().count())
+    else {
+        return HashSet::new();
+    };
+    let Some(disk) = disk_of(home_src) else {
+        return HashSet::new();
+    };
+    let mut disks: HashMap<&str, Option<String>> = HashMap::new();
+    top.iter()
+        .filter(|(p, _)| p != &home)
+        .filter(|(_, src)| disks.entry(src).or_insert_with(|| disk_of(src)).as_deref() == Some(disk.as_str()))
+        .map(|(p, _)| p.to_path_buf())
+        .collect()
+}
+
 /// Walk `root`, returning its tree with directory sizes summed.
 ///
 /// Returns `None` when `root` is not a readable directory. `cancel` is polled
@@ -136,6 +257,9 @@ pub fn scan(
     if !meta.is_dir() {
         return None;
     }
+    // Mount points are canonical, so the walk's paths must be too.
+    let real = std::fs::canonicalize(root).ok()?;
+    let mountinfo = std::fs::read_to_string("/proc/self/mountinfo").unwrap_or_default();
 
     let name = root
         .file_name()
@@ -143,7 +267,8 @@ pub fn scan(
         .unwrap_or_else(|| root.to_string_lossy().into_owned());
 
     let mut walker = Walker {
-        dev: meta.dev(),
+        crossable: crossable(&parse_mountinfo(&mountinfo), &real, disk_of),
+        seen: HashSet::from([(meta.dev(), meta.ino())]),
         cancel,
         files: 0,
         dirs: 0,
@@ -151,7 +276,7 @@ pub fn scan(
         on_progress,
         last_report: Instant::now(),
     };
-    let tree = walker.walk(root, name, 0);
+    let tree = walker.walk(&real, name, 0, meta.dev());
 
     Some(ScanResult {
         tree,
@@ -249,6 +374,49 @@ mod tests {
         fs::remove_dir_all(&root).unwrap();
     }
 
+    #[test]
+    fn mountinfo_parses_past_optional_fields_and_escapes() {
+        let text = "\
+29 1 0:31 /@ / rw,relatime shared:1 - btrfs /dev/nvme0n1p2 rw,ssd
+40 29 0:50 /@home /home rw,relatime shared:2 master:7 - btrfs /dev/nvme0n1p2 rw
+41 29 259:1 / /mnt/my\\040drive rw - ext4 /dev/sdb1 rw
+";
+        let m = parse_mountinfo(text);
+        assert_eq!(m.len(), 3);
+        assert_eq!(m[1], Mount { point: "/home".into(), source: "/dev/nvme0n1p2".into() });
+        assert_eq!(m[2].point, PathBuf::from("/mnt/my drive"));
+    }
+
+    #[test]
+    fn mounts_on_the_scanned_disk_are_crossed_and_nothing_else() {
+        let mount = |p: &str, s: &str| Mount { point: p.into(), source: s.into() };
+        let mounts = vec![
+            mount("/", "/dev/nvme0n1p2"),
+            mount("/home", "/dev/nvme0n1p2"),        // btrfs subvolume
+            mount("/boot", "/dev/nvme0n1p1"),        // another partition, same disk
+            mount("/mnt/backup", "/dev/sdb1"),       // another disk
+            mount("/tmp", "tmpfs"),
+            mount("/proc", "proc"),
+            mount("/srv/share", "nas:/export"),
+            mount("/home/me/cache", "/dev/nvme0n1p2"),
+            mount("/home/me/cache", "tmpfs"),        // stacked on top: the walk sees this
+        ];
+        let disk = |src: &str| match src {
+            "/dev/nvme0n1p1" | "/dev/nvme0n1p2" => Some("nvme0n1".to_string()),
+            "/dev/sdb1" => Some("sdb".to_string()),
+            _ => None,
+        };
+
+        let got = crossable(&mounts, Path::new("/"), disk);
+        let want: HashSet<PathBuf> = ["/home", "/boot"].iter().map(PathBuf::from).collect();
+        assert_eq!(got, want);
+
+        // Scanning inside the backup drive crosses nothing on the system disk.
+        assert!(crossable(&mounts, Path::new("/mnt/backup/x"), disk).is_empty());
+        // Nor does a scan rooted on a tmpfs.
+        assert!(crossable(&mounts, Path::new("/tmp/x"), disk).is_empty());
+    }
+
     #[test]
     fn non_directory_root_is_rejected() {
         let root = scratch("file");