file manager
git clone https://git.lucas.co/cce-files.git
fix(scan): enter mounts on the same disk, so a scan of / includes /home
The walk skipped every directory on another device than the scan
root's. On a btrfs layout each mounted subvolume is its own device: @
at / is 0:31, @home at /home 0:50, so a scan of / left out /home, most
of the disk, along with /var/log and the pacman cache. A separate /home
partition was dropped the same way.
The walk now enters a mount point when its topmost mount is on the same
physical disk as the mount the root lies in. Each source in
/proc/self/mountinfo is resolved through /sys/class/block, past a
partition to its disk and through device-mapper (LUKS, LVM) to what it
is built on. /proc, /sys, tmpfs, network shares and other drives have no
disk or another one and stay out, which is what the device check was
for. Unmounted nested subvolumes (snapshots, container layers) have a
device of their own and no mount, and stay out too, so a snapshot is not
counted as a second copy.
Crossing mounts makes a bind mount reachable, so every directory is now
entered once by (device, inode): a second sighting is skipped, counted
once, and cannot loop the walk. The root is canonicalised so its paths
match the mount table's.
On this machine a scan of / now crosses /home, /boot, /var/log and
/var/cache/pacman/pkg, and nothing else.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
src/services/scan.rs | 192 +++++++++++++++++++++++++++++++++++++++++++++++----
1 file changed, 180 insertions(+), 12 deletions(-)
diff --git a/src/services/scan.rs b/src/services/scan.rs
index 21ab615..e8c2d5e 100644
--- a/src/services/scan.rs
+++ b/src/services/scan.rs
@@ -6,7 +6,8 @@
//! on the `FsService` thread like every other request; the app sees it only as
//! progress messages followed by a completed tree.
-use std::path::Path;
+use std::collections::{HashMap, HashSet};
+use std::path::{Path, PathBuf};
use std::os::unix::fs::MetadataExt;
use std::sync::atomic::{AtomicBool, Ordering};
use std::time::{Duration, Instant};
@@ -48,10 +49,15 @@ pub struct ScanResult {
}
struct Walker<'a> {
- /// Device of the scan root. Entries on any other device are skipped, so
- /// scanning `/` does not wander into `/proc`, `/sys`, or a mounted backup
- /// drive — and cannot loop through a bind mount pointing back inside.
- dev: u64,
+ /// Mount points the walk may descend into although they sit on another
+ /// device than their parent — see [`crossable`]. Any other change of
+ /// device is skipped, so scanning `/` does not wander into `/proc`, `/sys`,
+ /// a tmpfs, or a backup drive.
+ crossable: HashSet<PathBuf>,
+ /// Every directory entered, by (device, inode). Directories have no hard
+ /// links, so a second sighting is a bind mount: skipped, it is counted once
+ /// and cannot loop the walk back into itself.
+ seen: HashSet<(u64, u64)>,
cancel: &'a AtomicBool,
files: u64,
dirs: u64,
@@ -61,7 +67,8 @@ struct Walker<'a> {
}
impl Walker<'_> {
- fn walk(&mut self, dir: &Path, name: String, depth: u32) -> TreeNode {
+ /// `dev` is the device `dir` lies on; a child on any other is a mount.
+ fn walk(&mut self, dir: &Path, name: String, depth: u32, dev: u64) -> TreeNode {
let mut node = TreeNode { name, size: 0, is_dir: true, children: Vec::new() };
if depth >= MAX_DEPTH || self.cancel.load(Ordering::Relaxed) {
return node;
@@ -81,17 +88,24 @@ impl Walker<'_> {
// subtree's total, and cannot loop the walk back into itself.
let Ok(meta) = entry.metadata() else { continue };
let ft = meta.file_type();
- if ft.is_symlink() || meta.dev() != self.dev {
+ if ft.is_symlink() {
continue;
}
let child_name = entry.file_name().to_string_lossy().into_owned();
if ft.is_dir() {
+ let path = entry.path();
+ if meta.dev() != dev && !self.crossable.contains(&path) {
+ continue;
+ }
+ if !self.seen.insert((meta.dev(), meta.ino())) {
+ continue;
+ }
self.dirs += 1;
- let child = self.walk(&entry.path(), child_name, depth + 1);
+ let child = self.walk(&path, child_name, depth + 1, meta.dev());
node.size += child.size;
node.children.push(child);
- } else if ft.is_file() {
+ } else if ft.is_file() && meta.dev() == dev {
let size = meta.len();
self.files += 1;
self.bytes += size;
@@ -104,7 +118,7 @@ impl Walker<'_> {
});
}
// Sockets, fifos, and device nodes occupy no meaningful space and
- // are dropped entirely.
+ // are dropped entirely, as is a file bind-mounted from elsewhere.
self.maybe_report();
}
@@ -120,6 +134,113 @@ impl Walker<'_> {
}
}
+/// One line of `/proc/self/mountinfo`, as much as the walk needs.
+#[derive(Debug, Clone, PartialEq)]
+struct Mount {
+ point: PathBuf,
+ /// What is mounted: a device path for a disk filesystem, a bare word
+ /// (`tmpfs`, `proc`) or `host:/path` for anything else.
+ source: String,
+}
+
+/// Parse mountinfo: `id parent maj:min root POINT opts [optional...] - fstype SOURCE superopts`.
+/// The optional fields vary in number, so the source is found after the ` - `.
+fn parse_mountinfo(text: &str) -> Vec<Mount> {
+ text.lines()
+ .filter_map(|line| {
+ let (left, right) = line.split_once(" - ")?;
+ let point = left.split(' ').nth(4)?;
+ let source = right.split(' ').nth(1)?;
+ Some(Mount { point: unescape_mount(point), source: unescape_mount(source).to_string_lossy().into_owned() })
+ })
+ .collect()
+}
+
+/// mountinfo writes a space, tab, newline or backslash in a path as a
+/// three-digit octal escape (`\040`).
+fn unescape_mount(s: &str) -> PathBuf {
+ use std::os::unix::ffi::OsStringExt;
+ let b = s.as_bytes();
+ let mut out = Vec::with_capacity(b.len());
+ let mut i = 0;
+ while i < b.len() {
+ if b[i] == b'\\' && i + 3 < b.len() && b[i + 1..i + 4].iter().all(|c| (b'0'..=b'7').contains(c)) {
+ out.push((b[i + 1] - b'0') * 64 + (b[i + 2] - b'0') * 8 + (b[i + 3] - b'0'));
+ i += 4;
+ } else {
+ out.push(b[i]);
+ i += 1;
+ }
+ }
+ PathBuf::from(std::ffi::OsString::from_vec(out))
+}
+
+/// The disk a mount source lives on: the device itself, a partition's
+/// parent, and through device-mapper (LUKS, LVM) the disk underneath.
+/// `None` for anything that is not a block device — tmpfs, proc, a share.
+fn disk_of(source: &str) -> Option<String> {
+ if !source.starts_with("/dev/") {
+ return None;
+ }
+ let dev = std::fs::canonicalize(source).ok()?;
+ disk_of_block(dev.file_name()?.to_str()?, 0)
+}
+
+fn disk_of_block(name: &str, depth: u32) -> Option<String> {
+ let sys = Path::new("/sys/class/block").join(name);
+ if !sys.exists() {
+ return None;
+ }
+ // A mapped device (dm-N) names what it is built on under `slaves/`.
+ if depth < 8 {
+ let slave = std::fs::read_dir(sys.join("slaves")).ok().and_then(|mut rd| rd.next()).and_then(|e| e.ok());
+ if let Some(slave) = slave {
+ return disk_of_block(&slave.file_name().to_string_lossy(), depth + 1);
+ }
+ }
+ // A partition's sysfs node sits inside its disk's.
+ if sys.join("partition").exists() {
+ let real = std::fs::canonicalize(&sys).ok()?;
+ return Some(real.parent()?.file_name()?.to_string_lossy().into_owned());
+ }
+ Some(name.to_string())
+}
+
+/// The mount points a scan of `root` may enter: every one whose topmost
+/// mount is on the same disk as the mount `root` lies in.
+///
+/// Comparing devices alone stopped at every mount, and on a btrfs layout
+/// (`@` at `/`, `@home` at `/home`) each subvolume is its own device: a scan
+/// of `/` left out `/home`, most of the disk, and a separate `/home`
+/// partition went the same way. Comparing disks keeps those and still keeps
+/// out what the device check was for — `/proc`, `/sys`, tmpfs, network
+/// shares, a backup drive. Nested btrfs subvolumes that are not mounted
+/// (snapshots, container layers) have a device of their own and no entry
+/// here, so they stay out, and a snapshot is not counted as a second copy.
+fn crossable(mounts: &[Mount], root: &Path, disk_of: impl Fn(&str) -> Option<String>) -> HashSet<PathBuf> {
+ // The last mount at a point is the one on top, the one the walk sees.
+ let mut top: HashMap<&Path, &str> = HashMap::new();
+ for m in mounts {
+ top.insert(&m.point, &m.source);
+ }
+ let Some((home, home_src)) = top
+ .iter()
+ .filter(|(p, _)| root.starts_with(p))
+ .max_by_key(|(p, _)| p.components().count())
+ else {
+ return HashSet::new();
+ };
+ let Some(disk) = disk_of(home_src) else {
+ return HashSet::new();
+ };
+ let mut disks: HashMap<&str, Option<String>> = HashMap::new();
+ top.iter()
+ .filter(|(p, _)| p != &home)
+ .filter(|(_, src)| disks.entry(src).or_insert_with(|| disk_of(src)).as_deref() == Some(disk.as_str()))
+ .map(|(p, _)| p.to_path_buf())
+ .collect()
+}
+
/// Walk `root`, returning its tree with directory sizes summed.
///
/// Returns `None` when `root` is not a readable directory. `cancel` is polled
@@ -136,6 +257,9 @@ pub fn scan(
if !meta.is_dir() {
return None;
}
+ // Mount points are canonical, so the walk's paths must be too.
+ let real = std::fs::canonicalize(root).ok()?;
+ let mountinfo = std::fs::read_to_string("/proc/self/mountinfo").unwrap_or_default();
let name = root
.file_name()
@@ -143,7 +267,8 @@ pub fn scan(
.unwrap_or_else(|| root.to_string_lossy().into_owned());
let mut walker = Walker {
- dev: meta.dev(),
+ crossable: crossable(&parse_mountinfo(&mountinfo), &real, disk_of),
+ seen: HashSet::from([(meta.dev(), meta.ino())]),
cancel,
files: 0,
dirs: 0,
@@ -151,7 +276,7 @@ pub fn scan(
on_progress,
last_report: Instant::now(),
};
- let tree = walker.walk(root, name, 0);
+ let tree = walker.walk(&real, name, 0, meta.dev());
Some(ScanResult {
tree,
@@ -249,6 +374,49 @@ mod tests {
fs::remove_dir_all(&root).unwrap();
}
+ #[test]
+ fn mountinfo_parses_past_optional_fields_and_escapes() {
+ let text = "\
+29 1 0:31 /@ / rw,relatime shared:1 - btrfs /dev/nvme0n1p2 rw,ssd
+40 29 0:50 /@home /home rw,relatime shared:2 master:7 - btrfs /dev/nvme0n1p2 rw
+41 29 259:1 / /mnt/my\\040drive rw - ext4 /dev/sdb1 rw
+";
+ let m = parse_mountinfo(text);
+ assert_eq!(m.len(), 3);
+ assert_eq!(m[1], Mount { point: "/home".into(), source: "/dev/nvme0n1p2".into() });
+ assert_eq!(m[2].point, PathBuf::from("/mnt/my drive"));
+ }
+
+ #[test]
+ fn mounts_on_the_scanned_disk_are_crossed_and_nothing_else() {
+ let mount = |p: &str, s: &str| Mount { point: p.into(), source: s.into() };
+ let mounts = vec![
+ mount("/", "/dev/nvme0n1p2"),
+ mount("/home", "/dev/nvme0n1p2"), // btrfs subvolume
+ mount("/boot", "/dev/nvme0n1p1"), // another partition, same disk
+ mount("/mnt/backup", "/dev/sdb1"), // another disk
+ mount("/tmp", "tmpfs"),
+ mount("/proc", "proc"),
+ mount("/srv/share", "nas:/export"),
+ mount("/home/me/cache", "/dev/nvme0n1p2"),
+ mount("/home/me/cache", "tmpfs"), // stacked on top: the walk sees this
+ ];
+ let disk = |src: &str| match src {
+ "/dev/nvme0n1p1" | "/dev/nvme0n1p2" => Some("nvme0n1".to_string()),
+ "/dev/sdb1" => Some("sdb".to_string()),
+ _ => None,
+ };
+
+ let got = crossable(&mounts, Path::new("/"), disk);
+ let want: HashSet<PathBuf> = ["/home", "/boot"].iter().map(PathBuf::from).collect();
+ assert_eq!(got, want);
+
+ // Scanning inside the backup drive crosses nothing on the system disk.
+ assert!(crossable(&mounts, Path::new("/mnt/backup/x"), disk).is_empty());
+ // Nor does a scan rooted on a tmpfs.
+ assert!(crossable(&mounts, Path::new("/tmp/x"), disk).is_empty());
+ }
+
#[test]
fn non_directory_root_is_rejected() {
let root = scratch("file");