Skip to main content

datui_lib/
locality.rs

1//! Where a dataset physically lives, and what that implies about opening it.
2//!
3//! The home screen's job is to let someone decide what to open. Size and row count
4//! answer "what is this"; they say nothing about "what will happen when I press
5//! Enter". A 2 GB file on tmpfs and a 2 GB file on a hotel-wifi NFS mount are the
6//! same row and a thousandfold different experience.
7//!
8//! Everything here is derived from `/proc/self/mountinfo`, which is a local read of a
9//! kernel-generated file: it cannot block on the filesystem it describes, which is
10//! the whole reason it is safe to consult about a share that has stopped answering.
11
12use std::path::Path;
13use std::sync::{Arc, Mutex};
14use std::time::{Duration, Instant};
15
16/// Filesystems whose reads cross a network.
17///
18/// The list is about *behaviour*, not about protocol families: what these have in
19/// common is that a read can stall for as long as the far end is unreachable, and on
20/// a `hard` NFS mount that stall is uninterruptible.
21pub const NETWORK_FILESYSTEMS: &[&str] = &[
22    "nfs",
23    "nfs4",
24    "cifs",
25    "smb3",
26    "smbfs",
27    "afs",
28    "9p",
29    "ceph",
30    "glusterfs",
31    "fuse.sshfs",
32    "fuse.rclone",
33    "fuse.s3fs",
34    "fuse.davfs",
35    "davfs",
36    "ftpfs",
37    // An automount point that has not been triggered yet blocks on first access,
38    // which is exactly what the marker is warning about. Once it triggers, the real
39    // filesystem shadows it in the mount table and is judged on its own merits.
40    "autofs",
41];
42
43/// Filesystems backed by RAM. Reads from these are free, which is worth saying when
44/// every other row on screen is not.
45const MEMORY_FILESYSTEMS: &[&str] = &["tmpfs", "ramfs", "devtmpfs"];
46
47/// How a dataset is reached, in the terms that predict what opening it costs.
48#[derive(Debug, Clone, PartialEq, Eq)]
49pub enum Locality {
50    /// A disk on this machine.
51    Local,
52    /// RAM. Reading is as fast as it gets.
53    Memory,
54    /// Reads cross a network and can stall.
55    Network,
56    /// An object store, reached by URL rather than by path.
57    Object,
58    /// Nothing in the mount table covered it.
59    Unknown,
60}
61
62impl Locality {
63    /// The locality a [`Source::fstype`] implies.
64    ///
65    /// The round trip exists because a row carries the filesystem name and not the
66    /// classification: `fstype` is what gets cached, and re-deriving it here is free,
67    /// whereas asking the mount table again means reading `/proc` on the thread that
68    /// draws.
69    ///
70    /// Object-store schemes are handled first. They never appear in the mount table, so
71    /// a classifier that only knew filesystems would call `s3` a local disk.
72    pub fn of_fstype(fstype: &str) -> Locality {
73        match fstype {
74            "s3" | "s3a" | "gs" | "gcs" | "az" | "cloud" | "http" | "https" => Locality::Object,
75            "" | "unknown" => Locality::Unknown,
76            other => classify(other),
77        }
78    }
79}
80
81/// Where something lives and what the kernel calls it.
82#[derive(Debug, Clone, PartialEq, Eq)]
83pub struct Source {
84    /// The filesystem type, or the URL scheme for an object store: `nfs4`, `ext4`,
85    /// `fuse.sshfs`, `tmpfs`, `s3`, `gs`.
86    pub fstype: String,
87    pub locality: Locality,
88}
89
90impl Source {
91    /// A short label for the interface. The filesystem's own name is the most
92    /// informative thing available and costs one word: `nfs4` and `fuse.sshfs` fail
93    /// in different ways, and neither behaves like `tmpfs`.
94    pub fn label(&self) -> &str {
95        &self.fstype
96    }
97
98    /// Whether this is worth flagging on a row. Local disk is the unremarkable case
99    /// and saying so on every row would be noise; everything else changes what
100    /// pressing Enter means.
101    pub fn notable(&self) -> bool {
102        !matches!(self.locality, Locality::Local)
103    }
104
105    pub fn network(&self) -> bool {
106        self.locality == Locality::Network
107    }
108
109    /// Classify a filesystem name on its own, for a source recorded earlier rather
110    /// than resolved from a path just now.
111    pub fn from_fstype(fstype: &str) -> Self {
112        if let Some(scheme) = ["s3", "gs", "http", "https", "az", "hdfs"]
113            .into_iter()
114            .find(|s| *s == fstype)
115        {
116            return Self {
117                fstype: scheme.to_string(),
118                locality: Locality::Object,
119            };
120        }
121        Self {
122            locality: classify(fstype),
123            fstype: fstype.to_string(),
124        }
125    }
126}
127
128/// How long a read of the mount table is reused by [`Mounts::cached`].
129///
130/// Mounts appear and disappear on a human timescale — someone plugs in a share, an
131/// automount triggers — and every caller of this module treats what it says as a
132/// hint, never as a gate. Half a second is below the point anyone notices a new
133/// mount arriving and far above the burst of per-row questions one listing asks.
134const MOUNTS_TTL: Duration = Duration::from_millis(500);
135
136/// The last read of the mount table, and when it was taken.
137static CACHED_MOUNTS: Mutex<Option<(Instant, Arc<Mounts>)>> = Mutex::new(None);
138
139/// The mount table, parsed once.
140///
141/// Resolving a path against this is string work, so a whole listing can be described
142/// from a single read rather than re-reading `/proc/self/mountinfo` per row.
143#[derive(Debug, Clone, Default)]
144pub struct Mounts {
145    /// (mount point, filesystem type), in the order the kernel listed them.
146    entries: Vec<(String, String)>,
147}
148
149impl Mounts {
150    /// Read the current mount table. An unreadable one yields an empty table, which
151    /// reports everything as unknown rather than as wrong.
152    pub fn current() -> Self {
153        std::fs::read_to_string("/proc/self/mountinfo")
154            .map(|s| Self::parse(&s))
155            .unwrap_or_default()
156    }
157
158    /// The mount table as of at most [`MOUNTS_TTL`] ago, shared between callers.
159    ///
160    /// For the callers that ask per row. `/proc/self/mountinfo` is generated by the
161    /// kernel on each open, so reading it is the expensive part of an answer — four
162    /// fifths of it — and a listing of five thousand rows asked the same unchanged
163    /// question five thousand times, once per row and again on every frame that drew
164    /// them. Sharing one read across the burst is what makes a per-row question
165    /// affordable.
166    pub fn cached() -> Arc<Self> {
167        let now = Instant::now();
168        // Read under the lock rather than around it: two threads arriving on a cold
169        // cache would otherwise both read `/proc`, and the loser's read would replace
170        // a table just as good as its own.
171        let mut slot = CACHED_MOUNTS
172            .lock()
173            .unwrap_or_else(std::sync::PoisonError::into_inner);
174        if let Some((read_at, mounts)) = slot.as_ref()
175            && now.duration_since(*read_at) < MOUNTS_TTL
176        {
177            return Arc::clone(mounts);
178        }
179        let mounts = Arc::new(Self::current());
180        *slot = Some((now, Arc::clone(&mounts)));
181        mounts
182    }
183
184    pub fn parse(mountinfo: &str) -> Self {
185        let mut entries = Vec::new();
186        for line in mountinfo.lines() {
187            // Fields before the separator end with the mount point at index 4; the
188            // filesystem type is the first field after it.
189            let Some((before, after)) = line.split_once(" - ") else {
190                continue;
191            };
192            let Some(point) = before.split_whitespace().nth(4) else {
193                continue;
194            };
195            let Some(fstype) = after.split_whitespace().next() else {
196                continue;
197            };
198            entries.push((point.to_string(), fstype.to_string()));
199        }
200        Self { entries }
201    }
202
203    /// The filesystem covering `path`.
204    ///
205    /// The deepest mount wins, and among mounts at the same point the *last* one
206    /// wins: mountinfo lists them in mount order, so a later entry shadows an earlier
207    /// one. An NFS share automounted at a path appears after the autofs entry
208    /// covering the same path, and it is the NFS entry that describes what a read
209    /// will actually do.
210    pub fn fstype_for(&self, path: &Path) -> Option<&str> {
211        // Mount points are absolute, so a relative path matches nothing and would
212        // report "unknown" for a file sitting on the disk under the caller's feet.
213        // Joining the working directory is pure string work -- unlike canonicalising,
214        // which touches the filesystem and is exactly what must not happen here.
215        let joined;
216        // `has_root`, not `is_absolute`. On Windows a path is absolute only with a
217        // drive or UNC prefix, so `/mnt/nas/data` is "relative" there -- and
218        // joining the working directory onto it turns an already-rooted path into
219        // nonsense that matches no mount at all.
220        let path = if path.has_root() {
221            path
222        } else {
223            match std::env::current_dir() {
224                Ok(cwd) => {
225                    joined = cwd.join(path);
226                    &joined
227                }
228                Err(_) => path,
229            }
230        };
231
232        let mut best: Option<(usize, &str)> = None;
233        for (point, fstype) in &self.entries {
234            if !path.starts_with(point) {
235                continue;
236            }
237            let len = point.len();
238            if best.is_none_or(|(n, _)| len >= n) {
239                best = Some((len, fstype));
240            }
241        }
242        best.map(|(_, f)| f)
243    }
244
245    /// How `path` is reached.
246    pub fn describe(&self, path: &Path) -> Source {
247        if let Some(scheme) = object_scheme(path) {
248            return Source {
249                fstype: scheme,
250                locality: Locality::Object,
251            };
252        }
253        match self.fstype_for(path) {
254            Some(fstype) => Source {
255                locality: classify(fstype),
256                fstype: fstype.to_string(),
257            },
258            None => Source {
259                fstype: "unknown".to_string(),
260                locality: Locality::Unknown,
261            },
262        }
263    }
264
265    pub fn is_network(&self, path: &Path) -> bool {
266        self.describe(path).network()
267    }
268}
269
270fn classify(fstype: &str) -> Locality {
271    if NETWORK_FILESYSTEMS.contains(&fstype) {
272        Locality::Network
273    } else if MEMORY_FILESYSTEMS.contains(&fstype) {
274        Locality::Memory
275    } else {
276        Locality::Local
277    }
278}
279
280/// The URL scheme of an object-store path, if it is one.
281///
282/// These never appear in the mount table and are never walked or measured: they are
283/// listed from history so that what you opened before is still findable.
284pub fn object_scheme(path: &Path) -> Option<String> {
285    match crate::source::input_source(path) {
286        crate::source::InputSource::Local(_) => None,
287        crate::source::InputSource::S3(_) => Some("s3".to_string()),
288        crate::source::InputSource::Gcs(_) => Some("gs".to_string()),
289        crate::source::InputSource::Azure(_) => Some("az".to_string()),
290        crate::source::InputSource::Http(_) => Some("http".to_string()),
291    }
292}
293
294#[cfg(test)]
295mod locality_of_fstype_tests {
296    use super::*;
297
298    #[test]
299    fn object_store_schemes_are_not_local_disks() {
300        // The case the round trip exists for. These never appear in the mount table, so
301        // a classifier that only knew filesystems would call every one of them a disk
302        // on this machine -- and the row would then claim a cloud object is local.
303        for scheme in ["s3", "s3a", "gs", "gcs", "http", "https"] {
304            assert_eq!(
305                Locality::of_fstype(scheme),
306                Locality::Object,
307                "{scheme} should be an object store"
308            );
309        }
310    }
311
312    #[test]
313    fn network_filesystems_are_network() {
314        for fstype in ["nfs", "nfs4", "cifs", "smb3"] {
315            assert_eq!(
316                Locality::of_fstype(fstype),
317                Locality::Network,
318                "{fstype} should be network"
319            );
320        }
321    }
322
323    #[test]
324    fn memory_filesystems_are_memory() {
325        assert_eq!(Locality::of_fstype("tmpfs"), Locality::Memory);
326    }
327
328    #[test]
329    fn ordinary_filesystems_are_local() {
330        for fstype in ["ext4", "btrfs", "xfs", "apfs", "ntfs"] {
331            assert_eq!(
332                Locality::of_fstype(fstype),
333                Locality::Local,
334                "{fstype} should be local"
335            );
336        }
337    }
338
339    #[test]
340    fn nothing_known_is_not_guessed_at() {
341        assert_eq!(Locality::of_fstype(""), Locality::Unknown);
342        assert_eq!(Locality::of_fstype("unknown"), Locality::Unknown);
343    }
344
345    /// What `describe` reports and what `of_fstype` makes of it have to agree, or a row
346    /// is classified one way for the detail pane and another for its marker.
347    #[test]
348    fn it_agrees_with_describe() {
349        let mounts = Mounts::parse("");
350        for path in ["s3://bucket/key.parquet", "gs://bucket/key.parquet"] {
351            let source = mounts.describe(std::path::Path::new(path));
352            assert_eq!(source.locality, Locality::of_fstype(&source.fstype));
353        }
354    }
355}