1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
//! Where a dataset physically lives, and what that implies about opening it.
//!
//! The home screen's job is to let someone decide what to open. Size and row count
//! answer "what is this"; they say nothing about "what will happen when I press
//! Enter". A 2 GB file on tmpfs and a 2 GB file on a hotel-wifi NFS mount are the
//! same row and a thousandfold different experience.
//!
//! Everything here is derived from `/proc/self/mountinfo`, which is a local read of a
//! kernel-generated file: it cannot block on the filesystem it describes, which is
//! the whole reason it is safe to consult about a share that has stopped answering.
use std::path::Path;
use std::sync::{Arc, Mutex};
use std::time::{Duration, Instant};
/// Filesystems whose reads cross a network.
///
/// The list is about *behaviour*, not about protocol families: what these have in
/// common is that a read can stall for as long as the far end is unreachable, and on
/// a `hard` NFS mount that stall is uninterruptible.
pub const NETWORK_FILESYSTEMS: &[&str] = &[
"nfs",
"nfs4",
"cifs",
"smb3",
"smbfs",
"afs",
"9p",
"ceph",
"glusterfs",
"fuse.sshfs",
"fuse.rclone",
"fuse.s3fs",
"fuse.davfs",
"davfs",
"ftpfs",
// An automount point that has not been triggered yet blocks on first access,
// which is exactly what the marker is warning about. Once it triggers, the real
// filesystem shadows it in the mount table and is judged on its own merits.
"autofs",
];
/// Filesystems backed by RAM. Reads from these are free, which is worth saying when
/// every other row on screen is not.
const MEMORY_FILESYSTEMS: &[&str] = &["tmpfs", "ramfs", "devtmpfs"];
/// How a dataset is reached, in the terms that predict what opening it costs.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Locality {
/// A disk on this machine.
Local,
/// RAM. Reading is as fast as it gets.
Memory,
/// Reads cross a network and can stall.
Network,
/// An object store, reached by URL rather than by path.
Object,
/// Nothing in the mount table covered it.
Unknown,
}
impl Locality {
/// The locality a [`Source::fstype`] implies.
///
/// The round trip exists because a row carries the filesystem name and not the
/// classification: `fstype` is what gets cached, and re-deriving it here is free,
/// whereas asking the mount table again means reading `/proc` on the thread that
/// draws.
///
/// Object-store schemes are handled first. They never appear in the mount table, so
/// a classifier that only knew filesystems would call `s3` a local disk.
pub fn of_fstype(fstype: &str) -> Locality {
match fstype {
"s3" | "s3a" | "gs" | "gcs" | "az" | "cloud" | "http" | "https" => Locality::Object,
"" | "unknown" => Locality::Unknown,
other => classify(other),
}
}
}
/// Where something lives and what the kernel calls it.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Source {
/// The filesystem type, or the URL scheme for an object store: `nfs4`, `ext4`,
/// `fuse.sshfs`, `tmpfs`, `s3`, `gs`.
pub fstype: String,
pub locality: Locality,
}
impl Source {
/// A short label for the interface. The filesystem's own name is the most
/// informative thing available and costs one word: `nfs4` and `fuse.sshfs` fail
/// in different ways, and neither behaves like `tmpfs`.
pub fn label(&self) -> &str {
&self.fstype
}
/// Whether this is worth flagging on a row. Local disk is the unremarkable case
/// and saying so on every row would be noise; everything else changes what
/// pressing Enter means.
pub fn notable(&self) -> bool {
!matches!(self.locality, Locality::Local)
}
pub fn network(&self) -> bool {
self.locality == Locality::Network
}
/// Classify a filesystem name on its own, for a source recorded earlier rather
/// than resolved from a path just now.
pub fn from_fstype(fstype: &str) -> Self {
if let Some(scheme) = ["s3", "gs", "http", "https", "az", "hdfs"]
.into_iter()
.find(|s| *s == fstype)
{
return Self {
fstype: scheme.to_string(),
locality: Locality::Object,
};
}
Self {
locality: classify(fstype),
fstype: fstype.to_string(),
}
}
}
/// How long a read of the mount table is reused by [`Mounts::cached`].
///
/// Mounts appear and disappear on a human timescale — someone plugs in a share, an
/// automount triggers — and every caller of this module treats what it says as a
/// hint, never as a gate. Half a second is below the point anyone notices a new
/// mount arriving and far above the burst of per-row questions one listing asks.
const MOUNTS_TTL: Duration = Duration::from_millis(500);
/// The last read of the mount table, and when it was taken.
static CACHED_MOUNTS: Mutex<Option<(Instant, Arc<Mounts>)>> = Mutex::new(None);
/// The mount table, parsed once.
///
/// Resolving a path against this is string work, so a whole listing can be described
/// from a single read rather than re-reading `/proc/self/mountinfo` per row.
#[derive(Debug, Clone, Default)]
pub struct Mounts {
/// (mount point, filesystem type), in the order the kernel listed them.
entries: Vec<(String, String)>,
}
impl Mounts {
/// Read the current mount table. An unreadable one yields an empty table, which
/// reports everything as unknown rather than as wrong.
pub fn current() -> Self {
std::fs::read_to_string("/proc/self/mountinfo")
.map(|s| Self::parse(&s))
.unwrap_or_default()
}
/// The mount table as of at most [`MOUNTS_TTL`] ago, shared between callers.
///
/// For the callers that ask per row. `/proc/self/mountinfo` is generated by the
/// kernel on each open, so reading it is the expensive part of an answer — four
/// fifths of it — and a listing of five thousand rows asked the same unchanged
/// question five thousand times, once per row and again on every frame that drew
/// them. Sharing one read across the burst is what makes a per-row question
/// affordable.
pub fn cached() -> Arc<Self> {
let now = Instant::now();
// Read under the lock rather than around it: two threads arriving on a cold
// cache would otherwise both read `/proc`, and the loser's read would replace
// a table just as good as its own.
let mut slot = CACHED_MOUNTS
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
if let Some((read_at, mounts)) = slot.as_ref()
&& now.duration_since(*read_at) < MOUNTS_TTL
{
return Arc::clone(mounts);
}
let mounts = Arc::new(Self::current());
*slot = Some((now, Arc::clone(&mounts)));
mounts
}
pub fn parse(mountinfo: &str) -> Self {
let mut entries = Vec::new();
for line in mountinfo.lines() {
// Fields before the separator end with the mount point at index 4; the
// filesystem type is the first field after it.
let Some((before, after)) = line.split_once(" - ") else {
continue;
};
let Some(point) = before.split_whitespace().nth(4) else {
continue;
};
let Some(fstype) = after.split_whitespace().next() else {
continue;
};
entries.push((point.to_string(), fstype.to_string()));
}
Self { entries }
}
/// The filesystem covering `path`.
///
/// The deepest mount wins, and among mounts at the same point the *last* one
/// wins: mountinfo lists them in mount order, so a later entry shadows an earlier
/// one. An NFS share automounted at a path appears after the autofs entry
/// covering the same path, and it is the NFS entry that describes what a read
/// will actually do.
pub fn fstype_for(&self, path: &Path) -> Option<&str> {
// Mount points are absolute, so a relative path matches nothing and would
// report "unknown" for a file sitting on the disk under the caller's feet.
// Joining the working directory is pure string work -- unlike canonicalising,
// which touches the filesystem and is exactly what must not happen here.
let joined;
// `has_root`, not `is_absolute`. On Windows a path is absolute only with a
// drive or UNC prefix, so `/mnt/nas/data` is "relative" there -- and
// joining the working directory onto it turns an already-rooted path into
// nonsense that matches no mount at all.
let path = if path.has_root() {
path
} else {
match std::env::current_dir() {
Ok(cwd) => {
joined = cwd.join(path);
&joined
}
Err(_) => path,
}
};
let mut best: Option<(usize, &str)> = None;
for (point, fstype) in &self.entries {
if !path.starts_with(point) {
continue;
}
let len = point.len();
if best.is_none_or(|(n, _)| len >= n) {
best = Some((len, fstype));
}
}
best.map(|(_, f)| f)
}
/// How `path` is reached.
pub fn describe(&self, path: &Path) -> Source {
if let Some(scheme) = object_scheme(path) {
return Source {
fstype: scheme,
locality: Locality::Object,
};
}
match self.fstype_for(path) {
Some(fstype) => Source {
locality: classify(fstype),
fstype: fstype.to_string(),
},
None => Source {
fstype: "unknown".to_string(),
locality: Locality::Unknown,
},
}
}
pub fn is_network(&self, path: &Path) -> bool {
self.describe(path).network()
}
}
fn classify(fstype: &str) -> Locality {
if NETWORK_FILESYSTEMS.contains(&fstype) {
Locality::Network
} else if MEMORY_FILESYSTEMS.contains(&fstype) {
Locality::Memory
} else {
Locality::Local
}
}
/// The URL scheme of an object-store path, if it is one.
///
/// These never appear in the mount table and are never walked or measured: they are
/// listed from history so that what you opened before is still findable.
pub fn object_scheme(path: &Path) -> Option<String> {
match crate::source::input_source(path) {
crate::source::InputSource::Local(_) => None,
crate::source::InputSource::S3(_) => Some("s3".to_string()),
crate::source::InputSource::Gcs(_) => Some("gs".to_string()),
crate::source::InputSource::Azure(_) => Some("az".to_string()),
crate::source::InputSource::Http(_) => Some("http".to_string()),
}
}
#[cfg(test)]
mod locality_of_fstype_tests {
use super::*;
#[test]
fn object_store_schemes_are_not_local_disks() {
// The case the round trip exists for. These never appear in the mount table, so
// a classifier that only knew filesystems would call every one of them a disk
// on this machine -- and the row would then claim a cloud object is local.
for scheme in ["s3", "s3a", "gs", "gcs", "http", "https"] {
assert_eq!(
Locality::of_fstype(scheme),
Locality::Object,
"{scheme} should be an object store"
);
}
}
#[test]
fn network_filesystems_are_network() {
for fstype in ["nfs", "nfs4", "cifs", "smb3"] {
assert_eq!(
Locality::of_fstype(fstype),
Locality::Network,
"{fstype} should be network"
);
}
}
#[test]
fn memory_filesystems_are_memory() {
assert_eq!(Locality::of_fstype("tmpfs"), Locality::Memory);
}
#[test]
fn ordinary_filesystems_are_local() {
for fstype in ["ext4", "btrfs", "xfs", "apfs", "ntfs"] {
assert_eq!(
Locality::of_fstype(fstype),
Locality::Local,
"{fstype} should be local"
);
}
}
#[test]
fn nothing_known_is_not_guessed_at() {
assert_eq!(Locality::of_fstype(""), Locality::Unknown);
assert_eq!(Locality::of_fstype("unknown"), Locality::Unknown);
}
/// What `describe` reports and what `of_fstype` makes of it have to agree, or a row
/// is classified one way for the detail pane and another for its marker.
#[test]
fn it_agrees_with_describe() {
let mounts = Mounts::parse("");
for path in ["s3://bucket/key.parquet", "gs://bucket/key.parquet"] {
let source = mounts.describe(std::path::Path::new(path));
assert_eq!(source.locality, Locality::of_fstype(&source.fstype));
}
}
}