Skip to main content

detcore/syscalls/
sysinfo.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9use procfs::process::Process;
10use reverie::Error;
11use reverie::Guest;
12use reverie::syscalls;
13use reverie::syscalls::Errno;
14use reverie::syscalls::MemoryAccess;
15
16use crate::Detcore;
17use crate::RecordOrReplay;
18use crate::tool_global::thread_observe_time;
19use crate::tool_local::ResourceLimit;
20
21// Linux exposes USER_HZ, not the kernel's configurable scheduler HZ, through times(2).
22const CLOCK_TICKS_PER_SECOND: u64 = 100;
23const NANOS_PER_CLOCK_TICK: u64 = 1_000_000_000 / CLOCK_TICKS_PER_SECOND;
24
25fn clock_ticks(duration: crate::types::LogicalTime) -> u64 {
26    duration.as_nanos() / NANOS_PER_CLOCK_TICK
27}
28
29fn clock_t_from_ticks(ticks: u64) -> libc::clock_t {
30    ticks as libc::clock_t
31}
32
33const NANOS_PER_SECOND: u64 = 1_000_000_000;
34const NANOS_PER_MICROSECOND: u64 = 1_000;
35
36/// Render a logical CPU duration as the `timeval` `getrusage(2)` reports.
37///
38/// Linux truncates rusage CPU times to microsecond granularity, so the sub-microsecond
39/// remainder of the logical duration is discarded rather than rounded. Truncation (not
40/// rounding) is what keeps the value monotonic: a duration that grows by less than a
41/// microsecond must never make the reported total go backwards, and rounding-to-nearest
42/// on a shrinking remainder can do exactly that.
43fn timeval_from_logical(duration: crate::types::LogicalTime) -> libc::timeval {
44    let nanos = duration.as_nanos();
45    libc::timeval {
46        tv_sec: (nanos / NANOS_PER_SECOND) as libc::time_t,
47        tv_usec: ((nanos % NANOS_PER_SECOND) / NANOS_PER_MICROSECOND) as libc::suseconds_t,
48    }
49}
50
51fn logical_clock_ticks(
52    now: crate::types::LogicalTime,
53    boot: crate::types::LogicalTime,
54    uptime_offset_seconds: u64,
55) -> libc::clock_t {
56    let ticks = uptime_offset_seconds
57        .wrapping_mul(CLOCK_TICKS_PER_SECOND)
58        .wrapping_add(clock_ticks(now - boot));
59    clock_t_from_ticks(ticks)
60}
61
62/// Project elapsed logical time into Linux's integer `sysinfo(2)` uptime.
63///
64/// Linux rounds a positive fractional boottime up to the next second. Subtract
65/// the exact internal epoch first so its calendar fraction cannot affect the
66/// result. This projection does not alter the underlying logical clock.
67fn sysinfo_uptime_seconds(
68    now: crate::types::LogicalTime,
69    epoch: crate::types::LogicalTime,
70    uptime_offset_seconds: u64,
71) -> Result<u64, Error> {
72    let now_ns = now.as_nanos();
73    let epoch_ns = epoch.as_nanos();
74    let elapsed_ns = now_ns.checked_sub(epoch_ns).ok_or_else(|| {
75        Error::Tool(anyhow::anyhow!(
76            "sysinfo observed logical time {now_ns} ns before epoch {epoch_ns} ns"
77        ))
78    })?;
79    // Divide before rounding: adding NANOS_PER_SECOND - 1 to elapsed_ns can
80    // overflow. The rounded quotient of any u64 nanosecond duration fits u64.
81    let seconds =
82        elapsed_ns / NANOS_PER_SECOND + u64::from(!elapsed_ns.is_multiple_of(NANOS_PER_SECOND));
83    // Config accepts the full u64 offset; preserve its existing release-build
84    // wrapping extension, including the subsequent SysInfo -> c_long ABI cast.
85    // Ordinary Linux uptime semantics apply within the nonnegative c_long range.
86    Ok(uptime_offset_seconds.wrapping_add(seconds))
87}
88
89/// Render the whole-second part of `/proc/uptime` from an absolute logical clock.
90///
91/// Linux's `uptime_proc_show` prints `tv_sec` followed by truncated
92/// centiseconds, so the integer part is the floor of the elapsed time.
93/// Subtract before truncating. Truncating `now` and `boot` separately makes a
94/// sub-second run appear one second old whenever it crosses an absolute-second
95/// boundary, even though less than one logical second elapsed.
96fn procfs_uptime_seconds(
97    now: crate::types::LogicalTime,
98    boot: crate::types::LogicalTime,
99    uptime_offset_seconds: u64,
100) -> u64 {
101    uptime_offset_seconds + (now - boot).as_secs()
102}
103
104/// Render `/proc/stat`'s `btime` from the logical boot instant.
105///
106/// Linux's `show_stat` prints the seconds of `getboottime64`, a fixed instant
107/// that moves only when the realtime clock or time-namespace offset changes.
108/// Derive it from the boot instant itself, never as `floor(now) - uptime`: for
109/// a fractional boot those two floors round independently, so ordinary
110/// elapsed time would move `btime` back and forth by one second.
111///
112/// Config accepts every u64 offset, and an offset above `i64::MAX` can still
113/// place the boot instant inside time64_t. Subtract in i128, where every such
114/// difference is exact, and range-check only the result. `None` means the
115/// boot instant itself precedes `i64::MIN` seconds.
116fn procfs_boot_time_seconds(
117    boot: crate::types::LogicalTime,
118    uptime_offset_seconds: u64,
119) -> Option<i64> {
120    i64::try_from(i128::from(boot.as_secs()) - i128::from(uptime_offset_seconds)).ok()
121}
122
123fn prlimit_targets_current_process(
124    target_pid: i32,
125    deterministic_pid: Option<i32>,
126    physical_pid: i32,
127) -> bool {
128    target_pid == 0 || target_pid == deterministic_pid.unwrap_or(physical_pid)
129}
130
131fn validate_resource_limit_mutation(
132    resource: u32,
133    previous: ResourceLimit,
134    requested: ResourceLimit,
135) -> Result<(), Errno> {
136    if requested.current > requested.maximum {
137        return Err(Errno::EINVAL);
138    }
139    // Linux accepts an exact no-op for every valid resource, including limits
140    // that an unprivileged process could not otherwise change. Recognize that
141    // case before applying Detcore's narrower virtual-mutation policy.
142    if requested == previous {
143        return Ok(());
144    }
145    // CORE is virtual compatibility state too: changing it cannot enable host
146    // core dumps, while Linux sanitizers routinely lower its soft limit.
147    if resource != libc::RLIMIT_STACK
148        && resource != libc::RLIMIT_NOFILE
149        && resource != libc::RLIMIT_CORE
150    {
151        return Err(Errno::EPERM);
152    }
153    if requested.maximum > previous.maximum {
154        return Err(Errno::EPERM);
155    }
156    Ok(())
157}
158
159impl<T: RecordOrReplay> Detcore<T> {
160    // AUTONOMOUS-BOT-IMPLEMENTED
161    // TODO-HUMAN-REVIEW(#663)
162    /// Return one deterministic process resource limit through the legacy ABI.
163    pub async fn handle_getrlimit<G: Guest<Self>>(
164        &self,
165        guest: &mut G,
166        call: syscalls::Getrlimit,
167    ) -> Result<i64, Error> {
168        let resource = u32::try_from(call.resource()).map_err(|_| Errno::EINVAL)?;
169        let address = call.rlim().ok_or(Errno::EFAULT)?;
170        let limit = guest
171            .thread_state()
172            .resource_limits
173            .lock()
174            .expect("resource limits mutex poisoned")
175            .get(resource)
176            .ok_or(Errno::EINVAL)?;
177        let result = libc::rlimit {
178            rlim_cur: limit.current,
179            rlim_max: limit.maximum,
180        };
181        guest.memory().write_value(address, &result)?;
182        Ok(0)
183    }
184
185    // AUTONOMOUS-BOT-IMPLEMENTED
186    // TODO-HUMAN-REVIEW(#663)
187    /// Update one virtual process resource limit through the legacy ABI.
188    pub async fn handle_setrlimit<G: Guest<Self>>(
189        &self,
190        guest: &mut G,
191        call: syscalls::Setrlimit,
192    ) -> Result<i64, Error> {
193        let resource = u32::try_from(call.resource()).map_err(|_| Errno::EINVAL)?;
194        let address = call.rlim().ok_or(Errno::EFAULT)?;
195        let requested: libc::rlimit = guest.memory().read_value(address)?;
196        let requested = ResourceLimit {
197            current: requested.rlim_cur,
198            maximum: requested.rlim_max,
199        };
200        let resource_limits = guest.thread_state().resource_limits.clone();
201        let mut limits = resource_limits
202            .lock()
203            .expect("resource limits mutex poisoned");
204        let previous = limits.get(resource).ok_or(Errno::EINVAL)?;
205        validate_resource_limit_mutation(resource, previous, requested)?;
206        if requested != previous {
207            limits.set(resource, requested);
208        }
209        Ok(0)
210    }
211
212    /// Virtualize `prlimit64(2)` for the current guest process.
213    ///
214    /// Queries return process-local deterministic values. Exact no-op updates
215    /// succeed for every valid resource, as on Linux. Changes are kept virtual
216    /// and restricted to limits that do not grant access to host resources or
217    /// affect host scheduling. Accepted changes update only guest-observable
218    /// compatibility state; they are not a sandbox boundary and do not ask the
219    /// host kernel to enforce the virtual limit.
220    // AUTONOMOUS-BOT-IMPLEMENTED
221    // TODO-HUMAN-REVIEW(#534)
222    pub async fn handle_prlimit64<G: Guest<Self>>(
223        &self,
224        guest: &mut G,
225        call: syscalls::Prlimit64,
226    ) -> Result<i64, Error> {
227        let resource = call.resource();
228        let resource_limits = guest.thread_state().resource_limits.clone();
229        if resource_limits
230            .lock()
231            .expect("resource limits mutex poisoned")
232            .get(resource)
233            .is_none()
234        {
235            return Err(Errno::EINVAL.into());
236        }
237
238        let requested = if let Some(address) = call.new_rlim() {
239            let limit: libc::rlimit64 = guest.memory().read_value(address)?;
240            Some(ResourceLimit {
241                current: limit.rlim_cur,
242                maximum: limit.rlim_max,
243            })
244        } else {
245            None
246        };
247
248        let pid = call.pid();
249        let deterministic_pid = guest.thread_state().detpid.map(|detpid| detpid.as_raw());
250        if !prlimit_targets_current_process(pid, deterministic_pid, guest.pid().as_raw()) {
251            return Err(Errno::EPERM.into());
252        }
253
254        let previous = {
255            let mut limits = resource_limits
256                .lock()
257                .expect("resource limits mutex poisoned");
258            let previous = limits
259                .get(resource)
260                .expect("resource validity changed while handling prlimit64");
261
262            if let Some(requested) = requested {
263                validate_resource_limit_mutation(resource, previous, requested)?;
264                if requested != previous {
265                    limits.set(resource, requested);
266                }
267            }
268
269            previous
270        };
271
272        if let Some(address) = call.old_rlim() {
273            let previous = libc::rlimit64 {
274                rlim_cur: previous.current,
275                rlim_max: previous.maximum,
276            };
277            guest.memory().write_value(address, &previous)?;
278        }
279
280        crate::detlog!(
281            "prlimit64: pid={pid}, resource={resource}, mutation={}, old={}:{}",
282            requested.is_some(),
283            previous.current,
284            previous.maximum
285        );
286        Ok(0)
287    }
288    /// Return a deterministic resource-usage snapshot.
289    ///
290    /// `ru_utime`/`ru_stime` come from the SAME logical CPU accounting that backs `times(2)`
291    /// (see [`Self::handle_times`]), not from host scheduler counters. Reporting them as zero,
292    /// as this did previously, was both a fidelity bug and an internal contradiction: a guest
293    /// that called `times(2)` saw advancing CPU time while `getrusage(2)` insisted the same
294    /// process had consumed none. Deriving both from `ProcessCpuSnapshot` makes the two
295    /// syscalls agree by construction rather than by coincidence.
296    ///
297    /// The `who` values report different aggregates, matching Linux:
298    /// - `RUSAGE_SELF` — this process, summed across its threads.
299    /// - `RUSAGE_THREAD` — the calling thread alone. This reads the thread's own logical CPU
300    ///   counters rather than the process totals; substituting the process aggregate would
301    ///   over-report for every multithreaded guest.
302    /// - `RUSAGE_CHILDREN` — reaped children only, which is exactly what the `children_*`
303    ///   fields accumulate on `wait`.
304    ///
305    /// `ru_maxrss` is populated for the process/thread cases with the guest's peak resident set
306    /// size so that programs which require a positive maximum RSS (e.g. rr's `rusage` test)
307    /// behave like they do on Linux. This remains a best-effort host-procfs observation on
308    /// backends where [`Guest::pid`] names a host process; it is separate from the configured
309    /// system-wide memory reported by `sysinfo(2)` and virtual `/proc/meminfo`.
310    ///
311    /// Page-fault and context-switch counts remain zero: Detcore does not model them, and
312    /// synthesizing a plausible-looking number would be worse than reporting none.
313    pub async fn handle_getrusage<G: Guest<Self>>(
314        &self,
315        guest: &mut G,
316        call: syscalls::Getrusage,
317    ) -> Result<i64, Error> {
318        let who = call.who();
319        match who {
320            libc::RUSAGE_SELF | libc::RUSAGE_CHILDREN | libc::RUSAGE_THREAD => {}
321            _ => return Err(Errno::EINVAL.into()),
322        }
323
324        let usage_addr = call.usage().ok_or(Errno::EFAULT)?;
325
326        // SAFETY: `libc::rusage` is a plain-old-data C struct that is valid when zero-initialized.
327        let mut usage: libc::rusage = unsafe { std::mem::zeroed() };
328
329        let (user, system) = match who {
330            libc::RUSAGE_THREAD => guest.thread_state_mut().thread_cpu_time(),
331            libc::RUSAGE_CHILDREN => {
332                let cpu = guest.thread_state_mut().process_cpu_time();
333                (cpu.children_user, cpu.children_system)
334            }
335            // RUSAGE_SELF
336            _ => {
337                let cpu = guest.thread_state_mut().process_cpu_time();
338                (cpu.user, cpu.system)
339            }
340        };
341        usage.ru_utime = timeval_from_logical(user);
342        usage.ru_stime = timeval_from_logical(system);
343
344        // RUSAGE_SELF/RUSAGE_THREAD report this process's peak RSS. RUSAGE_CHILDREN aggregates
345        // terminated children only; with no such accounting we leave it zero, matching Linux when
346        // no child has exited.
347        if matches!(who, libc::RUSAGE_SELF | libc::RUSAGE_THREAD) {
348            usage.ru_maxrss = self.guest_peak_rss_kb(guest) as libc::c_long;
349        }
350
351        guest.memory().write_value(usage_addr, &usage)?;
352        Ok(0)
353    }
354
355    // AUTONOMOUS-BOT-IMPLEMENTED
356    // TODO-HUMAN-REVIEW(#797): Review logical elapsed and process CPU accounting semantics.
357    /// Return deterministic elapsed ticks and process CPU accounting for `times(2)`.
358    ///
359    /// Linux's host boot epoch and scheduler CPU counters are nondeterministic. Detcore instead
360    /// derives the return value from its global logical clock. Per-process logical CPU accounting
361    /// aggregates user instruction and syscall-system time across threads; forked processes start
362    /// fresh counters and contribute their totals to the parent's child counters when reaped.
363    pub async fn handle_times<G: Guest<Self>>(
364        &self,
365        guest: &mut G,
366        call: syscalls::Times,
367    ) -> Result<i64, Error> {
368        let now = thread_observe_time(guest).await;
369        let boot = crate::types::DetTime::new(&self.cfg).as_nanos();
370        let ticks = logical_clock_ticks(now, boot, self.cfg.sysinfo_uptime_offset);
371        let cpu = guest.thread_state_mut().process_cpu_time();
372
373        if let Some(address) = call.buf() {
374            let usage = libc::tms {
375                tms_utime: clock_t_from_ticks(clock_ticks(cpu.user)),
376                tms_stime: clock_t_from_ticks(clock_ticks(cpu.system)),
377                tms_cutime: clock_t_from_ticks(clock_ticks(cpu.children_user)),
378                tms_cstime: clock_t_from_ticks(clock_ticks(cpu.children_system)),
379            };
380            guest.memory().write_value(address, &usage)?;
381        }
382
383        Ok(ticks as i64)
384    }
385
386    /// The guest's peak resident set size ("high water mark") in kibibytes, matching the units of
387    /// Linux `getrusage`'s `ru_maxrss`. This reads host procfs through [`Guest::pid`], which only
388    /// identifies the guest process on backends where it names a host process; always returns a
389    /// positive value so guests can rely on a nonzero maximum RSS even if the read fails.
390    fn guest_peak_rss_kb<G: Guest<Self>>(&self, guest: &G) -> u64 {
391        Process::new(guest.pid().as_raw())
392            .and_then(|process| process.status())
393            .ok()
394            .and_then(|status| status.vmhwm.or(status.vmrss))
395            .unwrap_or(0)
396            .max(1)
397    }
398
399    /// handle sysinfo syscall
400    pub async fn handle_sysinfo<G: Guest<Self>>(
401        &self,
402        guest: &mut G,
403        call: syscalls::Sysinfo,
404    ) -> Result<i64, Error> {
405        let info_addr = call.info().ok_or(Errno::EFAULT)?;
406        let sys_info = self.collect_sysinfo(guest).await?;
407        let mut memory = guest.memory();
408
409        memory.write_value(info_addr, &sys_info.into())?;
410        Ok(0)
411    }
412
413    /// Whole-second `/proc/uptime` value (floor of elapsed logical time).
414    pub(super) async fn calculate_procfs_uptime<G: Guest<Self>>(
415        &self,
416        guest: &mut G,
417    ) -> Result<u64, Error> {
418        let global_time = thread_observe_time(guest).await;
419        Ok(procfs_uptime_seconds(
420            global_time,
421            crate::types::DetTime::new(&self.cfg).as_nanos(),
422            self.cfg.sysinfo_uptime_offset,
423        ))
424    }
425
426    /// Whole-second `/proc/stat` `btime`, fixed for the life of the run.
427    ///
428    /// `EOVERFLOW` only when the boot instant precedes `i64::MIN` seconds.
429    /// Only snapshots that render `btime` may ask, so that refusal stays
430    /// local to `/proc/stat`.
431    pub(super) fn calculate_procfs_boot_time(&self) -> Result<i64, Error> {
432        procfs_boot_time_seconds(
433            crate::types::DetTime::new(&self.cfg).as_nanos(),
434            self.cfg.sysinfo_uptime_offset,
435        )
436        .ok_or_else(|| Errno::EOVERFLOW.into())
437    }
438
439    async fn collect_sysinfo<G: Guest<Self>>(
440        &self,
441        guest: &mut G,
442    ) -> Result<syscalls::SysInfo, Error> {
443        let memory = configured_memory(self.cfg.memory);
444        let now = thread_observe_time(guest).await;
445        let epoch = crate::types::DetTime::new(&self.cfg).as_nanos();
446        Ok(syscalls::SysInfo {
447            uptime: sysinfo_uptime_seconds(now, epoch, self.cfg.sysinfo_uptime_offset)?,
448            loads_1: 1,
449            loads_5: 1,
450            loads_15: 1,
451            total_ram: memory.total_ram,
452            free_ram: memory.free_ram,
453            buffer_ram: memory.buffer_ram,
454            shared_ram: memory.shared_ram,
455            total_swap: memory.total_swap,
456            free_swap: memory.free_swap,
457            procs: 1,
458            total_high: memory.total_high,
459            free_high: memory.free_high,
460            mem_unit: memory.mem_unit,
461        })
462    }
463}
464
465// AUTONOMOUS-BOT-IMPLEMENTED
466// TODO-HUMAN-REVIEW(PR-2979): Deterministic free-memory accounting for sysinfo(2).
467#[derive(Debug, PartialEq, Eq)]
468struct ConfiguredMemory {
469    total_ram: u64,
470    free_ram: u64,
471    buffer_ram: u64,
472    shared_ram: u64,
473    total_swap: u64,
474    free_swap: u64,
475    total_high: u64,
476    free_high: u64,
477    mem_unit: u32,
478}
479
480/// Report the configured guest memory consistently with virtual `/proc/meminfo`.
481///
482/// Linux `sysinfo(2)` describes system-wide memory, not one process's virtual
483/// mappings. Detcore does not model allocation pressure within its configured
484/// memory limit, so all configured memory remains available and the other
485/// modeled memory categories remain empty.
486fn configured_memory(memory: u64) -> ConfiguredMemory {
487    ConfiguredMemory {
488        total_ram: memory,
489        free_ram: memory,
490        buffer_ram: 0,
491        shared_ram: 0,
492        total_swap: 0,
493        free_swap: 0,
494        total_high: 0,
495        free_high: 0,
496        mem_unit: 1,
497    }
498}
499
500#[cfg(test)]
501mod tests {
502    use super::*;
503    use crate::types::LogicalTime;
504
505    #[test]
506    fn logical_clock_ticks_include_boot_offset_and_fractional_seconds() {
507        let boot = LogicalTime::from_secs(1_000);
508        let now = boot + LogicalTime::from_millis(25);
509
510        assert_eq!(logical_clock_ticks(now, boot, 120), 12_002);
511    }
512
513    #[test]
514    fn sysinfo_uptime_rounds_positive_elapsed_up() {
515        let epoch = LogicalTime::from_nanos(1_000_000_000_000);
516        for (elapsed_ns, expected_zero, expected_offset) in [
517            (0, 0, 120),
518            (1, 1, 121),
519            (999_999_999, 1, 121),
520            (1_000_000_000, 1, 121),
521            (1_000_000_001, 2, 122),
522            (1_200_000_000, 2, 122),
523        ] {
524            let now = epoch + LogicalTime::from_nanos(elapsed_ns);
525            assert_eq!(
526                sysinfo_uptime_seconds(now, epoch, 0).unwrap(),
527                expected_zero,
528                "elapsed {elapsed_ns} ns without boot offset"
529            );
530            assert_eq!(
531                sysinfo_uptime_seconds(now, epoch, 120).unwrap(),
532                expected_offset,
533                "elapsed {elapsed_ns} ns with boot offset"
534            );
535        }
536    }
537
538    #[test]
539    fn sysinfo_uptime_ignores_epoch_fraction() {
540        for fraction_ns in [0, 1, 1_000, 999_999_000, 999_999_999] {
541            let epoch = LogicalTime::from_nanos(1_000_000_000_000 + fraction_ns);
542            for (elapsed_ns, expected) in [
543                (0, 120),
544                (1, 121),
545                (999_999_999, 121),
546                (1_000_000_000, 121),
547                (1_000_000_001, 122),
548                (1_200_000_000, 122),
549            ] {
550                let now = epoch + LogicalTime::from_nanos(elapsed_ns);
551                assert_eq!(
552                    sysinfo_uptime_seconds(now, epoch, 120).unwrap(),
553                    expected,
554                    "epoch fraction {fraction_ns} ns, elapsed {elapsed_ns} ns"
555                );
556            }
557        }
558    }
559
560    #[test]
561    fn sysinfo_uptime_rounds_maximum_duration_without_overflow() {
562        let epoch = LogicalTime::from_nanos(0);
563        assert_eq!(
564            sysinfo_uptime_seconds(LogicalTime::MAX, epoch, 0).unwrap(),
565            18_446_744_074
566        );
567        assert_eq!(
568            sysinfo_uptime_seconds(
569                LogicalTime::from_nanos(18_446_744_073_000_000_000),
570                epoch,
571                120,
572            )
573            .unwrap(),
574            18_446_744_193
575        );
576        // An absolute timestamp at the representation limit need not have a
577        // large elapsed duration; subtraction must precede the projection.
578        assert_eq!(
579            sysinfo_uptime_seconds(LogicalTime::MAX, LogicalTime::from_nanos(u64::MAX - 1), 120,)
580                .unwrap(),
581            121
582        );
583    }
584
585    #[test]
586    fn sysinfo_uptime_preserves_wrapping_offset_extension() {
587        let epoch = LogicalTime::from_nanos(0);
588        for (elapsed_ns, offset, expected) in [
589            (0, u64::MAX, u64::MAX),
590            (1, u64::MAX, 0),
591            (1_000_000_000, u64::MAX, 0),
592            (1_000_000_001, u64::MAX, 1),
593            (1, u64::MAX - 1, u64::MAX),
594            (1_000_000_001, u64::MAX - 1, 0),
595        ] {
596            assert_eq!(
597                sysinfo_uptime_seconds(LogicalTime::from_nanos(elapsed_ns), epoch, offset).unwrap(),
598                expected,
599                "elapsed {elapsed_ns} ns, offset {offset}"
600            );
601        }
602    }
603
604    #[test]
605    fn sysinfo_uptime_preserves_signed_abi_conversion() {
606        let epoch = LogicalTime::from_nanos(0);
607        for (elapsed_ns, offset, expected) in [
608            (0, i64::MAX as u64, i64::MAX),
609            (1, i64::MAX as u64, i64::MIN),
610            (0, u64::MAX, -1),
611            (1, u64::MAX, 0),
612        ] {
613            let uptime =
614                sysinfo_uptime_seconds(LogicalTime::from_nanos(elapsed_ns), epoch, offset).unwrap();
615            let info: libc::sysinfo = syscalls::SysInfo {
616                uptime,
617                loads_1: 0,
618                loads_5: 0,
619                loads_15: 0,
620                total_ram: 0,
621                free_ram: 0,
622                shared_ram: 0,
623                buffer_ram: 0,
624                total_swap: 0,
625                free_swap: 0,
626                procs: 0,
627                total_high: 0,
628                free_high: 0,
629                mem_unit: 1,
630            }
631            .into();
632            assert_eq!(info.uptime, expected);
633        }
634    }
635
636    #[test]
637    fn sysinfo_uptime_rejects_time_before_epoch() {
638        let error = sysinfo_uptime_seconds(
639            LogicalTime::from_nanos(999),
640            LogicalTime::from_nanos(1_000),
641            120,
642        )
643        .unwrap_err();
644        let Error::Tool(error) = error else {
645            panic!("an impossible clock must be a tool failure, got {error:?}");
646        };
647        assert_eq!(
648            error.to_string(),
649            "sysinfo observed logical time 999 ns before epoch 1000 ns"
650        );
651    }
652
653    #[test]
654    fn procfs_uptime_subtracts_fractional_boot_before_truncating() {
655        let boot = LogicalTime::from_nanos(1_000_999_999_999);
656
657        assert_eq!(
658            procfs_uptime_seconds(boot + LogicalTime::from_nanos(1), boot, 120),
659            120
660        );
661        assert_eq!(
662            procfs_uptime_seconds(boot + LogicalTime::from_secs(1), boot, 120),
663            121
664        );
665    }
666
667    #[test]
668    fn procfs_boot_time_is_the_boot_instant_not_now_minus_uptime() {
669        // The reviewed reproducer: boot 1000.75s, offset 120s. Linux reports
670        // the seconds of the fixed boot instant, 880, for every sample.
671        let boot = LogicalTime::from_nanos(1_000_750_000_000);
672        assert_eq!(procfs_boot_time_seconds(boot, 120), Some(880));
673
674        // `floor(now) - uptime` is what the fix replaced: at these samples it
675        // yields 880, 881, 880. Pin that the old derivation really did move,
676        // so this test keeps meaning something if the helpers change.
677        let old_btime =
678            |now: LogicalTime| now.as_secs() as i64 - procfs_uptime_seconds(now, boot, 120) as i64;
679        let samples = [100, 400, 1_100].map(|millis| boot + LogicalTime::from_millis(millis));
680        assert_eq!(samples.map(old_btime), [880, 881, 880]);
681
682        // For an integral boot the two derivations agree, so whole-second
683        // epochs render exactly what they rendered before.
684        let integral_boot = LogicalTime::from_secs(1_000);
685        let integral_now = integral_boot + LogicalTime::from_millis(1_400);
686        assert_eq!(
687            procfs_boot_time_seconds(integral_boot, 120),
688            Some(
689                integral_now.as_secs() as i64
690                    - procfs_uptime_seconds(integral_now, integral_boot, 120) as i64
691            )
692        );
693
694        assert_eq!(procfs_boot_time_seconds(boot, u64::MAX), None);
695    }
696
697    #[test]
698    fn procfs_boot_time_is_exact_for_every_offset_whose_result_fits() {
699        // Config accepts every u64 offset. 2026-01-01T00:00:00Z minus 2^63 s
700        // is -9223372035087550208, which time64_t represents even though the
701        // offset alone exceeds i64::MAX.
702        let boot = LogicalTime::from_secs(1_767_225_600);
703        assert_eq!(procfs_boot_time_seconds(boot, 0), Some(1_767_225_600));
704        assert_eq!(
705            procfs_boot_time_seconds(boot, 1 << 63),
706            Some(-9_223_372_035_087_550_208)
707        );
708        // The result, not the offset, bounds the domain: exactly i64::MIN is
709        // representable and one second earlier is not.
710        assert_eq!(
711            procfs_boot_time_seconds(boot, 1_767_225_600 + (1 << 63)),
712            Some(i64::MIN)
713        );
714        assert_eq!(
715            procfs_boot_time_seconds(boot, 1_767_225_600 + (1 << 63) + 1),
716            None
717        );
718        assert_eq!(procfs_boot_time_seconds(boot, u64::MAX), None);
719    }
720
721    #[test]
722    fn sysinfo_uptime_rounds_fractional_elapsed_up_like_linux() {
723        // A boot instant with a fractional absolute second: rounding must see
724        // only the elapsed time, never the absolute boundary crossing.
725        let boot = LogicalTime::from_nanos(1_000_999_999_999);
726
727        assert_eq!(sysinfo_uptime_seconds(boot, boot, 120).unwrap(), 120);
728        assert_eq!(
729            sysinfo_uptime_seconds(boot + LogicalTime::from_nanos(1), boot, 120).unwrap(),
730            121
731        );
732        assert_eq!(
733            sysinfo_uptime_seconds(boot + LogicalTime::from_millis(999), boot, 120).unwrap(),
734            121
735        );
736        assert_eq!(
737            sysinfo_uptime_seconds(boot + LogicalTime::from_secs(1), boot, 120).unwrap(),
738            121
739        );
740        assert_eq!(
741            sysinfo_uptime_seconds(
742                boot + LogicalTime::from_secs(1) + LogicalTime::from_nanos(1),
743                boot,
744                120
745            )
746            .unwrap(),
747            122
748        );
749        // The floor-based procfs value differs exactly on fractional elapsed.
750        let fractional = boot + LogicalTime::from_millis(1_500);
751        assert_eq!(procfs_uptime_seconds(fractional, boot, 120), 121);
752        assert_eq!(sysinfo_uptime_seconds(fractional, boot, 120).unwrap(), 122);
753    }
754
755    #[test]
756    fn sysinfo_memory_matches_configured_memory() {
757        assert_eq!(
758            configured_memory(1_000_000_000),
759            ConfiguredMemory {
760                total_ram: 1_000_000_000,
761                free_ram: 1_000_000_000,
762                buffer_ram: 0,
763                shared_ram: 0,
764                total_swap: 0,
765                free_swap: 0,
766                total_high: 0,
767                free_high: 0,
768                mem_unit: 1,
769            },
770        );
771    }
772
773    #[test]
774    fn prlimit_self_target_prefers_deterministic_process_identity() {
775        assert!(prlimit_targets_current_process(3, Some(3), 10_003));
776        assert!(prlimit_targets_current_process(0, Some(3), 10_003));
777        assert!(!prlimit_targets_current_process(10_003, Some(3), 10_003));
778        assert!(!prlimit_targets_current_process(4, Some(3), 10_003));
779    }
780
781    #[test]
782    fn prlimit_self_target_falls_back_to_physical_identity_before_init() {
783        assert!(prlimit_targets_current_process(10_003, None, 10_003));
784        assert!(!prlimit_targets_current_process(3, None, 10_003));
785    }
786
787    #[test]
788    fn prlimit_accepts_exact_noop_for_restricted_resource() {
789        let limit = ResourceLimit {
790            current: 0,
791            maximum: 0,
792        };
793        assert_eq!(
794            validate_resource_limit_mutation(libc::RLIMIT_CPU, limit, limit),
795            Ok(())
796        );
797    }
798
799    #[test]
800    fn prlimit_accepts_core_soft_limit_change() {
801        let previous = ResourceLimit {
802            current: 1,
803            maximum: 1,
804        };
805        let requested = ResourceLimit {
806            current: 0,
807            maximum: 1,
808        };
809        assert_eq!(
810            validate_resource_limit_mutation(libc::RLIMIT_CORE, previous, requested),
811            Ok(())
812        );
813    }
814
815    #[test]
816    fn prlimit_rejects_actual_change_to_restricted_resource() {
817        let previous = ResourceLimit {
818            current: 1,
819            maximum: 1,
820        };
821        let requested = ResourceLimit {
822            current: 0,
823            maximum: 1,
824        };
825        assert_eq!(
826            validate_resource_limit_mutation(libc::RLIMIT_CPU, previous, requested),
827            Err(Errno::EPERM)
828        );
829    }
830
831    #[test]
832    fn prlimit_rejects_invalid_soft_limit_before_noop_policy() {
833        let previous = ResourceLimit {
834            current: 1,
835            maximum: 1,
836        };
837        let requested = ResourceLimit {
838            current: 2,
839            maximum: 1,
840        };
841        assert_eq!(
842            validate_resource_limit_mutation(libc::RLIMIT_CORE, previous, requested),
843            Err(Errno::EINVAL)
844        );
845    }
846
847    #[test]
848    fn prlimit_rejects_core_hard_limit_raise() {
849        let previous = ResourceLimit {
850            current: 1,
851            maximum: 1,
852        };
853        let requested = ResourceLimit {
854            current: 1,
855            maximum: 2,
856        };
857        assert_eq!(
858            validate_resource_limit_mutation(libc::RLIMIT_CORE, previous, requested),
859            Err(Errno::EPERM)
860        );
861    }
862
863    #[test]
864    fn logical_cpu_ticks_exclude_boot_epoch() {
865        assert_eq!(clock_ticks(LogicalTime::from_millis(25)), 2);
866    }
867
868    #[test]
869    fn rusage_timeval_splits_seconds_and_microseconds() {
870        let tv = timeval_from_logical(LogicalTime::from_millis(2_500));
871        assert_eq!(tv.tv_sec, 2);
872        assert_eq!(tv.tv_usec, 500_000);
873    }
874
875    #[test]
876    fn rusage_timeval_truncates_sub_microsecond_rather_than_rounding() {
877        // 1_999 ns is a hair under 2us. Truncating yields 1us; rounding to nearest would
878        // yield 2us and could make a later, larger duration report a SMALLER value once its
879        // remainder shrank -- i.e. CPU time going backwards. Pin truncation explicitly.
880        let tv = timeval_from_logical(LogicalTime::from_nanos(1_999));
881        assert_eq!(tv.tv_sec, 0);
882        assert_eq!(tv.tv_usec, 1);
883    }
884
885    #[test]
886    fn rusage_timeval_is_monotonic_in_the_logical_duration() {
887        // The property that matters to a guest: CPU time never goes backwards. Walk a range
888        // of nanosecond durations across microsecond and second boundaries and assert the
889        // rendered timeval is non-decreasing at every step.
890        let mut previous = (0_i64, 0_i64);
891        for nanos in (0..3_000_000u64).step_by(997) {
892            let tv = timeval_from_logical(LogicalTime::from_nanos(nanos));
893            let current = (tv.tv_sec, tv.tv_usec);
894            assert!(
895                current >= previous,
896                "rusage timeval went backwards at {nanos}ns: {previous:?} -> {current:?}"
897            );
898            previous = current;
899        }
900    }
901
902    #[test]
903    fn rusage_zero_cpu_time_renders_as_zero() {
904        let tv = timeval_from_logical(LogicalTime::ZERO);
905        assert_eq!(tv.tv_sec, 0);
906        assert_eq!(tv.tv_usec, 0);
907    }
908
909    #[test]
910    fn rusage_and_times_agree_within_one_clock_tick() {
911        // Both syscalls project the same logical duration, but times(2) is
912        // quantized to USER_HZ while getrusage(2) retains microseconds.
913        for nanos in [0u64, 1_000_000, 300_484_000, 7_000_000_000, 12_345_678_901] {
914            let duration = LogicalTime::from_nanos(nanos);
915            let tv = timeval_from_logical(duration);
916            let rusage_micros = tv.tv_sec as u64 * 1_000_000 + tv.tv_usec as u64;
917            let times_micros = clock_ticks(duration) * (NANOS_PER_CLOCK_TICK / 1_000);
918
919            assert!(rusage_micros >= times_micros);
920            assert!(rusage_micros - times_micros < NANOS_PER_CLOCK_TICK / 1_000);
921
922            // A tick-ALIGNED duration must agree EXACTLY, not merely to within
923            // one tick. The bounds above are satisfied at every sample by an
924            // implementation carrying a constant sub-tick offset, so without
925            // this the suite cannot distinguish that from a correct one.
926            if nanos % NANOS_PER_CLOCK_TICK == 0 {
927                assert_eq!(rusage_micros, times_micros);
928            }
929        }
930    }
931
932    #[test]
933    fn logical_clock_ticks_wrap_configured_offset_like_linux_clock_t() {
934        let boot = LogicalTime::from_secs(1_000);
935        let before = logical_clock_ticks(boot, boot, u64::MAX);
936        let after = logical_clock_ticks(boot + LogicalTime::from_millis(10), boot, u64::MAX);
937
938        assert_eq!(before, -100);
939        assert_eq!(after, -99);
940    }
941}