saluki_env/workload/helpers/
cgroups.rs

1#![allow(dead_code)]
2
3use std::{
4    collections::HashMap,
5    fs::{self, OpenOptions},
6    io::{self, BufRead as _, BufReader},
7    os::unix::fs::MetadataExt as _,
8    path::{Path, PathBuf},
9    sync::LazyLock,
10};
11
12use regex::Regex;
13use saluki_error::{generic_error, ErrorContext as _, GenericError};
14use stringtheory::{
15    interning::{GenericMapInterner, Interner as _},
16    MetaString,
17};
18use tracing::{debug, error, trace, warn};
19
20use crate::features::{Feature, FeatureDetector};
21
22const DEFAULT_PROCFS_ROOT: &str = "/proc";
23const DEFAULT_CGROUPFS_ROOT: &str = "/sys/fs/cgroup";
24const DEFAULT_LEGACY_CGROUPFS_ROOT: &str = "/cgroup";
25const DEFAULT_HOST_MAPPED_PROCFS_ROOT: &str = "/host/proc";
26const DEFAULT_HOST_MAPPED_CGROUPFS_ROOT: &str = "/host/sys/fs/cgroup";
27const CGROUPS_V1_BASE_CONTROLLER_NAME: &str = "memory";
28const CGROUPS_V2_CONTROLLERS_FILE: &str = "cgroup.controllers";
29const SELF_CGROUP_PATH: &str = "/proc/self/cgroup";
30const SELF_CGROUPFS_PATH: &str = "/sys/fs/cgroup";
31
32/// Highest inode number that can't refer to a specific cgroup controller.
33///
34/// Inodes 0 and 1 are never valid, and inode 2 is conventionally the root of a filesystem.
35const MAX_RESERVED_INODE: u64 = 2;
36
37/// Linux Control Groups-specific configuration.
38///
39/// Provides environment-specific paths to both "procfs" and "cgroupfs" filesystems, necessary for querying the Linux
40/// Control Groups v2 unified hierarchy.
41pub struct CgroupsConfiguration {
42    procfs_root: PathBuf,
43    cgroupfs_root: PathBuf,
44}
45
46impl CgroupsConfiguration {
47    /// Creates a new `CgroupsConfiguration` from the given filesystem roots.
48    ///
49    /// If a root is given, that path is used. Otherwise, each root falls back to its host-mapped default when its own
50    /// filesystem is detected as host-mapped, and to its local default when it isn't. The cgroupfs root has one more
51    /// fallback: the legacy `/cgroup` root, when that layout is detected.
52    pub fn new(
53        procfs_root: Option<PathBuf>, cgroupfs_root: Option<PathBuf>, feature_detector: &FeatureDetector,
54    ) -> Self {
55        let procfs_root = procfs_root.unwrap_or_else(|| {
56            if feature_detector.is_feature_available(Feature::HostMappedProcfs) {
57                PathBuf::from(DEFAULT_HOST_MAPPED_PROCFS_ROOT)
58            } else {
59                PathBuf::from(DEFAULT_PROCFS_ROOT)
60            }
61        });
62
63        let cgroupfs_root = cgroupfs_root.unwrap_or_else(|| {
64            // Detected separately from procfs: the two are independent mounts, and a deployment can map one
65            // without the other. Keying this off the procfs feature would point us at a cgroupfs path that isn't
66            // there, or make us miss the host's hierarchy in favor of our own container's.
67            if feature_detector.is_feature_available(Feature::HostMappedCgroupfs) {
68                PathBuf::from(DEFAULT_HOST_MAPPED_CGROUPFS_ROOT)
69            } else if feature_detector.is_feature_available(Feature::LegacyCgroupfsRoot) {
70                // Older Amazon Linux hosts put the cgroups v1 hierarchy at `/cgroup`. The Datadog Agent detects that
71                // layout and defaults `container_cgroup_root` to it, but it sends us the result as a default value,
72                // which we can't tell apart from a schema default, so we detect the layout ourselves.
73                PathBuf::from(DEFAULT_LEGACY_CGROUPFS_ROOT)
74            } else {
75                PathBuf::from(DEFAULT_CGROUPFS_ROOT)
76            }
77        });
78
79        Self {
80            procfs_root,
81            cgroupfs_root,
82        }
83    }
84
85    /// Returns the path to the "procfs" filesystem.
86    pub fn procfs_path(&self) -> &Path {
87        self.procfs_root.as_path()
88    }
89
90    /// Returns the path to the "cgroupfs" filesystem.
91    pub fn cgroupfs_path(&self) -> &Path {
92        self.cgroupfs_root.as_path()
93    }
94}
95
96/// Reader for querying control groups being used for containerization.
97///
98/// This reader is capable of querying both cgroups v1 and v2 hierarchies, and can be used to find cgroups -- either
99/// within the entire hierarchy, or for a specific process ID -- that are mapped specifically to containers. A simple
100/// naming heuristic is used to both identify and extract container IDs from cgroup names.
101#[derive(Clone)]
102pub struct CgroupsReader {
103    procfs_path: PathBuf,
104    hierarchy_reader: HierarchyReader,
105    interner: GenericMapInterner,
106}
107
108impl CgroupsReader {
109    /// Creates a new `CgroupsReader` from the given configuration and interner.
110    ///
111    /// If either a valid cgroups v1 or v2 hierarchy is found, `Ok(Some)` is returned with the reader. Otherwise,
112    /// `Ok(None)` is returned.
113    ///
114    /// The provided interner will be used exclusively for handling container IDs.
115    ///
116    /// # Errors
117    ///
118    /// If there is an I/O error while attempting to query the current cgroups hierarchy, an error will be returned.
119    pub fn try_from_config(
120        config: &CgroupsConfiguration, interner: GenericMapInterner,
121    ) -> Result<Option<Self>, GenericError> {
122        let hierarchy_reader = HierarchyReader::try_from_config(config)?;
123        Ok(hierarchy_reader.map(|hierarchy_reader| Self {
124            procfs_path: config.procfs_path().to_path_buf(),
125            hierarchy_reader,
126            interner,
127        }))
128    }
129
130    /// Returns whether the hierarchy being read is the cgroups v2 unified hierarchy.
131    ///
132    /// This distinguishes the layout, not just the version: under the unified hierarchy every cgroup lives in one
133    /// filesystem rooted at the cgroupfs mount, whereas cgroups v1 mounts a separate filesystem per controller
134    /// underneath it. Anything comparing inodes across the hierarchy depends on the former, since inode numbers only
135    /// identify a file within a single filesystem.
136    pub fn is_unified(&self) -> bool {
137        self.hierarchy_reader.is_unified()
138    }
139
140    fn try_cgroup_from_path(&self, cgroup_path: &Path) -> Option<Cgroup> {
141        let container_id = extract_container_id_from_path(cgroup_path, &self.interner)?;
142
143        let metadata = match cgroup_path.metadata() {
144            Ok(metadata) => metadata,
145            Err(e) => {
146                trace!(error = %e, cgroup_controller_path = %cgroup_path.display(), "Failed to get metadata for possible cgroup controller path.");
147                return None;
148            }
149        };
150
151        // A reserved inode can't be attributed to this controller specifically, so we have nothing usable to key an
152        // alias on. Drop the cgroup rather than reporting it with an inode that would resolve the wrong workload -- or
153        // none at all.
154        let controller_inode = metadata.ino();
155        if !is_usable_controller_inode(controller_inode) {
156            debug!(
157                controller_inode,
158                %container_id,
159                cgroup_controller_path = %cgroup_path.display(),
160                "Ignoring cgroup controller with reserved inode.",
161            );
162            return None;
163        }
164
165        trace!(
166            controller_inode,
167            %container_id,
168            cgroup_controller_path = %cgroup_path.display(),
169            "Found valid cgroups controller for container.",
170        );
171
172        Some(Cgroup {
173            ino: Some(controller_inode),
174            container_id,
175        })
176    }
177
178    /// Gets a cgroup for the given process ID.
179    ///
180    /// This method will attempt to find the cgroup for the given process ID by looking at the `/proc/<pid>/cgroup`
181    /// file. If the process ID doesn't exist or isn't attached to a cgroup, `None` will be returned.
182    pub fn get_cgroup_by_pid(&self, pid: u32) -> Option<Cgroup> {
183        // See if the given process ID exists in the proc filesystem _and_ if there's a cgroup path for it.
184        let proc_pid_cgroup_path = self.procfs_path.join(pid.to_string()).join("cgroup");
185        let lines = match read_lines(&proc_pid_cgroup_path) {
186            Ok(lines) => lines,
187            Err(e) => match e.kind() {
188                io::ErrorKind::NotFound => {
189                    debug!(pid, cgroup_lookup_path = %proc_pid_cgroup_path.display(), "Process does not exist or is not attached to a cgroup.");
190                    return None;
191                }
192                _ => {
193                    debug!(error = %e, pid, cgroup_lookup_path = %proc_pid_cgroup_path.display(), "Failed to read cgroup file for process.");
194                    return None;
195                }
196            },
197        };
198
199        let base_controller_name = self.hierarchy_reader.base_controller();
200
201        // We're looking for the first line that matches our base controller name, and then we'll see if it's attached
202        // to the container based on the name, and if so, return it.
203        for entry in lines.iter().filter_map(|s| CgroupControllerEntry::try_from_str(s)) {
204            if entry.name == base_controller_name {
205                // We explicitly try to extract the container ID from the reported cgroup controller path, rather than
206                // trying to stick it on the end of our configured root cgroups path. This is because unless we're in the
207                // host's cgroup namespace, the path we get here will be the leaf directory -- the part with the
208                // container ID in it -- but it will be relative in a way that doesn't allow it to be appended to the
209                // root cgroups path, and so trying to query it to get the controller inode, and all of that, will fail.
210                //
211                // The names in that path are all we need: matching them doesn't touch the filesystem, so it works just
212                // as well for a relative path as an absolute one.
213                if let Some(container_id) = extract_container_id_from_path(entry.path, &self.interner) {
214                    return Some(Cgroup {
215                        ino: None,
216                        container_id,
217                    });
218                }
219            } else {
220                debug!(pid, cgroup_lookup_path = %proc_pid_cgroup_path.display(), base_controller_name, "Found cgroup controller for process, but it doesn't match the base controller.");
221            }
222        }
223
224        debug!(pid, cgroup_lookup_path = %proc_pid_cgroup_path.display(), base_controller_name, "Could not find matching base cgroup controller for process.");
225
226        None
227    }
228
229    /// Gets all child cgroups in the current cgroups hierarchy.
230    ///
231    /// Individual paths that can't be traversed -- most commonly because a container exited and its cgroup was removed
232    /// while we were walking the hierarchy -- are skipped rather than aborting the traversal. If any of those skips
233    /// could have hidden a cgroup that still exists, the returned traversal is marked as incomplete. See
234    /// [`TraversalResult::is_complete`] for why that distinction matters.
235    pub fn get_child_cgroups(&self) -> TraversalResult {
236        // Walk the cgroups hierarchy and collect all cgroups that we can find that are related to containers..
237        let root_path = self.hierarchy_reader.root_path();
238
239        match visit_subdirectories(root_path, |path| self.try_cgroup_from_path(path)) {
240            Ok(traversal) => traversal,
241            Err(e) => {
242                // We only get here if the hierarchy root itself couldn't be read, which generally points at a
243                // misconfigured cgroupfs path rather than a transient condition.
244                warn!(error = %e, cgroups_root = %root_path.display(), "Failed to visit cgroups hierarchy.");
245
246                TraversalResult::unreadable()
247            }
248        }
249    }
250}
251
252/// The result of traversing the cgroups hierarchy.
253///
254/// This accumulates as the traversal runs: [`visit_subdirectories`] creates one, records each cgroup it's handed and
255/// each path it couldn't read, and returns it.
256#[derive(Default)]
257pub struct TraversalResult {
258    cgroups: Vec<Cgroup>,
259    skipped: usize,
260    obscured: usize,
261}
262
263impl TraversalResult {
264    /// Creates a traversal representing a hierarchy that couldn't be read at all.
265    ///
266    /// The result is empty and not [complete][Self::is_complete], since failing to read the root hides everything
267    /// beneath it.
268    fn unreadable() -> Self {
269        Self {
270            cgroups: Vec::new(),
271            skipped: 0,
272            obscured: 1,
273        }
274    }
275
276    /// Records a container cgroup found during the traversal.
277    fn record_cgroup(&mut self, cgroup: Cgroup) {
278        self.cgroups.push(cgroup);
279    }
280
281    /// Records a path that couldn't be read, classifying whether skipping it may have hidden existing cgroups.
282    fn record_skip(&mut self, e: &io::Error, path: &Path) {
283        self.skipped += 1;
284
285        match e.kind() {
286            // The directory is gone, so everything beneath it is gone too. There's nothing left for the skip to hide,
287            // and callers tracking those cgroups are right to consider them removed.
288            //
289            // This is routine rather than exceptional: cgroups are removed as the workloads attached to them exit, and
290            // we have no way to hold the hierarchy still while we walk it.
291            io::ErrorKind::NotFound => {
292                trace!(error = %e, path = %path.display(), "Path disappeared during traversal. Skipping.");
293            }
294
295            // We can't read this subtree, so we've never reported anything from it, so there's nothing a caller could
296            // be tracking for us to hide from them.
297            //
298            // This assumes the permissions aren't changing underneath us: a directory that was readable and becomes
299            // unreadable would be misclassified here. That's rare enough to accept, and the alternative -- treating
300            // every permission error as obscuring -- would permanently mark traversals unreliable whenever some part
301            // of the tree is simply not ours to read.
302            io::ErrorKind::PermissionDenied => {
303                debug!(error = %e, path = %path.display(), "Path is not readable. Skipping.");
304            }
305
306            // The subtree is still there and we just failed to read it this time, so anything beneath it is now
307            // invisible to us despite still existing.
308            _ => {
309                self.obscured += 1;
310                debug!(error = %e, path = %path.display(), "Failed to traverse path. Skipping.");
311            }
312        }
313    }
314
315    /// Returns whether absence from this traversal can be taken to mean a cgroup no longer exists.
316    ///
317    /// When this is `false`, part of the hierarchy that may still hold live cgroups couldn't be read, so the cgroups
318    /// reported are not exhaustive. The entries present are still valid, and callers can safely treat them as live, but
319    /// callers **MUST NOT** infer that a previously known cgroup was removed simply because it's absent here.
320    ///
321    /// Note that this can be `true` even when [`skipped`][Self::skipped] is non-zero: a skipped path that couldn't have
322    /// hidden a live cgroup -- because the path is gone, or because we've never been able to read it -- doesn't make
323    /// the set unreliable.
324    pub fn is_complete(&self) -> bool {
325        self.obscured == 0
326    }
327
328    /// Returns the number of paths skipped due to recoverable errors during the traversal.
329    ///
330    /// This counts every skip, including those that don't affect [`is_complete`][Self::is_complete], and is intended
331    /// for telemetry rather than for deciding how much to trust the result.
332    pub fn skipped(&self) -> usize {
333        self.skipped
334    }
335
336    /// Consumes `self` and returns the container cgroups found during the traversal.
337    pub fn into_cgroups(self) -> Vec<Cgroup> {
338        self.cgroups
339    }
340}
341
342#[derive(Clone)]
343enum HierarchyReader {
344    V1 {
345        base_controller_path: PathBuf,
346        controllers: HashMap<String, PathBuf>,
347    },
348
349    V2 {
350        root: PathBuf,
351        controllers: Vec<String>,
352    },
353}
354
355impl HierarchyReader {
356    fn try_from_config(config: &CgroupsConfiguration) -> Result<Option<Self>, GenericError> {
357        // Open the mount file from procfs to scan through and find any cgroups subsystems.
358        let mounts_path = config.procfs_path().join("mounts");
359        let mount_entries = read_lines(&mounts_path)
360            .with_error_context(|| format!("Failed to read mount entries from procfs ({})", mounts_path.display()))?;
361
362        let mut controllers = HashMap::new();
363        let mut maybe_cgroups_v2 = None;
364
365        // For each mount line, check if its of the `cgroup` or `cgroup2` type. Skip everything else.
366        for mount_entry in mount_entries {
367            // Split the line into fields, and take the second and third values. We always expect at least three fields
368            // in a line if it's a line that might possibly be a cgroup mount.
369            let mut fields = mount_entry.split_whitespace();
370            let maybe_cgroup_path = fields.nth(1);
371            let maybe_fs_type = fields.nth(0);
372
373            if let (Some(raw_cgroup_path), Some(fs_type)) = (maybe_cgroup_path, maybe_fs_type) {
374                let cgroup_path = Path::new(raw_cgroup_path);
375
376                // Make sure this path is rooted within our configured cgroupfs path.
377                //
378                // When we're inside a container that has a host-mapped cgroupfs path, the `mounts` file might end up
379                // having duplicate entries (like one set as `/sys/fs/cgroup` and another set as `/host/sys/fs/cgroup`,
380                // etc)... and we want to use the one that matches our configured cgroupfs path as that's the one that
381                // will actually have the cgroups we care about.
382                if !cgroup_path.starts_with(config.cgroupfs_path()) {
383                    continue;
384                }
385
386                match fs_type {
387                    // For cgroups v1, we have to go through all mounts we see to build a full list of enabled controlled.
388                    "cgroup" => process_cgroupv1_mount_entry(cgroup_path, &mut controllers),
389                    // For cgroups v2, we only need to find the unified root mountpoint, and then we can create our reader.
390                    "cgroup2" => maybe_cgroups_v2 = process_cgroupv2_mount_entry(cgroup_path)?,
391                    _ => {}
392                }
393            }
394        }
395
396        // If we didn't find any cgroups v1 controllers, then we potentially return the cgroups v2 hierarchy if found...
397        // otherwise, this will just return `None`.
398        if controllers.is_empty() {
399            if maybe_cgroups_v2.is_some() {
400                debug!("Using cgroups v2 hierarchy.");
401            }
402
403            return Ok(maybe_cgroups_v2);
404        }
405
406        // If we're here, we potentially have a cgroups v1 hierarchy.  Find our base controller -- the memory controller
407        // -- and once we do that, we can create our reader.
408        let base_controller_path = controllers
409            .get(CGROUPS_V1_BASE_CONTROLLER_NAME)
410            .cloned()
411            .ok_or_else(|| {
412                generic_error!(
413                    "Failed to find base controller ({}) in cgroups v1 hierarchy.",
414                    CGROUPS_V1_BASE_CONTROLLER_NAME
415                )
416            })?;
417
418        debug!(root = %base_controller_path.display(), controllers_len = controllers.len(), "Using cgroups v1 hierarchy.");
419
420        Ok(Some(HierarchyReader::V1 {
421            base_controller_path,
422            controllers,
423        }))
424    }
425
426    fn base_controller(&self) -> Option<&'static str> {
427        match self {
428            Self::V1 { .. } => Some(CGROUPS_V1_BASE_CONTROLLER_NAME),
429
430            // Since cgroups v2 is "unified", there's no base controller path.
431            Self::V2 { .. } => None,
432        }
433    }
434
435    fn root_path(&self) -> &Path {
436        match self {
437            Self::V1 {
438                base_controller_path, ..
439            } => base_controller_path.as_path(),
440            Self::V2 { root, .. } => root.as_path(),
441        }
442    }
443
444    fn is_unified(&self) -> bool {
445        matches!(self, Self::V2 { .. })
446    }
447}
448
449/// A container cgroup.
450pub struct Cgroup {
451    ino: Option<u64>,
452    container_id: MetaString,
453}
454
455impl Cgroup {
456    /// Returns the inode of the cgroup controller, if available.
457    pub fn inode(&self) -> Option<u64> {
458        self.ino
459    }
460
461    /// Consumes `self` and returns the container ID.
462    pub fn into_container_id(self) -> MetaString {
463        self.container_id
464    }
465}
466
467struct CgroupControllerEntry<'a> {
468    id: usize,
469    name: Option<&'a str>,
470    path: &'a Path,
471}
472
473impl<'a> CgroupControllerEntry<'a> {
474    fn try_from_str(line: &'a str) -> Option<Self> {
475        let mut fields = line.splitn(3, ':');
476
477        let id = fields.next()?.parse::<usize>().ok()?;
478        let name = fields.next().map(|s| if s.is_empty() { None } else { Some(s) })?;
479        let path = fields.next()?;
480
481        if path.is_empty() {
482            return None;
483        }
484
485        Some(Self {
486            id,
487            name,
488            path: Path::new(path),
489        })
490    }
491}
492
493fn process_cgroupv1_mount_entry(cgroup_path: &Path, controllers: &mut HashMap<String, PathBuf>) {
494    // Split the cgroup path, since there can be multiple controllers mounted at the same path.
495    let path_controllers = cgroup_path
496        .file_name()
497        .and_then(|s| s.to_str().map(|s| s.split(',')))
498        .into_iter()
499        .flatten();
500    for path_controller in path_controllers {
501        // If we have an existing path mapping for this controller, keep whichever one is the
502        // shortest, as we want the more generic path.
503        if let Some(existing_path) = controllers.get(path_controller) {
504            if existing_path.as_os_str().len() < cgroup_path.as_os_str().len() {
505                continue;
506            }
507        }
508
509        controllers.insert(path_controller.to_string(), PathBuf::from(cgroup_path));
510    }
511}
512
513fn process_cgroupv2_mount_entry(cgroup_path: &Path) -> Result<Option<HierarchyReader>, GenericError> {
514    // Read and get the list of active/enabled controllers.
515    let controllers_path = cgroup_path.join(CGROUPS_V2_CONTROLLERS_FILE);
516    let controllers = read_lines(&controllers_path)
517        .with_error_context(|| {
518            format!(
519                "Failed to read controllers from cgroups v2 hierarchy ({}).",
520                controllers_path.display()
521            )
522        })?
523        .into_iter()
524        .flat_map(|s| s.split_whitespace().map(|s| s.to_string()).collect::<Vec<_>>())
525        .collect::<Vec<_>>();
526
527    Ok(Some(HierarchyReader::V2 {
528        root: cgroup_path.to_path_buf(),
529        controllers,
530    }))
531}
532
533fn read_lines(path: &Path) -> io::Result<Vec<String>> {
534    let file = OpenOptions::new().read(true).open(path)?;
535
536    let reader = BufReader::new(file).lines();
537
538    let mut lines = Vec::new();
539    for line in reader {
540        lines.push(line?);
541    }
542
543    Ok(lines)
544}
545
546/// Visits every subdirectory beneath the given path, collecting the cgroups that `visit` identifies.
547///
548/// Subdirectories that can't be read are skipped, along with everything beneath them, and recorded in the returned
549/// [`TraversalResult`]. Callers that need to distinguish "this subdirectory is gone" from "we couldn't see this
550/// subdirectory" **MUST** check [`TraversalResult::is_complete`].
551///
552/// # Errors
553///
554/// If the given path itself can't be queried or listed, an error is returned: nothing was seen, so there's no result
555/// worth reporting. Failures below the given path are never fatal.
556fn visit_subdirectories<P, F>(path: P, mut visit: F) -> Result<TraversalResult, GenericError>
557where
558    P: AsRef<Path>,
559    F: FnMut(&Path) -> Option<Cgroup>,
560{
561    let root = path.as_ref();
562
563    // We can only visit directories, so if the initial path we're given isn't a directory, then we can't do anything.
564    let metadata = fs::metadata(root)
565        .with_error_context(|| format!("Failed to query metadata for traversal root ({}).", root.display()))?;
566    if !metadata.is_dir() {
567        return Ok(TraversalResult::default());
568    }
569
570    let mut traversal = TraversalResult::default();
571
572    // Do an initial pass on our path to get all of its subdirectories, which we'll visit, and then also use as the seed
573    // for further visiting.
574    let mut stack = vec![root.to_path_buf()];
575    while let Some(path) = stack.pop() {
576        // A directory can be removed between the point where we discovered it and the point where we pop it off the
577        // stack to read it, so failing here costs us that subtree but shouldn't stop us from walking the rest.
578        let dir_reader = match fs::read_dir(&path) {
579            Ok(dir_reader) => dir_reader,
580            Err(e) => {
581                // Failing on the root is fatal, unlike failing anywhere below it. Every other path costs us one
582                // subtree, but if we can't list the root then we haven't seen anything at all -- and an empty result
583                // that claims to be complete tells callers every cgroup they know about has gone away.
584                //
585                // Note that this is reachable even though we successfully stat'd the root above: listing a directory
586                // needs read permission, while stat'ing it only needs to traverse its parent.
587                if path.as_path() == root {
588                    return Err(e)
589                        .with_error_context(|| format!("Failed to read traversal root ({}).", root.display()));
590                }
591
592                traversal.record_skip(&e, &path);
593                continue;
594            }
595        };
596
597        for entry in dir_reader {
598            let entry = match entry {
599                Ok(entry) => entry,
600                Err(e) => {
601                    traversal.record_skip(&e, &path);
602                    continue;
603                }
604            };
605
606            let entry_path = entry.path();
607            let file_type = match entry.file_type() {
608                Ok(file_type) => file_type,
609                Err(e) => {
610                    traversal.record_skip(&e, &entry_path);
611                    continue;
612                }
613            };
614
615            if file_type.is_dir() {
616                if let Some(cgroup) = visit(&entry_path) {
617                    traversal.record_cgroup(cgroup);
618                }
619
620                stack.push(entry_path);
621            }
622        }
623    }
624
625    Ok(traversal)
626}
627
628/// Gets the current process's container ID from its local cgroup membership.
629///
630/// This intentionally reads the process namespace's `/proc/self/cgroup` instead of a configured procfs root, which may
631/// refer to the host namespace.
632pub(crate) fn get_self_container_id(interner: &GenericMapInterner) -> Option<MetaString> {
633    let lines = read_lines(Path::new(SELF_CGROUP_PATH)).ok()?;
634    get_container_id_from_cgroup_lines(&lines, interner)
635}
636
637/// Gets the inode of the cgroup controller the current process is attached to.
638///
639/// When the process runs in its own cgroup namespace, the namespace's root *is* the process's own cgroup, so this
640/// identifies the container without needing a path that names it. Inodes are the same on both sides of a namespace
641/// boundary, so the value matches what a traversal of the host's hierarchy reports for the same cgroup.
642///
643/// Returns `None` when the process shares the host's cgroup namespace, since the namespace root is then the hierarchy
644/// root rather than any particular container, and its inode is rejected as reserved.
645///
646/// Like [`get_self_container_id`], this intentionally reads the process's own `/sys/fs/cgroup` rather than a configured
647/// cgroupfs root, which may refer to the host.
648///
649/// Only meaningful under the cgroups v2 unified hierarchy, where that path is itself a cgroup. Under cgroups v1 it's
650/// the `tmpfs` the controllers are mounted into, and an inode from that filesystem doesn't identify anything in the
651/// hierarchy, so this returns `None` unless our own mount is the unified hierarchy.
652pub(crate) fn get_self_cgroup_controller_inode() -> Option<u64> {
653    unified_cgroup_controller_inode(Path::new(SELF_CGROUPFS_PATH))
654}
655
656/// Gets the inode of the given cgroupfs mount, if that mount is the cgroups v2 unified hierarchy.
657///
658/// Every cgroup in the unified hierarchy exposes [`CGROUPS_V2_CONTROLLERS_FILE`], so its presence at the root of the
659/// mount establishes that an inode read from there belongs to a cgroup2 filesystem. Under cgroups v1 the same path is
660/// the `tmpfs` the per-controller filesystems are mounted into, and the file isn't there.
661///
662/// This is deliberately a property of the mount being read, not of the hierarchy the cgroups metadata collector walks.
663/// Those are different mounts whenever the collector is reading a host-mapped cgroupfs, and only the former says
664/// anything about where this inode came from.
665fn unified_cgroup_controller_inode(path: &Path) -> Option<u64> {
666    if !path.join(CGROUPS_V2_CONTROLLERS_FILE).exists() {
667        debug!(
668            path = %path.display(),
669            "Own cgroupfs mount is not the unified hierarchy, so its inode can't identify a container.",
670        );
671        return None;
672    }
673
674    let metadata = match fs::metadata(path) {
675        Ok(metadata) => metadata,
676        Err(e) => {
677            debug!(error = %e, path = %path.display(), "Failed to query metadata for own cgroup controller.");
678            return None;
679        }
680    };
681
682    let controller_inode = metadata.ino();
683    if !is_usable_controller_inode(controller_inode) {
684        debug!(
685            controller_inode,
686            path = %path.display(),
687            "Own cgroup controller reports a reserved inode, which can't identify a container.",
688        );
689        return None;
690    }
691
692    Some(controller_inode)
693}
694
695fn get_container_id_from_cgroup_lines(lines: &[String], interner: &GenericMapInterner) -> Option<MetaString> {
696    lines
697        .iter()
698        .filter_map(|line| CgroupControllerEntry::try_from_str(line))
699        .filter_map(|entry| entry.path.file_name().and_then(|name| name.to_str()))
700        .find_map(|cgroup_name| extract_container_id(cgroup_name, interner))
701}
702
703/// Returns `true` if the given inode can identify a specific cgroup controller.
704///
705/// Reserved inodes -- see [`MAX_RESERVED_INODE`] -- are reported by some filesystems for paths that aren't a distinct
706/// object, so they can't be used to tell one controller apart from another.
707fn is_usable_controller_inode(inode: u64) -> bool {
708    inode > MAX_RESERVED_INODE
709}
710
711/// Matches a container ID anywhere within a cgroup name.
712///
713/// This regular expression is meant to capture:
714/// - 64 character hexadecimal strings (standard format for container IDs almost everywhere)
715/// - 32 character hexadecimal strings followed by a dash and a number (used by AWS ECS)
716/// - 8 character hexadecimal strings followed by up to four groups of 4 character hexadecimal strings separated by
717///   dashes (essentially a UUID, used by Pivotal Cloud Foundry's Garden technology)
718static CONTAINER_REGEX: LazyLock<Regex> =
719    LazyLock::new(|| Regex::new("([0-9a-f]{64})|([0-9a-f]{32}-\\d+)|([0-9a-f]{8}(-[0-9a-f]{4}){4}$)").unwrap());
720
721fn extract_container_id(cgroup_name: &str, interner: &GenericMapInterner) -> Option<MetaString> {
722    match match_container_id(cgroup_name, interner) {
723        ContainerIdMatch::Container(container_id) => Some(container_id),
724        ContainerIdMatch::Excluded | ContainerIdMatch::Uninternable | ContainerIdMatch::NoMatch => None,
725    }
726}
727
728/// What a single cgroup name turned out to be.
729enum ContainerIdMatch {
730    /// The cgroup belongs to a container, with the given ID.
731    Container(MetaString),
732
733    /// The cgroup is named after a container but doesn't represent one.
734    Excluded,
735
736    /// The cgroup belongs to a container, but interning its ID failed.
737    ///
738    /// We know which container this is and simply can't name it, which is different from not knowing: a caller walking
739    /// a path **MUST NOT** keep searching outwards, since the answer it found would be a different container.
740    Uninternable,
741
742    /// The cgroup isn't named after a container at all.
743    NoMatch,
744}
745
746/// Matches a single cgroup name against the container ID heuristic.
747///
748/// [`ContainerIdMatch::Excluded`] is reported separately from [`ContainerIdMatch::NoMatch`] because the two mean
749/// different things to a caller walking a path: a name that isn't a container tells you nothing about its ancestors,
750/// but a name that is deliberately excluded is a definitive answer for that cgroup.
751fn match_container_id(cgroup_name: &str, interner: &GenericMapInterner) -> ContainerIdMatch {
752    let container_id = match CONTAINER_REGEX.find(cgroup_name) {
753        Some(container_id) => container_id,
754        None => return ContainerIdMatch::NoMatch,
755    };
756
757    // Note that this is checked against the full cgroup name, not against the ID we just matched out of it: the match
758    // is a bare hexadecimal string, which can never carry any of these prefixes or suffixes.
759    if is_container_named_but_not_a_container(cgroup_name) {
760        return ContainerIdMatch::Excluded;
761    }
762
763    match interner.try_intern(container_id.as_str()) {
764        Some(interned) => ContainerIdMatch::Container(MetaString::from(interned)),
765        None => {
766            error!(container_id = %container_id.as_str(), "Failed to intern container ID.");
767            ContainerIdMatch::Uninternable
768        }
769    }
770}
771
772/// Resolves the container ID for a cgroup path, falling back to the path's ancestors.
773///
774/// A container's workload can sit in a cgroup nested below the one named for the container, in which case the leaf
775/// doesn't carry the ID but one of its ancestors does. The deepest ancestor that names a container wins, so the most
776/// specific enclosing container is the one reported.
777///
778/// The search stops early, returning `None`, in two cases where continuing outwards would answer with some *other*
779/// container:
780///
781/// - The leaf is named after a container but isn't one. Such a cgroup isn't part of the container's workload at all.
782/// - A container is identified but its ID can't be interned. We know which container it is and just can't name it,
783///   which is not the same as not knowing.
784fn extract_container_id_from_path(cgroup_path: &Path, interner: &GenericMapInterner) -> Option<MetaString> {
785    let leaf_name = cgroup_path.file_name().and_then(|s| s.to_str())?;
786
787    match match_container_id(leaf_name, interner) {
788        ContainerIdMatch::Container(container_id) => return Some(container_id),
789        ContainerIdMatch::Excluded | ContainerIdMatch::Uninternable => return None,
790        ContainerIdMatch::NoMatch => {}
791    }
792
793    // `ancestors` yields the path itself first, which we've already checked, so skip it.
794    for ancestor in cgroup_path.ancestors().skip(1) {
795        let ancestor_name = match ancestor.file_name().and_then(|s| s.to_str()) {
796            Some(ancestor_name) => ancestor_name,
797            None => continue,
798        };
799
800        match match_container_id(ancestor_name, interner) {
801            ContainerIdMatch::Container(container_id) => return Some(container_id),
802
803            // An excluded ancestor is a statement about that cgroup, not about the one we're resolving, so the
804            // container enclosing it can still be the right answer.
805            ContainerIdMatch::Excluded | ContainerIdMatch::NoMatch => {}
806
807            // Reporting the next container out would attribute this cgroup to the wrong one, so stop here.
808            ContainerIdMatch::Uninternable => return None,
809        }
810    }
811
812    None
813}
814
815/// Returns `true` if a cgroup is named after a container but doesn't represent that container's workload.
816fn is_container_named_but_not_a_container(cgroup_name: &str) -> bool {
817    // With the systemd cgroup driver, a `.mount` cgroup can sit alongside a container's own cgroup. It exists, but no
818    // process is ever attached to it, so it holds no stats.
819    //
820    // The `conmon` cgroups belong to the CRI-O/Podman monitor process supervising a container, rather than to the
821    // container itself.
822    cgroup_name.ends_with(".mount")
823        || cgroup_name.starts_with("crio-conmon-")
824        || cgroup_name.starts_with("libpod-conmon-")
825}
826
827#[cfg(test)]
828mod tests {
829    use std::{
830        collections::{HashMap, HashSet},
831        fs, io,
832        num::NonZeroUsize,
833        os::unix::fs::{MetadataExt as _, PermissionsExt as _},
834        path::{Path, PathBuf},
835    };
836
837    use stringtheory::{
838        interning::{GenericMapInterner, InternedString, Interner as _},
839        MetaString,
840    };
841    use tempfile::tempdir;
842
843    use super::{
844        extract_container_id, extract_container_id_from_path, get_container_id_from_cgroup_lines,
845        is_usable_controller_inode, unified_cgroup_controller_inode, visit_subdirectories, CgroupControllerEntry,
846        CgroupsConfiguration, CgroupsReader, Feature, FeatureDetector, HierarchyReader, TraversalResult,
847        CGROUPS_V1_BASE_CONTROLLER_NAME, CGROUPS_V2_CONTROLLERS_FILE, DEFAULT_CGROUPFS_ROOT,
848        DEFAULT_HOST_MAPPED_CGROUPFS_ROOT, DEFAULT_HOST_MAPPED_PROCFS_ROOT, DEFAULT_LEGACY_CGROUPFS_ROOT,
849        DEFAULT_PROCFS_ROOT,
850    };
851
852    #[test]
853    fn parse_controller_entry_cgroups_v1() {
854        let controller_id = 12;
855        let controller_name = "memory";
856        let controller_path_raw = "/kubepods.slice/kubepods-burstable.slice/kubepods-burstable-pod095a9475_4c4f_4726_912c_65743701ef3f.slice/cri-containerd-06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe.scope";
857        let controller_path = Path::new(controller_path_raw);
858        let raw = format!("{}:{}:{}", controller_id, controller_name, controller_path_raw);
859
860        let entry = CgroupControllerEntry::try_from_str(&raw).unwrap();
861        assert_eq!(entry.id, controller_id);
862        assert_eq!(entry.name, Some(controller_name));
863        assert_eq!(entry.path, controller_path);
864    }
865
866    #[test]
867    fn parse_controller_entry_cgroups_v2() {
868        let controller_id = 0;
869        let controller_path_raw =
870            "/system.slice/docker-0b96e72f48e169638a735c0a05adcfc9d6aba2bf6697b627f1635b4f00ea011d.scope";
871        let controller_path = Path::new(controller_path_raw);
872        let raw = format!("{}::{}", controller_id, controller_path_raw);
873
874        let entry = CgroupControllerEntry::try_from_str(&raw).unwrap();
875        assert_eq!(entry.id, controller_id);
876        assert_eq!(entry.name, None);
877        assert_eq!(entry.path, controller_path);
878    }
879
880    fn extract(raw: &str) -> Option<MetaString> {
881        let interner = GenericMapInterner::new(NonZeroUsize::new(1024).unwrap());
882        extract_container_id(raw, &interner)
883    }
884
885    fn extract_from_path(raw: &str) -> Option<MetaString> {
886        let interner = GenericMapInterner::new(NonZeroUsize::new(1024).unwrap());
887        extract_container_id_from_path(Path::new(raw), &interner)
888    }
889
890    #[test]
891    fn resolves_container_id_from_current_process_cgroup_format() {
892        let container_id = "06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe";
893        let cgroup_lines = vec![format!("0::/system.slice/cri-containerd-{container_id}.scope")];
894        let interner = GenericMapInterner::new(NonZeroUsize::new(1024).unwrap());
895
896        assert_eq!(
897            get_container_id_from_cgroup_lines(&cgroup_lines, &interner),
898            Some(MetaString::from(container_id))
899        );
900    }
901
902    #[test]
903    fn does_not_resolve_self_container_from_non_container_cgroup_fixture() {
904        let cgroup_lines = include_str!("testdata/non-container-proc-self-cgroup")
905            .lines()
906            .map(str::to_owned)
907            .collect::<Vec<_>>();
908        let interner = GenericMapInterner::new(NonZeroUsize::new(1024).unwrap());
909
910        assert_eq!(get_container_id_from_cgroup_lines(&cgroup_lines, &interner), None);
911    }
912
913    #[test]
914    fn extract_container_id_cri_containerd() {
915        let expected_container_id =
916            MetaString::from("06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe");
917        let raw = format!("cri-containerd-{}.scope", expected_container_id);
918
919        assert_eq!(extract(&raw), Some(expected_container_id));
920    }
921
922    // The exclusions below have to be checked against the full cgroup name. Checking them against the matched
923    // container ID -- a bare hexadecimal string -- can never fire, which is precisely the bug these tests guard.
924
925    #[test]
926    fn extract_container_id_excludes_dot_mount_cgroups() {
927        let container_id = "06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe";
928        let raw = format!("{}.mount", container_id);
929
930        assert_eq!(extract(&raw), None);
931    }
932
933    #[test]
934    fn extract_container_id_excludes_crio_conmon_cgroups() {
935        let container_id = "06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe";
936        let raw = format!("crio-conmon-{}.scope", container_id);
937
938        assert_eq!(extract(&raw), None);
939    }
940
941    #[test]
942    fn extract_container_id_excludes_libpod_conmon_cgroups() {
943        let container_id = "06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe";
944        let raw = format!("libpod-conmon-{}.scope", container_id);
945
946        assert_eq!(extract(&raw), None);
947    }
948
949    #[test]
950    fn extract_container_id_includes_libpod_container_cgroups() {
951        // Only the `conmon` monitor cgroup is excluded -- the container's own Podman cgroup shares the `libpod-`
952        // prefix and must still resolve.
953        let container_id = "06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe";
954        let raw = format!("libpod-{}.scope", container_id);
955
956        assert_eq!(extract(&raw), Some(MetaString::from(container_id)));
957    }
958
959    #[test]
960    fn reserved_inodes_are_not_usable_controller_inodes() {
961        // 0 and 1 are never valid inodes, and 2 is conventionally the root of a filesystem.
962        assert!(!is_usable_controller_inode(0));
963        assert!(!is_usable_controller_inode(1));
964        assert!(!is_usable_controller_inode(2));
965    }
966
967    #[test]
968    fn ordinary_inodes_are_usable_controller_inodes() {
969        assert!(is_usable_controller_inode(3));
970        assert!(is_usable_controller_inode(4_026_531_835));
971        assert!(is_usable_controller_inode(u64::MAX));
972    }
973
974    const CONTAINER_ID_A: &str = "06d914d2013e51a777feead523895935e33d8ad725b3251ac74c491b3d55d8fe";
975    const CONTAINER_ID_B: &str = "1a2b3c4d5e6f70819293a4b5c6d7e8f90a1b2c3d4e5f60718293a4b5c6d7e8f9";
976
977    #[test]
978    fn extract_from_path_prefers_the_leaf() {
979        let path = format!(
980            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/cri-containerd-{}.scope",
981            CONTAINER_ID_A, CONTAINER_ID_B
982        );
983
984        // The deepest container wins, so a nested container isn't attributed to the one enclosing it.
985        assert_eq!(extract_from_path(&path), Some(MetaString::from(CONTAINER_ID_B)));
986    }
987
988    #[test]
989    fn extract_from_path_falls_back_to_ancestors() {
990        // A container's workload can live in a cgroup nested below the one named for the container.
991        let path = format!(
992            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/init",
993            CONTAINER_ID_A
994        );
995
996        assert_eq!(extract_from_path(&path), Some(MetaString::from(CONTAINER_ID_A)));
997    }
998
999    #[test]
1000    fn extract_from_path_uses_the_deepest_matching_ancestor() {
1001        let path = format!(
1002            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/cri-containerd-{}.scope/init",
1003            CONTAINER_ID_A, CONTAINER_ID_B
1004        );
1005
1006        assert_eq!(extract_from_path(&path), Some(MetaString::from(CONTAINER_ID_B)));
1007    }
1008
1009    #[test]
1010    fn extract_from_path_returns_none_without_any_container_segment() {
1011        let path = "/sys/fs/cgroup/system.slice/systemd-journald.service";
1012
1013        assert_eq!(extract_from_path(path), None);
1014    }
1015
1016    #[test]
1017    fn extract_from_path_does_not_rescue_excluded_leaves_from_ancestors() {
1018        // A `.mount` or `conmon` cgroup isn't part of the container's workload, so even though an ancestor names a
1019        // container, attributing it there would be wrong.
1020        let mount_path = format!(
1021            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/{}.mount",
1022            CONTAINER_ID_A, CONTAINER_ID_B
1023        );
1024        let conmon_path = format!(
1025            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/crio-conmon-{}.scope",
1026            CONTAINER_ID_A, CONTAINER_ID_B
1027        );
1028
1029        assert_eq!(extract_from_path(&mount_path), None);
1030        assert_eq!(extract_from_path(&conmon_path), None);
1031    }
1032
1033    #[test]
1034    fn extract_from_path_skips_excluded_ancestors() {
1035        // An excluded ancestor doesn't claim the cgroup either; the search continues past it.
1036        let path = format!(
1037            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/crio-conmon-{}.scope/init",
1038            CONTAINER_ID_A, CONTAINER_ID_B
1039        );
1040
1041        assert_eq!(extract_from_path(&path), Some(MetaString::from(CONTAINER_ID_A)));
1042    }
1043
1044    /// Builds an interner that can still resolve `held_id` but has no room left for `blocked_id`.
1045    ///
1046    /// The returned handles have to be kept alive for as long as the interner is used: an entry is reclaimed as soon
1047    /// as its last handle drops, so dropping them would un-fill the interner.
1048    fn interner_holding_but_blocking(held_id: &str, blocked_id: &str) -> (GenericMapInterner, Vec<InternedString>) {
1049        let interner = GenericMapInterner::new(NonZeroUsize::new(1024).unwrap());
1050        let mut held = vec![interner.try_intern(held_id).expect("interner starts out empty")];
1051
1052        // The interner is sharded, so "full" is per-shard: we have to keep adding distinct strings until the shard
1053        // that `blocked_id` hashes to is the one that fills up. Probing with `blocked_id` itself is harmless, since
1054        // dropping the handle immediately gives the entry back.
1055        for filler in 0.. {
1056            if interner.try_intern(blocked_id).is_none() {
1057                break;
1058            }
1059
1060            assert!(filler < 100_000, "interner never filled up");
1061
1062            if let Some(interned) = interner.try_intern(&format!("filler-{:060}", filler)) {
1063                held.push(interned);
1064            }
1065        }
1066
1067        // `held_id` has to still be resolvable, otherwise the tests below would pass for the wrong reason: they need
1068        // the ancestor fallback to be *capable* of succeeding, so that declining to take it means something.
1069        assert!(interner.try_intern(held_id).is_some());
1070
1071        (interner, held)
1072    }
1073
1074    #[test]
1075    fn extract_from_path_does_not_fall_back_when_the_leaf_id_cannot_be_interned() {
1076        let (interner, _held) = interner_holding_but_blocking(CONTAINER_ID_A, CONTAINER_ID_B);
1077
1078        // The leaf names a container we can identify but can't name. Walking out to the enclosing container would
1079        // attribute the inner container's workload to the outer one, which is worse than reporting nothing.
1080        let path = format!(
1081            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/cri-containerd-{}.scope",
1082            CONTAINER_ID_A, CONTAINER_ID_B
1083        );
1084
1085        assert_eq!(extract_container_id_from_path(Path::new(&path), &interner), None);
1086    }
1087
1088    #[test]
1089    fn extract_from_path_stops_at_an_ancestor_whose_id_cannot_be_interned() {
1090        let (interner, _held) = interner_holding_but_blocking(CONTAINER_ID_A, CONTAINER_ID_B);
1091
1092        // Same reasoning one level up: the nearest enclosing container is the right answer, so failing to name it
1093        // means we have no answer, not that we should keep searching outwards.
1094        let path = format!(
1095            "/sys/fs/cgroup/system.slice/cri-containerd-{}.scope/cri-containerd-{}.scope/init",
1096            CONTAINER_ID_A, CONTAINER_ID_B
1097        );
1098
1099        assert_eq!(extract_container_id_from_path(Path::new(&path), &interner), None);
1100    }
1101
1102    #[test]
1103    fn extract_from_path_handles_relative_paths() {
1104        // `/proc/<pid>/cgroup` reports a path that's relative to the cgroup namespace root, so resolution can't depend
1105        // on the path being absolute or on it existing on disk.
1106        let path = format!("kubepods.slice/cri-containerd-{}.scope/init", CONTAINER_ID_A);
1107
1108        assert_eq!(extract_from_path(&path), Some(MetaString::from(CONTAINER_ID_A)));
1109    }
1110
1111    /// Collects the names of every visited path, relative to `root`.
1112    fn visited_names(root: &Path, visited: &[PathBuf]) -> HashSet<String> {
1113        visited
1114            .iter()
1115            .map(|path| path.strip_prefix(root).unwrap().to_string_lossy().into_owned())
1116            .collect()
1117    }
1118
1119    fn names(names: &[&str]) -> HashSet<String> {
1120        names.iter().map(|name| (*name).to_owned()).collect()
1121    }
1122
1123    /// Makes `path` unreadable, returning `false` if the caller can still read it anyway.
1124    ///
1125    /// Tests running as root can read a directory regardless of its mode, and so have nothing to assert.
1126    fn make_unreadable(path: &Path) -> bool {
1127        fs::set_permissions(path, fs::Permissions::from_mode(0o000)).unwrap();
1128
1129        if fs::read_dir(path).is_ok() {
1130            make_readable(path);
1131            return false;
1132        }
1133
1134        true
1135    }
1136
1137    /// Restores `path` to a readable mode, so that its parent temporary directory can be cleaned up.
1138    fn make_readable(path: &Path) {
1139        fs::set_permissions(path, fs::Permissions::from_mode(0o755)).unwrap();
1140    }
1141
1142    fn reader_rooted_at(root: &Path) -> CgroupsReader {
1143        CgroupsReader {
1144            procfs_path: PathBuf::from(DEFAULT_PROCFS_ROOT),
1145            hierarchy_reader: HierarchyReader::V2 {
1146                root: root.to_path_buf(),
1147                controllers: Vec::new(),
1148            },
1149            interner: GenericMapInterner::new(NonZeroUsize::new(1024).unwrap()),
1150        }
1151    }
1152
1153    fn v1_reader_rooted_at(base_controller_path: &Path) -> CgroupsReader {
1154        CgroupsReader {
1155            procfs_path: PathBuf::from(DEFAULT_PROCFS_ROOT),
1156            hierarchy_reader: HierarchyReader::V1 {
1157                base_controller_path: base_controller_path.to_path_buf(),
1158                controllers: HashMap::new(),
1159            },
1160            interner: GenericMapInterner::new(NonZeroUsize::new(1024).unwrap()),
1161        }
1162    }
1163
1164    #[test]
1165    fn visit_subdirectories_visits_every_subdirectory() {
1166        let root = tempdir().unwrap();
1167        fs::create_dir_all(root.path().join("a/aa")).unwrap();
1168        fs::create_dir(root.path().join("b")).unwrap();
1169        fs::write(root.path().join("b/file"), "not a directory").unwrap();
1170
1171        let mut visited = Vec::new();
1172        let traversal = visit_subdirectories(root.path(), |path| {
1173            visited.push(path.to_path_buf());
1174            None
1175        })
1176        .unwrap();
1177
1178        assert_eq!(traversal.skipped, 0);
1179        assert_eq!(traversal.obscured, 0);
1180        assert_eq!(visited_names(root.path(), &visited), names(&["a", "a/aa", "b"]));
1181    }
1182
1183    #[test]
1184    fn record_skip_classifies_by_error_kind() {
1185        let mut traversal = TraversalResult::default();
1186
1187        // A path that's gone takes its subdirectories with it, and a path we can't read never showed us any, so
1188        // neither can be hiding anything from us.
1189        traversal.record_skip(&io::Error::from(io::ErrorKind::NotFound), Path::new("/gone"));
1190        traversal.record_skip(&io::Error::from(io::ErrorKind::PermissionDenied), Path::new("/denied"));
1191
1192        assert_eq!(traversal.skipped, 2);
1193        assert_eq!(traversal.obscured, 0);
1194        assert!(traversal.is_complete());
1195
1196        // Any other failure leaves a subtree that still exists but that we couldn't see into.
1197        traversal.record_skip(&io::Error::from(io::ErrorKind::Other), Path::new("/unreadable"));
1198
1199        assert_eq!(traversal.skipped, 3);
1200        assert_eq!(traversal.obscured, 1);
1201        assert!(!traversal.is_complete());
1202    }
1203
1204    #[test]
1205    fn visit_subdirectories_errors_when_root_is_missing() {
1206        let root = tempdir().unwrap();
1207
1208        assert!(visit_subdirectories(root.path().join("missing"), |_| None).is_err());
1209    }
1210
1211    #[test]
1212    fn visit_subdirectories_errors_when_root_is_unreadable() {
1213        // Nest the traversal root inside the temporary directory so its mode can be restored for cleanup.
1214        let parent = tempdir().unwrap();
1215        let root = parent.path().join("root");
1216        fs::create_dir_all(root.join("child")).unwrap();
1217
1218        if !make_unreadable(&root) {
1219            return;
1220        }
1221
1222        // Stat'ing the root still succeeds -- that only needs to traverse its parent -- so this exercises the
1223        // `read_dir` failure specifically, which is the path that used to be recorded as an ordinary skip.
1224        assert!(fs::metadata(&root).is_ok());
1225
1226        let result = visit_subdirectories(&root, |_| None);
1227
1228        make_readable(&root);
1229
1230        // An unreadable root has to be an error rather than an empty-but-complete traversal: we saw nothing, so we
1231        // can't let a caller conclude that everything it knew about has gone away.
1232        assert!(result.is_err());
1233    }
1234
1235    #[test]
1236    fn visit_subdirectories_ignores_non_directory_root() {
1237        let root = tempdir().unwrap();
1238        let file_path = root.path().join("file");
1239        fs::write(&file_path, "not a directory").unwrap();
1240
1241        let mut visited = Vec::new();
1242        let traversal = visit_subdirectories(&file_path, |path| {
1243            visited.push(path.to_path_buf());
1244            None
1245        })
1246        .unwrap();
1247
1248        assert_eq!(traversal.skipped, 0);
1249        assert!(visited.is_empty());
1250    }
1251
1252    #[test]
1253    fn visit_subdirectories_skips_directories_removed_mid_traversal() {
1254        let root = tempdir().unwrap();
1255        for name in ["a", "b", "c"] {
1256            fs::create_dir(root.path().join(name)).unwrap();
1257        }
1258
1259        // Every subdirectory of `root` is visited and pushed onto the traversal stack before any of them is read back,
1260        // so removing one that was already visited guarantees that reading it later fails with `ENOENT`. That's the
1261        // same race we lose in production when a container exits mid-traversal, but without depending on any timing.
1262        let mut visited = Vec::new();
1263        let traversal = visit_subdirectories(root.path(), |path| {
1264            visited.push(path.to_path_buf());
1265            if visited.len() == 2 {
1266                fs::remove_dir(&visited[0]).unwrap();
1267            }
1268            None
1269        })
1270        .unwrap();
1271
1272        // The removed directory was still visited -- we saw it before it went away -- but reading it was skipped
1273        // rather than aborting the traversal, so its two siblings were still read.
1274        assert_eq!(traversal.skipped, 1);
1275
1276        // A directory that's gone can't be hiding anything, so the traversal is still trustworthy.
1277        assert_eq!(traversal.obscured, 0);
1278        assert!(traversal.is_complete());
1279        assert_eq!(visited_names(root.path(), &visited), names(&["a", "b", "c"]));
1280    }
1281
1282    #[test]
1283    fn visit_subdirectories_reports_obscured_paths_for_unexpected_errors() {
1284        let root = tempdir().unwrap();
1285        for name in ["a", "b"] {
1286            fs::create_dir(root.path().join(name)).unwrap();
1287        }
1288
1289        // Same trick as the removal test, but the already-visited directory is replaced with a regular file instead of
1290        // being deleted, so reading it back fails with `ENOTDIR` rather than `ENOENT`.
1291        let mut visited = Vec::new();
1292        let traversal = visit_subdirectories(root.path(), |path| {
1293            visited.push(path.to_path_buf());
1294            if visited.len() == 2 {
1295                fs::remove_dir(&visited[0]).unwrap();
1296                fs::write(&visited[0], "no longer a directory").unwrap();
1297            }
1298            None
1299        })
1300        .unwrap();
1301
1302        // Unlike a removal, this is a path we can't account for, so it counts against the traversal's reliability.
1303        assert_eq!(traversal.skipped, 1);
1304        assert_eq!(traversal.obscured, 1);
1305        assert!(!traversal.is_complete());
1306    }
1307
1308    #[test]
1309    fn visit_subdirectories_skips_unreadable_directories() {
1310        let root = tempdir().unwrap();
1311        let unreadable = root.path().join("unreadable");
1312        fs::create_dir(&unreadable).unwrap();
1313        fs::create_dir_all(root.path().join("readable/nested")).unwrap();
1314
1315        if !make_unreadable(&unreadable) {
1316            return;
1317        }
1318
1319        let mut visited = Vec::new();
1320        let traversal = visit_subdirectories(root.path(), |path| {
1321            visited.push(path.to_path_buf());
1322            None
1323        });
1324
1325        make_readable(&unreadable);
1326
1327        let traversal = traversal.unwrap();
1328        assert_eq!(traversal.skipped, 1);
1329
1330        // We've never been able to see into this subtree, so skipping it doesn't hide anything we'd previously
1331        // reported. Counting it as obscuring would permanently taint every traversal on a host where part of the tree
1332        // simply isn't ours to read.
1333        assert_eq!(traversal.obscured, 0);
1334        assert!(traversal.is_complete());
1335
1336        // The unreadable directory itself is still visited -- we only fail on its contents.
1337        assert_eq!(
1338            visited_names(root.path(), &visited),
1339            names(&["readable", "readable/nested", "unreadable"])
1340        );
1341    }
1342
1343    #[test]
1344    fn get_child_cgroups_reports_complete_traversal() {
1345        let root = tempdir().unwrap();
1346        fs::create_dir(root.path().join(format!("cri-containerd-{}.scope", CONTAINER_ID_A))).unwrap();
1347
1348        let traversal = reader_rooted_at(root.path()).get_child_cgroups();
1349
1350        assert!(traversal.is_complete());
1351        assert_eq!(traversal.skipped, 0);
1352        assert_eq!(traversal.cgroups.len(), 1);
1353        assert_eq!(traversal.cgroups[0].container_id, MetaString::from(CONTAINER_ID_A));
1354    }
1355
1356    #[test]
1357    fn get_child_cgroups_stays_complete_when_subdirectory_is_unreadable() {
1358        let root = tempdir().unwrap();
1359        fs::create_dir(root.path().join(format!("cri-containerd-{}.scope", CONTAINER_ID_A))).unwrap();
1360
1361        let unreadable = root.path().join("unreadable");
1362        fs::create_dir(&unreadable).unwrap();
1363        if !make_unreadable(&unreadable) {
1364            return;
1365        }
1366
1367        let traversal = reader_rooted_at(root.path()).get_child_cgroups();
1368
1369        make_readable(&unreadable);
1370
1371        // The skip is reported for telemetry, but it can't have hidden a live cgroup, so callers can still act on
1372        // what's absent. Marking this incomplete would stop the collector from ever reaping cgroups on a host where
1373        // some part of the hierarchy is permanently unreadable.
1374        assert!(traversal.is_complete());
1375        assert_eq!(traversal.skipped, 1);
1376        assert_eq!(traversal.cgroups.len(), 1);
1377        assert_eq!(traversal.cgroups[0].container_id, MetaString::from(CONTAINER_ID_A));
1378    }
1379
1380    #[test]
1381    fn get_child_cgroups_reports_incomplete_traversal_when_root_is_missing() {
1382        let root = tempdir().unwrap();
1383
1384        let traversal = reader_rooted_at(&root.path().join("missing")).get_child_cgroups();
1385
1386        assert!(!traversal.is_complete());
1387        assert_eq!(traversal.skipped, 0);
1388        assert!(traversal.cgroups.is_empty());
1389    }
1390
1391    #[test]
1392    fn get_child_cgroups_reports_incomplete_traversal_when_root_is_unreadable() {
1393        let parent = tempdir().unwrap();
1394        let root = parent.path().join("root");
1395        fs::create_dir(&root).unwrap();
1396        fs::create_dir(root.join(format!("cri-containerd-{}.scope", CONTAINER_ID_A))).unwrap();
1397
1398        if !make_unreadable(&root) {
1399            return;
1400        }
1401
1402        let traversal = reader_rooted_at(&root).get_child_cgroups();
1403
1404        make_readable(&root);
1405
1406        // The container cgroup underneath is real but invisible to us, so reporting this as complete would have the
1407        // collector reap every alias it holds.
1408        assert!(!traversal.is_complete());
1409        assert!(traversal.cgroups.is_empty());
1410    }
1411
1412    #[test]
1413    fn unified_cgroup_controller_inode_reports_a_real_directory_inode() {
1414        let root = tempdir().unwrap();
1415        fs::write(root.path().join(CGROUPS_V2_CONTROLLERS_FILE), "cpu memory pids").unwrap();
1416
1417        let inode = unified_cgroup_controller_inode(root.path()).expect("a real directory has a usable inode");
1418
1419        assert_eq!(inode, fs::metadata(root.path()).unwrap().ino());
1420        assert!(is_usable_controller_inode(inode));
1421    }
1422
1423    #[test]
1424    fn unified_cgroup_controller_inode_returns_none_for_a_missing_path() {
1425        let root = tempdir().unwrap();
1426
1427        assert_eq!(unified_cgroup_controller_inode(&root.path().join("missing")), None);
1428    }
1429
1430    #[test]
1431    fn unified_cgroup_controller_inode_returns_none_when_our_own_mount_is_not_unified() {
1432        // A cgroups v1 mount is the `tmpfs` the per-controller filesystems hang off of, so it has no
1433        // `cgroup.controllers` and its inode belongs to a filesystem the collector never walks. Handing that inode to
1434        // an alias map keyed on inode alone would at best resolve nothing and at worst name another container.
1435        let root = tempdir().unwrap();
1436        fs::create_dir_all(root.path().join(CGROUPS_V1_BASE_CONTROLLER_NAME)).unwrap();
1437
1438        assert_eq!(unified_cgroup_controller_inode(root.path()), None);
1439    }
1440
1441    #[test]
1442    fn is_unified_distinguishes_the_hierarchy_layout() {
1443        let root = tempdir().unwrap();
1444
1445        // Under the unified hierarchy every cgroup shares one filesystem rooted at the cgroupfs mount, so an inode
1446        // read from that mount identifies the same object the collector saw. Under v1 the controllers are separate
1447        // filesystems mounted beneath it, and that no longer holds.
1448        assert!(reader_rooted_at(root.path()).is_unified());
1449        assert!(!v1_reader_rooted_at(root.path()).is_unified());
1450    }
1451
1452    fn cgroups_config_with(detected: Feature) -> CgroupsConfiguration {
1453        CgroupsConfiguration::new(None, None, &FeatureDetector::from_detected_features(detected))
1454    }
1455
1456    #[test]
1457    fn cgroupfs_root_defaults_to_local_when_nothing_is_host_mapped() {
1458        let config = cgroups_config_with(Feature::none());
1459
1460        assert_eq!(config.procfs_path(), Path::new(DEFAULT_PROCFS_ROOT));
1461        assert_eq!(config.cgroupfs_path(), Path::new(DEFAULT_CGROUPFS_ROOT));
1462    }
1463
1464    #[test]
1465    fn cgroupfs_root_follows_host_mapped_cgroupfs() {
1466        let config = cgroups_config_with(Feature::HostMappedCgroupfs);
1467
1468        assert_eq!(config.cgroupfs_path(), Path::new(DEFAULT_HOST_MAPPED_CGROUPFS_ROOT));
1469    }
1470
1471    #[test]
1472    fn cgroupfs_root_ignores_host_mapped_procfs() {
1473        // procfs and cgroupfs are independent mounts. A deployment that maps one without the other used to get the
1474        // host cgroupfs path off the back of the procfs mount, pointing the reader at a path that isn't there.
1475        let config = cgroups_config_with(Feature::HostMappedProcfs);
1476
1477        assert_eq!(config.procfs_path(), Path::new(DEFAULT_HOST_MAPPED_PROCFS_ROOT));
1478        assert_eq!(config.cgroupfs_path(), Path::new(DEFAULT_CGROUPFS_ROOT));
1479    }
1480
1481    #[test]
1482    fn cgroupfs_root_follows_legacy_root() {
1483        let config = cgroups_config_with(Feature::LegacyCgroupfsRoot);
1484
1485        assert_eq!(config.cgroupfs_path(), Path::new(DEFAULT_LEGACY_CGROUPFS_ROOT));
1486        assert_eq!(config.procfs_path(), Path::new(DEFAULT_PROCFS_ROOT));
1487    }
1488
1489    #[test]
1490    fn host_mapped_cgroupfs_takes_precedence_over_legacy_root() {
1491        // The legacy root is a host layout, so a container that has the host cgroupfs mapped in reads the host
1492        // hierarchy through that mount rather than through a `/cgroup` path in its own filesystem. Feature detection
1493        // never reports both at once, since it only looks for the legacy root when it isn't containerized, but the
1494        // ordering here is what makes that safe.
1495        let config = cgroups_config_with(Feature::HostMappedCgroupfs | Feature::LegacyCgroupfsRoot);
1496
1497        assert_eq!(config.cgroupfs_path(), Path::new(DEFAULT_HOST_MAPPED_CGROUPFS_ROOT));
1498    }
1499
1500    #[test]
1501    fn procfs_root_ignores_host_mapped_cgroupfs() {
1502        let config = cgroups_config_with(Feature::HostMappedCgroupfs);
1503
1504        assert_eq!(config.procfs_path(), Path::new(DEFAULT_PROCFS_ROOT));
1505    }
1506}