fs: Move all files to fs subdirectory

Prepare to implement FsManager, which should be done in the following-up
commits. Apart from that, no code is not modified.

Signed-off-by: Xuewei Niu <niuxuewei.nxw@antgroup.com>
This commit is contained in:
Xuewei Niu
2025-07-01 10:25:51 +08:00
parent 6e273cd2fb
commit b8031f1a21
30 changed files with 1195 additions and 1190 deletions
+975
View File
@@ -0,0 +1,975 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 And Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `blkio` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/blkio-controller.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/blkio-controller.txt)
use std::io::Write;
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{read_string_from, read_u64_from};
use crate::fs::{
BlkIoResources, ControllIdentifier, ControllerInternal, Controllers, CustomizedAttribute,
Resources, Subsystem,
};
/// A controller that allows controlling the `blkio` subsystem of a Cgroup.
///
/// In essence, using the `blkio` controller one can limit and throttle the tasks' usage of block
/// devices in the control group.
#[derive(Debug, Clone)]
pub struct BlkIoController {
base: PathBuf,
path: PathBuf,
v2: bool,
}
#[derive(Eq, PartialEq, Debug)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
/// Per-device information
pub struct BlkIoData {
/// The major number of the device.
pub major: i16,
/// The minor number of the device.
pub minor: i16,
/// The data that is associated with the device.
pub data: u64,
}
#[derive(Eq, PartialEq, Debug, Default)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
/// Per-device activity from the control group.
pub struct IoService {
/// The major number of the device.
pub major: i16,
/// The minor number of the device.
pub minor: i16,
/// How many items were read from the device.
pub read: u64,
/// How many items were written to the device.
pub write: u64,
/// How many items were synchronously transferred.
pub sync: u64,
/// How many items were asynchronously transferred.
pub r#async: u64,
/// How many items were discarded.
pub discard: u64,
/// Total number of items transferred.
pub total: u64,
}
#[derive(Eq, PartialEq, Debug)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
/// Per-device activity from the control group.
/// Only for cgroup v2
pub struct IoStat {
/// The major number of the device.
pub major: i16,
/// The minor number of the device.
pub minor: i16,
/// How many bytes were read from the device.
pub rbytes: u64,
/// How many bytes were written to the device.
pub wbytes: u64,
/// How many iops were read from the device.
pub rios: u64,
/// How many iops were written to the device.
pub wios: u64,
/// How many discard bytes were read from the device.
pub dbytes: u64,
/// How many discard iops were written to the device.
pub dios: u64,
}
fn parse_io_service(s: String) -> Result<Vec<IoService>> {
let mut io_services = Vec::<IoService>::new();
let mut io_service = IoService::default();
let lines = s
.lines()
.filter(|x| x.split_whitespace().count() == 3)
.map(|x| {
let mut spl = x.split_whitespace();
(
spl.next().unwrap(),
spl.next().unwrap(),
spl.next().unwrap(),
)
})
.map(|(a, b, c)| {
let mut spl = a.split(':');
(
spl.next().unwrap().parse::<i16>(),
spl.next().unwrap().parse::<i16>(),
b,
c,
)
})
.collect::<Vec<_>>();
for (major_num, minor_num, op, val) in lines.iter() {
let major = *major_num.as_ref().map_err(|_| Error::new(ParseError))?;
let minor = *minor_num.as_ref().map_err(|_| Error::new(ParseError))?;
if (major != io_service.major || minor != io_service.minor) && io_service.major != 0 {
// new block device
io_services.push(io_service);
io_service = IoService::default();
}
io_service.major = major;
io_service.minor = minor;
let val = val.parse::<u64>().map_err(|_| Error::new(ParseError))?;
match *op {
"Read" => io_service.read = val,
"Write" => io_service.write = val,
"Sync" => io_service.sync = val,
"Async" => io_service.r#async = val,
"Discard" => io_service.discard = val,
"Total" => io_service.total = val,
_ => {}
}
}
if io_service.major != 0 {
io_services.push(io_service);
}
Ok(io_services)
}
fn get_value(s: &str) -> String {
let arr = s.split(':').collect::<Vec<&str>>();
if arr.len() != 2 {
return "0".to_string();
}
arr[1].to_string()
}
fn parse_io_stat(s: String) -> Vec<IoStat> {
// line:
// 8:0 rbytes=180224 wbytes=0 rios=3 wios=0 dbytes=0 dios=0
s.lines()
.filter(|x| x.split_whitespace().count() == 7)
.map(|x| {
let arr = x.split_whitespace().collect::<Vec<&str>>();
let device = arr[0].split(':').collect::<Vec<&str>>();
let (major, minor) = (device[0], device[1]);
IoStat {
major: major.parse::<i16>().unwrap(),
minor: minor.parse::<i16>().unwrap(),
rbytes: get_value(arr[1]).parse::<u64>().unwrap(),
wbytes: get_value(arr[2]).parse::<u64>().unwrap(),
rios: get_value(arr[3]).parse::<u64>().unwrap(),
wios: get_value(arr[4]).parse::<u64>().unwrap(),
dbytes: get_value(arr[5]).parse::<u64>().unwrap(),
dios: get_value(arr[6]).parse::<u64>().unwrap(),
}
})
.collect::<Vec<IoStat>>()
}
fn parse_io_service_total(s: String) -> Result<u64> {
s.lines()
.find_map(|line| {
let mut parts = line.split_whitespace();
match (parts.next(), parts.next(), parts.next()) {
(Some("Total"), Some(val), None) => val.parse::<u64>().ok(),
_ => None,
}
})
.ok_or_else(|| Error::new(ParseError))
}
fn parse_blkio_data(s: String) -> Result<Vec<BlkIoData>> {
let r = s
.chars()
.map(|x| if x == ':' { ' ' } else { x })
.collect::<String>();
let r = r
.lines()
.flat_map(|x| x.split_whitespace())
.collect::<Vec<_>>();
let r = r.chunks(3).collect::<Vec<_>>();
let mut res = Vec::new();
let err = r.iter().try_for_each(|x| match x {
[major, minor, data] => {
res.push(BlkIoData {
major: major.parse::<i16>().unwrap(),
minor: minor.parse::<i16>().unwrap(),
data: data.parse::<u64>().unwrap(),
});
Ok(())
}
_ => Err(Error::new(ParseError)),
});
if err.is_err() {
Err(Error::new(ParseError))
} else {
Ok(res)
}
}
/// Current state and statistics about how throttled are the block devices when accessed from the
/// controller's control group.
#[derive(Default, Debug)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct BlkIoThrottle {
/// Statistics about the bytes transferred between the block devices by the tasks in this
/// control group.
pub io_service_bytes: Vec<IoService>,
/// Total amount of bytes transferred to and from the block devices.
pub io_service_bytes_total: u64,
/// Same as `io_service_bytes`, but contains all descendant control groups.
pub io_service_bytes_recursive: Vec<IoService>,
/// Total amount of bytes transferred to and from the block devices, including all descendant
/// control groups.
pub io_service_bytes_recursive_total: u64,
/// The number of I/O operations performed on the devices as seen by the throttling policy.
pub io_serviced: Vec<IoService>,
/// The total number of I/O operations performed on the devices as seen by the throttling
/// policy.
pub io_serviced_total: u64,
/// Same as `io_serviced`, but contains all descendant control groups.
pub io_serviced_recursive: Vec<IoService>,
/// Same as `io_serviced`, but contains all descendant control groups and contains only the
/// total amount.
pub io_serviced_recursive_total: u64,
/// The upper limit of bytes per second rate of read operation on the block devices by the
/// control group's tasks.
pub read_bps_device: Vec<BlkIoData>,
/// The upper limit of I/O operation per second, when said operation is a read operation.
pub read_iops_device: Vec<BlkIoData>,
/// The upper limit of bytes per second rate of write operation on the block devices by the
/// control group's tasks.
pub write_bps_device: Vec<BlkIoData>,
/// The upper limit of I/O operation per second, when said operation is a write operation.
pub write_iops_device: Vec<BlkIoData>,
}
/// Statistics and state of the block devices.
#[derive(Default, Debug)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct BlkIo {
/// The number of BIOS requests merged into I/O requests by the control group's tasks.
pub io_merged: Vec<IoService>,
/// Same as `io_merged`, but only reports the total number.
pub io_merged_total: u64,
/// Same as `io_merged`, but contains all descendant control groups.
pub io_merged_recursive: Vec<IoService>,
/// Same as `io_merged_recursive`, but only reports the total number.
pub io_merged_recursive_total: u64,
/// The number of requests queued for I/O operations by the tasks of the control group.
pub io_queued: Vec<IoService>,
/// Same as `io_queued`, but only reports the total number.
pub io_queued_total: u64,
/// Same as `io_queued`, but contains all descendant control groups.
pub io_queued_recursive: Vec<IoService>,
/// Same as `io_queued_recursive`, but contains all descendant control groups.
pub io_queued_recursive_total: u64,
/// The number of bytes transferred from and to the block device (as seen by the CFQ I/O scheduler).
pub io_service_bytes: Vec<IoService>,
/// Same as `io_service_bytes`, but contains all descendant control groups.
pub io_service_bytes_total: u64,
/// Same as `io_service_bytes`, but contains all descendant control groups.
pub io_service_bytes_recursive: Vec<IoService>,
/// Total amount of bytes transferred between the tasks and block devices, including the
/// descendant control groups' numbers.
pub io_service_bytes_recursive_total: u64,
/// The number of I/O operations (as seen by the CFQ I/O scheduler) between the devices and the
/// control group's tasks.
pub io_serviced: Vec<IoService>,
/// The total number of I/O operations performed on the devices as seen by the throttling
/// policy.
pub io_serviced_total: u64,
/// Same as `io_serviced`, but contains all descendant control groups.
pub io_serviced_recursive: Vec<IoService>,
/// Same as `io_serviced`, but contains all descendant control groups and contains only the
/// total amount.
pub io_serviced_recursive_total: u64,
/// The total time spent between dispatch and request completion for I/O requests (as seen by
/// the CFQ I/O scheduler) by the control group's tasks.
pub io_service_time: Vec<IoService>,
/// Same as `io_service_time`, but contains all descendant control groups and contains only the
/// total amount.
pub io_service_time_total: u64,
/// Same as `io_service_time`, but contains all descendant control groups.
pub io_service_time_recursive: Vec<IoService>,
/// Same as `io_service_time_recursive`, but contains all descendant control groups and only
/// the total amount.
pub io_service_time_recursive_total: u64,
/// Total amount of time spent waiting for a free slot in the CFQ I/O scheduler's queue.
pub io_wait_time: Vec<IoService>,
/// Same as `io_wait_time`, but only reports the total amount.
pub io_wait_time_total: u64,
/// Same as `io_wait_time`, but contains all descendant control groups.
pub io_wait_time_recursive: Vec<IoService>,
/// Same as `io_wait_time_recursive`, but only reports the total amount.
pub io_wait_time_recursive_total: u64,
/// How much weight do the control group's tasks have when competing against the descendant
/// control group's tasks.
pub leaf_weight: u64,
/// Same as `leaf_weight`, but per-block-device.
pub leaf_weight_device: Vec<BlkIoData>,
/// Total number of sectors transferred between the block devices and the control group's
/// tasks.
pub sectors: Vec<BlkIoData>,
/// Same as `sectors`, but contains all descendant control groups.
pub sectors_recursive: Vec<BlkIoData>,
/// Similar statistics, but as seen by the throttle policy.
pub throttle: BlkIoThrottle,
/// The time the control group had access to the I/O devices.
pub time: Vec<BlkIoData>,
/// Same as `time`, but contains all descendant control groups.
pub time_recursive: Vec<BlkIoData>,
/// The weight of this control group.
pub weight: u64,
/// Same as `weight`, but per-block-device.
pub weight_device: Vec<BlkIoData>,
/// IoStat for cgroup v2
pub io_stat: Vec<IoStat>,
}
impl ControllerInternal for BlkIoController {
fn control_type(&self) -> Controllers {
Controllers::BlkIo
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn is_v2(&self) -> bool {
self.v2
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &BlkIoResources = &res.blkio;
if let Some(weight) = res.weight {
let _ = self.set_weight(weight as u64);
}
if let Some(leaf_weight) = res.leaf_weight {
let _ = self.set_leaf_weight(leaf_weight as u64);
}
for dev in &res.weight_device {
if let Some(weight) = dev.weight {
let _ = self.set_weight_for_device(dev.major, dev.minor, weight as u64);
}
if let Some(leaf_weight) = dev.leaf_weight {
let _ = self.set_leaf_weight_for_device(dev.major, dev.minor, leaf_weight as u64);
}
}
for dev in &res.throttle_read_bps_device {
let _ = self.throttle_read_bps_for_device(dev.major, dev.minor, dev.rate);
}
for dev in &res.throttle_write_bps_device {
let _ = self.throttle_write_bps_for_device(dev.major, dev.minor, dev.rate);
}
for dev in &res.throttle_read_iops_device {
let _ = self.throttle_read_iops_for_device(dev.major, dev.minor, dev.rate);
}
for dev in &res.throttle_write_iops_device {
let _ = self.throttle_write_iops_for_device(dev.major, dev.minor, dev.rate);
}
res.attrs.iter().for_each(|(k, v)| {
let _ = self.set(k, v);
});
Ok(())
}
}
impl ControllIdentifier for BlkIoController {
fn controller_type() -> Controllers {
Controllers::BlkIo
}
}
impl<'a> From<&'a Subsystem> for &'a BlkIoController {
fn from(sub: &'a Subsystem) -> &'a BlkIoController {
unsafe {
match sub {
Subsystem::BlkIo(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl BlkIoController {
/// Constructs a new `BlkIoController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
Self {
base: root,
path: point,
v2,
}
}
fn blkio_v2(&self) -> BlkIo {
BlkIo {
io_stat: self
.open_path("io.stat", false)
.and_then(read_string_from)
.map(parse_io_stat)
.unwrap_or_default(),
..Default::default()
}
}
/// Gathers statistics about and reports the state of the block devices used by the control
/// group's tasks.
pub fn blkio(&self) -> BlkIo {
if self.v2 {
return self.blkio_v2();
}
BlkIo {
io_merged: self
.open_path("blkio.io_merged", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_merged_total: self
.open_path("blkio.io_merged", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_merged_recursive: self
.open_path("blkio.io_merged_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_merged_recursive_total: self
.open_path("blkio.io_merged_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_queued: self
.open_path("blkio.io_queued", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_queued_total: self
.open_path("blkio.io_queued", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_queued_recursive: self
.open_path("blkio.io_queued_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_queued_recursive_total: self
.open_path("blkio.io_queued_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_service_bytes: self
.open_path("blkio.io_service_bytes", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_service_bytes_total: self
.open_path("blkio.io_service_bytes", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_service_bytes_recursive: self
.open_path("blkio.io_service_bytes_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_service_bytes_recursive_total: self
.open_path("blkio.io_service_bytes_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_serviced: self
.open_path("blkio.io_serviced", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_serviced_total: self
.open_path("blkio.io_serviced", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_serviced_recursive: self
.open_path("blkio.io_serviced_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_serviced_recursive_total: self
.open_path("blkio.io_serviced_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_service_time: self
.open_path("blkio.io_service_time", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_service_time_total: self
.open_path("blkio.io_service_time", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_service_time_recursive: self
.open_path("blkio.io_service_time_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_service_time_recursive_total: self
.open_path("blkio.io_service_time_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_wait_time: self
.open_path("blkio.io_wait_time", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_wait_time_total: self
.open_path("blkio.io_wait_time", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_wait_time_recursive: self
.open_path("blkio.io_wait_time_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_wait_time_recursive_total: self
.open_path("blkio.io_wait_time_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
leaf_weight: self
.open_path("blkio.leaf_weight", false)
.and_then(read_u64_from)
.unwrap_or(0u64),
leaf_weight_device: self
.open_path("blkio.leaf_weight_device", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
sectors: self
.open_path("blkio.sectors", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
sectors_recursive: self
.open_path("blkio.sectors_recursive", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
throttle: BlkIoThrottle {
io_service_bytes: self
.open_path("blkio.throttle.io_service_bytes", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_service_bytes_total: self
.open_path("blkio.throttle.io_service_bytes", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_service_bytes_recursive: self
.open_path("blkio.throttle.io_service_bytes_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_service_bytes_recursive_total: self
.open_path("blkio.throttle.io_service_bytes_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_serviced: self
.open_path("blkio.throttle.io_serviced", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_serviced_total: self
.open_path("blkio.throttle.io_serviced", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
io_serviced_recursive: self
.open_path("blkio.throttle.io_serviced_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service)
.unwrap_or_default(),
io_serviced_recursive_total: self
.open_path("blkio.throttle.io_serviced_recursive", false)
.and_then(read_string_from)
.and_then(parse_io_service_total)
.unwrap_or_default(),
read_bps_device: self
.open_path("blkio.throttle.read_bps_device", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
read_iops_device: self
.open_path("blkio.throttle.read_iops_device", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
write_bps_device: self
.open_path("blkio.throttle.write_bps_device", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
write_iops_device: self
.open_path("blkio.throttle.write_iops_device", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
},
time: self
.open_path("blkio.time", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
time_recursive: self
.open_path("blkio.time_recursive", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
weight: self
.open_path("blkio.weight", false)
.and_then(read_u64_from)
.unwrap_or(0u64),
weight_device: self
.open_path("blkio.weight_device", false)
.and_then(read_string_from)
.and_then(parse_blkio_data)
.unwrap_or_default(),
io_stat: Vec::new(),
}
}
/// Set the leaf weight on the control group's tasks, i.e., how are they weighted against the
/// descendant control groups' tasks.
pub fn set_leaf_weight(&self, w: u64) -> Result<()> {
self.open_path("blkio.leaf_weight", true)
.and_then(|mut file| {
file.write_all(w.to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("blkio.leaf_weight".to_string(), w.to_string()),
e,
)
})
})
}
/// Same as `set_leaf_weight()`, but settable per each block device.
pub fn set_leaf_weight_for_device(&self, major: u64, minor: u64, weight: u64) -> Result<()> {
self.open_path("blkio.leaf_weight_device", true)
.and_then(|mut file| {
file.write_all(format!("{}:{} {}", major, minor, weight).as_ref())
.map_err(|e| {
Error::with_cause(
WriteFailed(
"blkio.leaf_weight_device".to_string(),
format!("{}:{} {}", major, minor, weight),
),
e,
)
})
})
}
/// Reset the statistics the kernel has gathered so far and start fresh.
pub fn reset_stats(&self) -> Result<()> {
self.open_path("blkio.reset_stats", true)
.and_then(|mut file| {
file.write_all("1".to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("blkio.reset_stats".to_string(), "1".to_string()),
e,
)
})
})
}
/// Throttle the bytes per second rate of read operation affecting the block device
/// `major:minor` to `bps`.
pub fn throttle_read_bps_for_device(&self, major: u64, minor: u64, bps: u64) -> Result<()> {
let mut file_name = "blkio.throttle.read_bps_device";
let mut content = format!("{}:{} {}", major, minor, bps);
if self.v2 {
file_name = "io.max";
content = format!("{}:{} rbps={}", major, minor, bps);
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(content.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), content.to_string()), e)
})
})
}
/// Throttle the I/O operations per second rate of read operation affecting the block device
/// `major:minor` to `bps`.
pub fn throttle_read_iops_for_device(&self, major: u64, minor: u64, iops: u64) -> Result<()> {
let mut file_name = "blkio.throttle.read_iops_device";
let mut content = format!("{}:{} {}", major, minor, iops);
if self.v2 {
file_name = "io.max";
content = format!("{}:{} riops={}", major, minor, iops);
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(content.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), content.to_string()), e)
})
})
}
/// Throttle the bytes per second rate of write operation affecting the block device
/// `major:minor` to `bps`.
pub fn throttle_write_bps_for_device(&self, major: u64, minor: u64, bps: u64) -> Result<()> {
let mut file_name = "blkio.throttle.write_bps_device";
let mut content = format!("{}:{} {}", major, minor, bps);
if self.v2 {
file_name = "io.max";
content = format!("{}:{} wbps={}", major, minor, bps);
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(content.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), content.to_string()), e)
})
})
}
/// Throttle the I/O operations per second rate of write operation affecting the block device
/// `major:minor` to `bps`.
pub fn throttle_write_iops_for_device(&self, major: u64, minor: u64, iops: u64) -> Result<()> {
let mut file_name = "blkio.throttle.write_iops_device";
let mut content = format!("{}:{} {}", major, minor, iops);
if self.v2 {
file_name = "io.max";
content = format!("{}:{} wiops={}", major, minor, iops);
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(content.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), content.to_string()), e)
})
})
}
/// Set the weight of the control group's tasks.
pub fn set_weight(&self, w: u64) -> Result<()> {
// Attation: may not find in high kernel version.
let mut file_name = "blkio.weight";
if self.v2 {
file_name = "io.bfq.weight";
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(w.to_string().as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), w.to_string()), e)
})
})
}
/// Same as `set_weight()`, but settable per each block device.
pub fn set_weight_for_device(&self, major: u64, minor: u64, weight: u64) -> Result<()> {
let mut file_name = "blkio.weight_device";
if self.v2 {
// Attation: there is no weight for device in runc
// https://github.com/opencontainers/runc/blob/46be7b612e2533c494e6a251111de46d8e286ed5/libcontainer/cgroups/fs2/io.go#L30
// may depends on IO schedulers https://wiki.ubuntu.com/Kernel/Reference/IOSchedulers
file_name = "io.bfq.weight";
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(format!("{}:{} {}", major, minor, weight).as_ref())
.map_err(|e| {
Error::with_cause(
WriteFailed(
file_name.to_string(),
format!("{}:{} {}", major, minor, weight),
),
e,
)
})
})
}
}
impl CustomizedAttribute for BlkIoController {}
#[cfg(test)]
mod test {
use crate::fs::blkio::{parse_blkio_data, BlkIoData};
use crate::fs::blkio::{parse_io_service, parse_io_service_total, IoService};
use crate::fs::error::*;
static TEST_VALUE: &str = "\
8:32 Read 4280320
8:32 Write 0
8:32 Sync 4280320
8:32 Async 0
8:32 Discard 1
8:32 Total 4280320
8:48 Read 5705479168
8:48 Write 56096055296
8:48 Sync 11213923328
8:48 Async 50587611136
8:48 Total 61801534464
8:16 Read 10059776
8:16 Write 0
8:16 Sync 10059776
8:16 Async 0
8:16 Total 10059776
8:0 Read 7192576
8:0 Write 0
8:0 Sync 7192576
8:0 Async 0
8:0 Total 7192576
Total 61823067136
";
static TEST_BLKIO_DATA: &str = "\
8:48 454480833999
8:32 228392923193
8:16 772456885
8:0 559583764
";
#[test]
fn test_parse_io_service_total() {
let ok = parse_io_service_total(TEST_VALUE.to_string()).unwrap();
assert_eq!(ok, 61823067136);
}
#[test]
fn test_parse_io_service() {
let ok = parse_io_service(TEST_VALUE.to_string()).unwrap();
assert_eq!(
ok,
vec![
IoService {
major: 8,
minor: 32,
read: 4280320,
write: 0,
sync: 4280320,
r#async: 0,
discard: 1,
total: 4280320,
},
IoService {
major: 8,
minor: 48,
read: 5705479168,
write: 56096055296,
sync: 11213923328,
r#async: 50587611136,
discard: 0,
total: 61801534464,
},
IoService {
major: 8,
minor: 16,
read: 10059776,
write: 0,
sync: 10059776,
r#async: 0,
discard: 0,
total: 10059776,
},
IoService {
major: 8,
minor: 0,
read: 7192576,
write: 0,
sync: 7192576,
r#async: 0,
discard: 0,
total: 7192576,
}
]
);
let invalid_values = vec![
"\
8:32 Read 4280320
8:32 Write a
8:32 Async 1
",
"\
8:32 Read 4280320
b:32 Write 1
8:32 Async 1
",
"\
8:32 Read 4280320
8:32 Write 1
8:c Async 1
",
];
for value in invalid_values {
let err = parse_io_service(value.to_string()).unwrap_err();
assert_eq!(err.kind(), &ErrorKind::ParseError,);
}
}
#[test]
fn test_parse_blkio_data() {
assert_eq!(
parse_blkio_data(TEST_BLKIO_DATA.to_string()).unwrap(),
vec![
BlkIoData {
major: 8,
minor: 48,
data: 454480833999,
},
BlkIoData {
major: 8,
minor: 32,
data: 228392923193,
},
BlkIoData {
major: 8,
minor: 16,
data: 772456885,
},
BlkIoData {
major: 8,
minor: 0,
data: 559583764,
}
]
);
}
}
+636
View File
@@ -0,0 +1,636 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module handles cgroup operations. Start here!
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::hierarchies::V1;
use crate::fs::{CgroupPid, ControllIdentifier, Controller, Hierarchy, Resources, Subsystem};
use std::collections::HashMap;
use std::convert::From;
use std::fs;
use std::path::{Path, PathBuf};
pub const CGROUP_MODE_DOMAIN: &str = "domain";
pub const CGROUP_MODE_DOMAIN_THREADED: &str = "domain threaded";
pub const CGROUP_MODE_DOMAIN_INVALID: &str = "domain invalid";
pub const CGROUP_MODE_THREADED: &str = "threaded";
/// A control group is the central structure to this crate.
///
///
/// # What are control groups?
///
/// Lifting over from the Linux kernel sources:
///
/// > Control Groups provide a mechanism for aggregating/partitioning sets of
/// > tasks, and all their future children, into hierarchical groups with
/// > specialized behaviour.
///
/// This crate is an attempt at providing a Rust-native way of managing these cgroups.
#[derive(Debug)]
pub struct Cgroup {
/// The list of subsystems that control this cgroup
subsystems: Vec<Subsystem>,
/// The hierarchy.
hier: Box<dyn Hierarchy>,
path: String,
/// List of controllers specifically enabled in the control group.
specified_controllers: Option<Vec<String>>,
}
impl Clone for Cgroup {
fn clone(&self) -> Self {
Cgroup {
subsystems: self.subsystems.clone(),
hier: crate::fs::hierarchies::auto(),
path: self.path.clone(),
specified_controllers: None,
}
}
}
impl Default for Cgroup {
fn default() -> Self {
Cgroup {
subsystems: Vec::new(),
hier: crate::fs::hierarchies::auto(),
path: "".to_string(),
specified_controllers: None,
}
}
}
impl Cgroup {
pub fn v2(&self) -> bool {
self.hier.v2()
}
/// Return the path the cgroup is located at.
pub fn path(&self) -> &str {
&self.path
}
/// Create this control group.
pub fn create(&self) -> Result<()> {
if self.hier.v2() {
create_v2_cgroup(self.hier.root(), &self.path, &self.specified_controllers)
} else {
for subsystem in &self.subsystems {
subsystem.to_controller().create();
}
Ok(())
}
}
/// Create a new control group in the hierarchy `hier`, with name `path`.
///
/// Returns a handle to the control group that can be used to manipulate it.
pub fn new<P: AsRef<Path>>(hier: Box<dyn Hierarchy>, path: P) -> Result<Cgroup> {
let cg = Cgroup::load(hier, path);
cg.create()?;
Ok(cg)
}
/// Create a new control group in the hierarchy `hier`, with name `path`.
///
/// Returns a handle to the control group that can be used to manipulate it.
pub fn new_with_specified_controllers<P: AsRef<Path>>(
hier: Box<dyn Hierarchy>,
path: P,
specified_controllers: Option<Vec<String>>,
) -> Result<Cgroup> {
let cg = if let Some(sc) = specified_controllers {
Cgroup::load_with_specified_controllers(hier, path, sc)
} else {
Cgroup::load(hier, path)
};
cg.create()?;
Ok(cg)
}
/// Create a new control group in the hierarchy `hier`, with name `path` and `relative_paths`
///
/// Returns a handle to the control group that can be used to manipulate it.
///
/// Note that this method is only meaningful for cgroup v1, call it is equivalent to call `new` in the v2 mode.
pub fn new_with_relative_paths<P: AsRef<Path>>(
hier: Box<dyn Hierarchy>,
path: P,
relative_paths: HashMap<String, String>,
) -> Result<Cgroup> {
let cg = Cgroup::load_with_relative_paths(hier, path, relative_paths);
cg.create()?;
Ok(cg)
}
/// Create a handle for a control group in the hierarchy `hier`, with name `path`.
///
/// Returns a handle to the control group (that possibly does not exist until `create()` has
/// been called on the cgroup.
pub fn load<P: AsRef<Path>>(hier: Box<dyn Hierarchy>, path: P) -> Cgroup {
let path = path.as_ref();
let mut subsystems = hier.subsystems();
if path.as_os_str() != "" {
subsystems = subsystems
.into_iter()
.map(|x| x.enter(path))
.collect::<Vec<_>>();
}
Cgroup {
path: path.to_str().unwrap().to_string(),
subsystems,
hier,
specified_controllers: None,
}
}
/// Create a handle for a specified control group in the hierarchy `hier`, with name `path`.
///
/// Returns a handle to the control group (that possibly does not exist until `create()` has
/// been called on the cgroup.
pub fn load_with_specified_controllers<P: AsRef<Path>>(
hier: Box<dyn Hierarchy>,
path: P,
specified_controllers: Vec<String>,
) -> Cgroup {
let path = path.as_ref();
let mut subsystems = hier.subsystems();
if path.as_os_str() != "" {
subsystems = subsystems
.into_iter()
.filter(|x| specified_controllers.contains(&x.controller_name()))
.map(|x| x.enter(path))
.collect::<Vec<_>>();
}
Cgroup {
path: path.to_str().unwrap().to_string(),
subsystems,
hier,
specified_controllers: Some(specified_controllers),
}
}
/// Create a handle for a control group in the hierarchy `hier`, with name `path` and `relative_paths`
///
/// Returns a handle to the control group (that possibly does not exist until `create()` has
/// been called on the cgroup.
///
/// Note that this method is only meaningful for cgroup v1, call it is equivalent to call `load` in the v2 mode
pub fn load_with_relative_paths<P: AsRef<Path>>(
hier: Box<dyn Hierarchy>,
path: P,
relative_paths: HashMap<String, String>,
) -> Cgroup {
// relative_paths only valid for cgroup v1
if hier.v2() {
return Self::load(hier, path);
}
let path = path.as_ref();
let mut subsystems = hier.subsystems();
if path.as_os_str() != "" {
subsystems = subsystems
.into_iter()
.map(|x| {
let cn = x.controller_name();
if relative_paths.contains_key(&cn) {
let rp = relative_paths.get(&cn).unwrap();
let valid_path = rp.trim_start_matches('/').to_string();
let mut p = PathBuf::from(valid_path);
p.push(path);
x.enter(p.as_ref())
} else {
x.enter(path)
}
})
.collect::<Vec<_>>();
}
Cgroup {
subsystems,
hier,
path: path.to_str().unwrap().to_string(),
specified_controllers: None,
}
}
/// The list of subsystems that this control group supports.
pub fn subsystems(&self) -> &Vec<Subsystem> {
&self.subsystems
}
/// Deletes the control group.
///
/// Note that this function makes no effort in cleaning up the descendant and the underlying
/// system call will fail if there are any descendants. Thus, one should check whether it was
/// actually removed, and remove the descendants first if not. In the future, this behavior
/// will change.
pub fn delete(&self) -> Result<()> {
if self.v2() {
if !self.path.is_empty() {
let mut p = self.hier.root();
p.push(self.path.clone());
return fs::remove_dir(p).map_err(|e| Error::with_cause(RemoveFailed, e));
}
return Ok(());
}
self.subsystems.iter().try_for_each(|sub| match sub {
Subsystem::Pid(pidc) => pidc.delete(),
Subsystem::Mem(c) => c.delete(),
Subsystem::CpuSet(c) => c.delete(),
Subsystem::CpuAcct(c) => c.delete(),
Subsystem::Cpu(c) => c.delete(),
Subsystem::Devices(c) => c.delete(),
Subsystem::Freezer(c) => c.delete(),
Subsystem::NetCls(c) => c.delete(),
Subsystem::BlkIo(c) => c.delete(),
Subsystem::PerfEvent(c) => c.delete(),
Subsystem::NetPrio(c) => c.delete(),
Subsystem::HugeTlb(c) => c.delete(),
Subsystem::Rdma(c) => c.delete(),
Subsystem::Systemd(c) => c.delete(),
})
}
/// Apply a set of resource limits to the control group.
pub fn apply(&self, res: &Resources) -> Result<()> {
self.subsystems
.iter()
.try_fold((), |_, e| e.to_controller().apply(res))
}
/// Retrieve a container based on type inference.
///
/// ## Example:
///
/// ```text
/// let pids: &PidController = control_group.controller_of()
/// .expect("No pids controller attached!");
/// let cpu: &CpuController = control_group.controller_of()
/// .expect("No cpu controller attached!");
/// ```
pub fn controller_of<'a, T>(&'a self) -> Option<&'a T>
where
&'a T: From<&'a Subsystem>,
T: Controller + ControllIdentifier,
{
for i in &self.subsystems {
if i.to_controller().control_type() == T::controller_type() {
// N.B.:
// https://play.rust-lang.org/?gist=978b2846bacebdaa00be62374f4f4334&version=stable&mode=debug&edition=2015
return Some(i.into());
}
}
None
}
/// Removes tasks from the control group by thread group id.
///
/// Note that this means that the task will be moved back to the root control group in the
/// hierarchy and any rules applied to that control group will _still_ apply to the proc.
pub fn remove_task_by_tgid(&self, tgid: CgroupPid) -> Result<()> {
self.hier.root_control_group().add_task_by_tgid(tgid)
}
/// Removes a task from the control group.
///
/// Note that this means that the task will be moved back to the root control group in the
/// hierarchy and any rules applied to that control group will _still_ apply to the task.
pub fn remove_task(&self, tid: CgroupPid) -> Result<()> {
self.hier.root_control_group().add_task(tid)
}
/// Moves tasks to the parent control group by thread group id.
pub fn move_task_to_parent_by_tgid(&self, tgid: CgroupPid) -> Result<()> {
self.hier
.parent_control_group(&self.path)
.add_task_by_tgid(tgid)
}
/// Moves a task to the parent control group.
pub fn move_task_to_parent(&self, tid: CgroupPid) -> Result<()> {
self.hier.parent_control_group(&self.path).add_task(tid)
}
/// Return a handle to the parent control group in the hierarchy.
pub fn parent_control_group(&self) -> Cgroup {
self.hier.parent_control_group(&self.path)
}
/// Kill every process in the control group. Only supported for v2 cgroups and on
/// kernels 5.14+. This will fail with InvalidOperation if the 'cgroup.kill' file does
/// not exist.
pub fn kill(&self) -> Result<()> {
if !self.v2() {
return Err(Error::new(CgroupVersion));
}
let val = "1";
let file_name = "cgroup.kill";
let p = self.hier.root().join(self.path.clone()).join(file_name);
// If cgroup.kill doesn't exist they're not on 5.14+ so lets
// surface some error the caller can check against.
if !p.exists() {
return Err(Error::new(InvalidOperation));
}
fs::write(p, val)
.map_err(|e| Error::with_cause(WriteFailed(file_name.to_string(), val.to_string()), e))
}
/// Attach a task to the control group.
pub fn add_task(&self, tid: CgroupPid) -> Result<()> {
if self.v2() {
let subsystems = self.subsystems();
if !subsystems.is_empty() {
let c = subsystems[0].to_controller();
let cgroup_type = self.get_cgroup_type()?;
// In cgroup v2, writing to the cgroup.threads file is only supported in thread mode.
if cgroup_type == *CGROUP_MODE_DOMAIN_THREADED
|| cgroup_type == *CGROUP_MODE_THREADED
{
// It is used to move the threads of a process into a cgroup in thread mode.
c.add_task(&tid)
} else {
// When the cgroup type is domain or domain invalid,
// cgroup.threads cannot be written.
Err(Error::new(CgroupMode))
}
} else {
Err(Error::new(SubsystemsEmpty))
}
} else {
self.subsystems()
.iter()
.try_for_each(|sub| sub.to_controller().add_task(&tid))
}
}
/// Attach tasks to the control group by thread group id.
pub fn add_task_by_tgid(&self, tgid: CgroupPid) -> Result<()> {
if self.v2() {
let subsystems = self.subsystems();
if !subsystems.is_empty() {
let c = subsystems[0].to_controller();
// It is used to move a thread of the process to a cgroup,
// and other threads of the process will also move together.
c.add_task_by_tgid(&tgid)
} else {
Err(Error::new(SubsystemsEmpty))
}
} else {
self.subsystems()
.iter()
.try_for_each(|sub| sub.to_controller().add_task_by_tgid(&tgid))
}
}
/// set cgroup.type
pub fn set_cgroup_type(&self, cgroup_type: &str) -> Result<()> {
if self.v2() {
let subsystems = self.subsystems();
if !subsystems.is_empty() {
let c = subsystems[0].to_controller();
c.set_cgroup_type(cgroup_type)
} else {
Err(Error::new(SubsystemsEmpty))
}
} else {
Err(Error::new(CgroupVersion))
}
}
/// get cgroup.type
pub fn get_cgroup_type(&self) -> Result<String> {
if self.v2() {
let subsystems = self.subsystems();
if !subsystems.is_empty() {
let c = subsystems[0].to_controller();
let cgroup_type = c.get_cgroup_type()?;
Ok(cgroup_type)
} else {
Err(Error::new(SubsystemsEmpty))
}
} else {
Err(Error::new(CgroupVersion))
}
}
/// Set notify_on_release to the control group.
pub fn set_notify_on_release(&self, enable: bool) -> Result<()> {
self.subsystems()
.iter()
.try_for_each(|sub| sub.to_controller().set_notify_on_release(enable))
}
/// Set release_agent
pub fn set_release_agent(&self, path: &str) -> Result<()> {
self.hier
.root_control_group()
.subsystems()
.iter()
.try_for_each(|sub| sub.to_controller().set_release_agent(path))
}
/// Returns an Iterator that can be used to iterate over the procs that are currently in the
/// control group.
pub fn procs(&self) -> Vec<CgroupPid> {
// Collect the procs from all subsystems
let mut v = if self.v2() {
let subsystems = self.subsystems();
if !subsystems.is_empty() {
let c = subsystems[0].to_controller();
c.procs()
} else {
vec![]
}
} else {
self.subsystems()
.iter()
.map(|x| x.to_controller().procs())
.fold(vec![], |mut acc, mut x| {
acc.append(&mut x);
acc
})
};
v.sort();
v.dedup();
v
}
/// Returns an Iterator that can be used to iterate over the tasks that are currently in the
/// control group.
pub fn tasks(&self) -> Vec<CgroupPid> {
// Collect the tasks from all subsystems
let mut v = if self.v2() {
let subsystems = self.subsystems();
if !subsystems.is_empty() {
let c = subsystems[0].to_controller();
c.tasks()
} else {
vec![]
}
} else {
self.subsystems()
.iter()
.map(|x| x.to_controller().tasks())
.fold(vec![], |mut acc, mut x| {
acc.append(&mut x);
acc
})
};
v.sort();
v.dedup();
v
}
/// Checks if the cgroup exists.
///
/// Returns true if at least one subsystem exists.
pub fn exists(&self) -> bool {
self.subsystems().iter().any(|e| e.to_controller().exists())
}
}
pub const UNIFIED_MOUNTPOINT: &str = "/sys/fs/cgroup";
fn enable_controllers(controllers: &[String], path: &Path) {
let f = path.join("cgroup.subtree_control");
for c in controllers {
let body = format!("+{}", c);
let _rest = fs::write(f.as_path(), body.as_bytes());
}
}
fn supported_controllers() -> Vec<String> {
let p = format!("{}/{}", UNIFIED_MOUNTPOINT, "cgroup.controllers");
let ret = fs::read_to_string(p.as_str());
ret.unwrap_or_default()
.split(' ')
.map(|x| x.trim().to_string())
.collect::<Vec<String>>()
}
fn create_v2_cgroup(
root: PathBuf,
path: &str,
specified_controllers: &Option<Vec<String>>,
) -> Result<()> {
// controler list ["memory", "cpu"]
let controllers = if let Some(s_controllers) = specified_controllers.clone() {
if verify_supported_controllers(s_controllers.as_ref()) {
s_controllers
} else {
return Err(Error::new(ErrorKind::SpecifiedControllers));
}
} else {
supported_controllers()
};
let mut fp = root;
// enable for root
enable_controllers(&controllers, &fp);
// path: "a/b/c"
let elements = path.split('/').collect::<Vec<&str>>();
let last_index = elements.len() - 1;
for (i, ele) in elements.iter().enumerate() {
// ROOT/a
fp.push(ele);
// create dir, need not check if is a file or directory
if !fp.exists() {
if let Err(e) = std::fs::create_dir(fp.clone()) {
return Err(Error::with_cause(ErrorKind::FsError, e));
}
}
if i < last_index {
// enable controllers for substree
enable_controllers(&controllers, &fp);
}
}
Ok(())
}
pub fn verify_supported_controllers(controllers: &[String]) -> bool {
let sc = supported_controllers();
for controller in controllers.iter() {
if !sc.contains(controller) {
return false;
}
}
true
}
pub fn get_cgroups_relative_paths() -> Result<HashMap<String, String>> {
let path = "/proc/self/cgroup".to_string();
get_cgroups_relative_paths_by_path(path)
}
pub fn get_cgroups_relative_paths_by_pid(pid: u32) -> Result<HashMap<String, String>> {
let path = format!("/proc/{}/cgroup", pid);
get_cgroups_relative_paths_by_path(path)
}
fn get_cgroup_destination(mut mount_root: String, pidpath: String) -> String {
if mount_root == "/" {
mount_root = String::from("");
}
pidpath.trim_start_matches(&mount_root).to_string()
}
pub fn existing_path(paths: HashMap<String, String>) -> Result<HashMap<String, String>> {
let mount_roots_v1 = V1::new();
let mut mount_roots_subsystems_map = HashMap::new();
for s in mount_roots_v1.subsystems().iter() {
let controller_name = s.controller_name();
let path_from_cgroup = paths
.get(&controller_name)
.ok_or(Error::new(Common(format!(
"controller {} found in mountinfo, but not found in cgroup.",
controller_name
))))?;
let path_from_mountinfo = s.to_controller().base().to_string_lossy().to_string();
let des_path = get_cgroup_destination(path_from_mountinfo, path_from_cgroup.to_owned());
mount_roots_subsystems_map.insert(controller_name, des_path);
}
Ok(mount_roots_subsystems_map)
}
fn get_cgroups_relative_paths_by_path(path: String) -> Result<HashMap<String, String>> {
let mut m = HashMap::new();
let content =
fs::read_to_string(path.clone()).map_err(|e| Error::with_cause(ReadFailed(path), e))?;
// cgroup path may have ":" , likes
// "2:cpu,cpuacct:/system.slice/containerd.service/test.slice:cri-containerd:96b37a2edf84351487f42039e137427f1812f678850675fac214caf597ee5e4a"
for line in content.lines() {
if let Some((first_value_part, remaining_path)) =
line.split_once(':').unwrap_or_default().1.split_once(':')
{
let keys: Vec<&str> = first_value_part.split(',').collect();
keys.iter().for_each(|key| {
m.insert(key.to_string(), remaining_path.to_string());
});
}
}
Ok(m)
}
+405
View File
@@ -0,0 +1,405 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module allows the user to create a control group using the Builder pattern.
//! # Example
//!
//! The following example demonstrates how the control group builder looks like. The user
//! specifies the name of the control group (here: "hello") and the hierarchy it belongs to (here:
//! a V1 hierarchy). Next, the user selects a subsystem by calling functions like `memory()`,
//! `cpu()` and `devices()`. The user can then add restrictions and details via subsystem-specific
//! calls. To finalize a subsystem, the user may call `done()`. Finally, if the control group build
//! is done and all requirements/restrictions have been specified, the control group can be created
//! by a call to `build()`.
//!
//! ```rust,no_run
//! # use cgroups_rs::fs::*;
//! # use cgroups_rs::fs::devices::*;
//! # use cgroups_rs::fs::cgroup_builder::*;
//! let h = cgroups_rs::fs::hierarchies::auto();
//! let cgroup: Cgroup = CgroupBuilder::new("hello")
//! .memory()
//! .kernel_memory_limit(1024 * 1024)
//! .memory_hard_limit(1024 * 1024)
//! .done()
//! .cpu()
//! .shares(100)
//! .done()
//! .devices()
//! .device(1000, 10, DeviceType::Block, true,
//! vec![DevicePermissions::Read,
//! DevicePermissions::Write,
//! DevicePermissions::MkNod])
//! .device(6, 1, DeviceType::Char, false, vec![])
//! .done()
//! .network()
//! .class_id(1337)
//! .priority("eth0".to_string(), 100)
//! .priority("wl0".to_string(), 200)
//! .done()
//! .hugepages()
//! .limit("2M".to_string(), 0)
//! .limit("4M".to_string(), 4 * 1024 * 1024 * 100)
//! .limit("2G".to_string(), 2 * 1024 * 1024 * 1024)
//! .done()
//! .blkio()
//! .weight(123)
//! .leaf_weight(99)
//! .weight_device(6, 1, Some(100), Some(55))
//! .weight_device(6, 1, Some(100), Some(55))
//! .throttle_iops()
//! .read(6, 1, 10)
//! .write(11, 1, 100)
//! .throttle_bps()
//! .read(6, 1, 10)
//! .write(11, 1, 100)
//! .done()
//! .build(h).unwrap();
//! ```
use crate::fs::{
BlkIoDeviceResource, BlkIoDeviceThrottleResource, Cgroup, DeviceResource, Error, Hierarchy,
HugePageResource, MaxValue, NetworkPriority, Resources,
};
macro_rules! gen_setter {
($res:ident, $cont:ident, $func:ident, $name:ident, $ty:ty) => {
/// See the similarly named function in the respective controller.
pub fn $name(mut self, $name: $ty) -> Self {
self.cgroup.resources.$res.$name = Some($name);
self
}
};
}
/// A control group builder instance
pub struct CgroupBuilder {
name: String,
/// Internal, unsupported field: use the associated builders instead.
resources: Resources,
/// List of controllers specifically enabled in the control group.
specified_controllers: Option<Vec<String>>,
}
impl CgroupBuilder {
/// Start building a control group with the supplied hierarchy and name pair.
///
/// Note that this does not actually create the control group until `build()` is called.
pub fn new(name: &str) -> CgroupBuilder {
CgroupBuilder {
name: name.to_owned(),
resources: Resources::default(),
specified_controllers: None,
}
}
/// Builds the memory resources of the control group.
pub fn memory(self) -> MemoryResourceBuilder {
MemoryResourceBuilder { cgroup: self }
}
/// Builds the pid resources of the control group.
pub fn pid(self) -> PidResourceBuilder {
PidResourceBuilder { cgroup: self }
}
/// Builds the cpu resources of the control group.
pub fn cpu(self) -> CpuResourceBuilder {
CpuResourceBuilder { cgroup: self }
}
/// Builds the devices resources of the control group, disallowing or
/// allowing access to certain devices in the system.
pub fn devices(self) -> DeviceResourceBuilder {
DeviceResourceBuilder { cgroup: self }
}
/// Builds the network resources of the control group, setting class id, or
/// various priorities on networking interfaces.
pub fn network(self) -> NetworkResourceBuilder {
NetworkResourceBuilder { cgroup: self }
}
/// Builds the hugepage/hugetlb resources available to the control group.
pub fn hugepages(self) -> HugepagesResourceBuilder {
HugepagesResourceBuilder { cgroup: self }
}
/// Builds the block I/O resources available for the control group.
pub fn blkio(self) -> BlkIoResourcesBuilder {
BlkIoResourcesBuilder {
cgroup: self,
throttling_iops: false,
}
}
/// Finalize the control group, consuming the builder and creating the control group.
pub fn build(self, hier: Box<dyn Hierarchy>) -> Result<Cgroup, Error> {
if let Some(controllers) = self.specified_controllers {
let cg = Cgroup::new_with_specified_controllers(hier, self.name, Some(controllers))?;
cg.apply(&self.resources)?;
Ok(cg)
} else {
let cg = Cgroup::new(hier, self.name)?;
cg.apply(&self.resources)?;
Ok(cg)
}
}
/// Specifically enable some controllers in the control group.
pub fn set_specified_controllers(mut self, specified_controllers: Vec<String>) -> Self {
self.specified_controllers = Some(specified_controllers);
self
}
}
/// A builder that configures the memory controller of a control group.
pub struct MemoryResourceBuilder {
cgroup: CgroupBuilder,
}
impl MemoryResourceBuilder {
gen_setter!(
memory,
MemController,
set_kmem_limit,
kernel_memory_limit,
i64
);
gen_setter!(memory, MemController, set_limit, memory_hard_limit, i64);
gen_setter!(
memory,
MemController,
set_soft_limit,
memory_soft_limit,
i64
);
gen_setter!(
memory,
MemController,
set_tcp_limit,
kernel_tcp_memory_limit,
i64
);
gen_setter!(
memory,
MemController,
set_memswap_limit,
memory_swap_limit,
i64
);
gen_setter!(memory, MemController, set_swappiness, swappiness, u64);
/// Finish the construction of the memory resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
/// A builder that configures the pid controller of a control group.
pub struct PidResourceBuilder {
cgroup: CgroupBuilder,
}
impl PidResourceBuilder {
gen_setter!(
pid,
PidController,
set_pid_max,
maximum_number_of_processes,
MaxValue
);
/// Finish the construction of the pid resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
/// A builder that configures the cpuset & cpu controllers of a control group.
pub struct CpuResourceBuilder {
cgroup: CgroupBuilder,
}
impl CpuResourceBuilder {
gen_setter!(cpu, CpuSetController, set_cpus, cpus, String);
gen_setter!(cpu, CpuSetController, set_mems, mems, String);
gen_setter!(cpu, CpuController, set_shares, shares, u64);
gen_setter!(cpu, CpuController, set_cfs_quota, quota, i64);
gen_setter!(cpu, CpuController, set_cfs_period, period, u64);
gen_setter!(cpu, CpuController, set_rt_runtime, realtime_runtime, i64);
gen_setter!(cpu, CpuController, set_rt_period, realtime_period, u64);
/// Finish the construction of the cpu resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
/// A builder that configures the devices controller of a control group.
pub struct DeviceResourceBuilder {
cgroup: CgroupBuilder,
}
impl DeviceResourceBuilder {
/// Restrict (or allow) a device to the tasks inside the control group.
pub fn device(
mut self,
major: i64,
minor: i64,
devtype: crate::fs::devices::DeviceType,
allow: bool,
access: Vec<crate::fs::devices::DevicePermissions>,
) -> DeviceResourceBuilder {
self.cgroup.resources.devices.devices.push(DeviceResource {
allow,
devtype,
major,
minor,
access,
});
self
}
/// Finish the construction of the devices resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
/// A builder that configures the net_cls & net_prio controllers of a control group.
pub struct NetworkResourceBuilder {
cgroup: CgroupBuilder,
}
impl NetworkResourceBuilder {
gen_setter!(network, NetclsController, set_class, class_id, u64);
/// Set the priority of the tasks when operating on a networking device defined by `name` to be
/// `priority`.
pub fn priority(mut self, name: String, priority: u64) -> NetworkResourceBuilder {
self.cgroup
.resources
.network
.priorities
.push(NetworkPriority { name, priority });
self
}
/// Finish the construction of the network resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
/// A builder that configures the hugepages controller of a control group.
pub struct HugepagesResourceBuilder {
cgroup: CgroupBuilder,
}
impl HugepagesResourceBuilder {
/// Limit the usage of certain hugepages (determined by `size`) to be at most `limit` bytes.
pub fn limit(mut self, size: String, limit: u64) -> HugepagesResourceBuilder {
self.cgroup
.resources
.hugepages
.limits
.push(HugePageResource { size, limit });
self
}
/// Finish the construction of the network resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
/// A builder that configures the blkio controller of a control group.
pub struct BlkIoResourcesBuilder {
cgroup: CgroupBuilder,
throttling_iops: bool,
}
impl BlkIoResourcesBuilder {
gen_setter!(blkio, BlkIoController, set_weight, weight, u16);
gen_setter!(blkio, BlkIoController, set_leaf_weight, leaf_weight, u16);
/// Set the weight of a certain device.
pub fn weight_device(
mut self,
major: u64,
minor: u64,
weight: Option<u16>,
leaf_weight: Option<u16>,
) -> BlkIoResourcesBuilder {
self.cgroup
.resources
.blkio
.weight_device
.push(BlkIoDeviceResource {
major,
minor,
weight,
leaf_weight,
});
self
}
/// Start configuring the I/O operations per second metric.
pub fn throttle_iops(mut self) -> BlkIoResourcesBuilder {
self.throttling_iops = true;
self
}
/// Start configuring the bytes per second metric.
pub fn throttle_bps(mut self) -> BlkIoResourcesBuilder {
self.throttling_iops = false;
self
}
/// Limit the read rate of the current metric for a certain device.
pub fn read(mut self, major: u64, minor: u64, rate: u64) -> BlkIoResourcesBuilder {
let throttle = BlkIoDeviceThrottleResource { major, minor, rate };
if self.throttling_iops {
self.cgroup
.resources
.blkio
.throttle_read_iops_device
.push(throttle);
} else {
self.cgroup
.resources
.blkio
.throttle_read_bps_device
.push(throttle);
}
self
}
/// Limit the write rate of the current metric for a certain device.
pub fn write(mut self, major: u64, minor: u64, rate: u64) -> BlkIoResourcesBuilder {
let throttle = BlkIoDeviceThrottleResource { major, minor, rate };
if self.throttling_iops {
self.cgroup
.resources
.blkio
.throttle_write_iops_device
.push(throttle);
} else {
self.cgroup
.resources
.blkio
.throttle_write_bps_device
.push(throttle);
}
self
}
/// Finish the construction of the blkio resources of a control group.
pub fn done(self) -> CgroupBuilder {
self.cgroup
}
}
+320
View File
@@ -0,0 +1,320 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `cpu` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/scheduler/sched-design-CFS.txt](https://www.kernel.org/doc/Documentation/scheduler/sched-design-CFS.txt)
//! paragraph 7 ("GROUP SCHEDULER EXTENSIONS TO CFS").
use std::fs::File;
use std::io::{Read, Write};
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{parse_max_value, read_i64_from, read_u64_from};
use crate::fs::{
ControllIdentifier, ControllerInternal, Controllers, CpuResources, CustomizedAttribute,
MaxValue, Resources, Subsystem,
};
/// A controller that allows controlling the `cpu` subsystem of a Cgroup.
///
/// In essence, it allows gathering information about how much the tasks inside the control group
/// are using the CPU and creating rules that limit their usage. Note that this crate does not yet
/// support managing realtime tasks.
#[derive(Debug, Clone)]
pub struct CpuController {
base: PathBuf,
path: PathBuf,
v2: bool,
}
/// The current state of the control group and its processes.
#[derive(Debug)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct Cpu {
/// Reports CPU time statistics.
///
/// Corresponds the `cpu.stat` file in `cpu` control group.
pub stat: String,
}
/// The current state of the control group and its processes.
#[derive(Debug)]
struct CfsQuotaAndPeriod {
quota: MaxValue,
period: u64,
}
impl ControllerInternal for CpuController {
fn control_type(&self) -> Controllers {
Controllers::Cpu
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn is_v2(&self) -> bool {
self.v2
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &CpuResources = &res.cpu;
update_and_test!(self, set_shares, res.shares, shares);
update_and_test!(self, set_cfs_period, res.period, cfs_period);
update_and_test!(self, set_cfs_quota, res.quota, cfs_quota);
res.attrs.iter().for_each(|(k, v)| {
let _ = self.set(k, v);
});
// TODO: rt properties (CONFIG_RT_GROUP_SCHED) are not yet supported
Ok(())
}
}
impl ControllIdentifier for CpuController {
fn controller_type() -> Controllers {
Controllers::Cpu
}
}
impl<'a> From<&'a Subsystem> for &'a CpuController {
fn from(sub: &'a Subsystem) -> &'a CpuController {
unsafe {
match sub {
Subsystem::Cpu(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl CpuController {
/// Contructs a new `CpuController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
Self {
base: root,
path: point,
v2,
}
}
/// Returns CPU time statistics based on the processes in the control group.
pub fn cpu(&self) -> Cpu {
Cpu {
stat: self
.open_path("cpu.stat", false)
.and_then(|mut file| {
let mut s = String::new();
let res = file.read_to_string(&mut s);
match res {
Ok(_) => Ok(s),
Err(e) => Err(Error::with_cause(ReadFailed("cpu.stat".to_string()), e)),
}
})
.unwrap_or_default(),
}
}
/// Configures the CPU bandwidth (in relative relation to other control groups and this control
/// group's parent).
///
/// For example, setting control group `A`'s `shares` to `100`, and control group `B`'s
/// `shares` to `200` ensures that control group `B` receives twice as much as CPU bandwidth.
/// (Assuming both `A` and `B` are of the same parent)
pub fn set_shares(&self, shares: u64) -> Result<()> {
let mut file_name = "cpu.shares";
if self.v2 {
file_name = "cpu.weight";
}
// NOTE: .CpuShares is not used here. Conversion is the caller's responsibility.
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(shares.to_string().as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), shares.to_string()), e)
})
})
}
/// Retrieve the CPU bandwidth that this control group (relative to other control groups and
/// this control group's parent) can use.
pub fn shares(&self) -> Result<u64> {
let mut file = "cpu.shares";
if self.v2 {
file = "cpu.weight";
}
self.open_path(file, false).and_then(read_u64_from)
}
/// Specify a period (when using the CFS scheduler) of time in microseconds for how often this
/// control group's access to the CPU should be reallocated.
pub fn set_cfs_period(&self, us: u64) -> Result<()> {
if self.v2 {
return self.set_cfs_quota_and_period(None, Some(us));
}
self.open_path("cpu.cfs_period_us", true)
.and_then(|mut file| {
file.write_all(us.to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("cpu.cfs_period_us".to_string(), us.to_string()),
e,
)
})
})
}
/// Retrieve the period of time of how often this cgroup's access to the CPU should be
/// reallocated in microseconds.
pub fn cfs_period(&self) -> Result<u64> {
if self.v2 {
let current_value = self
.open_path("cpu.max", false)
.and_then(parse_cfs_quota_and_period)?;
return Ok(current_value.period);
}
self.open_path("cpu.cfs_period_us", false)
.and_then(read_u64_from)
}
/// Specify a quota (when using the CFS scheduler) of time in microseconds for which all tasks
/// in this control group can run during one period (see: `set_cfs_period()`).
pub fn set_cfs_quota(&self, us: i64) -> Result<()> {
if self.v2 {
return self.set_cfs_quota_and_period(Some(us), None);
}
self.open_path("cpu.cfs_quota_us", true)
.and_then(|mut file| {
file.write_all(us.to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("cpu.cfs_quota_us".to_string(), us.to_string()),
e,
)
})
})
}
/// Retrieve the quota of time for which all tasks in this cgroup can run during one period, in
/// microseconds.
pub fn cfs_quota(&self) -> Result<i64> {
if self.v2 {
let current_value = self
.open_path("cpu.max", false)
.and_then(parse_cfs_quota_and_period)?;
return Ok(current_value.quota.to_i64());
}
self.open_path("cpu.cfs_quota_us", false)
.and_then(read_i64_from)
}
pub fn set_cfs_quota_and_period(&self, quota: Option<i64>, period: Option<u64>) -> Result<()> {
if !self.v2 {
if let Some(q) = quota {
self.set_cfs_quota(q)?;
}
if let Some(p) = period {
self.set_cfs_period(p)?;
}
return Ok(());
}
// https://www.kernel.org/doc/html/latest/admin-guide/cgroup-v2.html
// cpu.max
// A read-write two value file which exists on non-root cgroups. The default is “max 100000”.
// The maximum bandwidth limit. Its in the following format:
// $MAX $PERIOD
// which indicates that the group may consume upto $MAX in each $PERIOD duration.
// “max” for $MAX indicates no limit. If only one number is written, $MAX is updated.
let current_value = self
.open_path("cpu.max", false)
.and_then(parse_cfs_quota_and_period)?;
let new_quota = if let Some(q) = quota {
if q > 0 {
q.to_string()
} else {
"max".to_string()
}
} else {
current_value.quota.to_string()
};
let new_period = if let Some(p) = period {
p.to_string()
} else {
current_value.period.to_string()
};
let line = format!("{} {}", new_quota, new_period);
self.open_path("cpu.max", true).and_then(|mut file| {
file.write_all(line.as_ref())
.map_err(|e| Error::with_cause(WriteFailed("cpu.max".to_string(), line), e))
})
}
pub fn set_rt_runtime(&self, us: i64) -> Result<()> {
self.open_path("cpu.rt_runtime_us", true)
.and_then(|mut file| {
file.write_all(us.to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("cpu.rt_runtime_us".to_string(), us.to_string()),
e,
)
})
})
}
pub fn set_rt_period_us(&self, us: u64) -> Result<()> {
self.open_path("cpu.rt_period_us", true)
.and_then(|mut file| {
file.write_all(us.to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("cpu.rt_period_us".to_string(), us.to_string()),
e,
)
})
})
}
}
impl CustomizedAttribute for CpuController {}
fn parse_cfs_quota_and_period(mut file: File) -> Result<CfsQuotaAndPeriod> {
let mut content = String::new();
file.read_to_string(&mut content)
.map_err(|e| Error::with_cause(ReadFailed("cpu.max".to_string()), e))?;
let fields = content.trim().split(' ').collect::<Vec<&str>>();
if fields.len() != 2 {
return Err(Error::from_string(format!("invaild format: {}", content)));
}
let quota = parse_max_value(fields[0])?;
let period = fields[1]
.parse::<u64>()
.map_err(|e| Error::with_cause(ParseError, e))?;
Ok(CfsQuotaAndPeriod { quota, period })
}
+156
View File
@@ -0,0 +1,156 @@
// Copyright (c) 2018 Levente Kurusa
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `cpuacct` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/cpuacct.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/cpuacct.txt)
use std::io::Write;
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{read_string_from, read_u64_from};
use crate::fs::{ControllIdentifier, ControllerInternal, Controllers, Resources, Subsystem};
/// A controller that allows controlling the `cpuacct` subsystem of a Cgroup.
///
/// In essence, this control group provides accounting (hence the name `cpuacct`) for CPU usage of
/// the tasks in the control group.
#[derive(Debug, Clone)]
pub struct CpuAcctController {
base: PathBuf,
path: PathBuf,
}
/// Represents the statistics retrieved from the control group.
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct CpuAcct {
/// Divides the time used by the tasks into `user` time and `system` time.
pub stat: String,
/// Total CPU time (in nanoseconds) spent by the tasks.
pub usage: u64,
/// Total CPU time (in nanoseconds) spent by the tasks, broken down by CPU and by whether the
/// time spent is `user` time or `system` time.
///
/// An example is as follows:
/// ```text
/// cpu user system
/// 0 8348363768 0
/// 1 8324369100 0
/// 2 8598185449 0
/// 3 8648262473 0
/// ```
pub usage_all: String,
/// CPU time (in nanoseconds) spent by the tasks, broken down by each CPU.
/// Times spent in each CPU are separated by a space.
pub usage_percpu: String,
/// As for `usage_percpu`, but the `system` time spent.
pub usage_percpu_sys: String,
/// As for `usage_percpu`, but the `user` time spent.
pub usage_percpu_user: String,
/// CPU time (in nanoseconds) spent by the tasks that counted for `system` time.
pub usage_sys: u64,
/// CPU time (in nanoseconds) spent by the tasks that counted for `user` time.
pub usage_user: u64,
}
impl ControllerInternal for CpuAcctController {
fn control_type(&self) -> Controllers {
Controllers::CpuAcct
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, _res: &Resources) -> Result<()> {
Ok(())
}
}
impl ControllIdentifier for CpuAcctController {
fn controller_type() -> Controllers {
Controllers::CpuAcct
}
}
impl<'a> From<&'a Subsystem> for &'a CpuAcctController {
fn from(sub: &'a Subsystem) -> &'a CpuAcctController {
unsafe {
match sub {
Subsystem::CpuAcct(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl CpuAcctController {
/// Contructs a new `CpuAcctController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf) -> Self {
Self {
base: root,
path: point,
}
}
/// Gathers the statistics that are available in the control group into a `CpuAcct` structure.
pub fn cpuacct(&self) -> CpuAcct {
CpuAcct {
stat: self
.open_path("cpuacct.stat", false)
.and_then(read_string_from)
.unwrap_or_default(),
usage: self
.open_path("cpuacct.usage", false)
.and_then(read_u64_from)
.unwrap_or(0),
usage_all: self
.open_path("cpuacct.usage_all", false)
.and_then(read_string_from)
.unwrap_or_default(),
usage_percpu: self
.open_path("cpuacct.usage_percpu", false)
.and_then(read_string_from)
.unwrap_or_default(),
usage_percpu_sys: self
.open_path("cpuacct.usage_percpu_sys", false)
.and_then(read_string_from)
.unwrap_or_default(),
usage_percpu_user: self
.open_path("cpuacct.usage_percpu_user", false)
.and_then(read_string_from)
.unwrap_or_default(),
usage_sys: self
.open_path("cpuacct.usage_sys", false)
.and_then(read_u64_from)
.unwrap_or(0),
usage_user: self
.open_path("cpuacct.usage_user", false)
.and_then(read_u64_from)
.unwrap_or(0),
}
}
/// Reset the statistics the kernel has gathered about the control group.
pub fn reset(&self) -> Result<()> {
self.open_path("cpuacct.usage", true).and_then(|mut file| {
file.write_all(b"0").map_err(|e| {
Error::with_cause(WriteFailed("cpuacct.usage".to_string(), "0".to_string()), e)
})
})
}
}
+621
View File
@@ -0,0 +1,621 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `cpuset` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/cpusets.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/cpusets.txt)
use log::*;
use std::io::Write;
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{read_string_from, read_u64_from};
use crate::fs::{
ControllIdentifier, ControllerInternal, Controllers, CpuResources, Resources, Subsystem,
};
/// A controller that allows controlling the `cpuset` subsystem of a Cgroup.
///
/// In essence, this controller is responsible for restricting the tasks in the control group to a
/// set of CPUs and/or memory nodes.
#[derive(Debug, Clone)]
pub struct CpuSetController {
base: PathBuf,
path: PathBuf,
v2: bool,
}
/// The current state of the `cpuset` controller for this control group.
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct CpuSet {
/// If true, no other control groups can share the CPUs listed in the `cpus` field.
pub cpu_exclusive: bool,
/// The list of CPUs the tasks of the control group can run on.
///
/// This is a vector of `(start, end)` tuples, where each tuple is a range of CPUs where the
/// control group is allowed to run on. Both sides of the range are inclusive.
pub cpus: Vec<(u64, u64)>,
/// The list of CPUs that the tasks can effectively run on. This removes the list of CPUs that
/// the parent (and all of its parents) cannot run on from the `cpus` field of this control
/// group.
pub effective_cpus: Vec<(u64, u64)>,
/// The list of memory nodes that the tasks can effectively use. This removes the list of nodes that
/// the parent (and all of its parents) cannot use from the `mems` field of this control
/// group.
pub effective_mems: Vec<(u64, u64)>,
/// If true, no other control groups can share the memory nodes listed in the `mems` field.
pub mem_exclusive: bool,
/// If true, the control group is 'hardwalled'. Kernel memory allocations (except for a few
/// minor exceptions) are made from the memory nodes designated in the `mems` field.
pub mem_hardwall: bool,
/// If true, whenever `mems` is changed via `set_mems()`, the memory stored on the previous
/// nodes are migrated to the new nodes selected by the new `mems`.
pub memory_migrate: bool,
/// Running average of the memory pressured faced by the tasks in the control group.
pub memory_pressure: u64,
/// This field is only at the root control group and controls whether the kernel will compute
/// the memory pressure for control groups or not.
pub memory_pressure_enabled: Option<bool>,
/// If true, filesystem buffers are spread across evenly between the nodes specified in `mems`.
pub memory_spread_page: bool,
/// If true, kernel slab caches for file I/O are spread across evenly between the nodes
/// specified in `mems`.
pub memory_spread_slab: bool,
/// The list of memory nodes the tasks of the control group can use.
///
/// The format is the same as the `cpus`, `effective_cpus` and `effective_mems` fields.
pub mems: Vec<(u64, u64)>,
/// If true, the kernel will attempt to rebalance the load between the CPUs specified in the
/// `cpus` field of this control group.
pub sched_load_balance: bool,
/// Represents how much work the kernel should do to rebalance this cpuset.
///
/// | `sched_load_balance` | Effect |
/// | -------------------- | ------ |
/// | -1 | Use the system default value |
/// | 0 | Only balance loads periodically |
/// | 1 | Immediately balance the load across tasks on the same core |
/// | 2 | Immediately balance the load across cores in the same CPU package |
/// | 4 | Immediately balance the load across CPUs on the same node |
/// | 5 | Immediately balance the load between CPUs even if the system is NUMA |
/// | 6 | Immediately balance the load between all CPUs |
pub sched_relax_domain_level: u64,
}
impl ControllerInternal for CpuSetController {
fn control_type(&self) -> Controllers {
Controllers::CpuSet
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn is_v2(&self) -> bool {
self.v2
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &CpuResources = &res.cpu;
update!(self, set_cpus, res.cpus.as_ref());
update!(self, set_mems, res.mems.as_ref());
Ok(())
}
fn post_create(&self) {
if self.is_v2() {
return;
}
let current = self.get_path();
if current != self.get_base() {
match copy_from_parent(current.to_str().unwrap(), "cpuset.cpus") {
Ok(_) => (),
Err(err) => error!("error create_dir for cpuset.cpus {:?}", err),
}
match copy_from_parent(current.to_str().unwrap(), "cpuset.mems") {
Ok(_) => (),
Err(err) => error!("error create_dir for cpuset.mems {:?}", err),
}
}
}
}
fn find_no_empty_parent(from: &str, file: &str) -> Result<(String, Vec<PathBuf>)> {
let mut current_path = ::std::path::Path::new(from).to_path_buf();
let mut v = vec![];
loop {
let current_value =
match ::std::fs::read_to_string(current_path.clone().join(file).to_str().unwrap()) {
Ok(cpus) => String::from(cpus.trim()),
Err(e) => {
return Err(Error::with_cause(
ReadFailed(current_path.display().to_string()),
e,
))
}
};
if !current_value.is_empty() {
return Ok((current_value, v));
}
v.push(current_path.clone());
let parent = match current_path.parent() {
Some(p) => p,
None => return Ok(("".to_string(), v)),
};
// next loop, find parent
current_path = parent.to_path_buf();
}
}
/// copy_from_parent copy the cpuset.cpus and cpuset.mems from the parent
/// directory to the current directory if the file's contents are 0
fn copy_from_parent(current: &str, file: &str) -> Result<()> {
// find not empty cpus/memes from current directory.
let (value, parents) = find_no_empty_parent(current, file)?;
if value.is_empty() || parents.is_empty() {
return Ok(());
}
for p in parents.iter().rev() {
let mut pb = p.clone();
pb.push(file);
match ::std::fs::write(pb.to_str().unwrap(), value.as_bytes()) {
Ok(_) => (),
Err(e) => {
return Err(Error::with_cause(
WriteFailed(pb.display().to_string(), pb.display().to_string()),
e,
))
}
}
}
Ok(())
}
impl ControllIdentifier for CpuSetController {
fn controller_type() -> Controllers {
Controllers::CpuSet
}
}
impl<'a> From<&'a Subsystem> for &'a CpuSetController {
fn from(sub: &'a Subsystem) -> &'a CpuSetController {
unsafe {
match sub {
Subsystem::CpuSet(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
/// Parse a string like "1,2,4-5,8" into a list of (start, end) tuples.
fn parse_range(s: String) -> Result<Vec<(u64, u64)>> {
let mut fin = Vec::new();
if s.is_empty() {
return Ok(fin);
}
// first split by commas
let comma_split = s.split(',');
for sp in comma_split {
if sp.contains('-') {
// this is a true range
let dash_split = sp.split('-').collect::<Vec<_>>();
if dash_split.len() != 2 {
return Err(Error::new(ParseError));
}
let first = dash_split[0].parse::<u64>();
let second = dash_split[1].parse::<u64>();
if first.is_err() || second.is_err() {
return Err(Error::new(ParseError));
}
fin.push((first.unwrap(), second.unwrap()));
} else {
// this is just a single number
let num = sp.parse::<u64>();
if num.is_err() {
return Err(Error::new(ParseError));
}
fin.push((num.clone().unwrap(), num.clone().unwrap()));
}
}
Ok(fin)
}
impl CpuSetController {
/// Contructs a new `CpuSetController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
Self {
base: root,
path: point,
v2,
}
}
/// Returns the statistics gathered by the kernel for this control group. See the struct for
/// more information on what information this entails.
pub fn cpuset(&self) -> CpuSet {
CpuSet {
cpu_exclusive: {
self.open_path("cpuset.cpu_exclusive", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
cpus: {
self.open_path("cpuset.cpus", false)
.and_then(read_string_from)
.and_then(parse_range)
.unwrap_or_default()
},
effective_cpus: {
self.open_path("cpuset.effective_cpus", false)
.and_then(read_string_from)
.and_then(parse_range)
.unwrap_or_default()
},
effective_mems: {
self.open_path("cpuset.effective_mems", false)
.and_then(read_string_from)
.and_then(parse_range)
.unwrap_or_default()
},
mem_exclusive: {
self.open_path("cpuset.mem_exclusive", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
mem_hardwall: {
self.open_path("cpuset.mem_hardwall", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
memory_migrate: {
self.open_path("cpuset.memory_migrate", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
memory_pressure: {
self.open_path("cpuset.memory_pressure", false)
.and_then(read_u64_from)
.unwrap_or(0)
},
memory_pressure_enabled: {
self.open_path("cpuset.memory_pressure_enabled", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.ok()
},
memory_spread_page: {
self.open_path("cpuset.memory_spread_page", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
memory_spread_slab: {
self.open_path("cpuset.memory_spread_slab", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
mems: {
self.open_path("cpuset.mems", false)
.and_then(read_string_from)
.and_then(parse_range)
.unwrap_or_default()
},
sched_load_balance: {
self.open_path("cpuset.sched_load_balance", false)
.and_then(read_u64_from)
.map(|x| x == 1)
.unwrap_or(false)
},
sched_relax_domain_level: {
self.open_path("cpuset.sched_relax_domain_level", false)
.and_then(read_u64_from)
.unwrap_or(0)
},
}
}
/// Control whether the CPUs selected via `set_cpus()` should be exclusive to this control
/// group or not.
pub fn set_cpu_exclusive(&self, b: bool) -> Result<()> {
self.open_path("cpuset.cpu_exclusive", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.cpu_exclusive".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.cpu_exclusive".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Control whether the memory nodes selected via `set_memss()` should be exclusive to this control
/// group or not.
pub fn set_mem_exclusive(&self, b: bool) -> Result<()> {
self.open_path("cpuset.mem_exclusive", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.mem_exclusive".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.mem_exclusive".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Set the CPUs that the tasks in this control group can run on.
///
/// Syntax is a comma separated list of CPUs, with an additional extension that ranges can
/// be represented via dashes.
pub fn set_cpus(&self, cpus: &str) -> Result<()> {
self.open_path("cpuset.cpus", true).and_then(|mut file| {
file.write_all(cpus.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed("cpuset.cpus".to_string(), cpus.to_string()), e)
})
})
}
/// Set the memory nodes that the tasks in this control group can use.
///
/// Syntax is the same as with `set_cpus()`.
pub fn set_mems(&self, mems: &str) -> Result<()> {
self.open_path("cpuset.mems", true).and_then(|mut file| {
file.write_all(mems.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed("cpuset.mems".to_string(), mems.to_string()), e)
})
})
}
/// Controls whether the control group should be "hardwalled", i.e., whether kernel allocations
/// should exclusively use the memory nodes set via `set_mems()`.
///
/// Note that some kernel allocations, most notably those that are made in interrupt handlers
/// may disregard this.
pub fn set_hardwall(&self, b: bool) -> Result<()> {
self.open_path("cpuset.mem_hardwall", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.mem_hardwall".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.mem_hardwall".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Controls whether the kernel should attempt to rebalance the load between the CPUs specified in the
/// `cpus` field of this control group.
pub fn set_load_balancing(&self, b: bool) -> Result<()> {
self.open_path("cpuset.sched_load_balance", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.sched_load_balance".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.sched_load_balance".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Contorl how much effort the kernel should invest in rebalacing the control group.
///
/// See @CpuSet 's similar field for more information.
pub fn set_rebalance_relax_domain_level(&self, i: i64) -> Result<()> {
self.open_path("cpuset.sched_relax_domain_level", true)
.and_then(|mut file| {
file.write_all(i.to_string().as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.sched_relax_domain_level".to_string(), i.to_string()),
e,
)
})
})
}
/// Control whether when using `set_mems()` the existing memory used by the tasks should be
/// migrated over to the now-selected nodes.
pub fn set_memory_migration(&self, b: bool) -> Result<()> {
self.open_path("cpuset.memory_migrate", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.memory_migrate".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.memory_migrate".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Control whether filesystem buffers should be evenly split across the nodes selected via
/// `set_mems()`.
pub fn set_memory_spread_page(&self, b: bool) -> Result<()> {
self.open_path("cpuset.memory_spread_page", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.memory_spread_page".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.memory_spread_page".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Control whether the kernel's slab cache for file I/O should be evenly split across the
/// nodes selected via `set_mems()`.
pub fn set_memory_spread_slab(&self, b: bool) -> Result<()> {
self.open_path("cpuset.memory_spread_slab", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.memory_spread_slab".to_string(), "1".to_string()),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed("cpuset.memory_spread_slab".to_string(), "0".to_string()),
e,
)
})
}
})
}
/// Control whether the kernel should collect information to calculate memory pressure for
/// control groups.
///
/// Note: This will fail with `InvalidOperation` if the current congrol group is not the root
/// control group.
pub fn set_enable_memory_pressure(&self, b: bool) -> Result<()> {
if !self.path_exists("cpuset.memory_pressure_enabled") {
return Err(Error::new(InvalidOperation));
}
self.open_path("cpuset.memory_pressure_enabled", true)
.and_then(|mut file| {
if b {
file.write_all(b"1").map_err(|e| {
Error::with_cause(
WriteFailed(
"cpuset.memory_pressure_enabled".to_string(),
"1".to_string(),
),
e,
)
})
} else {
file.write_all(b"0").map_err(|e| {
Error::with_cause(
WriteFailed(
"cpuset.memory_pressure_enabled".to_string(),
"0".to_string(),
),
e,
)
})
}
})
}
}
#[cfg(test)]
mod tests {
use crate::fs::cpuset;
#[test]
fn test_parse_range() {
let test_cases = vec![
"1,2,4-6,9".to_string(),
"".to_string(),
"1".to_string(),
"1-111".to_string(),
"1,2,3,4".to_string(),
"1-5,6-7,8-9".to_string(),
];
let expecteds = [
vec![(1, 1), (2, 2), (4, 6), (9, 9)],
vec![],
vec![(1, 1)],
vec![(1, 111)],
vec![(1, 1), (2, 2), (3, 3), (4, 4)],
vec![(1, 5), (6, 7), (8, 9)],
];
for (i, case) in test_cases.into_iter().enumerate() {
let range = cpuset::parse_range(case.clone());
println!("{:?} => {:?}", case, range);
assert!(range.is_ok());
assert_eq!(range.unwrap(), expecteds[i]);
}
}
}
+342
View File
@@ -0,0 +1,342 @@
// Copyright (c) 2018 Levente Kurusa
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `devices` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/devices.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/devices.txt)
use std::io::{Read, Write};
use std::path::PathBuf;
use log::*;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{
ControllIdentifier, ControllerInternal, Controllers, DeviceResource, DeviceResources,
Resources, Subsystem,
};
/// A controller that allows controlling the `devices` subsystem of a Cgroup.
///
/// In essence, using the devices controller, it is possible to allow or disallow sets of devices to
/// be used by the control group's tasks.
#[derive(Debug, Clone)]
pub struct DevicesController {
base: PathBuf,
path: PathBuf,
}
/// An enum holding the different types of devices that can be manipulated using this controller.
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
#[cfg_attr(
feature = "serde",
derive(serde::Serialize, serde::Deserialize),
serde(rename_all = "snake_case")
)]
pub enum DeviceType {
/// The rule applies to all devices.
All,
/// The rule only applies to character devices.
Char,
/// The rule only applies to block devices.
Block,
}
#[allow(clippy::derivable_impls)]
impl Default for DeviceType {
fn default() -> Self {
DeviceType::All
}
}
impl DeviceType {
/// Convert a DeviceType into the character that the kernel recognizes.
#[allow(clippy::should_implement_trait, clippy::wrong_self_convention)]
pub fn to_char(&self) -> char {
match self {
DeviceType::All => 'a',
DeviceType::Char => 'c',
DeviceType::Block => 'b',
}
}
/// Convert the kenrel's representation into the DeviceType type.
pub fn from_char(c: Option<char>) -> Option<DeviceType> {
match c {
Some('a') => Some(DeviceType::All),
Some('c') => Some(DeviceType::Char),
Some('b') => Some(DeviceType::Block),
_ => None,
}
}
}
/// An enum with the permissions that can be allowed/denied to the control group.
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
#[cfg_attr(
feature = "serde",
derive(serde::Serialize, serde::Deserialize),
serde(rename_all = "snake_case")
)]
pub enum DevicePermissions {
/// Permission to read from the device.
Read,
/// Permission to write to the device.
Write,
/// Permission to execute the `mknod(2)` system call with the device's major and minor numbers.
/// That is, the permission to create a special file that refers to the device node.
MkNod,
}
impl DevicePermissions {
/// Convert a DevicePermissions into the character that the kernel recognizes.
#[allow(clippy::should_implement_trait, clippy::wrong_self_convention)]
pub fn to_char(&self) -> char {
match self {
DevicePermissions::Read => 'r',
DevicePermissions::Write => 'w',
DevicePermissions::MkNod => 'm',
}
}
/// Convert a char to a DevicePermission if there is such a mapping.
pub fn from_char(c: char) -> Option<DevicePermissions> {
match c {
'r' => Some(DevicePermissions::Read),
'w' => Some(DevicePermissions::Write),
'm' => Some(DevicePermissions::MkNod),
_ => None,
}
}
/// Checks whether the string is a valid descriptor of DevicePermissions.
pub fn is_valid(s: &str) -> bool {
if s.is_empty() {
return false;
}
for i in s.chars() {
if i != 'r' && i != 'w' && i != 'm' {
return false;
}
}
true
}
/// Returns a Vec will all the permissions that a device can have.
pub fn all() -> Vec<DevicePermissions> {
vec![
DevicePermissions::Read,
DevicePermissions::Write,
DevicePermissions::MkNod,
]
}
/// Convert a string into DevicePermissions.
#[allow(clippy::should_implement_trait)]
pub fn from_str(s: &str) -> Result<Vec<DevicePermissions>> {
let mut v = Vec::new();
if s.is_empty() {
return Ok(v);
}
for e in s.chars() {
let perm = DevicePermissions::from_char(e).ok_or_else(|| Error::new(ParseError))?;
v.push(perm);
}
Ok(v)
}
}
impl ControllerInternal for DevicesController {
fn control_type(&self) -> Controllers {
Controllers::Devices
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &DeviceResources = &res.devices;
for i in &res.devices {
if i.allow {
self.allow_device(i.devtype, i.major, i.minor, &i.access)?;
} else {
self.deny_device(i.devtype, i.major, i.minor, &i.access)?;
}
}
Ok(())
}
}
impl ControllIdentifier for DevicesController {
fn controller_type() -> Controllers {
Controllers::Devices
}
}
impl<'a> From<&'a Subsystem> for &'a DevicesController {
fn from(sub: &'a Subsystem) -> &'a DevicesController {
unsafe {
match sub {
Subsystem::Devices(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl DevicesController {
/// Constructs a new `DevicesController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf) -> Self {
Self {
base: root,
path: point,
}
}
/// Allow a (possibly, set of) device(s) to be used by the tasks in the control group.
///
/// When `-1` is passed as `major` or `minor`, the kernel interprets that value as "any",
/// meaning that it will match any device.
pub fn allow_device(
&self,
devtype: DeviceType,
major: i64,
minor: i64,
perm: &[DevicePermissions],
) -> Result<()> {
let perms = perm
.iter()
.map(DevicePermissions::to_char)
.collect::<String>();
let minor = if minor == -1 {
"*".to_string()
} else {
format!("{}", minor)
};
let major = if major == -1 {
"*".to_string()
} else {
format!("{}", major)
};
let final_str = format!("{} {}:{} {}", devtype.to_char(), major, minor, perms);
self.open_path("devices.allow", true).and_then(|mut file| {
file.write_all(final_str.as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed(
self.get_path().join("devices.allow").display().to_string(),
final_str,
),
e,
)
})
})
}
/// Deny the control group's tasks access to the devices covered by `dev`.
///
/// When `-1` is passed as `major` or `minor`, the kernel interprets that value as "any",
/// meaning that it will match any device.
pub fn deny_device(
&self,
devtype: DeviceType,
major: i64,
minor: i64,
perm: &[DevicePermissions],
) -> Result<()> {
let perms = perm
.iter()
.map(DevicePermissions::to_char)
.collect::<String>();
let minor = if minor == -1 {
"*".to_string()
} else {
format!("{}", minor)
};
let major = if major == -1 {
"*".to_string()
} else {
format!("{}", major)
};
let final_str = format!("{} {}:{} {}", devtype.to_char(), major, minor, perms);
self.open_path("devices.deny", true).and_then(|mut file| {
file.write_all(final_str.as_ref()).map_err(|e| {
Error::with_cause(
WriteFailed(
self.get_path().join("devices.deny").display().to_string(),
final_str,
),
e,
)
})
})
}
/// Get the current list of allowed devices.
pub fn allowed_devices(&self) -> Result<Vec<DeviceResource>> {
self.open_path("devices.list", false).and_then(|mut file| {
let mut s = String::new();
let res = file.read_to_string(&mut s);
match res {
Ok(_) => s
.lines()
.map(|line| parse_device_line(line, true))
.collect(),
Err(e) => Err(Error::with_cause(ReadFailed("devices.list".to_string()), e)),
}
})
}
}
fn parse_device_number(s: &str) -> Result<i64> {
if s == "*" {
Ok(-1)
} else {
s.parse::<i64>().map_err(|_| Error::new(ParseError))
}
}
fn parse_device_line(line: &str, allow: bool) -> Result<DeviceResource> {
let parts: Vec<&str> = line.split([' ', ':']).collect();
if parts.len() != 4 {
error!("allowed_devices: invalid line format: {:?}", line);
return Err(Error::new(ParseError));
}
let devtype = DeviceType::from_char(parts[0].chars().next()).ok_or_else(|| {
error!("allowed_devices: invalid device type: {:?}", parts[0]);
Error::new(ParseError)
})?;
let major = parse_device_number(parts[1]).inspect_err(|_| {
error!("allowed_devices: invalid major number: {:?}", parts[1]);
})?;
let minor = parse_device_number(parts[2]).inspect_err(|_| {
error!("allowed_devices: invalid minor number: {:?}", parts[2]);
})?;
let access = DevicePermissions::from_str(parts[3])?;
Ok(DeviceResource {
allow,
devtype,
major,
minor,
access,
})
}
+127
View File
@@ -0,0 +1,127 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
use std::error::Error as StdError;
use std::fmt;
/// The different types of errors that can occur while manipulating control groups.
#[derive(thiserror::Error, Debug, Eq, PartialEq)]
pub enum ErrorKind {
#[error("fs error")]
FsError,
#[error("common error: {0}")]
Common(String),
/// An error occured while writing to a control group file.
#[error("unable to write to a control group file {0}, value {1}")]
WriteFailed(String, String),
/// An error occured while trying to read from a control group file.
#[error("unable to read a control group file {0}")]
ReadFailed(String),
/// An error occured while trying to remove a control group.
#[error("unable to remove a control group")]
RemoveFailed,
/// An error occured while trying to parse a value from a control group file.
///
/// In the future, there will be some information attached to this field.
#[error("unable to parse control group file")]
ParseError,
/// You tried to do something invalid.
///
/// This could be because you tried to set a value in a control group that is not a root
/// control group. Or, when using unified hierarchy, you tried to add a task in a leaf node.
#[error("the requested operation is invalid")]
InvalidOperation,
/// The path of the control group was invalid.
///
/// This could be caused by trying to escape the control group filesystem via a string of "..".
/// This crate checks against this and operations will fail with this error.
#[error("the given path is invalid")]
InvalidPath,
#[error("invalid bytes size")]
InvalidBytesSize,
/// The specified controller is not in the list of supported controllers.
#[error("specified controller is not in the list of supported controllers")]
SpecifiedControllers,
/// Using method in wrong cgroup version.
#[error("using method in wrong cgroup version")]
CgroupVersion,
/// Using method in wrong cgroup mode.
#[error("using method in wrong cgroup mode.")]
CgroupMode,
/// Subsystems is empty.
#[error("subsystems is empty")]
SubsystemsEmpty,
/// An unknown error has occured.
#[error("an unknown error")]
Other,
}
#[derive(Debug)]
pub struct Error {
kind: ErrorKind,
cause: Option<Box<dyn StdError + Send + Sync>>,
}
impl fmt::Display for Error {
fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
if let Some(cause) = &self.cause {
write!(f, "{} caused by: {:?}", &self.kind, cause)
} else {
write!(f, "{}", &self.kind)
}
}
}
impl StdError for Error {
fn source(&self) -> Option<&(dyn StdError + 'static)> {
#[allow(clippy::manual_map)]
match self.cause {
Some(ref x) => Some(&**x),
None => None,
}
}
}
impl Error {
pub(crate) fn from_string(s: String) -> Self {
Self {
kind: ErrorKind::Common(s),
cause: None,
}
}
pub(crate) fn new(kind: ErrorKind) -> Self {
Self { kind, cause: None }
}
pub(crate) fn with_cause<E>(kind: ErrorKind, cause: E) -> Self
where
E: 'static + Send + Sync + StdError,
{
Self {
kind,
cause: Some(Box::new(cause)),
}
}
pub fn kind(&self) -> &ErrorKind {
&self.kind
}
}
pub type Result<T> = ::std::result::Result<T, Error>;
+92
View File
@@ -0,0 +1,92 @@
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
use eventfd::{eventfd, EfdFlags};
use nix::sys::eventfd;
use std::fs::{self, File};
use std::io::Read;
use std::os::unix::io::{AsRawFd, FromRawFd};
use std::path::Path;
use std::sync::mpsc::{self, Receiver};
use std::thread;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
// notify_on_oom returns channel on which you can expect event about OOM,
// if process died without OOM this channel will be closed.
pub fn notify_on_oom_v2(key: &str, dir: &Path) -> Result<Receiver<String>> {
register_memory_event(key, dir, "memory.oom_control", "")
}
// notify_on_oom returns channel on which you can expect event about OOM,
// if process died without OOM this channel will be closed.
pub fn notify_on_oom_v1(key: &str, dir: &Path) -> Result<Receiver<String>> {
register_memory_event(key, dir, "memory.oom_control", "")
}
// level is one of "low", "medium", or "critical"
pub fn notify_memory_pressure(key: &str, dir: &Path, level: &str) -> Result<Receiver<String>> {
if level != "low" && level != "medium" && level != "critical" {
return Err(Error::from_string(format!(
"invalid pressure level {}",
level
)));
}
register_memory_event(key, dir, "memory.pressure_level", level)
}
fn register_memory_event(
key: &str,
cg_dir: &Path,
event_name: &str,
arg: &str,
) -> Result<Receiver<String>> {
let path = cg_dir.join(event_name);
let event_file = File::open(path.clone())
.map_err(|e| Error::with_cause(ReadFailed(path.display().to_string()), e))?;
let eventfd = eventfd(0, EfdFlags::EFD_CLOEXEC)
.map_err(|e| Error::with_cause(ReadFailed("eventfd".to_string()), e))?;
let event_control_path = cg_dir.join("cgroup.event_control");
let data = if arg.is_empty() {
format!("{} {}", eventfd, event_file.as_raw_fd())
} else {
format!("{} {} {}", eventfd, event_file.as_raw_fd(), arg)
};
// write to file and set mode to 0700(FIXME)
fs::write(&event_control_path, data.clone()).map_err(|e| {
Error::with_cause(
WriteFailed(event_control_path.display().to_string(), data),
e,
)
})?;
let mut eventfd_file = unsafe { File::from_raw_fd(eventfd) };
let (sender, receiver) = mpsc::channel();
let key = key.to_string();
thread::spawn(move || {
loop {
let mut buf = [0; 8];
if eventfd_file.read(&mut buf).is_err() {
return;
}
// When a cgroup is destroyed, an event is sent to eventfd.
// So if the control path is gone, return instead of notifying.
if !Path::new(&event_control_path).exists() {
return;
}
sender.send(key.clone()).unwrap();
}
});
Ok(receiver)
}
+145
View File
@@ -0,0 +1,145 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `freezer` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/freezer-subsystem.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/freezer-subsystem.txt)
use std::io::{Read, Write};
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{ControllIdentifier, ControllerInternal, Controllers, Resources, Subsystem};
/// A controller that allows controlling the `freezer` subsystem of a Cgroup.
///
/// In essence, this subsystem allows the user to freeze and thaw (== "un-freeze") the processes in
/// the control group. This is done _transparently_ so that neither the parent, nor the children of
/// the processes can observe the freeze.
///
/// Note that if the control group is currently in the `Frozen` or `Freezing` state, then no
/// processes can be added to it.
#[derive(Debug, Clone)]
pub struct FreezerController {
base: PathBuf,
path: PathBuf,
v2: bool,
}
/// The current state of the control group
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub enum FreezerState {
/// The processes in the control group are _not_ frozen.
Thawed,
/// The processes in the control group are in the processes of being frozen.
Freezing,
/// The processes in the control group are frozen.
Frozen,
}
impl ControllerInternal for FreezerController {
fn control_type(&self) -> Controllers {
Controllers::Freezer
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, _res: &Resources) -> Result<()> {
Ok(())
}
}
impl ControllIdentifier for FreezerController {
fn controller_type() -> Controllers {
Controllers::Freezer
}
}
impl<'a> From<&'a Subsystem> for &'a FreezerController {
fn from(sub: &'a Subsystem) -> &'a FreezerController {
unsafe {
match sub {
Subsystem::Freezer(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl FreezerController {
/// Contructs a new `FreezerController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
Self {
base: root,
path: point,
v2,
}
}
/// Freezes the processes in the control group.
pub fn freeze(&self) -> Result<()> {
let mut file_name = "freezer.state";
let mut content = "FROZEN".to_string();
if self.v2 {
file_name = "cgroup.freeze";
content = "1".to_string();
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(content.as_ref())
.map_err(|e| Error::with_cause(WriteFailed(file_name.to_string(), content), e))
})
}
/// Thaws, that is, unfreezes the processes in the control group.
pub fn thaw(&self) -> Result<()> {
let mut file_name = "freezer.state";
let mut content = "THAWED".to_string();
if self.v2 {
file_name = "cgroup.freeze";
content = "0".to_string();
}
self.open_path(file_name, true).and_then(|mut file| {
file.write_all(content.as_ref())
.map_err(|e| Error::with_cause(WriteFailed(file_name.to_string(), content), e))
})
}
/// Retrieve the state of processes in the control group.
pub fn state(&self) -> Result<FreezerState> {
let mut file_name = "freezer.state";
if self.v2 {
file_name = "cgroup.freeze";
}
self.open_path(file_name, false).and_then(|mut file| {
let mut s = String::new();
let res = file.read_to_string(&mut s);
match res {
Ok(_) => match s.trim() {
"FROZEN" => Ok(FreezerState::Frozen),
"THAWED" => Ok(FreezerState::Thawed),
"1" => Ok(FreezerState::Frozen),
"0" => Ok(FreezerState::Thawed),
"FREEZING" => Ok(FreezerState::Freezing),
_ => Err(Error::new(ParseError)),
},
Err(e) => Err(Error::with_cause(ReadFailed(file_name.to_string()), e)),
}
})
}
}
+392
View File
@@ -0,0 +1,392 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module represents the various control group hierarchies the Linux kernel supports.
use std::fs;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::path::{Path, PathBuf};
use crate::fs::blkio::BlkIoController;
use crate::fs::cpu::CpuController;
use crate::fs::cpuacct::CpuAcctController;
use crate::fs::cpuset::CpuSetController;
use crate::fs::devices::DevicesController;
use crate::fs::freezer::FreezerController;
use crate::fs::hugetlb::HugeTlbController;
use crate::fs::memory::MemController;
use crate::fs::net_cls::NetClsController;
use crate::fs::net_prio::NetPrioController;
use crate::fs::perf_event::PerfEventController;
use crate::fs::pid::PidController;
use crate::fs::rdma::RdmaController;
use crate::fs::systemd::SystemdController;
use crate::fs::{Controllers, Hierarchy, Subsystem};
use crate::fs::cgroup::Cgroup;
/// Process mounts information.
///
/// See `proc(5)` for format details.
#[derive(Debug, PartialEq, Eq, Hash, Clone)]
pub struct Mountinfo {
/// Mount root directory of the file system.
pub mount_root: PathBuf,
/// Mount pathname relative to the process's root.
pub mount_point: PathBuf,
/// Filesystem type (main type with optional sub-type).
pub fs_type: (String, Option<String>),
/// Superblock options.
pub super_opts: Vec<String>,
}
pub(crate) fn parse_mountinfo_for_line(line: &str) -> Option<Mountinfo> {
let s_values: Vec<_> = line.split(" - ").collect();
if s_values.len() != 2 {
return None;
}
let s0_values: Vec<_> = s_values[0].trim().split(' ').collect();
let s1_values: Vec<_> = s_values[1].trim().split(' ').collect();
if s0_values.len() < 6 || s1_values.len() < 3 {
return None;
}
let mount_point = PathBuf::from(s0_values[4]);
let mount_root = PathBuf::from(s0_values[3]);
let fs_type_values: Vec<_> = s1_values[0].trim().split('.').collect();
let fs_type = match fs_type_values.len() {
1 => (fs_type_values[0].to_string(), None),
2 => (
fs_type_values[0].to_string(),
Some(fs_type_values[1].to_string()),
),
_ => return None,
};
let super_opts: Vec<String> = s1_values[2].trim().split(',').map(String::from).collect();
Some(Mountinfo {
mount_root,
mount_point,
fs_type,
super_opts,
})
}
/// Parses the provided mountinfo file.
fn mountinfo_file(file: &mut File) -> Vec<Mountinfo> {
let mut r = Vec::new();
for line in BufReader::new(file).lines() {
match line {
Ok(line) => {
if let Some(mi) = parse_mountinfo_for_line(&line) {
if mi.fs_type.0 == "cgroup" {
r.push(mi);
}
}
}
Err(_) => break,
}
}
r
}
/// Returns mounts information for the current process.
pub fn mountinfo_self() -> Vec<Mountinfo> {
match File::open("/proc/self/mountinfo") {
Ok(mut file) => mountinfo_file(&mut file),
Err(_) => vec![],
}
}
/// The standard, original cgroup implementation. Often referred to as "cgroupv1".
#[derive(Debug, Clone)]
pub struct V1 {
mountinfo: Vec<Mountinfo>,
}
#[derive(Debug, Clone)]
pub struct V2 {
root: String,
}
impl Hierarchy for V1 {
fn v2(&self) -> bool {
false
}
fn subsystems(&self) -> Vec<Subsystem> {
let mut subs = vec![];
// The cgroup writeback feature requires cooperation between memcgs and blkcgs
// To avoid exceptions, we should add_task for blkcg before memcg(push BlkIo before Mem)
// For more Information: https://www.alibabacloud.com/help/doc-detail/155509.htm
if let Some((point, root)) = self.get_mount_point(Controllers::BlkIo) {
subs.push(Subsystem::BlkIo(BlkIoController::new(point, root, false)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Mem) {
subs.push(Subsystem::Mem(MemController::new(point, root, false)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Pids) {
subs.push(Subsystem::Pid(PidController::new(point, root, false)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::CpuSet) {
subs.push(Subsystem::CpuSet(CpuSetController::new(point, root, false)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::CpuAcct) {
subs.push(Subsystem::CpuAcct(CpuAcctController::new(point, root)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Cpu) {
subs.push(Subsystem::Cpu(CpuController::new(point, root, false)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Devices) {
subs.push(Subsystem::Devices(DevicesController::new(point, root)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Freezer) {
subs.push(Subsystem::Freezer(FreezerController::new(
point, root, false,
)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::NetCls) {
subs.push(Subsystem::NetCls(NetClsController::new(point, root)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::PerfEvent) {
subs.push(Subsystem::PerfEvent(PerfEventController::new(point, root)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::NetPrio) {
subs.push(Subsystem::NetPrio(NetPrioController::new(point, root)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::HugeTlb) {
subs.push(Subsystem::HugeTlb(HugeTlbController::new(
point, root, false,
)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Rdma) {
subs.push(Subsystem::Rdma(RdmaController::new(point, root)));
}
if let Some((point, root)) = self.get_mount_point(Controllers::Systemd) {
subs.push(Subsystem::Systemd(SystemdController::new(
point, root, false,
)));
}
subs
}
fn root_control_group(&self) -> Cgroup {
Cgroup::load(auto(), "")
}
fn parent_control_group(&self, path: &str) -> Cgroup {
let path = Path::new(path);
let parent_path = path.parent().unwrap().to_string_lossy().to_string();
Cgroup::load(auto(), parent_path)
}
fn root(&self) -> PathBuf {
self.mountinfo
.iter()
.find_map(|m| {
if m.fs_type.0 == "cgroup" {
return Some(m.mount_point.parent().unwrap());
}
None
})
.unwrap()
.to_path_buf()
}
}
impl Hierarchy for V2 {
fn v2(&self) -> bool {
true
}
fn subsystems(&self) -> Vec<Subsystem> {
let p = format!("{}/{}", UNIFIED_MOUNTPOINT, "cgroup.controllers");
let ret = fs::read_to_string(p.as_str());
if ret.is_err() {
return vec![];
}
let mut subs = vec![];
let controllers = ret.unwrap().trim().to_string();
let mut controller_list: Vec<&str> = controllers.split(' ').collect();
// The freezer functionality is present in V2, but not as a controller,
// but apparently as a core functionality. FreezerController supports
// that, but we must explicitly fake the controller here.
controller_list.push("freezer");
for s in controller_list {
match s {
"cpu" => {
subs.push(Subsystem::Cpu(CpuController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
"io" => {
subs.push(Subsystem::BlkIo(BlkIoController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
"cpuset" => {
subs.push(Subsystem::CpuSet(CpuSetController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
"memory" => {
subs.push(Subsystem::Mem(MemController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
"pids" => {
subs.push(Subsystem::Pid(PidController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
"freezer" => {
subs.push(Subsystem::Freezer(FreezerController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
"hugetlb" => {
subs.push(Subsystem::HugeTlb(HugeTlbController::new(
self.root(),
PathBuf::from(""),
true,
)));
}
_ => {}
}
}
subs
}
fn root_control_group(&self) -> Cgroup {
Cgroup::load(auto(), "")
}
fn parent_control_group(&self, path: &str) -> Cgroup {
let path = Path::new(path);
let parent_path = path.parent().unwrap().to_string_lossy().to_string();
Cgroup::load(auto(), parent_path)
}
fn root(&self) -> PathBuf {
PathBuf::from(self.root.clone())
}
}
impl V1 {
/// Finds where control groups are mounted to and returns a hierarchy in which control groups
/// can be created.
pub fn new() -> V1 {
V1 {
mountinfo: mountinfo_self(),
}
}
pub fn get_mount_point(&self, controller: Controllers) -> Option<(PathBuf, PathBuf)> {
self.mountinfo.iter().find_map(|m| {
if m.fs_type.0 == "cgroup" && m.super_opts.contains(&controller.to_string()) {
return Some((m.mount_point.to_owned(), m.mount_root.to_owned()));
}
None
})
}
}
impl Default for V1 {
fn default() -> Self {
Self::new()
}
}
impl V2 {
/// Finds where control groups are mounted to and returns a hierarchy in which control groups
/// can be created.
pub fn new() -> V2 {
V2 {
root: String::from(UNIFIED_MOUNTPOINT),
}
}
}
impl Default for V2 {
fn default() -> Self {
Self::new()
}
}
pub const UNIFIED_MOUNTPOINT: &str = "/sys/fs/cgroup";
pub fn is_cgroup2_unified_mode() -> bool {
use nix::sys::statfs;
let path = std::path::Path::new(UNIFIED_MOUNTPOINT);
let fs_stat = match statfs::statfs(path) {
Ok(fs_stat) => fs_stat,
Err(_) => return false,
};
fs_stat.filesystem_type() == statfs::CGROUP2_SUPER_MAGIC
}
pub fn auto() -> Box<dyn Hierarchy> {
if is_cgroup2_unified_mode() {
Box::new(V2::new())
} else {
Box::new(V1::new())
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_parse_mount() {
let mountinfo = vec![
("29 26 0:26 / /sys/fs/cgroup/cpuset,cpu,cpuacct rw,nosuid,nodev,noexec,relatime shared:10 - cgroup cgroup rw,cpuset,cpu,cpuacct",
Mountinfo{mount_root: PathBuf::from("/"), mount_point: PathBuf::from("/sys/fs/cgroup/cpuset,cpu,cpuacct"), fs_type: ("cgroup".to_string(), None), super_opts: vec![
"rw".to_string(),
"cpuset".to_string(),
"cpu".to_string(),
"cpuacct".to_string(),
]}),
("121 1731 0:42 / /shm rw,nosuid,nodev,noexec,relatime shared:68 master:66 - tmpfs shm rw,size=65536k",
Mountinfo{mount_root: PathBuf::from("/"), mount_point: PathBuf::from("/shm"), fs_type: ("tmpfs".to_string(), None), super_opts: vec![
"rw".to_string(),
"size=65536k".to_string(),
]}),
("121 1731 0:42 / /shm rw,nosuid,nodev,noexec,relatime shared:68 master:66 - tmpfs.123 shm rw,size=65536k",
Mountinfo{mount_root: PathBuf::from("/"), mount_point: PathBuf::from("/shm"), fs_type: ("tmpfs".to_string(), Some("123".to_string())), super_opts: vec![
"rw".to_string(),
"size=65536k".to_string(),
]}),
];
for mi in mountinfo {
let info = parse_mountinfo_for_line(mi.0).unwrap();
assert_eq!(info, mi.1)
}
}
}
+372
View File
@@ -0,0 +1,372 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `hugetlb` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/hugetlb.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/hugetlb.txt)
use log::warn;
use std::io::Write;
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::{flat_keyed_to_vec, read_u64_from};
use crate::fs::{
ControllIdentifier, ControllerInternal, Controllers, HugePageResources, Resources, Subsystem,
};
/// A controller that allows controlling the `hugetlb` subsystem of a Cgroup.
///
/// In essence, using this controller it is possible to limit the use of hugepages in the tasks of
/// the control group.
#[derive(Debug, Clone)]
pub struct HugeTlbController {
base: PathBuf,
path: PathBuf,
sizes: Vec<String>,
v2: bool,
}
impl ControllerInternal for HugeTlbController {
fn control_type(&self) -> Controllers {
Controllers::HugeTlb
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn is_v2(&self) -> bool {
self.v2
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &HugePageResources = &res.hugepages;
for i in &res.limits {
let _ = self.set_limit_in_bytes(&i.size, i.limit);
if self.limit_in_bytes(&i.size)? != i.limit {
return Err(Error::new(Other));
}
}
Ok(())
}
}
impl ControllIdentifier for HugeTlbController {
fn controller_type() -> Controllers {
Controllers::HugeTlb
}
}
impl<'a> From<&'a Subsystem> for &'a HugeTlbController {
fn from(sub: &'a Subsystem) -> &'a HugeTlbController {
unsafe {
match sub {
Subsystem::HugeTlb(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl HugeTlbController {
/// Constructs a new `HugeTlbController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
let sizes = get_hugepage_sizes();
Self {
base: root,
path: point,
sizes,
v2,
}
}
/// Whether the system supports `hugetlb_size` hugepages.
pub fn size_supported(&self, hugetlb_size: &str) -> bool {
for s in &self.sizes {
if s == hugetlb_size {
return true;
}
}
false
}
pub fn get_sizes(&self) -> Vec<String> {
self.sizes.clone()
}
fn failcnt_v2(&self, hugetlb_size: &str) -> Result<u64> {
self.open_path(&format!("hugetlb.{}.events", hugetlb_size), false)
.and_then(flat_keyed_to_vec)
.and_then(|x| {
if x.is_empty() {
return Err(Error::from_string(format!(
"get empty from hugetlb.{}.events",
hugetlb_size
)));
}
Ok(x[0].1 as u64)
})
}
/// Check how many times has the limit of `hugetlb_size` hugepages been hit.
pub fn failcnt(&self, hugetlb_size: &str) -> Result<u64> {
if self.v2 {
return self.failcnt_v2(hugetlb_size);
}
self.open_path(&format!("hugetlb.{}.failcnt", hugetlb_size), false)
.and_then(read_u64_from)
}
/// Get the limit (in bytes) of how much memory can be backed by hugepages of a certain size
/// (`hugetlb_size`).
pub fn limit_in_bytes(&self, hugetlb_size: &str) -> Result<u64> {
let mut file_name = format!("hugetlb.{}.limit_in_bytes", hugetlb_size);
if self.v2 {
file_name = format!("hugetlb.{}.max", hugetlb_size);
}
self.open_path(&file_name, false).and_then(read_u64_from)
}
/// Get the current usage of memory that is backed by hugepages of a certain size
/// (`hugetlb_size`).
pub fn usage_in_bytes(&self, hugetlb_size: &str) -> Result<u64> {
let mut file = format!("hugetlb.{}.usage_in_bytes", hugetlb_size);
if self.v2 {
file = format!("hugetlb.{}.current", hugetlb_size);
}
self.open_path(&file, false).and_then(read_u64_from)
}
/// Get the maximum observed usage of memory that is backed by hugepages of a certain size
/// (`hugetlb_size`).
pub fn max_usage_in_bytes(&self, hugetlb_size: &str) -> Result<u64> {
self.open_path(
&format!("hugetlb.{}.max_usage_in_bytes", hugetlb_size),
false,
)
.and_then(read_u64_from)
}
/// Set the limit (in bytes) of how much memory can be backed by hugepages of a certain size
/// (`hugetlb_size`).
pub fn set_limit_in_bytes(&self, hugetlb_size: &str, limit: u64) -> Result<()> {
let mut file_name = format!("hugetlb.{}.limit_in_bytes", hugetlb_size);
if self.v2 {
file_name = format!("hugetlb.{}.max", hugetlb_size);
}
self.open_path(&file_name, true).and_then(|mut file| {
file.write_all(limit.to_string().as_ref()).map_err(|e| {
Error::with_cause(WriteFailed(file_name.to_string(), limit.to_string()), e)
})
})
}
}
pub const HUGEPAGESIZE_DIR: &str = "/sys/kernel/mm/hugepages";
use std::collections::HashMap;
use std::fs;
fn get_hugepage_sizes() -> Vec<String> {
let dirs = fs::read_dir(HUGEPAGESIZE_DIR);
if dirs.is_err() {
return Vec::new();
}
dirs.unwrap()
.filter_map(|e| {
let entry = e.map_err(|e| warn!("readdir error: {:?}", e)).ok()?;
let name = entry.file_name().into_string().unwrap();
let parts: Vec<&str> = name.split('-').collect();
if parts.len() != 2 {
return None;
}
let bmap = get_binary_size_map();
let size = parse_size(parts[1], &bmap)
.map_err(|e| warn!("parse_size error: {:?}", e))
.ok()?;
let dabbrs = get_decimal_abbrs();
Some(custom_size(size as f64, 1024.0, &dabbrs))
})
.collect()
}
pub const KB: u128 = 1000;
pub const MB: u128 = 1000 * KB;
pub const GB: u128 = 1000 * MB;
pub const TB: u128 = 1000 * GB;
pub const PB: u128 = 1000 * TB;
#[allow(non_upper_case_globals)]
pub const KiB: u128 = 1024;
#[allow(non_upper_case_globals)]
pub const MiB: u128 = 1024 * KiB;
#[allow(non_upper_case_globals)]
pub const GiB: u128 = 1024 * MiB;
#[allow(non_upper_case_globals)]
pub const TiB: u128 = 1024 * GiB;
#[allow(non_upper_case_globals)]
pub const PiB: u128 = 1024 * TiB;
pub fn get_binary_size_map() -> HashMap<String, u128> {
let mut m = HashMap::new();
m.insert("k".to_string(), KiB);
m.insert("m".to_string(), MiB);
m.insert("g".to_string(), GiB);
m.insert("t".to_string(), TiB);
m.insert("p".to_string(), PiB);
m
}
pub fn get_decimal_size_map() -> HashMap<String, u128> {
let mut m = HashMap::new();
m.insert("k".to_string(), KB);
m.insert("m".to_string(), MB);
m.insert("g".to_string(), GB);
m.insert("t".to_string(), TB);
m.insert("p".to_string(), PB);
m
}
pub fn get_decimal_abbrs() -> Vec<String> {
let m = vec![
"B".to_string(),
"KB".to_string(),
"MB".to_string(),
"GB".to_string(),
"TB".to_string(),
"PB".to_string(),
"EB".to_string(),
"ZB".to_string(),
"YB".to_string(),
];
m
}
fn parse_size(s: &str, m: &HashMap<String, u128>) -> Result<u128> {
// Remove leading/trailing whitespace.
let s = s.trim();
// Remove an optional trailing 'b' or 'B'
let s = if let Some(stripped) = s.strip_suffix('b').or_else(|| s.strip_suffix('B')) {
stripped
} else {
s
};
// Ensure that the string is not empty after stripping.
if s.is_empty() {
return Err(Error::new(InvalidBytesSize));
}
// The last character should be the multiplier letter.
let last_char = s.chars().last().unwrap();
if !"kKmMgGtTpP".contains(last_char) {
return Err(Error::new(InvalidBytesSize));
}
// The numeric part is everything before the multiplier letter.
let num_part = &s[..s.len() - last_char.len_utf8()];
if num_part.trim().is_empty() {
return Err(Error::new(InvalidBytesSize));
}
// Parse the numeric part into a u128.
let number: u128 = num_part
.trim()
.parse()
.map_err(|_| Error::new(InvalidBytesSize))?;
// Look up the multiplier in the provided HashMap.
let multiplier_key = last_char.to_string();
let multiplier = m
.get(&multiplier_key)
.ok_or_else(|| Error::new(InvalidBytesSize))?;
Ok(number * multiplier)
}
fn custom_size(mut size: f64, base: f64, m: &[String]) -> String {
let mut i = 0;
while size >= base && i < m.len() - 1 {
size /= base;
i += 1;
}
format!("{}{}", size, m[i].as_str())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_binary_size_valid() {
let m = get_binary_size_map();
// Valid inputs must include a multiplier letter.
assert_eq!(parse_size("1k", &m).unwrap(), KiB);
assert_eq!(parse_size("2m", &m).unwrap(), 2 * MiB);
assert_eq!(parse_size("3g", &m).unwrap(), 3 * GiB);
assert_eq!(parse_size("4t", &m).unwrap(), 4 * TiB);
assert_eq!(parse_size("5p", &m).unwrap(), 5 * PiB);
}
#[test]
fn test_decimal_size_valid() {
let m = get_decimal_size_map();
assert_eq!(parse_size("1k", &m).unwrap(), KB);
assert_eq!(parse_size("2m", &m).unwrap(), 2 * MB);
assert_eq!(parse_size("3g", &m).unwrap(), 3 * GB);
assert_eq!(parse_size("4t", &m).unwrap(), 4 * TB);
assert_eq!(parse_size("5p", &m).unwrap(), 5 * PB);
}
#[test]
fn test_trailing_b_suffix() {
let m = get_binary_size_map();
// Trailing 'b' or 'B' should be accepted.
assert_eq!(parse_size("1kb", &m).unwrap(), KiB);
assert_eq!(parse_size("2mB", &m).unwrap(), 2 * MiB);
}
#[test]
fn test_invalid_inputs() {
let m = get_binary_size_map();
// Missing multiplier letter results in error.
assert!(parse_size("1", &m).is_err());
// Invalid multiplier letter.
assert!(parse_size("10x", &m).is_err());
// Non-numeric input.
assert!(parse_size("abc", &m).is_err());
// Only multiplier letter with no number.
assert!(parse_size("k", &m).is_err());
// Number with an invalid trailing character.
assert!(parse_size("123z", &m).is_err());
}
#[test]
fn test_uppercase_multiplier_fails() {
let m = get_binary_size_map();
// Although the regex matches uppercase letters, the provided map only contains lowercase keys.
// Therefore, "1K" does not match any key and should produce an error.
assert!(parse_size("1K", &m).is_err());
}
}
+1219
View File
File diff suppressed because it is too large Load Diff
+1032
View File
File diff suppressed because it is too large Load Diff
+102
View File
@@ -0,0 +1,102 @@
// Copyright (c) 2018 Levente Kurusa
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `net_cls` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/net_cls.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/net_cls.txt)
use std::io::Write;
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::read_u64_from;
use crate::fs::{
ControllIdentifier, ControllerInternal, Controllers, NetworkResources, Resources, Subsystem,
};
/// A controller that allows controlling the `net_cls` subsystem of a Cgroup.
///
/// In esssence, using the `net_cls` controller, one can attach a custom class to the network
/// packets emitted by the control group's tasks. This can then later be used in iptables to have
/// custom firewall rules, QoS, etc.
#[derive(Debug, Clone)]
pub struct NetClsController {
base: PathBuf,
path: PathBuf,
}
impl ControllerInternal for NetClsController {
fn control_type(&self) -> Controllers {
Controllers::NetCls
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &NetworkResources = &res.network;
update_and_test!(self, set_class, res.class_id, get_class);
Ok(())
}
}
impl ControllIdentifier for NetClsController {
fn controller_type() -> Controllers {
Controllers::NetCls
}
}
impl<'a> From<&'a Subsystem> for &'a NetClsController {
fn from(sub: &'a Subsystem) -> &'a NetClsController {
unsafe {
match sub {
Subsystem::NetCls(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl NetClsController {
/// Constructs a new `NetClsController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf) -> Self {
Self {
base: root,
path: point,
}
}
/// Set the network class id of the outgoing packets of the control group's tasks.
pub fn set_class(&self, class: u64) -> Result<()> {
self.open_path("net_cls.classid", true)
.and_then(|mut file| {
let s = format!("{:#08X}", class);
file.write_all(s.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed("net_cls.classid".to_string(), s), e)
})
})
}
/// Get the network class id of the outgoing packets of the control group's tasks.
pub fn get_class(&self) -> Result<u64> {
self.open_path("net_cls.classid", false)
.and_then(read_u64_from)
}
}
+136
View File
@@ -0,0 +1,136 @@
// Copyright (c) 2018 Levente Kurusa
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `net_prio` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/net_prio.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/net_prio.txt)
use std::collections::HashMap;
use std::io::{BufRead, BufReader, Write};
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::read_u64_from;
use crate::fs::{
ControllIdentifier, ControllerInternal, Controllers, NetworkResources, Resources, Subsystem,
};
/// A controller that allows controlling the `net_prio` subsystem of a Cgroup.
///
/// In essence, using `net_prio` one can set the priority of the packets emitted from the control
/// group's tasks. This can then be used to have QoS restrictions on certain control groups and
/// thus, prioritizing certain tasks.
#[derive(Debug, Clone)]
pub struct NetPrioController {
base: PathBuf,
path: PathBuf,
}
impl ControllerInternal for NetPrioController {
fn control_type(&self) -> Controllers {
Controllers::NetPrio
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let res: &NetworkResources = &res.network;
for i in &res.priorities {
let _ = self.set_if_prio(&i.name, i.priority);
}
Ok(())
}
}
impl ControllIdentifier for NetPrioController {
fn controller_type() -> Controllers {
Controllers::NetPrio
}
}
impl<'a> From<&'a Subsystem> for &'a NetPrioController {
fn from(sub: &'a Subsystem) -> &'a NetPrioController {
unsafe {
match sub {
Subsystem::NetPrio(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl NetPrioController {
/// Constructs a new `NetPrioController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf) -> Self {
Self {
base: root,
path: point,
}
}
/// Retrieves the current priority of the emitted packets.
pub fn prio_idx(&self) -> u64 {
self.open_path("net_prio.prioidx", false)
.and_then(read_u64_from)
.unwrap_or(0)
}
/// A map of priorities for each network interface.
pub fn ifpriomap(&self) -> Result<HashMap<String, u64>> {
self.open_path("net_prio.ifpriomap", false)
.and_then(|file| {
let bf = BufReader::new(file);
bf.lines()
.map(|line| {
let line = line.map_err(|_| Error::new(ParseError))?;
let mut parts = line.split_whitespace();
let ifname = parts.next().ok_or(Error::new(ParseError))?;
let ifprio_str = parts.next().ok_or(Error::new(ParseError))?;
let ifprio = ifprio_str
.trim()
.parse()
.map_err(|e| Error::with_cause(ParseError, e))?;
Ok((ifname.to_string(), ifprio))
})
.collect::<Result<HashMap<String, _>>>()
})
}
/// Set the priority of the network traffic on `eif` to be `prio`.
pub fn set_if_prio(&self, eif: &str, prio: u64) -> Result<()> {
self.open_path("net_prio.ifpriomap", true)
.and_then(|mut file| {
file.write_all(format!("{} {}", eif, prio).as_ref())
.map_err(|e| {
Error::with_cause(
WriteFailed(
"net_prio.ifpriomap".to_string(),
format!("{} {}", eif, prio),
),
e,
)
})
})
}
}
+74
View File
@@ -0,0 +1,74 @@
// Copyright (c) 2018 Levente Kurusa
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `perf_event` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [tools/perf/Documentation/perf-record.txt](https://raw.githubusercontent.com/torvalds/linux/master/tools/perf/Documentation/perf-record.txt)
use std::path::PathBuf;
use crate::fs::error::*;
use crate::fs::{ControllIdentifier, ControllerInternal, Controllers, Resources, Subsystem};
/// A controller that allows controlling the `perf_event` subsystem of a Cgroup.
///
/// In essence, when processes belong to the same `perf_event` controller, they can be monitored
/// together using the `perf` performance monitoring and reporting tool.
#[derive(Debug, Clone)]
pub struct PerfEventController {
base: PathBuf,
path: PathBuf,
}
impl ControllerInternal for PerfEventController {
fn control_type(&self) -> Controllers {
Controllers::PerfEvent
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, _res: &Resources) -> Result<()> {
Ok(())
}
}
impl ControllIdentifier for PerfEventController {
fn controller_type() -> Controllers {
Controllers::PerfEvent
}
}
impl<'a> From<&'a Subsystem> for &'a PerfEventController {
fn from(sub: &'a Subsystem) -> &'a PerfEventController {
unsafe {
match sub {
Subsystem::PerfEvent(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl PerfEventController {
/// Constructs a new `PerfEventController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf) -> Self {
Self {
base: root,
path: point,
}
}
}
+155
View File
@@ -0,0 +1,155 @@
// Copyright (c) 2018 Levente Kurusa
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `pids` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroups-v1/pids.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/pids.txt)
use std::io::{Read, Write};
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::read_u64_from;
use crate::fs::{
parse_max_value, ControllIdentifier, ControllerInternal, Controllers, MaxValue, PidResources,
Resources, Subsystem,
};
/// A controller that allows controlling the `pids` subsystem of a Cgroup.
#[derive(Debug, Clone)]
pub struct PidController {
base: PathBuf,
path: PathBuf,
v2: bool,
}
impl ControllerInternal for PidController {
fn control_type(&self) -> Controllers {
Controllers::Pids
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn is_v2(&self) -> bool {
self.v2
}
fn apply(&self, res: &Resources) -> Result<()> {
// get the resources that apply to this controller
let pidres: &PidResources = &res.pid;
// apply pid_max
update_and_test!(
self,
set_pid_max,
pidres.maximum_number_of_processes,
get_pid_max
);
Ok(())
}
}
// impl<'a> ControllIdentifier for &'a PidController {
// fn controller_type() -> Controllers {
// Controllers::Pids
// }
// }
impl ControllIdentifier for PidController {
fn controller_type() -> Controllers {
Controllers::Pids
}
}
impl<'a> From<&'a Subsystem> for &'a PidController {
fn from(sub: &'a Subsystem) -> &'a PidController {
unsafe {
match sub {
Subsystem::Pid(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl PidController {
/// Constructors a new `PidController` instance, with `root` serving as the controller's root
/// directory.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
Self {
base: root,
path: point,
v2,
}
}
/// The number of times `fork` failed because the limit was hit.
pub fn get_pid_events(&self) -> Result<u64> {
self.open_path("pids.events", false).and_then(|mut file| {
let mut string = String::new();
match file.read_to_string(&mut string) {
Ok(_) => match string.split_whitespace().nth(1) {
Some(elem) => match elem.parse() {
Ok(val) => Ok(val),
Err(e) => Err(Error::with_cause(ParseError, e)),
},
None => Err(Error::new(ParseError)),
},
Err(e) => Err(Error::with_cause(ReadFailed("pids.events".to_string()), e)),
}
})
}
/// The number of processes currently.
pub fn get_pid_current(&self) -> Result<u64> {
self.open_path("pids.current", false)
.and_then(read_u64_from)
}
/// The maximum number of processes that can exist at one time in the control group.
pub fn get_pid_max(&self) -> Result<MaxValue> {
self.open_path("pids.max", false).and_then(|mut file| {
let mut string = String::new();
let res = file.read_to_string(&mut string);
match res {
Ok(_) => parse_max_value(&string),
Err(e) => Err(Error::with_cause(ReadFailed("pids.max".to_string()), e)),
}
})
}
/// Set the maximum number of processes that can exist in this control group.
///
/// Note that if `get_pid_current()` returns a higher number than what you
/// are about to set (`max_pid`), then no processess will be killed. Additonally, attaching
/// extra processes to a control group disregards the limit.
pub fn set_pid_max(&self, max_pid: MaxValue) -> Result<()> {
self.open_path("pids.max", true).and_then(|mut file| {
let string_to_write = max_pid.to_string();
match file.write_all(string_to_write.as_ref()) {
Ok(_) => Ok(()),
Err(e) => Err(Error::with_cause(
WriteFailed("pids.max".to_string(), format!("{:?}", max_pid)),
e,
)),
}
})
}
}
+97
View File
@@ -0,0 +1,97 @@
// Copyright (c) 2018 Levente Kurusa
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `rdma` cgroup subsystem.
//!
//! See the Kernel's documentation for more information about this subsystem, found at:
//! [Documentation/cgroup-v1/rdma.txt](https://www.kernel.org/doc/Documentation/cgroup-v1/rdma.txt)
use std::io::Write;
use std::path::PathBuf;
use crate::fs::error::ErrorKind::*;
use crate::fs::error::*;
use crate::fs::read_string_from;
use crate::fs::{ControllIdentifier, ControllerInternal, Controllers, Resources, Subsystem};
/// A controller that allows controlling the `rdma` subsystem of a Cgroup.
///
/// In essence, using this controller one can limit the RDMA/IB specific resources that the tasks
/// in the control group can use.
#[derive(Debug, Clone)]
pub struct RdmaController {
base: PathBuf,
path: PathBuf,
}
impl ControllerInternal for RdmaController {
fn control_type(&self) -> Controllers {
Controllers::Rdma
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, _res: &Resources) -> Result<()> {
Ok(())
}
}
impl ControllIdentifier for RdmaController {
fn controller_type() -> Controllers {
Controllers::Rdma
}
}
impl<'a> From<&'a Subsystem> for &'a RdmaController {
fn from(sub: &'a Subsystem) -> &'a RdmaController {
unsafe {
match sub {
Subsystem::Rdma(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl RdmaController {
/// Constructs a new `RdmaController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf) -> Self {
Self {
base: root,
path: point,
}
}
/// Returns the current usage of RDMA/IB specific resources.
pub fn current(&self) -> Result<String> {
self.open_path("rdma.current", false)
.and_then(read_string_from)
}
/// Returns the max usage of RDMA/IB specific resources.
pub fn max(&self) -> Result<String> {
self.open_path("rdma.max", false).and_then(read_string_from)
}
/// Set a maximum usage for each RDMA/IB resource.
pub fn set_max(&self, max: &str) -> Result<()> {
self.open_path("rdma.max", true).and_then(|mut file| {
file.write_all(max.as_ref()).map_err(|e| {
Error::with_cause(WriteFailed("rdma.max".to_string(), max.to_string()), e)
})
})
}
}
+72
View File
@@ -0,0 +1,72 @@
// Copyright (c) 2020 Ant Group
//
// SPDX-License-Identifier: Apache-2.0 or MIT
//
//! This module contains the implementation of the `systemd` cgroup subsystem.
//!
use std::path::PathBuf;
use crate::fs::error::*;
use crate::fs::{ControllIdentifier, ControllerInternal, Controllers, Resources, Subsystem};
/// A controller that allows controlling the `systemd` subsystem of a Cgroup.
///
#[derive(Debug, Clone)]
pub struct SystemdController {
base: PathBuf,
path: PathBuf,
_v2: bool,
}
impl ControllerInternal for SystemdController {
fn control_type(&self) -> Controllers {
Controllers::Systemd
}
fn get_path(&self) -> &PathBuf {
&self.path
}
fn get_path_mut(&mut self) -> &mut PathBuf {
&mut self.path
}
fn get_base(&self) -> &PathBuf {
&self.base
}
fn apply(&self, _res: &Resources) -> Result<()> {
Ok(())
}
}
impl ControllIdentifier for SystemdController {
fn controller_type() -> Controllers {
Controllers::Systemd
}
}
impl<'a> From<&'a Subsystem> for &'a SystemdController {
fn from(sub: &'a Subsystem) -> &'a SystemdController {
unsafe {
match sub {
Subsystem::Systemd(c) => c,
_ => {
assert_eq!(1, 0);
let v = std::mem::MaybeUninit::uninit();
v.assume_init()
}
}
}
}
}
impl SystemdController {
/// Constructs a new `SystemdController` with `root` serving as the root of the control group.
pub fn new(point: PathBuf, root: PathBuf, v2: bool) -> Self {
Self {
base: root,
path: point,
_v2: v2,
}
}
}