mirror of
https://github.com/cloud-hypervisor/cloud-hypervisor.git
synced 2026-08-05 02:19:16 +00:00
block: Move raw format files into formats/raw/
Move raw format implementation into a structured directory layout: raw_disk.rs -> formats/raw/mod.rs (RawDisk) raw_sync.rs -> formats/raw/worker/sync.rs (RawSync) raw_async.rs -> formats/raw/worker/async_uring.rs (RawAsync) raw_async_aio.rs -> formats/raw/worker/async_aio.rs (RawAio) raw_async_io_tests.rs -> formats/raw/worker/tests.rs Update imports in fixed_vhd_sync.rs and fixed_vhd_async.rs to use the new paths. Re-export formats::raw as raw_disk in lib.rs to preserve the external API. Signed-off-by: Anatol Belski <anbelski@linux.microsoft.com>
This commit is contained in:
committed by
Rob Bradford
parent
b68349f8b3
commit
a11f551572
10
block/src/formats/mod.rs
Normal file
10
block/src/formats/mod.rs
Normal file
@@ -0,0 +1,10 @@
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
//! Disk format implementations.
|
||||
//!
|
||||
//! Each format lives in its own submodule with a `DiskFile` wrapper,
|
||||
//! format specific internals, and sync/async I/O workers.
|
||||
|
||||
pub mod raw;
|
||||
266
block/src/formats/raw/mod.rs
Normal file
266
block/src/formats/raw/mod.rs
Normal file
@@ -0,0 +1,266 @@
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
//! Raw disk image format.
|
||||
//!
|
||||
//! Provides [`RawDisk`], the `DiskFile` wrapper for flat disk images
|
||||
//! with no metadata or copy on write layer.
|
||||
|
||||
use std::fs::File;
|
||||
use std::io;
|
||||
use std::os::unix::fs::FileTypeExt;
|
||||
use std::os::unix::io::AsRawFd;
|
||||
|
||||
use log::warn;
|
||||
|
||||
use self::worker::async_aio::RawAio;
|
||||
#[cfg(feature = "io_uring")]
|
||||
use self::worker::async_uring::RawAsync;
|
||||
use self::worker::sync::RawSync;
|
||||
use crate::async_io::{AsyncIo, BorrowedDiskFd, DiskFileError};
|
||||
use crate::error::{BlockError, BlockErrorKind, BlockResult};
|
||||
use crate::{DiskTopology, disk_file, probe_sparse_support, query_device_size};
|
||||
|
||||
pub(crate) mod worker;
|
||||
|
||||
/// Selects which async I/O backend a `RawDisk` uses.
|
||||
#[derive(Clone, Copy, Debug, PartialEq)]
|
||||
pub enum RawBackend {
|
||||
/// Blocking I/O where the caller waits for completion.
|
||||
Sync,
|
||||
/// Modern asynchronous I/O using shared submission and completion
|
||||
/// rings for lower overhead operation dispatch and completion handling.
|
||||
#[cfg(feature = "io_uring")]
|
||||
IoUring,
|
||||
/// Legacy asynchronous I/O where requests are handed to the kernel
|
||||
/// and completions are collected later.
|
||||
Aio,
|
||||
}
|
||||
|
||||
/// Unified DiskFile wrapper for raw disk images.
|
||||
///
|
||||
/// Owns the underlying file and delegates async I/O creation to the
|
||||
/// backend selected at construction time via [`RawBackend`].
|
||||
#[derive(Debug)]
|
||||
pub struct RawDisk {
|
||||
file: File,
|
||||
backend: RawBackend,
|
||||
}
|
||||
|
||||
impl RawDisk {
|
||||
pub fn new(file: File, backend: RawBackend) -> Self {
|
||||
Self { file, backend }
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::DiskSize for RawDisk {
|
||||
fn logical_size(&self) -> BlockResult<u64> {
|
||||
query_device_size(&self.file)
|
||||
.map(|(logical_size, _)| logical_size)
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, DiskFileError::Size(e)))
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::PhysicalSize for RawDisk {
|
||||
fn physical_size(&self) -> BlockResult<u64> {
|
||||
query_device_size(&self.file)
|
||||
.map(|(_, physical_size)| physical_size)
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, DiskFileError::Size(e)))
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::DiskFd for RawDisk {
|
||||
fn fd(&self) -> BorrowedDiskFd<'_> {
|
||||
BorrowedDiskFd::new(self.file.as_raw_fd())
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::Geometry for RawDisk {
|
||||
fn topology(&self) -> DiskTopology {
|
||||
DiskTopology::probe(&self.file).unwrap_or_else(|_| {
|
||||
warn!("Unable to get device topology. Using default topology");
|
||||
DiskTopology::default()
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::SparseCapable for RawDisk {
|
||||
fn supports_sparse_operations(&self) -> bool {
|
||||
probe_sparse_support(&self.file)
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::Resizable for RawDisk {
|
||||
fn resize(&mut self, size: u64) -> BlockResult<()> {
|
||||
let fd_metadata = self
|
||||
.file
|
||||
.metadata()
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, DiskFileError::ResizeError(e)))?;
|
||||
|
||||
if fd_metadata.file_type().is_block_device() {
|
||||
// Block devices cannot be resized via ftruncate; they are resized
|
||||
// externally (LVM, losetup, etc.). Verify the size matches.
|
||||
let (actual_size, _) = query_device_size(&self.file)
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, DiskFileError::ResizeError(e)))?;
|
||||
if actual_size != size {
|
||||
return Err(BlockError::new(
|
||||
BlockErrorKind::Io,
|
||||
DiskFileError::ResizeError(io::Error::other(format!(
|
||||
"Block device size {actual_size} does not match requested size {size}"
|
||||
))),
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
} else {
|
||||
self.file
|
||||
.set_len(size)
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, DiskFileError::ResizeError(e)))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::DiskFile for RawDisk {}
|
||||
|
||||
impl disk_file::AsyncDiskFile for RawDisk {
|
||||
fn try_clone(&self) -> BlockResult<Box<dyn disk_file::AsyncDiskFile>> {
|
||||
let file = self
|
||||
.file
|
||||
.try_clone()
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, DiskFileError::Clone(e)))?;
|
||||
Ok(Box::new(RawDisk {
|
||||
file,
|
||||
backend: self.backend,
|
||||
}))
|
||||
}
|
||||
|
||||
fn create_async_io(&self, ring_depth: u32) -> BlockResult<Box<dyn AsyncIo>> {
|
||||
match self.backend {
|
||||
RawBackend::Sync => Ok(Box::new(RawSync::new(self.file.as_raw_fd()))),
|
||||
#[cfg(feature = "io_uring")]
|
||||
RawBackend::IoUring => Ok(Box::new(RawAsync::new(self.file.as_raw_fd(), ring_depth)?)),
|
||||
RawBackend::Aio => Ok(Box::new(RawAio::new(self.file.as_raw_fd(), ring_depth)?)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod unit_tests {
|
||||
use std::fs::File;
|
||||
|
||||
use vmm_sys_util::tempfile::TempFile;
|
||||
|
||||
use super::*;
|
||||
use crate::async_io::AsyncIo;
|
||||
use crate::disk_file::{AsyncDiskFile, DiskSize, PhysicalSize, Resizable};
|
||||
|
||||
const TEST_SIZE: u64 = 0x1122_3344;
|
||||
|
||||
fn make_raw_file() -> File {
|
||||
let file: File = TempFile::new().unwrap().into_file();
|
||||
file.set_len(TEST_SIZE).unwrap();
|
||||
file
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_sync_returns_correct_size() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::Sync);
|
||||
assert_eq!(disk.logical_size().unwrap(), TEST_SIZE);
|
||||
}
|
||||
|
||||
fn assert_async_io_from_dyn(disk: &dyn AsyncDiskFile, expect_backend: RawBackend) {
|
||||
let io: Box<dyn AsyncIo> = disk.create_async_io(128).unwrap();
|
||||
cfg_if::cfg_if! {
|
||||
if #[cfg(feature = "io_uring")] {
|
||||
let expected_batch_requests = expect_backend == RawBackend::IoUring;
|
||||
} else {
|
||||
let _ = expect_backend;
|
||||
let expected_batch_requests = false;
|
||||
}
|
||||
}
|
||||
assert_eq!(io.batch_requests_enabled(), expected_batch_requests);
|
||||
}
|
||||
|
||||
fn assert_sync_backend(disk: &RawDisk) {
|
||||
assert_eq!(disk.backend, RawBackend::Sync);
|
||||
assert_async_io_from_dyn(disk, RawBackend::Sync);
|
||||
}
|
||||
|
||||
fn assert_aio_backend(disk: &RawDisk) {
|
||||
assert_eq!(disk.backend, RawBackend::Aio);
|
||||
assert_async_io_from_dyn(disk, RawBackend::Aio);
|
||||
}
|
||||
|
||||
#[cfg(feature = "io_uring")]
|
||||
fn assert_io_uring_backend(disk: &RawDisk) {
|
||||
assert_eq!(disk.backend, RawBackend::IoUring);
|
||||
assert_async_io_from_dyn(disk, RawBackend::IoUring);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sync_backend_disables_batch_requests() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::Sync);
|
||||
assert_sync_backend(&disk);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn aio_backend_disables_batch_requests() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::Aio);
|
||||
assert_aio_backend(&disk);
|
||||
}
|
||||
|
||||
#[cfg(feature = "io_uring")]
|
||||
#[test]
|
||||
fn io_uring_backend_enables_batch_requests() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::IoUring);
|
||||
assert_io_uring_backend(&disk);
|
||||
}
|
||||
|
||||
fn assert_try_clone(disk: &RawDisk, expect_backend: RawBackend) {
|
||||
let cloned = disk.try_clone().unwrap();
|
||||
assert_async_io_from_dyn(cloned.as_ref(), expect_backend);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_clone_preserves_sync_backend() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::Sync);
|
||||
assert_try_clone(&disk, RawBackend::Sync);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_clone_preserves_aio_backend() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::Aio);
|
||||
assert_try_clone(&disk, RawBackend::Aio);
|
||||
}
|
||||
|
||||
#[cfg(feature = "io_uring")]
|
||||
#[test]
|
||||
fn try_clone_preserves_io_uring_backend() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::IoUring);
|
||||
assert_try_clone(&disk, RawBackend::IoUring);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_changes_file_size() {
|
||||
let file = make_raw_file();
|
||||
let mut disk = RawDisk::new(file, RawBackend::Aio);
|
||||
let new_size = TEST_SIZE * 2;
|
||||
disk.resize(new_size).unwrap();
|
||||
assert_eq!(disk.logical_size().unwrap(), new_size);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn physical_size_reports_allocated_blocks() {
|
||||
let file = make_raw_file();
|
||||
let disk = RawDisk::new(file, RawBackend::Aio);
|
||||
// Sparse file: physical size is less than logical size.
|
||||
assert!(disk.physical_size().unwrap() < disk.logical_size().unwrap());
|
||||
}
|
||||
}
|
||||
135
block/src/formats/raw/worker/async_aio.rs
Normal file
135
block/src/formats/raw/worker/async_aio.rs
Normal file
@@ -0,0 +1,135 @@
|
||||
// Copyright © 2023 Intel Corporation
|
||||
//
|
||||
// Copyright (c) Meta Platforms, Inc. and affiliates.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
//
|
||||
// Copyright © 2023 Crusoe Energy Systems LLC
|
||||
//
|
||||
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
use vmm_sys_util::eventfd::EventFd;
|
||||
|
||||
use crate::async_io::{
|
||||
AioDataIo, AsyncIo, AsyncIoCompletion, AsyncIoError, AsyncIoOperation, AsyncIoResult,
|
||||
};
|
||||
use crate::error::{BlockError, BlockErrorKind, BlockResult};
|
||||
use crate::sparse::{punch_hole, write_zeroes};
|
||||
use crate::{SECTOR_SIZE, is_block_device};
|
||||
|
||||
pub struct RawAio {
|
||||
fd: RawFd,
|
||||
data_io: AioDataIo,
|
||||
alignment: u64,
|
||||
is_block_device: bool,
|
||||
}
|
||||
|
||||
impl RawAio {
|
||||
pub fn new(fd: RawFd, queue_depth: u32) -> BlockResult<Self> {
|
||||
let data_io =
|
||||
AioDataIo::new(queue_depth).map_err(|e| BlockError::new(BlockErrorKind::Io, e))?;
|
||||
let is_block_device = is_block_device(fd);
|
||||
|
||||
Ok(RawAio {
|
||||
fd,
|
||||
data_io,
|
||||
alignment: SECTOR_SIZE,
|
||||
is_block_device,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl AsyncIo for RawAio {
|
||||
fn notifier(&self) -> &EventFd {
|
||||
self.data_io.notifier()
|
||||
}
|
||||
|
||||
fn alignment(&self) -> u64 {
|
||||
self.alignment
|
||||
}
|
||||
|
||||
fn submit_data_operation(&mut self, op: AsyncIoOperation) -> AsyncIoResult<()> {
|
||||
let is_read = op.is_read();
|
||||
self.data_io.submit_operation(self.fd, op).map_err(|e| {
|
||||
if is_read {
|
||||
AsyncIoError::ReadVectored(e)
|
||||
} else {
|
||||
AsyncIoError::WriteVectored(e)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn fsync(&mut self, user_data: Option<u64>) -> AsyncIoResult<()> {
|
||||
if let Some(user_data) = user_data {
|
||||
self.data_io
|
||||
.submit_fsync(self.fd, user_data)
|
||||
.map_err(AsyncIoError::Fsync)?;
|
||||
} else {
|
||||
// SAFETY: FFI call with a valid fd
|
||||
unsafe { libc::fsync(self.fd) };
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn next_completed_request(&mut self) -> Option<AsyncIoCompletion> {
|
||||
self.data_io.next_completion()
|
||||
}
|
||||
|
||||
fn punch_hole(&mut self, offset: u64, length: u64, user_data: u64) -> AsyncIoResult<()> {
|
||||
// Linux AIO has no IOCB command for fallocate, so perform the
|
||||
// operation synchronously and signal completion via the completion
|
||||
// list, matching the pattern used by the sync backend (RawSync).
|
||||
punch_hole(self.fd, self.is_block_device, offset, length)
|
||||
.map_err(AsyncIoError::PunchHole)?;
|
||||
self.data_io
|
||||
.inject_completion(AsyncIoCompletion::new(user_data, 0, None));
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_zeroes(&mut self, offset: u64, length: u64, user_data: u64) -> AsyncIoResult<()> {
|
||||
// Same as punch_hole().
|
||||
write_zeroes(self.fd, self.is_block_device, offset, length)
|
||||
.map_err(AsyncIoError::WriteZeroes)?;
|
||||
self.data_io
|
||||
.inject_completion(AsyncIoCompletion::new(user_data, 0, None));
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod unit_tests {
|
||||
use std::os::unix::io::AsRawFd;
|
||||
|
||||
use vmm_sys_util::tempfile::TempFile;
|
||||
|
||||
use super::*;
|
||||
use crate::raw_disk::worker::tests;
|
||||
|
||||
#[test]
|
||||
fn test_punch_hole() {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
let mut file = temp_file.into_file();
|
||||
let mut async_io = RawAio::new(file.as_raw_fd(), 128).unwrap();
|
||||
tests::test_punch_hole(&mut async_io, &mut file);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_write_zeroes() {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
let mut file = temp_file.into_file();
|
||||
let mut async_io = RawAio::new(file.as_raw_fd(), 128).unwrap();
|
||||
tests::test_write_zeroes(&mut async_io, &mut file);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_punch_hole_multiple_operations() {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
let mut file = temp_file.into_file();
|
||||
let mut async_io = RawAio::new(file.as_raw_fd(), 128).unwrap();
|
||||
tests::test_punch_hole_multiple_operations(&mut async_io, &mut file);
|
||||
}
|
||||
}
|
||||
127
block/src/formats/raw/worker/async_uring.rs
Normal file
127
block/src/formats/raw/worker/async_uring.rs
Normal file
@@ -0,0 +1,127 @@
|
||||
// Copyright © 2021 Intel Corporation
|
||||
//
|
||||
// Copyright (c) Meta Platforms, Inc. and affiliates.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
use libc::{FALLOC_FL_KEEP_SIZE, FALLOC_FL_PUNCH_HOLE, FALLOC_FL_ZERO_RANGE};
|
||||
use vmm_sys_util::eventfd::EventFd;
|
||||
|
||||
use crate::async_io::{
|
||||
AsyncIo, AsyncIoCompletion, AsyncIoError, AsyncIoOperation, AsyncIoResult, UringDataIo,
|
||||
};
|
||||
use crate::error::{BlockError, BlockErrorKind, BlockResult};
|
||||
use crate::sparse::{blkdiscard, blkzeroout};
|
||||
use crate::{SECTOR_SIZE, is_block_device};
|
||||
|
||||
pub struct RawAsync {
|
||||
fd: RawFd,
|
||||
data_io: UringDataIo,
|
||||
alignment: u64,
|
||||
is_block_device: bool,
|
||||
}
|
||||
|
||||
impl RawAsync {
|
||||
pub fn new(fd: RawFd, ring_depth: u32) -> BlockResult<Self> {
|
||||
let data_io =
|
||||
UringDataIo::new(ring_depth).map_err(|e| BlockError::new(BlockErrorKind::Io, e))?;
|
||||
let is_block_device = is_block_device(fd);
|
||||
|
||||
Ok(RawAsync {
|
||||
fd,
|
||||
data_io,
|
||||
alignment: SECTOR_SIZE,
|
||||
is_block_device,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl AsyncIo for RawAsync {
|
||||
fn notifier(&self) -> &EventFd {
|
||||
self.data_io.notifier()
|
||||
}
|
||||
|
||||
fn alignment(&self) -> u64 {
|
||||
self.alignment
|
||||
}
|
||||
|
||||
fn submit_data_operation(&mut self, op: AsyncIoOperation) -> AsyncIoResult<()> {
|
||||
let is_read = op.is_read();
|
||||
self.data_io.submit_operation(self.fd, op).map_err(|e| {
|
||||
if is_read {
|
||||
AsyncIoError::ReadVectored(e)
|
||||
} else {
|
||||
AsyncIoError::WriteVectored(e)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn fsync(&mut self, user_data: Option<u64>) -> AsyncIoResult<()> {
|
||||
if let Some(user_data) = user_data {
|
||||
self.data_io
|
||||
.submit_fsync(self.fd, user_data)
|
||||
.map_err(AsyncIoError::Fsync)?;
|
||||
} else {
|
||||
// SAFETY: FFI call with a valid fd
|
||||
unsafe { libc::fsync(self.fd) };
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn next_completed_request(&mut self) -> Option<AsyncIoCompletion> {
|
||||
self.data_io.next_completion()
|
||||
}
|
||||
|
||||
fn batch_requests_enabled(&self) -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
fn submit_batch_requests(&mut self, batch_request: Vec<AsyncIoOperation>) -> AsyncIoResult<()> {
|
||||
self.data_io
|
||||
.submit_batch(self.fd, batch_request)
|
||||
.map_err(AsyncIoError::SubmitBatchRequests)
|
||||
}
|
||||
|
||||
fn punch_hole(&mut self, offset: u64, length: u64, user_data: u64) -> AsyncIoResult<()> {
|
||||
// Some block devices don't support fallocate(). Use ioctl instead. The assumption is that
|
||||
// this happens rarely and we don't need to introduce unnecessary complexity by submitting
|
||||
// a fallocate request, reaping ENOTSUPP in the completion routine, and reissuing the
|
||||
// request with an ioctl.
|
||||
if self.is_block_device {
|
||||
blkdiscard(self.fd, offset, length).map_err(AsyncIoError::PunchHole)?;
|
||||
// Deliver the completion through the normal io_uring path by
|
||||
// queuing a NOP carrying `user_data`. The registered eventfd will
|
||||
// fire when it completes, just like any other request.
|
||||
return self
|
||||
.data_io
|
||||
.submit_nop(user_data)
|
||||
.map_err(AsyncIoError::PunchHole);
|
||||
}
|
||||
|
||||
let mode = FALLOC_FL_PUNCH_HOLE | FALLOC_FL_KEEP_SIZE;
|
||||
|
||||
self.data_io
|
||||
.submit_fallocate(self.fd, offset, length, mode, user_data)
|
||||
.map_err(AsyncIoError::PunchHole)
|
||||
}
|
||||
|
||||
fn write_zeroes(&mut self, offset: u64, length: u64, user_data: u64) -> AsyncIoResult<()> {
|
||||
// Same rationale as punch_hole().
|
||||
if self.is_block_device {
|
||||
blkzeroout(self.fd, offset, length).map_err(AsyncIoError::WriteZeroes)?;
|
||||
return self
|
||||
.data_io
|
||||
.submit_nop(user_data)
|
||||
.map_err(AsyncIoError::WriteZeroes);
|
||||
}
|
||||
|
||||
let mode = FALLOC_FL_ZERO_RANGE | FALLOC_FL_KEEP_SIZE;
|
||||
|
||||
self.data_io
|
||||
.submit_fallocate(self.fd, offset, length, mode, user_data)
|
||||
.map_err(AsyncIoError::WriteZeroes)
|
||||
}
|
||||
}
|
||||
15
block/src/formats/raw/worker/mod.rs
Normal file
15
block/src/formats/raw/worker/mod.rs
Normal file
@@ -0,0 +1,15 @@
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
//! Sync/async I/O workers for raw images.
|
||||
//!
|
||||
//! Each backend implements the [`AsyncIo`](crate::async_io::AsyncIo)
|
||||
//! trait.
|
||||
|
||||
pub(crate) mod async_aio;
|
||||
#[cfg(feature = "io_uring")]
|
||||
pub(crate) mod async_uring;
|
||||
pub(crate) mod sync;
|
||||
#[cfg(test)]
|
||||
pub(crate) mod tests;
|
||||
163
block/src/formats/raw/worker/sync.rs
Normal file
163
block/src/formats/raw/worker/sync.rs
Normal file
@@ -0,0 +1,163 @@
|
||||
// Copyright © 2021 Intel Corporation
|
||||
//
|
||||
// Copyright (c) Meta Platforms, Inc. and affiliates.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
use std::collections::VecDeque;
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
use vmm_sys_util::eventfd::EventFd;
|
||||
|
||||
use crate::async_io::{AsyncIo, AsyncIoCompletion, AsyncIoError, AsyncIoOperation, AsyncIoResult};
|
||||
use crate::sparse::{punch_hole, write_zeroes};
|
||||
use crate::{SECTOR_SIZE, is_block_device};
|
||||
|
||||
pub struct RawSync {
|
||||
fd: RawFd,
|
||||
eventfd: EventFd,
|
||||
completion_list: VecDeque<AsyncIoCompletion>,
|
||||
alignment: u64,
|
||||
is_block_device: bool,
|
||||
}
|
||||
|
||||
impl RawSync {
|
||||
pub fn new(fd: RawFd) -> Self {
|
||||
let is_block_device = is_block_device(fd);
|
||||
RawSync {
|
||||
fd,
|
||||
eventfd: EventFd::new(libc::EFD_NONBLOCK).expect("Failed creating EventFd for RawFile"),
|
||||
completion_list: VecDeque::new(),
|
||||
alignment: SECTOR_SIZE,
|
||||
is_block_device,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl AsyncIo for RawSync {
|
||||
fn notifier(&self) -> &EventFd {
|
||||
&self.eventfd
|
||||
}
|
||||
|
||||
fn alignment(&self) -> u64 {
|
||||
self.alignment
|
||||
}
|
||||
|
||||
fn submit_data_operation(&mut self, op: AsyncIoOperation) -> AsyncIoResult<()> {
|
||||
let offset = op.offset();
|
||||
let is_read = op.is_read();
|
||||
let iovecs = op.iovecs();
|
||||
|
||||
let result = if is_read {
|
||||
// SAFETY: the memory pointed to by `iovecs` is backed by the op,
|
||||
// and valid for the kernel to write to by construction of
|
||||
// AsyncIoOperation.
|
||||
unsafe {
|
||||
libc::preadv(
|
||||
self.fd as libc::c_int,
|
||||
iovecs.as_ptr(),
|
||||
iovecs.len() as libc::c_int,
|
||||
offset,
|
||||
)
|
||||
}
|
||||
} else {
|
||||
// SAFETY: the memory pointed to by `iovecs` is backed by the op,
|
||||
// and valid for the kernel to read from by construction of
|
||||
// AsyncIoOperation.
|
||||
unsafe {
|
||||
libc::pwritev(
|
||||
self.fd as libc::c_int,
|
||||
iovecs.as_ptr(),
|
||||
iovecs.len() as libc::c_int,
|
||||
offset,
|
||||
)
|
||||
}
|
||||
};
|
||||
if result < 0 {
|
||||
let error = std::io::Error::last_os_error();
|
||||
return Err(if is_read {
|
||||
AsyncIoError::ReadVectored(error)
|
||||
} else {
|
||||
AsyncIoError::WriteVectored(error)
|
||||
});
|
||||
}
|
||||
|
||||
self.completion_list
|
||||
.push_back(AsyncIoCompletion::from_operation(op, result as i32));
|
||||
self.eventfd.write(1).unwrap();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn fsync(&mut self, user_data: Option<u64>) -> AsyncIoResult<()> {
|
||||
// SAFETY: FFI call
|
||||
let result = unsafe { libc::fsync(self.fd as libc::c_int) };
|
||||
if result < 0 {
|
||||
return Err(AsyncIoError::Fsync(std::io::Error::last_os_error()));
|
||||
}
|
||||
|
||||
if let Some(user_data) = user_data {
|
||||
self.completion_list
|
||||
.push_back(AsyncIoCompletion::new(user_data, result, None));
|
||||
self.eventfd.write(1).unwrap();
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn next_completed_request(&mut self) -> Option<AsyncIoCompletion> {
|
||||
self.completion_list.pop_front()
|
||||
}
|
||||
|
||||
fn punch_hole(&mut self, offset: u64, length: u64, user_data: u64) -> AsyncIoResult<()> {
|
||||
punch_hole(self.fd, self.is_block_device, offset, length)
|
||||
.map_err(AsyncIoError::PunchHole)?;
|
||||
self.completion_list
|
||||
.push_back(AsyncIoCompletion::new(user_data, 0, None));
|
||||
self.eventfd.write(1).unwrap();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_zeroes(&mut self, offset: u64, length: u64, user_data: u64) -> AsyncIoResult<()> {
|
||||
write_zeroes(self.fd, self.is_block_device, offset, length)
|
||||
.map_err(AsyncIoError::WriteZeroes)?;
|
||||
self.completion_list
|
||||
.push_back(AsyncIoCompletion::new(user_data, 0, None));
|
||||
self.eventfd.write(1).unwrap();
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod unit_tests {
|
||||
use std::os::unix::io::AsRawFd;
|
||||
|
||||
use vmm_sys_util::tempfile::TempFile;
|
||||
|
||||
use super::*;
|
||||
use crate::raw_disk::worker::tests;
|
||||
|
||||
#[test]
|
||||
fn test_punch_hole() {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
let mut file = temp_file.into_file();
|
||||
let mut async_io = RawSync::new(file.as_raw_fd());
|
||||
tests::test_punch_hole(&mut async_io, &mut file);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_write_zeroes() {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
let mut file = temp_file.into_file();
|
||||
let mut async_io = RawSync::new(file.as_raw_fd());
|
||||
tests::test_write_zeroes(&mut async_io, &mut file);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_punch_hole_multiple_operations() {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
let mut file = temp_file.into_file();
|
||||
let mut async_io = RawSync::new(file.as_raw_fd());
|
||||
tests::test_punch_hole_multiple_operations(&mut async_io, &mut file);
|
||||
}
|
||||
}
|
||||
169
block/src/formats/raw/worker/tests.rs
Normal file
169
block/src/formats/raw/worker/tests.rs
Normal file
@@ -0,0 +1,169 @@
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// Copyright (c) Meta Platforms, Inc. and affiliates.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
//! Shared test helpers for [`AsyncIo`] backends.
|
||||
//!
|
||||
//! Each helper takes a `&mut dyn AsyncIo` together with the [`File`] handle
|
||||
//! that backs the I/O object, so the same logic exercises every backend with
|
||||
//! only the constructor differing.
|
||||
|
||||
use std::fs::File;
|
||||
use std::io::{Read, Seek, SeekFrom, Write};
|
||||
|
||||
use crate::async_io::{AsyncIo, AsyncIoError};
|
||||
|
||||
fn next_completion(async_io: &mut dyn AsyncIo) -> (u64, i32) {
|
||||
let completion = async_io.next_completed_request().expect("No completion");
|
||||
(completion.user_data, completion.result)
|
||||
}
|
||||
|
||||
/// Tests punching a hole in the middle of a 4 MB file and verifying data
|
||||
/// integrity around the hole.
|
||||
pub fn test_punch_hole(async_io: &mut dyn AsyncIo, file: &mut File) {
|
||||
// Write 4MB of data
|
||||
let data = vec![0xAA; 4 * 1024 * 1024];
|
||||
file.write_all(&data).unwrap();
|
||||
file.sync_all().unwrap();
|
||||
|
||||
// Punch hole in the middle (1MB at offset 1MB)
|
||||
let offset = 1024 * 1024;
|
||||
let length = 1024 * 1024;
|
||||
async_io.punch_hole(offset, length, 1).unwrap();
|
||||
|
||||
// Check completion
|
||||
let (user_data, result) = next_completion(async_io);
|
||||
assert_eq!(user_data, 1);
|
||||
assert_eq!(result, 0);
|
||||
|
||||
// Verify the hole reads as zeros
|
||||
file.seek(SeekFrom::Start(offset)).unwrap();
|
||||
let mut read_buf = vec![0; length as usize];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(
|
||||
read_buf.iter().all(|&b| b == 0),
|
||||
"Punched hole should read as zeros"
|
||||
);
|
||||
|
||||
// Verify data before hole is intact
|
||||
file.seek(SeekFrom::Start(0)).unwrap();
|
||||
let mut read_buf = vec![0; 1024];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(
|
||||
read_buf.iter().all(|&b| b == 0xAA),
|
||||
"Data before hole should be intact"
|
||||
);
|
||||
|
||||
// Verify data after hole is intact
|
||||
file.seek(SeekFrom::Start(offset + length)).unwrap();
|
||||
let mut read_buf = vec![0; 1024];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(
|
||||
read_buf.iter().all(|&b| b == 0xAA),
|
||||
"Data after hole should be intact"
|
||||
);
|
||||
}
|
||||
|
||||
/// Tests writing zeroes to a 512 KB region inside a 4 MB file and verifying
|
||||
/// surrounding data is preserved. Gracefully skips when the filesystem does
|
||||
/// not support `FALLOC_FL_ZERO_RANGE`.
|
||||
pub fn test_write_zeroes(async_io: &mut dyn AsyncIo, file: &mut File) {
|
||||
// Write 4MB of data
|
||||
let data = vec![0xBB; 4 * 1024 * 1024];
|
||||
file.write_all(&data).unwrap();
|
||||
file.sync_all().unwrap();
|
||||
|
||||
// Write zeros in the middle (512KB at offset 2MB)
|
||||
let offset = 2 * 1024 * 1024;
|
||||
let length = 512 * 1024;
|
||||
let write_zeroes_result = async_io.write_zeroes(offset, length, 2);
|
||||
|
||||
// FALLOC_FL_ZERO_RANGE might not be supported on all filesystems (e.g., tmpfs)
|
||||
// If it fails with ENOTSUP, skip the test
|
||||
if let Err(AsyncIoError::WriteZeroes(ref e)) = write_zeroes_result
|
||||
&& (e.raw_os_error() == Some(libc::EOPNOTSUPP) || e.raw_os_error() == Some(libc::ENOTSUP))
|
||||
{
|
||||
eprintln!("Skipping test_write_zeroes: filesystem doesn't support FALLOC_FL_ZERO_RANGE");
|
||||
return;
|
||||
}
|
||||
write_zeroes_result.unwrap();
|
||||
|
||||
// Check completion
|
||||
let (user_data, result) = next_completion(async_io);
|
||||
assert_eq!(user_data, 2);
|
||||
assert_eq!(result, 0);
|
||||
|
||||
// Verify the zeroed region reads as zeros
|
||||
file.seek(SeekFrom::Start(offset)).unwrap();
|
||||
let mut read_buf = vec![0; length as usize];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(
|
||||
read_buf.iter().all(|&b| b == 0),
|
||||
"Zeroed region should read as zeros"
|
||||
);
|
||||
|
||||
// Verify data before zeroed region is intact
|
||||
file.seek(SeekFrom::Start(offset - 1024)).unwrap();
|
||||
let mut read_buf = vec![0; 1024];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(
|
||||
read_buf.iter().all(|&b| b == 0xBB),
|
||||
"Data before zeroed region should be intact"
|
||||
);
|
||||
|
||||
// Verify data after zeroed region is intact
|
||||
file.seek(SeekFrom::Start(offset + length)).unwrap();
|
||||
let mut read_buf = vec![0; 1024];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(
|
||||
read_buf.iter().all(|&b| b == 0xBB),
|
||||
"Data after zeroed region should be intact"
|
||||
);
|
||||
}
|
||||
|
||||
/// Tests punching multiple holes in an 8 MB file and verifying each hole
|
||||
/// independently reads as zeroes.
|
||||
pub fn test_punch_hole_multiple_operations(async_io: &mut dyn AsyncIo, file: &mut File) {
|
||||
// Write 8MB of data
|
||||
let data = vec![0xCC; 8 * 1024 * 1024];
|
||||
file.write_all(&data).unwrap();
|
||||
file.sync_all().unwrap();
|
||||
|
||||
// Punch multiple holes
|
||||
async_io.punch_hole(1024 * 1024, 512 * 1024, 10).unwrap();
|
||||
async_io
|
||||
.punch_hole(3 * 1024 * 1024, 512 * 1024, 11)
|
||||
.unwrap();
|
||||
async_io
|
||||
.punch_hole(5 * 1024 * 1024, 512 * 1024, 12)
|
||||
.unwrap();
|
||||
|
||||
// Check all completions
|
||||
let (user_data, result) = next_completion(async_io);
|
||||
assert_eq!(user_data, 10);
|
||||
assert_eq!(result, 0);
|
||||
|
||||
let (user_data, result) = next_completion(async_io);
|
||||
assert_eq!(user_data, 11);
|
||||
assert_eq!(result, 0);
|
||||
|
||||
let (user_data, result) = next_completion(async_io);
|
||||
assert_eq!(user_data, 12);
|
||||
assert_eq!(result, 0);
|
||||
|
||||
// Verify all holes read as zeros
|
||||
file.seek(SeekFrom::Start(1024 * 1024)).unwrap();
|
||||
let mut read_buf = vec![0; 512 * 1024];
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(read_buf.iter().all(|&b| b == 0));
|
||||
|
||||
file.seek(SeekFrom::Start(3 * 1024 * 1024)).unwrap();
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(read_buf.iter().all(|&b| b == 0));
|
||||
|
||||
file.seek(SeekFrom::Start(5 * 1024 * 1024)).unwrap();
|
||||
file.read_exact(&mut read_buf).unwrap();
|
||||
assert!(read_buf.iter().all(|&b| b == 0));
|
||||
}
|
||||
Reference in New Issue
Block a user