diff --git a/block/src/aligned_buffer.rs b/block/src/aligned_buffer.rs index dbe8c6279..232fb9ef9 100644 --- a/block/src/aligned_buffer.rs +++ b/block/src/aligned_buffer.rs @@ -102,7 +102,6 @@ impl AlignedBuffer { } /// Read the full aligned region from `f` into this buffer. - #[cfg_attr(not(test), expect(dead_code))] pub fn read_exact_from(&mut self, f: &impl FileExt) -> io::Result<()> { let offset = self.aligned_offset; f.read_exact_at(self.full_mut_slice(), offset) diff --git a/block/src/formats/qcow/common.rs b/block/src/formats/qcow/common.rs index 1e946fe60..74a506629 100644 --- a/block/src/formats/qcow/common.rs +++ b/block/src/formats/qcow/common.rs @@ -10,9 +10,8 @@ //! //! Position-independent I/O helpers used by both `qcow_sync` and `qcow_async`. -use std::alloc::{Layout, alloc_zeroed, dealloc}; +use std::io; use std::os::fd::RawFd; -use std::{io, slice}; #[cfg(test)] use super::internal; @@ -99,98 +98,6 @@ pub fn pwrite_all(fd: RawFd, buf: &[u8], offset: u64) -> io::Result<()> { Ok(()) } -/// RAII wrapper for an aligned heap buffer required by O_DIRECT. -pub struct AlignedBuf { - ptr: *mut u8, - layout: Layout, -} - -impl AlignedBuf { - pub fn new(size: usize, alignment: usize) -> io::Result { - let size = size.max(1).next_multiple_of(alignment); - let layout = Layout::from_size_align(size, alignment) - .map_err(|e| io::Error::other(format!("invalid aligned layout: {e}")))?; - // SAFETY: layout has non-zero size. - let ptr = unsafe { alloc_zeroed(layout) }; - if ptr.is_null() { - return Err(io::Error::new( - io::ErrorKind::OutOfMemory, - "aligned allocation failed", - )); - } - Ok(AlignedBuf { ptr, layout }) - } - - pub fn as_mut_slice(&mut self, len: usize) -> &mut [u8] { - let len = len.min(self.layout.size()); - // SAFETY: ptr is valid for layout.size() bytes; len <= layout.size(). - unsafe { slice::from_raw_parts_mut(self.ptr, len) } - } - - pub fn as_slice(&self, len: usize) -> &[u8] { - let len = len.min(self.layout.size()); - // SAFETY: ptr is valid for layout.size() bytes; len <= layout.size(). - unsafe { slice::from_raw_parts(self.ptr, len) } - } - - #[cfg(test)] - pub fn layout(&self) -> &Layout { - &self.layout - } - - #[cfg(test)] - pub fn ptr(&self) -> *const u8 { - self.ptr - } -} - -impl Drop for AlignedBuf { - fn drop(&mut self) { - // SAFETY: ptr was allocated by alloc_zeroed with self.layout. - unsafe { dealloc(self.ptr, self.layout) }; - } -} - -/// Read into `buf` via an aligned bounce buffer when O_DIRECT requires it. -pub fn aligned_pread(fd: RawFd, buf: &mut [u8], offset: u64, alignment: usize) -> io::Result<()> { - if alignment == 0 - || ((buf.as_ptr() as usize).is_multiple_of(alignment) - && buf.len().is_multiple_of(alignment) - && (offset as usize).is_multiple_of(alignment)) - { - return pread_exact(fd, buf, offset); - } - - let aligned_offset = offset & !(alignment as u64 - 1); - let head = (offset - aligned_offset) as usize; - let aligned_len = (head + buf.len()).next_multiple_of(alignment); - let mut bounce = AlignedBuf::new(aligned_len, alignment)?; - pread_exact(fd, bounce.as_mut_slice(aligned_len), aligned_offset)?; - buf.copy_from_slice(&bounce.as_slice(aligned_len)[head..head + buf.len()]); - Ok(()) -} - -/// Write `buf` via an aligned bounce buffer when O_DIRECT requires it. -pub fn aligned_pwrite(fd: RawFd, buf: &[u8], offset: u64, alignment: usize) -> io::Result<()> { - if alignment == 0 - || ((buf.as_ptr() as usize).is_multiple_of(alignment) - && buf.len().is_multiple_of(alignment) - && (offset as usize).is_multiple_of(alignment)) - { - return pwrite_all(fd, buf, offset); - } - - let aligned_offset = offset & !(alignment as u64 - 1); - let head = (offset - aligned_offset) as usize; - let aligned_len = (head + buf.len()).next_multiple_of(alignment); - let mut bounce = AlignedBuf::new(aligned_len, alignment)?; - - // Read-modify-write: read the existing aligned region, overlay our data. - pread_exact(fd, bounce.as_mut_slice(aligned_len), aligned_offset)?; - bounce.as_mut_slice(aligned_len)[head..head + buf.len()].copy_from_slice(buf); - pwrite_all(fd, bounce.as_slice(aligned_len), aligned_offset) -} - #[cfg(test)] pub(crate) mod unit_tests { use std::fs::File; diff --git a/block/src/formats/qcow/worker/async_uring.rs b/block/src/formats/qcow/worker/async_uring.rs index d795040fe..4844aa8da 100644 --- a/block/src/formats/qcow/worker/async_uring.rs +++ b/block/src/formats/qcow/worker/async_uring.rs @@ -16,16 +16,14 @@ use std::sync::Arc; use vmm_sys_util::eventfd::EventFd; use vmm_sys_util::write_zeroes::{PunchHole, WriteZeroesAt}; -use super::common::{ - AlignedBuf, aligned_pread, aligned_pwrite, decompress_cluster, pread_alloc, pread_exact, - pwrite_all, -}; +use super::common::{decompress_cluster, pread_alloc, pread_exact, pwrite_all}; use super::internal::decoder::Decoder; use super::internal::metadata::{ BackingRead, ClusterReadMapping, ClusterWriteMapping, DeallocAction, QcowMetadata, }; use super::internal::qcow_raw_file::QcowRawFile; use crate::SECTOR_SIZE; +use crate::aligned_buffer::AlignedBuffer; use crate::async_io::{ AsyncIo, AsyncIoCompletion, AsyncIoError, AsyncIoOperation, AsyncIoResult, UringDataIo, }; @@ -393,16 +391,11 @@ impl QcowAsync { } => { let len = length as usize; if alignment > 0 { - let mut abuf = - AlignedBuf::new(len, alignment).map_err(AsyncIoError::ReadVectored)?; - aligned_pread( - data_file.as_raw_fd(), - abuf.as_mut_slice(len), - host_offset, - alignment, - ) - .map_err(AsyncIoError::ReadVectored)?; - op.write_bytes_at(buf_offset, abuf.as_slice(len)) + let mut abuf = AlignedBuffer::new(host_offset, len, alignment) + .map_err(AsyncIoError::ReadVectored)?; + abuf.read_exact_from(data_file.file()) + .map_err(AsyncIoError::ReadVectored)?; + op.write_bytes_at(buf_offset, abuf.as_slice()) .map_err(AsyncIoError::ReadVectored)?; } else { let mut buf = vec![0u8; len]; @@ -493,20 +486,15 @@ impl QcowAsync { offset: host_offset, } => { if alignment > 0 { - // O_DIRECT, gather directly into aligned buffer. - let mut abuf = AlignedBuf::new(count, alignment) + let mut abuf = AlignedBuffer::new(host_offset, count, alignment) .map_err(AsyncIoError::WriteVectored)?; - op.read_bytes_at(buf_offset, abuf.as_mut_slice(count)) + abuf.read_exact_from(data_file.file()) + .map_err(AsyncIoError::WriteVectored)?; + op.read_bytes_at(buf_offset, abuf.as_mut_slice()) + .map_err(AsyncIoError::WriteVectored)?; + abuf.write_to(data_file.file()) .map_err(AsyncIoError::WriteVectored)?; - aligned_pwrite( - data_file.as_raw_fd(), - abuf.as_slice(count), - host_offset, - alignment, - ) - .map_err(AsyncIoError::WriteVectored)?; } else { - // No O_DIRECT, plain buffer is fine. let mut buf = vec![0u8; count]; op.read_bytes_at(buf_offset, &mut buf) .map_err(AsyncIoError::WriteVectored)?; diff --git a/block/src/formats/qcow/worker/sync.rs b/block/src/formats/qcow/worker/sync.rs index 4ed6170df..ac60726de 100644 --- a/block/src/formats/qcow/worker/sync.rs +++ b/block/src/formats/qcow/worker/sync.rs @@ -12,15 +12,13 @@ use std::sync::Arc; use vmm_sys_util::eventfd::EventFd; use vmm_sys_util::write_zeroes::{PunchHole, WriteZeroesAt}; -use super::common::{ - AlignedBuf, aligned_pread, aligned_pwrite, decompress_cluster, pread_alloc, pread_exact, - pwrite_all, -}; +use super::common::{decompress_cluster, pread_alloc, pread_exact, pwrite_all}; use super::internal::decoder::Decoder; use super::internal::metadata::{ BackingRead, ClusterReadMapping, ClusterWriteMapping, DeallocAction, QcowMetadata, }; use super::internal::qcow_raw_file::QcowRawFile; +use crate::aligned_buffer::AlignedBuffer; use crate::async_io::{AsyncIo, AsyncIoCompletion, AsyncIoError, AsyncIoOperation, AsyncIoResult}; pub struct QcowSync { @@ -103,20 +101,13 @@ impl QcowSync { } => { let len = length as usize; if self.alignment > 0 { - // O_DIRECT, aligned buffer avoids bounce copy. - let mut abuf = AlignedBuf::new(len, self.alignment) + let mut abuf = AlignedBuffer::new(host_offset, len, self.alignment) .map_err(AsyncIoError::ReadVectored)?; - aligned_pread( - self.data_file.as_raw_fd(), - abuf.as_mut_slice(len), - host_offset, - self.alignment, - ) - .map_err(AsyncIoError::ReadVectored)?; - op.write_bytes_at(buf_offset, abuf.as_slice(len)) + abuf.read_exact_from(self.data_file.file()) + .map_err(AsyncIoError::ReadVectored)?; + op.write_bytes_at(buf_offset, abuf.as_slice()) .map_err(AsyncIoError::ReadVectored)?; } else { - // No O_DIRECT, plain buffer is fine. let mut buf = vec![0u8; len]; pread_exact(self.data_file.as_raw_fd(), &mut buf, host_offset) .map_err(AsyncIoError::ReadVectored)?; @@ -202,20 +193,15 @@ impl QcowSync { offset: host_offset, } => { if self.alignment > 0 { - // O_DIRECT, gather directly into aligned buffer. - let mut abuf = AlignedBuf::new(count, self.alignment) + let mut abuf = AlignedBuffer::new(host_offset, count, self.alignment) .map_err(AsyncIoError::WriteVectored)?; - op.read_bytes_at(buf_offset, abuf.as_mut_slice(count)) + abuf.read_exact_from(self.data_file.file()) + .map_err(AsyncIoError::WriteVectored)?; + op.read_bytes_at(buf_offset, abuf.as_mut_slice()) + .map_err(AsyncIoError::WriteVectored)?; + abuf.write_to(self.data_file.file()) .map_err(AsyncIoError::WriteVectored)?; - aligned_pwrite( - self.data_file.as_raw_fd(), - abuf.as_slice(count), - host_offset, - self.alignment, - ) - .map_err(AsyncIoError::WriteVectored)?; } else { - // No O_DIRECT, plain buffer is fine. let mut buf = vec![0u8; count]; op.read_bytes_at(buf_offset, &mut buf) .map_err(AsyncIoError::WriteVectored)?; @@ -340,7 +326,6 @@ impl AsyncIo for QcowSync { mod unit_tests { use std::fs::{File, OpenOptions, create_dir}; use std::io::{Read, Seek, SeekFrom, Write}; - use std::os::fd::RawFd; use std::path::Path; use std::sync::Arc; use std::{env, thread}; @@ -1994,198 +1979,6 @@ mod unit_tests { test_multi_iovec_read_write_impl(true); } - // -- Low level aligned I/O function tests -- - // - // Test aligned_pread and aligned_pwrite directly with controlled - // alignment values on a plain temp file. - - /// Create a temp file filled with a repeating pattern of the given size. - /// Returns the TempFile (must be kept alive) and the raw fd. - fn create_pattern_file(size: usize) -> (TempFile, RawFd) { - let tf = TempFile::new().unwrap(); - let pattern: Vec = (0..size).map(|i| (i % 251) as u8).collect(); - tf.as_file().write_all(&pattern).unwrap(); - tf.as_file().sync_all().unwrap(); - let fd = tf.as_file().as_raw_fd(); - (tf, fd) - } - - #[test] - fn test_aligned_pread_pass_through() { - // When buffer address, length, and offset are all aligned, - // aligned_pread should take the fast path (no bounce buffer). - let size = 4096usize; - let (_tf, fd) = create_pattern_file(size); - let alignment = 512; - - // Use AlignedBuf to guarantee buffer address alignment. - let mut abuf = AlignedBuf::new(size, alignment).unwrap(); - aligned_pread(fd, abuf.as_mut_slice(size), 0, alignment).unwrap(); - - let expected: Vec = (0..size).map(|i| (i % 251) as u8).collect(); - assert_eq!(abuf.as_slice(size), &expected[..]); - } - - #[test] - fn test_aligned_pread_bounce_unaligned_buffer() { - // Force a misaligned buffer so aligned_pread must take the - // bounce path. A plain vec![0u8; 4096] is often page-aligned - // by the allocator, which would skip the bounce entirely. - let size = 4096usize; - let (_tf, fd) = create_pattern_file(size); - let alignment = 512; - - let mut backing = vec![0u8; size + 1]; - let buf = &mut backing[1..size + 1]; - aligned_pread(fd, buf, 0, alignment).unwrap(); - - let expected: Vec = (0..size).map(|i| (i % 251) as u8).collect(); - assert_eq!(buf, &expected[..]); - } - - #[test] - fn test_aligned_pread_unaligned_offset() { - // Read at an offset that is not a multiple of alignment. - // aligned_pread should round down the offset, read an aligned - // region, then copy the correct slice into the caller buffer. - let file_size = 8192usize; - let (_tf, fd) = create_pattern_file(file_size); - let alignment = 512; - - let offset = 100u64; - let len = 200usize; - let mut buf = vec![0u8; len]; - aligned_pread(fd, &mut buf, offset, alignment).unwrap(); - - let expected: Vec = (offset as usize..offset as usize + len) - .map(|i| (i % 251) as u8) - .collect(); - assert_eq!(buf, expected); - } - - #[test] - fn test_aligned_pwrite_pass_through() { - // When buffer address, length, and offset are all aligned, - // aligned_pwrite should take the fast path. - let size = 4096usize; - let (_tf, fd) = create_pattern_file(size); - let alignment = 512; - - let data: Vec = (0..size).map(|i| ((i + 1) % 251) as u8).collect(); - let mut abuf = AlignedBuf::new(size, alignment).unwrap(); - abuf.as_mut_slice(size).copy_from_slice(&data); - aligned_pwrite(fd, abuf.as_slice(size), 0, alignment).unwrap(); - - let mut readback = vec![0u8; size]; - pread_exact(fd, &mut readback, 0).unwrap(); - assert_eq!(readback, data); - } - - #[test] - fn test_aligned_pwrite_bounce_unaligned_buffer() { - // Force a misaligned buffer so aligned_pwrite must take the - // bounce path. A plain vec![0u8; 4096] is often page-aligned - // by the allocator, which would skip the bounce entirely. - let size = 4096usize; - let (_tf, fd) = create_pattern_file(size); - let alignment = 512; - - let backing: Vec = (0..size + 1).map(|i| ((i + 1) % 251) as u8).collect(); - let data = &backing[1..size + 1]; - aligned_pwrite(fd, data, 0, alignment).unwrap(); - - let mut readback = vec![0u8; size]; - pread_exact(fd, &mut readback, 0).unwrap(); - assert_eq!(readback, data); - } - - #[test] - fn test_aligned_pwrite_unaligned_offset() { - // Write at an offset that is not a multiple of alignment. - // aligned_pwrite should do read-modify-write and preserve - // surrounding data. - let file_size = 8192usize; - let (_tf, fd) = create_pattern_file(file_size); - let alignment = 512; - - let offset = 100u64; - let len = 200usize; - let data: Vec = (0..len).map(|i| ((i + 1) % 239) as u8).collect(); - aligned_pwrite(fd, &data, offset, alignment).unwrap(); - - // Read entire file and verify the written region plus untouched areas. - let mut whole = vec![0u8; file_size]; - pread_exact(fd, &mut whole, 0).unwrap(); - - // Before the write region: original pattern. - let before: Vec = (0..offset as usize).map(|i| (i % 251) as u8).collect(); - assert_eq!(&whole[..offset as usize], &before[..]); - - // The written region. - assert_eq!(&whole[offset as usize..offset as usize + len], &data[..]); - - // After the write region: original pattern. - let after_start = offset as usize + len; - let after: Vec = (after_start..file_size).map(|i| (i % 251) as u8).collect(); - assert_eq!(&whole[after_start..], &after[..]); - } - - #[test] - fn test_aligned_pread_pwrite_4096_alignment() { - // Exercise aligned I/O with 4096 byte alignment. - let file_size = 16384usize; - let (_tf, fd) = create_pattern_file(file_size); - let alignment = 4096; - - // Write 4096 bytes at offset 4096 via unaligned Vec. - let offset = 4096u64; - let len = 4096usize; - let data: Vec = (0..len).map(|i| ((i + 1) % 239) as u8).collect(); - aligned_pwrite(fd, &data, offset, alignment).unwrap(); - - // Read back the written region via unaligned Vec. - let mut buf = vec![0u8; len]; - aligned_pread(fd, &mut buf, offset, alignment).unwrap(); - assert_eq!(buf, data); - - // Verify untouched regions. - let mut whole = vec![0u8; file_size]; - pread_exact(fd, &mut whole, 0).unwrap(); - let before: Vec = (0..offset as usize).map(|i| (i % 251) as u8).collect(); - assert_eq!(&whole[..offset as usize], &before[..]); - let after_start = offset as usize + len; - let after: Vec = (after_start..file_size).map(|i| (i % 251) as u8).collect(); - assert_eq!(&whole[after_start..], &after[..]); - } - - #[test] - fn test_aligned_buf_allocation_and_access() { - for alignment in [512, 4096] { - let size = 1024usize; - let mut abuf = AlignedBuf::new(size, alignment).unwrap(); - let aligned_size = size.next_multiple_of(alignment); - - assert!( - (abuf.ptr() as usize).is_multiple_of(alignment), - "ptr not aligned to {alignment}" - ); - assert!(abuf.as_slice(aligned_size).iter().all(|&b| b == 0)); - - let pattern: Vec = (0..size).map(|i| (i % 251) as u8).collect(); - abuf.as_mut_slice(size).copy_from_slice(&pattern); - assert_eq!(abuf.as_slice(size), &pattern[..]); - } - } - - #[test] - fn test_aligned_buf_size_rounds_up() { - let abuf = AlignedBuf::new(1, 512).unwrap(); - assert_eq!(abuf.layout().size(), 512); - - let abuf = AlignedBuf::new(513, 512).unwrap(); - assert_eq!(abuf.layout().size(), 1024); - } - #[test] fn test_compressed_read() { let cluster_size = 65536usize;