mirror of
https://github.com/cloud-hypervisor/cloud-hypervisor.git
synced 2026-08-05 02:19:16 +00:00
block: Move QCOW2 format files into formats/qcow/
Move QCOW2 format implementation into a structured directory layout: qcow/ -> formats/qcow/internal/ (filenames unchanged) qcow_disk.rs -> formats/qcow/mod.rs (QcowDisk) qcow_sync.rs -> formats/qcow/worker/sync.rs (QcowSync) qcow_async.rs -> formats/qcow/worker/async_uring.rs (QcowAsync) qcow_common.rs -> formats/qcow/common.rs All internal cross references continue to resolve through re-exports in lib.rs: formats::qcow::internal as qcow, formats::qcow as qcow_disk, and formats::qcow::common as qcow_common. Signed-off-by: Anatol Belski <anbelski@linux.microsoft.com>
This commit is contained in:
committed by
Rob Bradford
parent
b595f1dbc3
commit
25afb8898c
@@ -7,6 +7,7 @@
|
||||
//! Each format lives in its own submodule with a `DiskFile` wrapper,
|
||||
//! format specific internals, and sync/async I/O workers.
|
||||
|
||||
pub mod qcow;
|
||||
pub mod raw;
|
||||
pub mod vhd;
|
||||
pub mod vhdx;
|
||||
|
||||
345
block/src/formats/qcow/common.rs
Normal file
345
block/src/formats/qcow/common.rs
Normal file
@@ -0,0 +1,345 @@
|
||||
// Copyright © 2021 Intel Corporation
|
||||
//
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// Copyright (c) Meta Platforms, Inc. and affiliates.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
//! Shared helpers for QCOW2 sync and async backends.
|
||||
//!
|
||||
//! Position-independent I/O helpers used by both `qcow_sync` and `qcow_async`.
|
||||
|
||||
use std::alloc::{Layout, alloc_zeroed, dealloc};
|
||||
use std::os::fd::RawFd;
|
||||
use std::{io, slice};
|
||||
|
||||
#[cfg(test)]
|
||||
use super::internal;
|
||||
use super::internal::decoder::Decoder;
|
||||
|
||||
// -- Position independent I/O helpers --
|
||||
//
|
||||
// Duplicated file descriptors share the kernel file description and thus the
|
||||
// file position. Using seek then read from multiple queues races on that
|
||||
// shared position. pread64 and pwrite64 are atomic and never touch the position.
|
||||
|
||||
/// Read exactly the requested bytes at offset, looping on short reads.
|
||||
pub fn pread_exact(fd: RawFd, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
let mut total = 0usize;
|
||||
while total < buf.len() {
|
||||
// SAFETY: buf and fd are valid for the lifetime of the call.
|
||||
let ret = unsafe {
|
||||
libc::pread64(
|
||||
fd,
|
||||
buf[total..].as_mut_ptr().cast(),
|
||||
buf.len() - total,
|
||||
(offset + total as u64) as libc::off_t,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
if ret == 0 {
|
||||
return Err(io::Error::from(io::ErrorKind::UnexpectedEof));
|
||||
}
|
||||
total += ret as usize;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Allocate a buffer and pread exactly `len` bytes at `offset`.
|
||||
pub fn pread_alloc(fd: RawFd, offset: u64, len: usize) -> io::Result<Vec<u8>> {
|
||||
let mut buf = vec![0u8; len];
|
||||
pread_exact(fd, &mut buf, offset)?;
|
||||
Ok(buf)
|
||||
}
|
||||
|
||||
/// Decompress a full QCOW2 cluster from compressed data.
|
||||
///
|
||||
/// Returns a `cluster_size` byte buffer with the decompressed cluster
|
||||
/// content. Fails if the decoder does not produce exactly `cluster_size`
|
||||
/// bytes.
|
||||
pub fn decompress_cluster(
|
||||
compressed: &[u8],
|
||||
cluster_size: usize,
|
||||
decoder: &dyn Decoder,
|
||||
) -> io::Result<Vec<u8>> {
|
||||
let mut decompressed = vec![0u8; cluster_size];
|
||||
let n = decoder
|
||||
.decode(compressed, &mut decompressed)
|
||||
.map_err(|_| io::Error::from_raw_os_error(libc::EIO))?;
|
||||
if n != cluster_size {
|
||||
return Err(io::Error::from_raw_os_error(libc::EIO));
|
||||
}
|
||||
Ok(decompressed)
|
||||
}
|
||||
|
||||
/// Write all bytes to fd at offset, looping on short writes.
|
||||
pub fn pwrite_all(fd: RawFd, buf: &[u8], offset: u64) -> io::Result<()> {
|
||||
let mut total = 0usize;
|
||||
while total < buf.len() {
|
||||
// SAFETY: buf and fd are valid for the lifetime of the call.
|
||||
let ret = unsafe {
|
||||
libc::pwrite64(
|
||||
fd,
|
||||
buf[total..].as_ptr().cast(),
|
||||
buf.len() - total,
|
||||
(offset + total as u64) as libc::off_t,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
if ret == 0 {
|
||||
return Err(io::Error::other("pwrite64 wrote 0 bytes"));
|
||||
}
|
||||
total += ret as usize;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// RAII wrapper for an aligned heap buffer required by O_DIRECT.
|
||||
pub struct AlignedBuf {
|
||||
ptr: *mut u8,
|
||||
layout: Layout,
|
||||
}
|
||||
|
||||
impl AlignedBuf {
|
||||
pub fn new(size: usize, alignment: usize) -> io::Result<Self> {
|
||||
let size = size.max(1).next_multiple_of(alignment);
|
||||
let layout = Layout::from_size_align(size, alignment)
|
||||
.map_err(|e| io::Error::other(format!("invalid aligned layout: {e}")))?;
|
||||
// SAFETY: layout has non-zero size.
|
||||
let ptr = unsafe { alloc_zeroed(layout) };
|
||||
if ptr.is_null() {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::OutOfMemory,
|
||||
"aligned allocation failed",
|
||||
));
|
||||
}
|
||||
Ok(AlignedBuf { ptr, layout })
|
||||
}
|
||||
|
||||
pub fn as_mut_slice(&mut self, len: usize) -> &mut [u8] {
|
||||
let len = len.min(self.layout.size());
|
||||
// SAFETY: ptr is valid for layout.size() bytes; len <= layout.size().
|
||||
unsafe { slice::from_raw_parts_mut(self.ptr, len) }
|
||||
}
|
||||
|
||||
pub fn as_slice(&self, len: usize) -> &[u8] {
|
||||
let len = len.min(self.layout.size());
|
||||
// SAFETY: ptr is valid for layout.size() bytes; len <= layout.size().
|
||||
unsafe { slice::from_raw_parts(self.ptr, len) }
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub fn layout(&self) -> &Layout {
|
||||
&self.layout
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub fn ptr(&self) -> *const u8 {
|
||||
self.ptr
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for AlignedBuf {
|
||||
fn drop(&mut self) {
|
||||
// SAFETY: ptr was allocated by alloc_zeroed with self.layout.
|
||||
unsafe { dealloc(self.ptr, self.layout) };
|
||||
}
|
||||
}
|
||||
|
||||
/// Read into `buf` via an aligned bounce buffer when O_DIRECT requires it.
|
||||
pub fn aligned_pread(fd: RawFd, buf: &mut [u8], offset: u64, alignment: usize) -> io::Result<()> {
|
||||
if alignment == 0
|
||||
|| ((buf.as_ptr() as usize).is_multiple_of(alignment)
|
||||
&& buf.len().is_multiple_of(alignment)
|
||||
&& (offset as usize).is_multiple_of(alignment))
|
||||
{
|
||||
return pread_exact(fd, buf, offset);
|
||||
}
|
||||
|
||||
let aligned_offset = offset & !(alignment as u64 - 1);
|
||||
let head = (offset - aligned_offset) as usize;
|
||||
let aligned_len = (head + buf.len()).next_multiple_of(alignment);
|
||||
let mut bounce = AlignedBuf::new(aligned_len, alignment)?;
|
||||
pread_exact(fd, bounce.as_mut_slice(aligned_len), aligned_offset)?;
|
||||
buf.copy_from_slice(&bounce.as_slice(aligned_len)[head..head + buf.len()]);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Write `buf` via an aligned bounce buffer when O_DIRECT requires it.
|
||||
pub fn aligned_pwrite(fd: RawFd, buf: &[u8], offset: u64, alignment: usize) -> io::Result<()> {
|
||||
if alignment == 0
|
||||
|| ((buf.as_ptr() as usize).is_multiple_of(alignment)
|
||||
&& buf.len().is_multiple_of(alignment)
|
||||
&& (offset as usize).is_multiple_of(alignment))
|
||||
{
|
||||
return pwrite_all(fd, buf, offset);
|
||||
}
|
||||
|
||||
let aligned_offset = offset & !(alignment as u64 - 1);
|
||||
let head = (offset - aligned_offset) as usize;
|
||||
let aligned_len = (head + buf.len()).next_multiple_of(alignment);
|
||||
let mut bounce = AlignedBuf::new(aligned_len, alignment)?;
|
||||
|
||||
// Read-modify-write: read the existing aligned region, overlay our data.
|
||||
pread_exact(fd, bounce.as_mut_slice(aligned_len), aligned_offset)?;
|
||||
bounce.as_mut_slice(aligned_len)[head..head + buf.len()].copy_from_slice(buf);
|
||||
pwrite_all(fd, bounce.as_slice(aligned_len), aligned_offset)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) mod unit_tests {
|
||||
use std::fs::File;
|
||||
use std::io::{Read, Seek, SeekFrom, Write};
|
||||
use std::os::unix::fs::FileExt;
|
||||
use std::os::unix::io::AsRawFd;
|
||||
|
||||
use byteorder::{BigEndian, ReadBytesExt, WriteBytesExt};
|
||||
use flate2::Compression;
|
||||
use flate2::write::DeflateEncoder;
|
||||
use vmm_sys_util::tempfile::TempFile;
|
||||
|
||||
use super::internal::decoder::ZlibDecoder;
|
||||
use super::{decompress_cluster, pread_alloc};
|
||||
|
||||
const COMPRESSED_FLAG: u64 = 1 << 62;
|
||||
const CLUSTER_USED_FLAG: u64 = 1 << 63;
|
||||
const COMPRESSED_SECTOR_SIZE: u64 = 512;
|
||||
|
||||
const HEADER_CLUSTER_BITS_OFFSET: u64 = 20;
|
||||
const HEADER_L1_SIZE_OFFSET: u64 = 36;
|
||||
const HEADER_L1_TABLE_OFFSET: u64 = 40;
|
||||
|
||||
const L1_L2_ADDR_MASK: u64 = 0x00ff_ffff_ffff_fe00;
|
||||
|
||||
fn make_compressed_l2_entry(host_offset: u64, compressed_len: usize, cluster_bits: u32) -> u64 {
|
||||
let compressed_size_shift = 62 - (cluster_bits - 8);
|
||||
let intra_sector_offset = host_offset & (COMPRESSED_SECTOR_SIZE - 1);
|
||||
let total_bytes = compressed_len as u64 + intra_sector_offset;
|
||||
let nsectors = total_bytes.div_ceil(COMPRESSED_SECTOR_SIZE);
|
||||
let addr_part = host_offset & ((1 << compressed_size_shift) - 1);
|
||||
let size_part = (nsectors - 1) << compressed_size_shift;
|
||||
COMPRESSED_FLAG | size_part | addr_part
|
||||
}
|
||||
|
||||
/// Compress every allocated cluster in a QCOW2 image file in place.
|
||||
///
|
||||
/// Walks L1 -> L2 tables, compresses each standard cluster with raw
|
||||
/// deflate, appends the compressed payload at the end of the file,
|
||||
/// and rewrites the L2 entry with the compressed layout.
|
||||
pub fn compress_allocated_clusters(file: &mut File) {
|
||||
file.seek(SeekFrom::Start(HEADER_CLUSTER_BITS_OFFSET))
|
||||
.unwrap();
|
||||
let cluster_bits = file.read_u32::<BigEndian>().unwrap();
|
||||
let cluster_size = 1u64 << cluster_bits;
|
||||
|
||||
file.seek(SeekFrom::Start(HEADER_L1_SIZE_OFFSET)).unwrap();
|
||||
let l1_size = file.read_u32::<BigEndian>().unwrap();
|
||||
|
||||
file.seek(SeekFrom::Start(HEADER_L1_TABLE_OFFSET)).unwrap();
|
||||
let l1_table_offset = file.read_u64::<BigEndian>().unwrap();
|
||||
|
||||
let entries_per_l2 = cluster_size / 8;
|
||||
|
||||
let mut append_offset = file.seek(SeekFrom::End(0)).unwrap();
|
||||
append_offset = (append_offset + 511) & !511;
|
||||
|
||||
for l1_idx in 0..l1_size as u64 {
|
||||
let l1_entry_offset = l1_table_offset + l1_idx * 8;
|
||||
file.seek(SeekFrom::Start(l1_entry_offset)).unwrap();
|
||||
let l1_entry = file.read_u64::<BigEndian>().unwrap();
|
||||
|
||||
let l2_table_addr = l1_entry & L1_L2_ADDR_MASK;
|
||||
if l2_table_addr == 0 {
|
||||
continue;
|
||||
}
|
||||
|
||||
for l2_idx in 0..entries_per_l2 {
|
||||
let l2_entry_offset = l2_table_addr + l2_idx * 8;
|
||||
file.seek(SeekFrom::Start(l2_entry_offset)).unwrap();
|
||||
let l2_entry = file.read_u64::<BigEndian>().unwrap();
|
||||
|
||||
if l2_entry & CLUSTER_USED_FLAG == 0 || l2_entry & COMPRESSED_FLAG != 0 {
|
||||
continue;
|
||||
}
|
||||
|
||||
let host_cluster_addr = l2_entry & L1_L2_ADDR_MASK;
|
||||
if host_cluster_addr == 0 {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut cluster_data = vec![0u8; cluster_size as usize];
|
||||
file.seek(SeekFrom::Start(host_cluster_addr)).unwrap();
|
||||
file.read_exact(&mut cluster_data).unwrap();
|
||||
|
||||
let mut encoder = DeflateEncoder::new(Vec::new(), Compression::default());
|
||||
encoder.write_all(&cluster_data).unwrap();
|
||||
let compressed = encoder.finish().unwrap();
|
||||
|
||||
file.seek(SeekFrom::Start(append_offset)).unwrap();
|
||||
file.write_all(&compressed).unwrap();
|
||||
|
||||
// The L2 entry encodes the compressed size in units of
|
||||
// 512 byte sectors. The reader decodes the sector count
|
||||
// back and computes: nsectors * 512 - (addr & 511).
|
||||
// Because addr is 512 aligned, this yields nsectors * 512
|
||||
// which rounds up to the next sector boundary. The file
|
||||
// must contain enough bytes for that rounded up pread.
|
||||
let padded_len = (compressed.len() + 511) & !511;
|
||||
if padded_len > compressed.len() {
|
||||
let padding = vec![0u8; padded_len - compressed.len()];
|
||||
file.write_all(&padding).unwrap();
|
||||
}
|
||||
|
||||
let new_entry =
|
||||
make_compressed_l2_entry(append_offset, compressed.len(), cluster_bits);
|
||||
file.seek(SeekFrom::Start(l2_entry_offset)).unwrap();
|
||||
file.write_u64::<BigEndian>(new_entry).unwrap();
|
||||
|
||||
append_offset += padded_len as u64;
|
||||
}
|
||||
}
|
||||
|
||||
file.flush().unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_pread_alloc() {
|
||||
let temp = TempFile::new().unwrap();
|
||||
let file = temp.as_file();
|
||||
let data: Vec<u8> = (0..=255).cycle().take(4096).collect();
|
||||
file.write_all_at(&data, 0).unwrap();
|
||||
|
||||
let buf = pread_alloc(file.as_raw_fd(), 0, 4096).unwrap();
|
||||
assert_eq!(buf, data);
|
||||
|
||||
let buf = pread_alloc(file.as_raw_fd(), 100, 200).unwrap();
|
||||
assert_eq!(buf, &data[100..300]);
|
||||
|
||||
pread_alloc(file.as_raw_fd(), 4000, 200).unwrap_err();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_decompress_cluster() {
|
||||
let cluster_size = 65536;
|
||||
let original: Vec<u8> = (0..=255).cycle().take(cluster_size).collect();
|
||||
|
||||
let mut encoder = DeflateEncoder::new(Vec::new(), Compression::default());
|
||||
encoder.write_all(&original).unwrap();
|
||||
let compressed = encoder.finish().unwrap();
|
||||
|
||||
let result = decompress_cluster(&compressed, cluster_size, &ZlibDecoder {}).unwrap();
|
||||
assert_eq!(result, original);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_decompress_cluster_corrupt_input() {
|
||||
let corrupt = vec![0xffu8; 64];
|
||||
let err = decompress_cluster(&corrupt, 65536, &ZlibDecoder {}).unwrap_err();
|
||||
assert_eq!(err.raw_os_error(), Some(libc::EIO));
|
||||
}
|
||||
}
|
||||
184
block/src/formats/qcow/internal/backing.rs
Normal file
184
block/src/formats/qcow/internal/backing.rs
Normal file
@@ -0,0 +1,184 @@
|
||||
// Copyright © 2021 Intel Corporation
|
||||
//
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
//! Thread safe backing file readers for QCOW2 images.
|
||||
|
||||
use std::io;
|
||||
use std::os::fd::{AsFd, AsRawFd, BorrowedFd, OwnedFd};
|
||||
use std::sync::Arc;
|
||||
|
||||
use super::decoder::Decoder;
|
||||
use super::metadata::{BackingRead, ClusterReadMapping, QcowMetadata};
|
||||
use super::{BackingFile, BackingKind, Error as QcowError};
|
||||
use crate::error::{BlockError, BlockErrorKind, BlockResult, ErrorOp};
|
||||
use crate::formats::qcow::common::{decompress_cluster, pread_alloc, pread_exact};
|
||||
|
||||
/// Raw backing file using pread64 on a duplicated fd.
|
||||
pub(crate) struct RawBacking {
|
||||
pub(crate) fd: OwnedFd,
|
||||
pub(crate) virtual_size: u64,
|
||||
}
|
||||
|
||||
// SAFETY: The only I/O operation is pread64 which is position independent
|
||||
// and safe for concurrent use from multiple threads.
|
||||
unsafe impl Sync for RawBacking {}
|
||||
|
||||
impl BackingRead for RawBacking {
|
||||
fn read_at(&self, address: u64, buf: &mut [u8]) -> io::Result<()> {
|
||||
if address >= self.virtual_size {
|
||||
buf.fill(0);
|
||||
return Ok(());
|
||||
}
|
||||
let available = (self.virtual_size - address) as usize;
|
||||
if available >= buf.len() {
|
||||
pread_exact(self.fd.as_raw_fd(), buf, address)
|
||||
} else {
|
||||
pread_exact(self.fd.as_raw_fd(), &mut buf[..available], address)?;
|
||||
buf[available..].fill(0);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// QCOW2 image used as a backing file for another QCOW2 image.
|
||||
///
|
||||
/// Resolves guest offsets through the QCOW2 cluster mapping (L1/L2
|
||||
/// tables, refcounts) before reading the underlying data. Read only
|
||||
/// because backing files never receive writes. Nested backing chains
|
||||
/// are handled recursively via the optional `backing_file` field.
|
||||
pub(crate) struct Qcow2Backing {
|
||||
pub(crate) metadata: Arc<QcowMetadata>,
|
||||
pub(crate) data_fd: OwnedFd,
|
||||
pub(crate) backing_file: Option<Arc<dyn BackingRead>>,
|
||||
pub(crate) cluster_size: u64,
|
||||
pub(crate) decoder: Arc<dyn Decoder>,
|
||||
}
|
||||
|
||||
// SAFETY: All reads go through QcowMetadata which uses RwLock
|
||||
// and pread64 which is position independent and thread safe.
|
||||
unsafe impl Sync for Qcow2Backing {}
|
||||
|
||||
impl BackingRead for Qcow2Backing {
|
||||
fn read_at(&self, address: u64, buf: &mut [u8]) -> io::Result<()> {
|
||||
let virtual_size = self.metadata.virtual_size();
|
||||
if address >= virtual_size {
|
||||
buf.fill(0);
|
||||
return Ok(());
|
||||
}
|
||||
let available = (virtual_size - address) as usize;
|
||||
if available < buf.len() {
|
||||
self.read_clusters(address, &mut buf[..available])?;
|
||||
buf[available..].fill(0);
|
||||
return Ok(());
|
||||
}
|
||||
self.read_clusters(address, buf)
|
||||
}
|
||||
}
|
||||
|
||||
impl Qcow2Backing {
|
||||
/// Resolve cluster mappings via metadata then read allocated clusters
|
||||
/// with pread64.
|
||||
fn read_clusters(&self, address: u64, buf: &mut [u8]) -> io::Result<()> {
|
||||
let total_len = buf.len();
|
||||
let has_backing = self.backing_file.is_some();
|
||||
|
||||
let mappings = self
|
||||
.metadata
|
||||
.map_clusters_for_read(address, total_len, has_backing)?;
|
||||
|
||||
let mut buf_offset = 0usize;
|
||||
for mapping in mappings {
|
||||
match mapping {
|
||||
ClusterReadMapping::Zero { length } => {
|
||||
buf[buf_offset..buf_offset + length as usize].fill(0);
|
||||
buf_offset += length as usize;
|
||||
}
|
||||
ClusterReadMapping::Allocated {
|
||||
offset: host_offset,
|
||||
length,
|
||||
} => {
|
||||
pread_exact(
|
||||
self.data_fd.as_raw_fd(),
|
||||
&mut buf[buf_offset..buf_offset + length as usize],
|
||||
host_offset,
|
||||
)?;
|
||||
buf_offset += length as usize;
|
||||
}
|
||||
ClusterReadMapping::Compressed {
|
||||
host_offset,
|
||||
compressed_size,
|
||||
cluster_offset,
|
||||
length,
|
||||
} => {
|
||||
let compressed =
|
||||
pread_alloc(self.data_fd.as_raw_fd(), host_offset, compressed_size)?;
|
||||
let decompressed = decompress_cluster(
|
||||
&compressed,
|
||||
self.cluster_size as usize,
|
||||
&*self.decoder,
|
||||
)?;
|
||||
buf[buf_offset..buf_offset + length]
|
||||
.copy_from_slice(&decompressed[cluster_offset..cluster_offset + length]);
|
||||
buf_offset += length;
|
||||
}
|
||||
ClusterReadMapping::Backing {
|
||||
offset: backing_offset,
|
||||
length,
|
||||
} => {
|
||||
self.backing_file.as_ref().unwrap().read_at(
|
||||
backing_offset,
|
||||
&mut buf[buf_offset..buf_offset + length as usize],
|
||||
)?;
|
||||
buf_offset += length as usize;
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for Qcow2Backing {
|
||||
fn drop(&mut self) {
|
||||
self.metadata.shutdown();
|
||||
}
|
||||
}
|
||||
|
||||
/// Construct a thread safe backing file reader.
|
||||
pub fn shared_backing_from(bf: BackingFile) -> BlockResult<Arc<dyn BackingRead>> {
|
||||
let (kind, virtual_size) = bf.into_kind();
|
||||
|
||||
let dup_fd = |fd: BorrowedFd<'_>| -> BlockResult<OwnedFd> {
|
||||
fd.try_clone_to_owned().map_err(|e| {
|
||||
BlockError::new(
|
||||
BlockErrorKind::Io,
|
||||
QcowError::BackingFileIo(String::new(), e),
|
||||
)
|
||||
.with_op(ErrorOp::DupBackingFd)
|
||||
})
|
||||
};
|
||||
|
||||
match kind {
|
||||
BackingKind::Raw(raw_file) => {
|
||||
let fd = dup_fd(raw_file.as_fd())?;
|
||||
Ok(Arc::new(RawBacking { fd, virtual_size }))
|
||||
}
|
||||
BackingKind::Qcow { inner, backing } => {
|
||||
let data_fd = dup_fd(inner.raw_file.as_fd())?;
|
||||
let metadata = Arc::new(QcowMetadata::new(*inner));
|
||||
Ok(Arc::new(Qcow2Backing {
|
||||
cluster_size: metadata.cluster_size(),
|
||||
decoder: metadata.decoder(),
|
||||
metadata,
|
||||
data_fd,
|
||||
backing_file: backing.map(|bf| shared_backing_from(*bf)).transpose()?,
|
||||
}))
|
||||
}
|
||||
#[cfg(test)]
|
||||
BackingKind::QcowFile(_) => {
|
||||
unreachable!("QcowFile variant is only used by set_backing_file() in tests")
|
||||
}
|
||||
}
|
||||
}
|
||||
87
block/src/formats/qcow/internal/decoder.rs
Normal file
87
block/src/formats/qcow/internal/decoder.rs
Normal file
@@ -0,0 +1,87 @@
|
||||
// Copyright 2025 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
pub enum Error {
|
||||
#[error("Zlib decompress error")]
|
||||
ZlibDecompress(#[source] flate2::DecompressError),
|
||||
#[error("Zlib unexpected status: {0:?}")]
|
||||
ZlibUnexpectedStatus(flate2::Status),
|
||||
#[error("Zstd decompress error")]
|
||||
ZstdDecompress(#[source] std::io::Error),
|
||||
#[error("Zstd: failed to fill buffer")]
|
||||
ZstdFillBuffer(#[source] std::io::Error),
|
||||
}
|
||||
|
||||
pub type Result<T> = std::result::Result<T, Error>;
|
||||
|
||||
/// Generic trait for decoding zlib/zstd formats
|
||||
pub trait Decoder: Send + Sync {
|
||||
fn decode(&self, input: &[u8], output: &mut [u8]) -> Result<usize>;
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ZlibDecoder {}
|
||||
|
||||
impl Decoder for ZlibDecoder {
|
||||
fn decode(&self, input: &[u8], output: &mut [u8]) -> Result<usize> {
|
||||
use flate2::{Decompress, FlushDecompress, Status};
|
||||
|
||||
let mut decompressor = Decompress::new(false);
|
||||
let status = decompressor
|
||||
.decompress(input, output, FlushDecompress::Finish)
|
||||
.map_err(Error::ZlibDecompress)?;
|
||||
if status == Status::StreamEnd {
|
||||
Ok(decompressor.total_out() as usize)
|
||||
} else {
|
||||
Err(Error::ZlibUnexpectedStatus(status))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ZstdDecoder {}
|
||||
|
||||
impl Decoder for ZstdDecoder {
|
||||
fn decode(&self, input: &[u8], output: &mut [u8]) -> Result<usize> {
|
||||
use std::io::Read;
|
||||
|
||||
let mut decoder = zstd::stream::read::Decoder::new(input).map_err(Error::ZstdDecompress)?;
|
||||
let decoded_size = decoder.read(output).map_err(Error::ZstdFillBuffer)?;
|
||||
Ok(decoded_size)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_zlib_decode() {
|
||||
let d = ZlibDecoder::default();
|
||||
let valid_input = vec![99, 96, 100, 98, 6, 0];
|
||||
let mut output1 = vec![0; 4];
|
||||
d.decode(&valid_input, &mut output1).unwrap();
|
||||
assert_eq!(&output1, b"\x00\x01\x02\x03");
|
||||
|
||||
let invalid_input = vec![1, 2, 3, 4];
|
||||
let mut output2 = vec![0; 1024];
|
||||
d.decode(&invalid_input, &mut output2).unwrap_err();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_zstd_decode() {
|
||||
let d = ZstdDecoder::default();
|
||||
let valid_input = vec![40, 181, 47, 253, 32, 2, 17, 0, 0, 1, 254];
|
||||
let mut output1 = vec![0; 2];
|
||||
d.decode(&valid_input, &mut output1).unwrap();
|
||||
assert_eq!(&output1, b"\x01\xfe");
|
||||
|
||||
let invalid_input = vec![1, 2, 3, 4];
|
||||
let mut output2 = vec![0; 1024];
|
||||
d.decode(&invalid_input, &mut output2).unwrap_err();
|
||||
}
|
||||
}
|
||||
605
block/src/formats/qcow/internal/header.rs
Normal file
605
block/src/formats/qcow/internal/header.rs
Normal file
@@ -0,0 +1,605 @@
|
||||
// Copyright 2018 The Chromium OS Authors. All rights reserved.
|
||||
// Use of this source code is governed by a BSD-style license that can be
|
||||
// found in the LICENSE-BSD-3-Clause file.
|
||||
//
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
//! QCOW2 header parsing, validation, and creation.
|
||||
|
||||
use std::fmt::{Display, Formatter, Result as FmtResult};
|
||||
use std::io::{Read, Seek, SeekFrom, Write};
|
||||
use std::mem::size_of;
|
||||
use std::str::FromStr;
|
||||
|
||||
use bitflags::bitflags;
|
||||
use vmm_sys_util::file_traits::FileSync;
|
||||
|
||||
use super::decoder::{Decoder, ZlibDecoder, ZstdDecoder};
|
||||
use super::qcow_raw_file::BeUint;
|
||||
use super::raw_file::RawFile;
|
||||
use super::{Error, Result, div_round_up_u32, div_round_up_u64};
|
||||
use crate::error::{BlockError, BlockErrorKind, BlockResult};
|
||||
|
||||
#[derive(Copy, Clone, Debug, PartialEq, Eq)]
|
||||
pub enum ImageType {
|
||||
Raw,
|
||||
Qcow2,
|
||||
}
|
||||
|
||||
impl Display for ImageType {
|
||||
fn fmt(&self, f: &mut Formatter<'_>) -> FmtResult {
|
||||
match self {
|
||||
ImageType::Raw => write!(f, "raw"),
|
||||
ImageType::Qcow2 => write!(f, "qcow2"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl FromStr for ImageType {
|
||||
type Err = Error;
|
||||
|
||||
fn from_str(s: &str) -> Result<Self> {
|
||||
match s {
|
||||
"raw" => Ok(ImageType::Raw),
|
||||
"qcow2" => Ok(ImageType::Qcow2),
|
||||
_ => Err(Error::UnsupportedBackingFileFormat(s.to_string())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub enum CompressionType {
|
||||
Zlib,
|
||||
Zstd,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct BackingFileConfig {
|
||||
pub path: String,
|
||||
// If this is None, we will autodetect it.
|
||||
pub format: Option<ImageType>,
|
||||
}
|
||||
|
||||
// Maximum data size supported.
|
||||
pub(super) const MAX_QCOW_FILE_SIZE: u64 = 0x01 << 44; // 16 TB.
|
||||
|
||||
// QCOW magic constant that starts the header.
|
||||
pub(super) const QCOW_MAGIC: u32 = 0x5146_49fb;
|
||||
// Default to a cluster size of 2^DEFAULT_CLUSTER_BITS
|
||||
pub(super) const DEFAULT_CLUSTER_BITS: u32 = 16;
|
||||
// Limit clusters to reasonable sizes. Choose the same limits as qemu. Making the clusters smaller
|
||||
// increases the amount of overhead for book keeping.
|
||||
pub(super) const MIN_CLUSTER_BITS: u32 = 9;
|
||||
pub(super) const MAX_CLUSTER_BITS: u32 = 21;
|
||||
// The L1 and RefCount table are kept in RAM, only handle files that require less than 35M entries.
|
||||
// This easily covers 1 TB files. When support for bigger files is needed the assumptions made to
|
||||
// keep these tables in RAM needs to be thrown out.
|
||||
pub(super) const MAX_RAM_POINTER_TABLE_SIZE: u64 = 35_000_000;
|
||||
// 16-bit refcounts.
|
||||
pub(super) const DEFAULT_REFCOUNT_ORDER: u32 = 4;
|
||||
|
||||
pub(super) const V2_BARE_HEADER_SIZE: u32 = 72;
|
||||
pub(super) const V3_BARE_HEADER_SIZE: u32 = 104;
|
||||
pub(super) const AUTOCLEAR_FEATURES_OFFSET: u64 = 88;
|
||||
|
||||
pub(super) const COMPATIBLE_FEATURES_LAZY_REFCOUNTS: u64 = 1;
|
||||
|
||||
// Compression types as defined in https://www.qemu.org/docs/master/interop/qcow2.html
|
||||
const COMPRESSION_TYPE_ZLIB: u64 = 0; // zlib/deflate <https://www.ietf.org/rfc/rfc1951.txt>
|
||||
const COMPRESSION_TYPE_ZSTD: u64 = 1; // zstd <http://github.com/facebook/zstd>
|
||||
|
||||
// Header extension types
|
||||
pub(super) const HEADER_EXT_END: u32 = 0x00000000;
|
||||
// Backing file format name (raw, qcow2)
|
||||
pub(super) const HEADER_EXT_BACKING_FORMAT: u32 = 0xe2792aca;
|
||||
// Feature name table
|
||||
const HEADER_EXT_FEATURE_NAME_TABLE: u32 = 0x6803f857;
|
||||
|
||||
// Feature name table entry type incompatible
|
||||
const FEAT_TYPE_INCOMPATIBLE: u8 = 0;
|
||||
|
||||
bitflags! {
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct IncompatFeatures: u64 {
|
||||
const DIRTY = 1 << 0;
|
||||
const CORRUPT = 1 << 1;
|
||||
const DATA_FILE = 1 << 2;
|
||||
const COMPRESSION = 1 << 3;
|
||||
const EXTENDED_L2 = 1 << 4;
|
||||
}
|
||||
}
|
||||
|
||||
impl IncompatFeatures {
|
||||
/// Features supported by this implementation.
|
||||
pub(super) const SUPPORTED: IncompatFeatures = IncompatFeatures::DIRTY
|
||||
.union(IncompatFeatures::CORRUPT)
|
||||
.union(IncompatFeatures::COMPRESSION);
|
||||
|
||||
/// Get the fallback name for a known feature bit.
|
||||
fn flag_name(bit: u8) -> Option<&'static str> {
|
||||
Some(match Self::from_bits_truncate(1u64 << bit) {
|
||||
Self::DIRTY => "dirty bit",
|
||||
Self::CORRUPT => "corrupt bit",
|
||||
Self::DATA_FILE => "external data file",
|
||||
Self::EXTENDED_L2 => "extended L2 entries",
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Error type for unsupported incompatible features.
|
||||
#[derive(Debug, Clone, thiserror::Error)]
|
||||
pub struct MissingFeatureError {
|
||||
/// Unsupported feature bits.
|
||||
features: IncompatFeatures,
|
||||
/// Feature name table from the qcow2 image.
|
||||
feature_names: Vec<(u8, String)>,
|
||||
}
|
||||
|
||||
impl MissingFeatureError {
|
||||
pub(super) fn new(features: IncompatFeatures, feature_names: Vec<(u8, String)>) -> Self {
|
||||
Self {
|
||||
features,
|
||||
feature_names,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Display for MissingFeatureError {
|
||||
fn fmt(&self, f: &mut Formatter<'_>) -> FmtResult {
|
||||
let names: Vec<String> = (0u8..64)
|
||||
.filter(|&bit| self.features.bits() & (1u64 << bit) != 0)
|
||||
.map(|bit| {
|
||||
// First try the image's feature name table
|
||||
self.feature_names
|
||||
.iter()
|
||||
.find(|(b, _)| *b == bit)
|
||||
.map(|(_, name)| name.clone())
|
||||
// Then try hardcoded fallback names
|
||||
.or_else(|| IncompatFeatures::flag_name(bit).map(|s| s.to_string()))
|
||||
// Finally, use generic description
|
||||
.unwrap_or_else(|| format!("unknown feature bit {bit}"))
|
||||
})
|
||||
.collect();
|
||||
write!(f, "Missing features: {}", names.join(", "))
|
||||
}
|
||||
}
|
||||
|
||||
// The format supports a "header extension area", that crosvm does not use.
|
||||
const QCOW_EMPTY_HEADER_EXTENSION_SIZE: u32 = 8;
|
||||
|
||||
// Defined by the specification
|
||||
const MAX_BACKING_FILE_SIZE: u32 = 1023;
|
||||
|
||||
/// Contains the information from the header of a qcow file.
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct QcowHeader {
|
||||
pub magic: u32,
|
||||
pub version: u32,
|
||||
|
||||
pub backing_file_offset: u64,
|
||||
pub backing_file_size: u32,
|
||||
|
||||
pub cluster_bits: u32,
|
||||
pub size: u64,
|
||||
pub crypt_method: u32,
|
||||
|
||||
pub l1_size: u32,
|
||||
pub l1_table_offset: u64,
|
||||
|
||||
pub refcount_table_offset: u64,
|
||||
pub refcount_table_clusters: u32,
|
||||
|
||||
pub nb_snapshots: u32,
|
||||
pub snapshots_offset: u64,
|
||||
|
||||
// v3 entries
|
||||
pub incompatible_features: u64,
|
||||
pub compatible_features: u64,
|
||||
pub autoclear_features: u64,
|
||||
pub refcount_order: u32,
|
||||
pub header_size: u32,
|
||||
pub compression_type: CompressionType,
|
||||
|
||||
// Post-header entries
|
||||
pub backing_file: Option<BackingFileConfig>,
|
||||
}
|
||||
|
||||
impl QcowHeader {
|
||||
/// Read header extensions, optionally collecting feature names for error reporting.
|
||||
pub(super) fn read_header_extensions(
|
||||
f: &mut RawFile,
|
||||
header: &mut QcowHeader,
|
||||
mut feature_table: Option<&mut Vec<(u8, String)>>,
|
||||
) -> Result<()> {
|
||||
// Extensions start directly after the header
|
||||
f.seek(SeekFrom::Start(header.header_size as u64))
|
||||
.map_err(Error::ReadingHeader)?;
|
||||
|
||||
loop {
|
||||
let ext_type = u32::read_be(f).map_err(Error::ReadingHeader)?;
|
||||
if ext_type == HEADER_EXT_END {
|
||||
break;
|
||||
}
|
||||
|
||||
let ext_length = u32::read_be(f).map_err(Error::ReadingHeader)?;
|
||||
|
||||
match ext_type {
|
||||
HEADER_EXT_BACKING_FORMAT => {
|
||||
let mut format_bytes = vec![0u8; ext_length as usize];
|
||||
f.read_exact(&mut format_bytes)
|
||||
.map_err(Error::ReadingHeader)?;
|
||||
let format_str = String::from_utf8(format_bytes)
|
||||
.map_err(|err| Error::InvalidBackingFileName(err.utf8_error()))?;
|
||||
if let Some(backing_file) = &mut header.backing_file {
|
||||
backing_file.format = Some(format_str.parse()?);
|
||||
}
|
||||
}
|
||||
HEADER_EXT_FEATURE_NAME_TABLE if feature_table.is_some() => {
|
||||
const FEATURE_NAME_ENTRY_SIZE: usize = 1 + 1 + 46; // type + bit + name
|
||||
let mut data = vec![0u8; ext_length as usize];
|
||||
f.read_exact(&mut data).map_err(Error::ReadingHeader)?;
|
||||
let table = feature_table.as_mut().unwrap();
|
||||
for entry in data.chunks_exact(FEATURE_NAME_ENTRY_SIZE) {
|
||||
if entry[0] == FEAT_TYPE_INCOMPATIBLE {
|
||||
let bit_number = entry[1];
|
||||
let name_bytes = &entry[2..];
|
||||
let name_len = name_bytes.iter().position(|&b| b == 0).unwrap_or(46);
|
||||
let name = String::from_utf8_lossy(&name_bytes[..name_len]).to_string();
|
||||
table.push((bit_number, name));
|
||||
}
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
// Skip unknown extension
|
||||
f.seek(SeekFrom::Current(ext_length as i64))
|
||||
.map_err(Error::ReadingHeader)?;
|
||||
}
|
||||
}
|
||||
|
||||
// Skip to the next 8 byte boundary
|
||||
let padding = (8 - (ext_length % 8)) % 8;
|
||||
f.seek(SeekFrom::Current(padding as i64))
|
||||
.map_err(Error::ReadingHeader)?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Creates a QcowHeader from a reference to a file.
|
||||
pub fn new(f: &mut RawFile) -> Result<QcowHeader> {
|
||||
f.rewind().map_err(Error::ReadingHeader)?;
|
||||
let magic = u32::read_be(f).map_err(Error::ReadingHeader)?;
|
||||
if magic != QCOW_MAGIC {
|
||||
return Err(Error::InvalidMagic);
|
||||
}
|
||||
|
||||
// Reads the next u32 from the file.
|
||||
fn read_u32_be(f: &mut RawFile) -> Result<u32> {
|
||||
u32::read_be(f).map_err(Error::ReadingHeader)
|
||||
}
|
||||
|
||||
// Reads the next u64 from the file.
|
||||
fn read_u64_be(f: &mut RawFile) -> Result<u64> {
|
||||
u64::read_be(f).map_err(Error::ReadingHeader)
|
||||
}
|
||||
|
||||
let version = read_u32_be(f)?;
|
||||
|
||||
let mut header = QcowHeader {
|
||||
magic,
|
||||
version,
|
||||
backing_file_offset: read_u64_be(f)?,
|
||||
backing_file_size: read_u32_be(f)?,
|
||||
cluster_bits: read_u32_be(f)?,
|
||||
size: read_u64_be(f)?,
|
||||
crypt_method: read_u32_be(f)?,
|
||||
l1_size: read_u32_be(f)?,
|
||||
l1_table_offset: read_u64_be(f)?,
|
||||
refcount_table_offset: read_u64_be(f)?,
|
||||
refcount_table_clusters: read_u32_be(f)?,
|
||||
nb_snapshots: read_u32_be(f)?,
|
||||
snapshots_offset: read_u64_be(f)?,
|
||||
incompatible_features: if version == 2 { 0 } else { read_u64_be(f)? },
|
||||
compatible_features: if version == 2 { 0 } else { read_u64_be(f)? },
|
||||
autoclear_features: if version == 2 { 0 } else { read_u64_be(f)? },
|
||||
refcount_order: if version == 2 {
|
||||
DEFAULT_REFCOUNT_ORDER
|
||||
} else {
|
||||
read_u32_be(f)?
|
||||
},
|
||||
header_size: if version == 2 {
|
||||
V2_BARE_HEADER_SIZE
|
||||
} else {
|
||||
read_u32_be(f)?
|
||||
},
|
||||
compression_type: CompressionType::Zlib,
|
||||
backing_file: None,
|
||||
};
|
||||
if version == 3 && header.header_size > V3_BARE_HEADER_SIZE {
|
||||
let raw_compression_type = read_u64_be(f)? >> (64 - 8);
|
||||
header.compression_type = if raw_compression_type == COMPRESSION_TYPE_ZLIB {
|
||||
Ok(CompressionType::Zlib)
|
||||
} else if raw_compression_type == COMPRESSION_TYPE_ZSTD {
|
||||
Ok(CompressionType::Zstd)
|
||||
} else {
|
||||
Err(Error::UnsupportedCompressionType)
|
||||
}?;
|
||||
}
|
||||
if header.backing_file_size > MAX_BACKING_FILE_SIZE {
|
||||
return Err(Error::BackingFileTooLong(header.backing_file_size as usize));
|
||||
}
|
||||
if header.backing_file_offset != 0 {
|
||||
f.seek(SeekFrom::Start(header.backing_file_offset))
|
||||
.map_err(Error::ReadingHeader)?;
|
||||
let mut backing_file_name_bytes = vec![0u8; header.backing_file_size as usize];
|
||||
f.read_exact(&mut backing_file_name_bytes)
|
||||
.map_err(Error::ReadingHeader)?;
|
||||
let path = String::from_utf8(backing_file_name_bytes)
|
||||
.map_err(|err| Error::InvalidBackingFileName(err.utf8_error()))?;
|
||||
header.backing_file = Some(BackingFileConfig { path, format: None });
|
||||
}
|
||||
|
||||
if version == 3 {
|
||||
// Check for unsupported incompatible features first
|
||||
let features = IncompatFeatures::from_bits_retain(header.incompatible_features);
|
||||
let unsupported = features - IncompatFeatures::SUPPORTED;
|
||||
if !unsupported.is_empty() {
|
||||
// Read extensions only to get feature names for error reporting
|
||||
let mut feature_table = Vec::new();
|
||||
if header.header_size > V3_BARE_HEADER_SIZE {
|
||||
let _ = Self::read_header_extensions(f, &mut header, Some(&mut feature_table));
|
||||
}
|
||||
return Err(Error::UnsupportedFeature(MissingFeatureError::new(
|
||||
unsupported,
|
||||
feature_table,
|
||||
)));
|
||||
}
|
||||
|
||||
// Features OK, now read extensions normally
|
||||
if header.header_size > V3_BARE_HEADER_SIZE {
|
||||
Self::read_header_extensions(f, &mut header, None)?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(header)
|
||||
}
|
||||
|
||||
pub fn get_decoder(&self) -> Box<dyn Decoder> {
|
||||
match self.compression_type {
|
||||
CompressionType::Zlib => Box::new(ZlibDecoder {}),
|
||||
CompressionType::Zstd => Box::new(ZstdDecoder {}),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn create_for_size_and_path(
|
||||
version: u32,
|
||||
size: u64,
|
||||
backing_file: Option<&str>,
|
||||
) -> Result<QcowHeader> {
|
||||
let header_size = if version == 2 {
|
||||
V2_BARE_HEADER_SIZE
|
||||
} else {
|
||||
V3_BARE_HEADER_SIZE + QCOW_EMPTY_HEADER_EXTENSION_SIZE
|
||||
};
|
||||
let cluster_bits: u32 = DEFAULT_CLUSTER_BITS;
|
||||
let cluster_size: u32 = 0x01 << cluster_bits;
|
||||
let max_length: usize = (cluster_size - header_size) as usize;
|
||||
if let Some(path) = backing_file
|
||||
&& path.len() > max_length
|
||||
{
|
||||
return Err(Error::BackingFileTooLong(path.len() - max_length));
|
||||
}
|
||||
|
||||
// L2 blocks are always one cluster long. They contain cluster_size/sizeof(u64) addresses.
|
||||
let entries_per_cluster: u32 = cluster_size / size_of::<u64>() as u32;
|
||||
let num_clusters: u32 = div_round_up_u64(size, u64::from(cluster_size)) as u32;
|
||||
let num_l2_clusters: u32 = div_round_up_u32(num_clusters, entries_per_cluster);
|
||||
let l1_clusters: u32 = div_round_up_u32(num_l2_clusters, entries_per_cluster);
|
||||
let header_clusters = div_round_up_u32(size_of::<QcowHeader>() as u32, cluster_size);
|
||||
Ok(QcowHeader {
|
||||
magic: QCOW_MAGIC,
|
||||
version,
|
||||
backing_file_offset: backing_file.map_or(0, |_| {
|
||||
header_size
|
||||
+ if version == 3 {
|
||||
QCOW_EMPTY_HEADER_EXTENSION_SIZE
|
||||
} else {
|
||||
0
|
||||
}
|
||||
}) as u64,
|
||||
backing_file_size: backing_file.map_or(0, |x| x.len()) as u32,
|
||||
cluster_bits: DEFAULT_CLUSTER_BITS,
|
||||
size,
|
||||
crypt_method: 0,
|
||||
l1_size: num_l2_clusters,
|
||||
l1_table_offset: u64::from(cluster_size),
|
||||
// The refcount table is after l1 + header.
|
||||
refcount_table_offset: u64::from(cluster_size * (l1_clusters + 1)),
|
||||
refcount_table_clusters: {
|
||||
// Pre-allocate enough clusters for the entire refcount table as it must be
|
||||
// continuous in the file. Allocate enough space to refcount all clusters, including
|
||||
// the refcount clusters.
|
||||
let max_refcount_clusters = max_refcount_clusters(
|
||||
DEFAULT_REFCOUNT_ORDER,
|
||||
cluster_size,
|
||||
num_clusters + l1_clusters + num_l2_clusters + header_clusters,
|
||||
) as u32;
|
||||
// The refcount table needs to store the offset of each refcount cluster.
|
||||
div_round_up_u32(
|
||||
max_refcount_clusters * size_of::<u64>() as u32,
|
||||
cluster_size,
|
||||
)
|
||||
},
|
||||
nb_snapshots: 0,
|
||||
snapshots_offset: 0,
|
||||
incompatible_features: 0,
|
||||
compatible_features: 0,
|
||||
autoclear_features: 0,
|
||||
refcount_order: DEFAULT_REFCOUNT_ORDER,
|
||||
header_size,
|
||||
compression_type: CompressionType::Zlib,
|
||||
backing_file: backing_file.map(|path| BackingFileConfig {
|
||||
path: String::from(path),
|
||||
format: None,
|
||||
}),
|
||||
})
|
||||
}
|
||||
|
||||
/// Write the header to `file`.
|
||||
pub fn write_to<F: Write + Seek>(&self, file: &mut F) -> Result<()> {
|
||||
// Writes the next u32 to the file.
|
||||
fn write_u32_be<F: Write>(f: &mut F, value: u32) -> Result<()> {
|
||||
u32::write_be(f, value).map_err(Error::WritingHeader)
|
||||
}
|
||||
|
||||
// Writes the next u64 to the file.
|
||||
fn write_u64_be<F: Write>(f: &mut F, value: u64) -> Result<()> {
|
||||
u64::write_be(f, value).map_err(Error::WritingHeader)
|
||||
}
|
||||
|
||||
write_u32_be(file, self.magic)?;
|
||||
write_u32_be(file, self.version)?;
|
||||
write_u64_be(file, self.backing_file_offset)?;
|
||||
write_u32_be(file, self.backing_file_size)?;
|
||||
write_u32_be(file, self.cluster_bits)?;
|
||||
write_u64_be(file, self.size)?;
|
||||
write_u32_be(file, self.crypt_method)?;
|
||||
write_u32_be(file, self.l1_size)?;
|
||||
write_u64_be(file, self.l1_table_offset)?;
|
||||
write_u64_be(file, self.refcount_table_offset)?;
|
||||
write_u32_be(file, self.refcount_table_clusters)?;
|
||||
write_u32_be(file, self.nb_snapshots)?;
|
||||
write_u64_be(file, self.snapshots_offset)?;
|
||||
|
||||
if self.version == 3 {
|
||||
write_u64_be(file, self.incompatible_features)?;
|
||||
write_u64_be(file, self.compatible_features)?;
|
||||
write_u64_be(file, self.autoclear_features)?;
|
||||
write_u32_be(file, self.refcount_order)?;
|
||||
write_u32_be(file, self.header_size)?;
|
||||
|
||||
if self.header_size > V3_BARE_HEADER_SIZE {
|
||||
write_u64_be(file, 0)?; // no compression
|
||||
}
|
||||
|
||||
write_u32_be(file, 0)?; // header extension type: end of header extension area
|
||||
write_u32_be(file, 0)?; // length of header extension data: 0
|
||||
}
|
||||
|
||||
if let Some(backing_file_path) = self.backing_file.as_ref().map(|bf| &bf.path) {
|
||||
if self.backing_file_offset > 0 {
|
||||
file.seek(SeekFrom::Start(self.backing_file_offset))
|
||||
.map_err(Error::WritingHeader)?;
|
||||
}
|
||||
write!(file, "{backing_file_path}").map_err(Error::WritingHeader)?;
|
||||
}
|
||||
|
||||
// Set the file length by seeking and writing a zero to the last byte. This avoids needing
|
||||
// a `File` instead of anything that implements seek as the `file` argument.
|
||||
// Zeros out the l1 and refcount table clusters.
|
||||
let cluster_size = 0x01u64 << self.cluster_bits;
|
||||
let refcount_blocks_size = u64::from(self.refcount_table_clusters) * cluster_size;
|
||||
file.seek(SeekFrom::Start(
|
||||
self.refcount_table_offset + refcount_blocks_size - 2,
|
||||
))
|
||||
.map_err(Error::WritingHeader)?;
|
||||
file.write(&[0u8]).map_err(Error::WritingHeader)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Write only the incompatible_features field to the file at its fixed offset.
|
||||
fn write_incompatible_features<F: Seek + Write>(&self, file: &mut F) -> BlockResult<()> {
|
||||
if self.version != 3 {
|
||||
return Ok(());
|
||||
}
|
||||
file.seek(SeekFrom::Start(V2_BARE_HEADER_SIZE as u64))
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, Error::WritingHeader(e)))?;
|
||||
u64::write_be(file, self.incompatible_features)
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, Error::WritingHeader(e)))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Set or clear the dirty bit for QCOW2 v3 images.
|
||||
///
|
||||
/// When `dirty` is true, sets the bit to indicate the image is in use.
|
||||
/// When `dirty` is false, clears the bit to indicate a clean shutdown.
|
||||
pub fn set_dirty_bit<F: Seek + Write + FileSync>(
|
||||
&mut self,
|
||||
file: &mut F,
|
||||
dirty: bool,
|
||||
) -> BlockResult<()> {
|
||||
if self.version == 3 {
|
||||
if dirty {
|
||||
self.incompatible_features |= IncompatFeatures::DIRTY.bits();
|
||||
} else {
|
||||
self.incompatible_features &= !IncompatFeatures::DIRTY.bits();
|
||||
}
|
||||
self.write_incompatible_features(file)?;
|
||||
file.fsync()
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, Error::SyncingHeader(e)))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Set the corrupt bit for QCOW2 v3 images.
|
||||
///
|
||||
/// This marks the image as corrupted. Once set, the image can only be
|
||||
/// opened read-only until repaired.
|
||||
pub fn set_corrupt_bit<F: Seek + Write + FileSync>(&mut self, file: &mut F) -> BlockResult<()> {
|
||||
if self.version == 3 {
|
||||
self.incompatible_features |= IncompatFeatures::CORRUPT.bits();
|
||||
self.write_incompatible_features(file)?;
|
||||
file.fsync()
|
||||
.map_err(|e| BlockError::new(BlockErrorKind::Io, Error::SyncingHeader(e)))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn is_corrupt(&self) -> bool {
|
||||
IncompatFeatures::from_bits_truncate(self.incompatible_features)
|
||||
.contains(IncompatFeatures::CORRUPT)
|
||||
}
|
||||
|
||||
/// Clear all autoclear feature bits for QCOW2 v3 images.
|
||||
///
|
||||
/// These bits indicate features that can be safely disabled when modified
|
||||
/// by software that doesn't understand them.
|
||||
pub fn clear_autoclear_features<F: Seek + Write + FileSync>(
|
||||
&mut self,
|
||||
file: &mut F,
|
||||
) -> Result<()> {
|
||||
if self.version == 3 && self.autoclear_features != 0 {
|
||||
self.autoclear_features = 0;
|
||||
file.seek(SeekFrom::Start(AUTOCLEAR_FEATURES_OFFSET))
|
||||
.map_err(Error::WritingHeader)?;
|
||||
u64::write_be(file, 0).map_err(Error::WritingHeader)?;
|
||||
file.fsync().map_err(Error::SyncingHeader)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn max_refcount_clusters(
|
||||
refcount_order: u32,
|
||||
cluster_size: u32,
|
||||
num_clusters: u32,
|
||||
) -> u64 {
|
||||
// Use u64 as the product of the u32 inputs can overflow.
|
||||
let refcount_bits = 0x01u64 << u64::from(refcount_order);
|
||||
let cluster_bits = u64::from(cluster_size) * 8;
|
||||
let for_data = div_round_up_u64(u64::from(num_clusters) * refcount_bits, cluster_bits);
|
||||
let for_refcounts = div_round_up_u64(for_data * refcount_bits, cluster_bits);
|
||||
for_data + for_refcounts
|
||||
}
|
||||
|
||||
/// Returns an Error if the given offset doesn't align to a cluster boundary.
|
||||
pub(super) fn offset_is_cluster_boundary(offset: u64, cluster_bits: u32) -> Result<()> {
|
||||
if offset & ((0x01 << cluster_bits) - 1) != 0 {
|
||||
return Err(Error::InvalidOffset(offset));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
1116
block/src/formats/qcow/internal/metadata.rs
Normal file
1116
block/src/formats/qcow/internal/metadata.rs
Normal file
File diff suppressed because it is too large
Load Diff
4832
block/src/formats/qcow/internal/mod.rs
Normal file
4832
block/src/formats/qcow/internal/mod.rs
Normal file
File diff suppressed because it is too large
Load Diff
369
block/src/formats/qcow/internal/qcow_raw_file.rs
Normal file
369
block/src/formats/qcow/internal/qcow_raw_file.rs
Normal file
@@ -0,0 +1,369 @@
|
||||
// Copyright 2018 The Chromium OS Authors. All rights reserved.
|
||||
// Use of this source code is governed by a BSD-style license that can be
|
||||
// found in the LICENSE-BSD-3-Clause file.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
use std::fmt::Debug;
|
||||
use std::io::{self, BufWriter, Read, Seek, SeekFrom, Write};
|
||||
use std::mem::size_of;
|
||||
use std::os::fd::{AsFd, AsRawFd, BorrowedFd, RawFd};
|
||||
|
||||
use byteorder::{BigEndian, ReadBytesExt, WriteBytesExt};
|
||||
use vmm_sys_util::write_zeroes::WriteZeroes;
|
||||
|
||||
use super::RawFile;
|
||||
|
||||
// Type aliases for the refcount read/write function pointers
|
||||
type RefcountReader = fn(&mut RawFile, usize) -> io::Result<Vec<u64>>;
|
||||
type RefcountWriter = fn(&mut RawFile, &[u64]) -> io::Result<()>;
|
||||
|
||||
/// Big-endian file access trait.
|
||||
pub(super) trait BeUint: Sized + Copy {
|
||||
fn from_be_slice(bytes: &[u8]) -> u64;
|
||||
fn read_be<R: Read>(r: &mut R) -> io::Result<Self>;
|
||||
fn write_be<W: Write>(w: &mut W, val: Self) -> io::Result<()>;
|
||||
}
|
||||
|
||||
impl BeUint for u8 {
|
||||
#[inline(always)]
|
||||
fn from_be_slice(bytes: &[u8]) -> u64 {
|
||||
bytes[0] as u64
|
||||
}
|
||||
#[inline(always)]
|
||||
fn read_be<R: Read>(r: &mut R) -> io::Result<Self> {
|
||||
r.read_u8()
|
||||
}
|
||||
#[inline(always)]
|
||||
fn write_be<W: Write>(w: &mut W, val: Self) -> io::Result<()> {
|
||||
w.write_u8(val)
|
||||
}
|
||||
}
|
||||
|
||||
impl BeUint for u16 {
|
||||
#[inline(always)]
|
||||
fn from_be_slice(bytes: &[u8]) -> u64 {
|
||||
u16::from_be_bytes([bytes[0], bytes[1]]) as u64
|
||||
}
|
||||
#[inline(always)]
|
||||
fn read_be<R: Read>(r: &mut R) -> io::Result<Self> {
|
||||
r.read_u16::<BigEndian>()
|
||||
}
|
||||
#[inline(always)]
|
||||
fn write_be<W: Write>(w: &mut W, val: Self) -> io::Result<()> {
|
||||
w.write_u16::<BigEndian>(val)
|
||||
}
|
||||
}
|
||||
|
||||
impl BeUint for u32 {
|
||||
#[inline(always)]
|
||||
fn from_be_slice(bytes: &[u8]) -> u64 {
|
||||
u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]) as u64
|
||||
}
|
||||
#[inline(always)]
|
||||
fn read_be<R: Read>(r: &mut R) -> io::Result<Self> {
|
||||
r.read_u32::<BigEndian>()
|
||||
}
|
||||
#[inline(always)]
|
||||
fn write_be<W: Write>(w: &mut W, val: Self) -> io::Result<()> {
|
||||
w.write_u32::<BigEndian>(val)
|
||||
}
|
||||
}
|
||||
|
||||
impl BeUint for u64 {
|
||||
#[inline(always)]
|
||||
fn from_be_slice(bytes: &[u8]) -> u64 {
|
||||
u64::from_be_bytes([
|
||||
bytes[0], bytes[1], bytes[2], bytes[3], bytes[4], bytes[5], bytes[6], bytes[7],
|
||||
])
|
||||
}
|
||||
#[inline(always)]
|
||||
fn read_be<R: Read>(r: &mut R) -> io::Result<Self> {
|
||||
r.read_u64::<BigEndian>()
|
||||
}
|
||||
#[inline(always)]
|
||||
fn write_be<W: Write>(w: &mut W, val: Self) -> io::Result<()> {
|
||||
w.write_u64::<BigEndian>(val)
|
||||
}
|
||||
}
|
||||
|
||||
/// Read byte-aligned refcounts.
|
||||
fn read_refcount<T: BeUint>(file: &mut RawFile, count: usize) -> io::Result<Vec<u64>> {
|
||||
let bytes_per_entry = size_of::<T>();
|
||||
let mut data = vec![0u8; count * bytes_per_entry];
|
||||
file.read_exact(&mut data)?;
|
||||
Ok(data
|
||||
.chunks_exact(bytes_per_entry)
|
||||
.map(T::from_be_slice)
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// Write byte-aligned refcounts.
|
||||
fn write_refcount<T: BeUint + TryFrom<u64>>(file: &mut RawFile, table: &[u64]) -> io::Result<()>
|
||||
where
|
||||
<T as TryFrom<u64>>::Error: Debug,
|
||||
{
|
||||
let bytes_per_entry = size_of::<T>();
|
||||
let mut buffer = BufWriter::with_capacity(table.len() * bytes_per_entry, file);
|
||||
for &val in table {
|
||||
let converted = T::try_from(val).expect("refcount values are validated on increment");
|
||||
T::write_be(&mut buffer, converted)?;
|
||||
}
|
||||
buffer.flush()
|
||||
}
|
||||
|
||||
/// Read sub-byte refcounts. Bit 0 is the least significant bit.
|
||||
fn read_refcount_subbyte<const BITS: usize>(
|
||||
file: &mut RawFile,
|
||||
count: usize,
|
||||
) -> io::Result<Vec<u64>> {
|
||||
const { assert!(BITS == 1 || BITS == 2 || BITS == 4) };
|
||||
let entries_per_byte = 8 / BITS;
|
||||
let mask = (1u64 << BITS) - 1;
|
||||
let bytes_needed = count.div_ceil(entries_per_byte);
|
||||
let mut bytes = vec![0u8; bytes_needed];
|
||||
file.read_exact(&mut bytes)?;
|
||||
|
||||
let mut table = vec![0u64; count];
|
||||
for (i, val) in table.iter_mut().enumerate() {
|
||||
let byte_idx = i / entries_per_byte;
|
||||
let bit_offset = (i % entries_per_byte) * BITS;
|
||||
*val = (bytes[byte_idx] as u64 >> bit_offset) & mask;
|
||||
}
|
||||
Ok(table)
|
||||
}
|
||||
|
||||
/// Write sub-byte refcounts. Bit 0 is the least significant bit.
|
||||
fn write_refcount_subbyte<const BITS: usize>(file: &mut RawFile, table: &[u64]) -> io::Result<()> {
|
||||
const { assert!(BITS == 1 || BITS == 2 || BITS == 4) };
|
||||
let entries_per_byte = 8 / BITS;
|
||||
let mask = (1u64 << BITS) - 1;
|
||||
let mut buffer = BufWriter::with_capacity(table.len().div_ceil(entries_per_byte), file);
|
||||
|
||||
for chunk in table.chunks(entries_per_byte) {
|
||||
let mut byte = 0u8;
|
||||
for (i, &val) in chunk.iter().enumerate() {
|
||||
let bit_offset = i * BITS;
|
||||
byte |= ((val & mask) << bit_offset) as u8;
|
||||
}
|
||||
buffer.write_u8(byte)?;
|
||||
}
|
||||
buffer.flush()
|
||||
}
|
||||
|
||||
/// A qcow file. Allows reading/writing clusters and appending clusters.
|
||||
#[derive(Debug)]
|
||||
pub struct QcowRawFile {
|
||||
file: RawFile,
|
||||
cluster_size: u64,
|
||||
cluster_mask: u64,
|
||||
refcount_block_entries: u64,
|
||||
read_refcount_fn: RefcountReader,
|
||||
write_refcount_fn: RefcountWriter,
|
||||
}
|
||||
|
||||
impl QcowRawFile {
|
||||
/// Creates a `QcowRawFile` from the given `File`, `None` is returned if `cluster_size` is not
|
||||
/// a power of two or refcount_bits is invalid.
|
||||
pub fn from(file: RawFile, cluster_size: u64, refcount_bits: u64) -> Option<Self> {
|
||||
if !cluster_size.is_power_of_two() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let (read_refcount_fn, write_refcount_fn): (RefcountReader, RefcountWriter) =
|
||||
match refcount_bits {
|
||||
1 => (read_refcount_subbyte::<1>, write_refcount_subbyte::<1>),
|
||||
2 => (read_refcount_subbyte::<2>, write_refcount_subbyte::<2>),
|
||||
4 => (read_refcount_subbyte::<4>, write_refcount_subbyte::<4>),
|
||||
8 => (read_refcount::<u8>, write_refcount::<u8>),
|
||||
16 => (read_refcount::<u16>, write_refcount::<u16>),
|
||||
32 => (read_refcount::<u32>, write_refcount::<u32>),
|
||||
64 => (read_refcount::<u64>, write_refcount::<u64>),
|
||||
_ => return None,
|
||||
};
|
||||
|
||||
// For sub-byte refcounts (1,2,4 bits), entries pack multiple per byte
|
||||
let refcount_block_entries = cluster_size * 8 / refcount_bits;
|
||||
|
||||
Some(QcowRawFile {
|
||||
file,
|
||||
cluster_size,
|
||||
cluster_mask: cluster_size - 1,
|
||||
refcount_block_entries,
|
||||
read_refcount_fn,
|
||||
write_refcount_fn,
|
||||
})
|
||||
}
|
||||
|
||||
/// Reads `count` 64 bit offsets and returns them as a vector.
|
||||
/// `mask` optionally `&`s out some of the bits on the file.
|
||||
pub fn read_pointer_table(
|
||||
&mut self,
|
||||
offset: u64,
|
||||
count: u64,
|
||||
mask: Option<u64>,
|
||||
) -> io::Result<Vec<u64>> {
|
||||
let mut table = vec![0; count as usize];
|
||||
self.file.seek(SeekFrom::Start(offset))?;
|
||||
self.file.read_u64_into::<BigEndian>(&mut table)?;
|
||||
if let Some(m) = mask {
|
||||
for ptr in &mut table {
|
||||
*ptr &= m;
|
||||
}
|
||||
}
|
||||
Ok(table)
|
||||
}
|
||||
|
||||
/// Reads a cluster's worth of 64 bit offsets and returns them as a vector.
|
||||
/// `mask` optionally `&`s out some of the bits on the file.
|
||||
pub fn read_pointer_cluster(&mut self, offset: u64, mask: Option<u64>) -> io::Result<Vec<u64>> {
|
||||
let count = self.cluster_size / size_of::<u64>() as u64;
|
||||
self.read_pointer_table(offset, count, mask)
|
||||
}
|
||||
|
||||
/// Internal helper for creating a buffered writer for pointer tables.
|
||||
#[inline]
|
||||
fn setup_pointer_table_writer<T>(
|
||||
&mut self,
|
||||
offset: u64,
|
||||
entries: &impl Iterator<Item = T>,
|
||||
) -> io::Result<BufWriter<RawFile>> {
|
||||
self.file.seek(SeekFrom::Start(offset))?;
|
||||
let my_file = self.file.try_clone()?;
|
||||
let capacity = entries.size_hint().0 * size_of::<u64>();
|
||||
Ok(BufWriter::with_capacity(capacity, my_file))
|
||||
}
|
||||
|
||||
/// Writes a pointer table to `offset` in the file.
|
||||
/// Entries are computed on-the-fly by the callback.
|
||||
pub fn write_pointer_table<'a, T: Copy + 'a>(
|
||||
&mut self,
|
||||
offset: u64,
|
||||
entries: impl Iterator<Item = &'a T>,
|
||||
mut f: impl FnMut(&mut QcowRawFile, T) -> io::Result<u64>,
|
||||
) -> io::Result<()> {
|
||||
let mut buffer = self.setup_pointer_table_writer(offset, &entries)?;
|
||||
|
||||
for addr in entries {
|
||||
let entry = f(self, *addr)?;
|
||||
u64::write_be(&mut buffer, entry)?;
|
||||
}
|
||||
buffer.flush()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Writes a pointer table directly without transforming values.
|
||||
pub fn write_pointer_table_direct<'a>(
|
||||
&mut self,
|
||||
offset: u64,
|
||||
entries: impl Iterator<Item = &'a u64>,
|
||||
) -> io::Result<()> {
|
||||
let mut buffer = self.setup_pointer_table_writer(offset, &entries)?;
|
||||
|
||||
for &entry in entries {
|
||||
u64::write_be(&mut buffer, entry)?;
|
||||
}
|
||||
buffer.flush()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read a refcount block from the file and returns a Vec containing the block.
|
||||
/// Always returns a cluster's worth of data.
|
||||
#[inline]
|
||||
pub fn read_refcount_block(&mut self, offset: u64) -> io::Result<Vec<u64>> {
|
||||
self.file.seek(SeekFrom::Start(offset))?;
|
||||
(self.read_refcount_fn)(&mut self.file, self.refcount_block_entries as usize)
|
||||
}
|
||||
|
||||
/// Writes a refcount block to the file.
|
||||
#[inline]
|
||||
pub fn write_refcount_block(&mut self, offset: u64, table: &[u64]) -> io::Result<()> {
|
||||
self.file.seek(SeekFrom::Start(offset))?;
|
||||
(self.write_refcount_fn)(&mut self.file, table)
|
||||
}
|
||||
|
||||
/// Allocates a new cluster at the end of the current file, return the address.
|
||||
pub fn add_cluster_end(&mut self, max_valid_cluster_offset: u64) -> io::Result<Option<u64>> {
|
||||
// Determine where the new end of the file should be and set_len, which
|
||||
// translates to truncate(2).
|
||||
let file_end: u64 = self.file.seek(SeekFrom::End(0))?;
|
||||
let new_cluster_address: u64 = (file_end + self.cluster_size - 1) & !self.cluster_mask;
|
||||
|
||||
if new_cluster_address > max_valid_cluster_offset {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
self.file.set_len(new_cluster_address + self.cluster_size)?;
|
||||
|
||||
Ok(Some(new_cluster_address))
|
||||
}
|
||||
|
||||
/// Returns a reference to the underlying file.
|
||||
pub fn file(&self) -> &RawFile {
|
||||
&self.file
|
||||
}
|
||||
|
||||
/// Returns a mutable reference to the underlying file.
|
||||
pub fn file_mut(&mut self) -> &mut RawFile {
|
||||
&mut self.file
|
||||
}
|
||||
|
||||
/// Returns the size of the file's clusters.
|
||||
pub fn cluster_size(&self) -> u64 {
|
||||
self.cluster_size
|
||||
}
|
||||
|
||||
/// Returns the offset of `address` within a cluster.
|
||||
pub fn cluster_offset(&self, address: u64) -> u64 {
|
||||
address & self.cluster_mask
|
||||
}
|
||||
|
||||
/// Returns the base address of the cluster containing `address`.
|
||||
pub fn cluster_address(&self, address: u64) -> u64 {
|
||||
address & !self.cluster_mask
|
||||
}
|
||||
|
||||
/// Zeros out a cluster in the file.
|
||||
pub fn zero_cluster(&mut self, address: u64) -> io::Result<()> {
|
||||
let cluster_size = self.cluster_size as usize;
|
||||
self.file.seek(SeekFrom::Start(address))?;
|
||||
self.file.write_zeroes(cluster_size)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Writes
|
||||
pub fn write_cluster(&mut self, address: u64, data: &[u8]) -> io::Result<()> {
|
||||
let cluster_size = self.cluster_size as usize;
|
||||
self.file.seek(SeekFrom::Start(address))?;
|
||||
self.file.write_all(&data[0..cluster_size])
|
||||
}
|
||||
|
||||
pub fn physical_size(&self) -> Result<u64, std::io::Error> {
|
||||
self.file.metadata().map(|m| m.len())
|
||||
}
|
||||
}
|
||||
|
||||
impl Clone for QcowRawFile {
|
||||
fn clone(&self) -> Self {
|
||||
QcowRawFile {
|
||||
file: self.file.try_clone().expect("QcowRawFile cloning failed"),
|
||||
cluster_size: self.cluster_size,
|
||||
cluster_mask: self.cluster_mask,
|
||||
refcount_block_entries: self.refcount_block_entries,
|
||||
read_refcount_fn: self.read_refcount_fn,
|
||||
write_refcount_fn: self.write_refcount_fn,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl AsRawFd for QcowRawFile {
|
||||
fn as_raw_fd(&self) -> RawFd {
|
||||
self.file.as_raw_fd()
|
||||
}
|
||||
}
|
||||
|
||||
impl AsFd for QcowRawFile {
|
||||
fn as_fd(&self) -> BorrowedFd<'_> {
|
||||
self.file.as_fd()
|
||||
}
|
||||
}
|
||||
406
block/src/formats/qcow/internal/raw_file.rs
Normal file
406
block/src/formats/qcow/internal/raw_file.rs
Normal file
@@ -0,0 +1,406 @@
|
||||
// Copyright 2018 Amazon.com, Inc. or its affiliates. All Rights Reserved.
|
||||
//
|
||||
// Portions Copyright 2017 The Chromium OS Authors. All rights reserved.
|
||||
// Use of this source code is governed by a BSD-style license that can be
|
||||
// found in the LICENSE-BSD-3-Clause file.
|
||||
//
|
||||
// Copyright © 2020 Intel Corporation
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
use std::alloc::{Layout, alloc_zeroed, dealloc};
|
||||
use std::fs::{File, Metadata};
|
||||
use std::io::{self, Read, Seek, SeekFrom, Write};
|
||||
use std::os::fd::{AsFd, BorrowedFd};
|
||||
use std::os::unix::io::{AsRawFd, RawFd};
|
||||
use std::slice;
|
||||
|
||||
use vmm_sys_util::file_traits::FileSync;
|
||||
use vmm_sys_util::seek_hole::SeekHole;
|
||||
use vmm_sys_util::write_zeroes::{PunchHole, WriteZeroesAt};
|
||||
|
||||
use crate::{BlockBackend, query_device_size};
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct RawFile {
|
||||
file: File,
|
||||
alignment: usize,
|
||||
position: u64,
|
||||
direct_io: bool,
|
||||
}
|
||||
|
||||
const BLK_ALIGNMENTS: [usize; 2] = [512, 4096];
|
||||
|
||||
fn is_valid_alignment(fd: RawFd, alignment: usize) -> bool {
|
||||
let layout = Layout::from_size_align(alignment, alignment).unwrap();
|
||||
// SAFETY: layout has non-zero size
|
||||
let ptr = unsafe { alloc_zeroed(layout) };
|
||||
assert!(!ptr.is_null());
|
||||
|
||||
// SAFETY: FFI call
|
||||
let ret = unsafe { ::libc::pread(fd, ptr.cast(), alignment, alignment.try_into().unwrap()) };
|
||||
|
||||
// SAFETY: ptr was allocated by alloc_zeroed with layout
|
||||
unsafe { dealloc(ptr, layout) };
|
||||
|
||||
ret >= 0
|
||||
}
|
||||
|
||||
impl RawFile {
|
||||
pub fn new(file: File, direct_io: bool) -> Self {
|
||||
// Assume no alignment restrictions if we aren't using O_DIRECT.
|
||||
let mut alignment = 0;
|
||||
if direct_io {
|
||||
for align in &BLK_ALIGNMENTS {
|
||||
if is_valid_alignment(file.as_raw_fd(), *align) {
|
||||
alignment = *align;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
RawFile {
|
||||
file,
|
||||
alignment,
|
||||
position: 0,
|
||||
direct_io,
|
||||
}
|
||||
}
|
||||
|
||||
fn round_up(&self, offset: u64) -> u64 {
|
||||
let align: u64 = self.alignment.try_into().unwrap();
|
||||
offset.div_ceil(align) * align
|
||||
}
|
||||
|
||||
fn round_down(&self, offset: u64) -> u64 {
|
||||
let align: u64 = self.alignment.try_into().unwrap();
|
||||
(offset / align) * align
|
||||
}
|
||||
|
||||
fn is_aligned(&self, buf: &[u8]) -> bool {
|
||||
if self.alignment == 0 {
|
||||
return true;
|
||||
}
|
||||
|
||||
let align64: u64 = self.alignment.try_into().unwrap();
|
||||
|
||||
self.position.is_multiple_of(align64)
|
||||
&& (buf.as_ptr() as usize).is_multiple_of(self.alignment)
|
||||
&& buf.len().is_multiple_of(self.alignment)
|
||||
}
|
||||
|
||||
pub fn set_len(&self, size: u64) -> std::io::Result<()> {
|
||||
self.file.set_len(size)
|
||||
}
|
||||
|
||||
pub fn metadata(&self) -> std::io::Result<Metadata> {
|
||||
self.file.metadata()
|
||||
}
|
||||
|
||||
pub fn try_clone(&self) -> std::io::Result<RawFile> {
|
||||
Ok(RawFile {
|
||||
file: self.file.try_clone().expect("RawFile cloning failed"),
|
||||
alignment: self.alignment,
|
||||
position: self.position,
|
||||
direct_io: self.direct_io,
|
||||
})
|
||||
}
|
||||
|
||||
pub fn sync_all(&self) -> std::io::Result<()> {
|
||||
self.file.sync_all()
|
||||
}
|
||||
|
||||
pub fn sync_data(&self) -> std::io::Result<()> {
|
||||
self.file.sync_data()
|
||||
}
|
||||
|
||||
pub fn is_direct(&self) -> bool {
|
||||
self.direct_io
|
||||
}
|
||||
|
||||
pub fn alignment(&self) -> usize {
|
||||
self.alignment
|
||||
}
|
||||
|
||||
/// Returns true if the file was opened with write access.
|
||||
pub fn is_writable(&self) -> bool {
|
||||
// SAFETY: fcntl with F_GETFL is safe and doesn't modify the file descriptor
|
||||
let flags = unsafe { libc::fcntl(self.file.as_raw_fd(), libc::F_GETFL) };
|
||||
if flags < 0 {
|
||||
return false;
|
||||
}
|
||||
let access_mode = flags & libc::O_ACCMODE;
|
||||
access_mode == libc::O_WRONLY || access_mode == libc::O_RDWR
|
||||
}
|
||||
}
|
||||
|
||||
impl Read for RawFile {
|
||||
fn read(&mut self, buf: &mut [u8]) -> std::io::Result<usize> {
|
||||
if self.is_aligned(buf) {
|
||||
match self.file.read(buf) {
|
||||
Ok(r) => {
|
||||
self.position = self.position.checked_add(r.try_into().unwrap()).unwrap();
|
||||
Ok(r)
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
} else {
|
||||
let rounded_pos: u64 = self.round_down(self.position);
|
||||
let file_offset: usize = self
|
||||
.position
|
||||
.checked_sub(rounded_pos)
|
||||
.unwrap()
|
||||
.try_into()
|
||||
.unwrap();
|
||||
let buf_len: usize = buf.len();
|
||||
let rounded_len: usize = self
|
||||
.round_up(
|
||||
file_offset
|
||||
.checked_add(buf_len)
|
||||
.unwrap()
|
||||
.try_into()
|
||||
.unwrap(),
|
||||
)
|
||||
.try_into()
|
||||
.unwrap();
|
||||
|
||||
let layout = Layout::from_size_align(rounded_len, self.alignment).unwrap();
|
||||
// SAFETY: layout has non-zero size
|
||||
let tmp_ptr = unsafe { alloc_zeroed(layout) };
|
||||
if tmp_ptr.is_null() {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
// SAFETY: tmp_ptr is valid and at least rounded_len long
|
||||
let tmp_buf = unsafe { slice::from_raw_parts_mut(tmp_ptr, rounded_len) };
|
||||
|
||||
// This can eventually replaced with read_at once its interface
|
||||
// has been stabilized.
|
||||
// SAFETY: FFI call. All parameters are valid.
|
||||
let ret = unsafe {
|
||||
::libc::pread64(
|
||||
self.file.as_raw_fd(),
|
||||
tmp_buf.as_mut_ptr().cast(),
|
||||
tmp_buf.len(),
|
||||
rounded_pos.try_into().unwrap(),
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
// SAFETY: tmp_ptr was allocated by alloc_zeroed with layout
|
||||
unsafe { dealloc(tmp_ptr, layout) };
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
let read: usize = ret.try_into().unwrap();
|
||||
if read < file_offset {
|
||||
// SAFETY: tmp_ptr was allocated by alloc_zeroed with layout
|
||||
unsafe { dealloc(tmp_ptr, layout) };
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
let mut to_copy = read - file_offset;
|
||||
if to_copy > buf_len {
|
||||
to_copy = buf_len;
|
||||
}
|
||||
|
||||
buf.copy_from_slice(&tmp_buf[file_offset..(file_offset + buf_len)]);
|
||||
// SAFETY: tmp_ptr was allocated by alloc_zeroed with layout
|
||||
unsafe { dealloc(tmp_ptr, layout) };
|
||||
|
||||
self.seek(SeekFrom::Current(to_copy.try_into().unwrap()))
|
||||
.unwrap();
|
||||
Ok(to_copy)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Write for RawFile {
|
||||
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
|
||||
if self.is_aligned(buf) {
|
||||
match self.file.write(buf) {
|
||||
Ok(r) => {
|
||||
self.position = self.position.checked_add(r.try_into().unwrap()).unwrap();
|
||||
Ok(r)
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
} else {
|
||||
let rounded_pos: u64 = self.round_down(self.position);
|
||||
let file_offset: usize = self
|
||||
.position
|
||||
.checked_sub(rounded_pos)
|
||||
.unwrap()
|
||||
.try_into()
|
||||
.unwrap();
|
||||
let buf_len: usize = buf.len();
|
||||
let rounded_len: usize = self
|
||||
.round_up(
|
||||
file_offset
|
||||
.checked_add(buf_len)
|
||||
.unwrap()
|
||||
.try_into()
|
||||
.unwrap(),
|
||||
)
|
||||
.try_into()
|
||||
.unwrap();
|
||||
|
||||
let layout = Layout::from_size_align(rounded_len, self.alignment).unwrap();
|
||||
// SAFETY: layout has non-zero size
|
||||
let tmp_ptr = unsafe { alloc_zeroed(layout) };
|
||||
if tmp_ptr.is_null() {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
// SAFETY: tmp_ptr is at least rounded_len long
|
||||
let tmp_buf = unsafe { slice::from_raw_parts_mut(tmp_ptr, rounded_len) };
|
||||
|
||||
// This can eventually replaced with read_at once its interface
|
||||
// has been stabilized.
|
||||
// SAFETY: FFI call
|
||||
let ret = unsafe {
|
||||
::libc::pread64(
|
||||
self.file.as_raw_fd(),
|
||||
tmp_buf.as_mut_ptr().cast(),
|
||||
tmp_buf.len(),
|
||||
rounded_pos.try_into().unwrap(),
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
// SAFETY: tmp_ptr was allocated by alloc_zeroed with layout
|
||||
unsafe { dealloc(tmp_ptr, layout) };
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
tmp_buf[file_offset..(file_offset + buf_len)].copy_from_slice(buf);
|
||||
|
||||
// This can eventually replaced with write_at once its interface
|
||||
// has been stabilized.
|
||||
// SAFETY: FFI call
|
||||
let ret = unsafe {
|
||||
::libc::pwrite64(
|
||||
self.file.as_raw_fd(),
|
||||
tmp_buf.as_ptr().cast(),
|
||||
tmp_buf.len(),
|
||||
rounded_pos.try_into().unwrap(),
|
||||
)
|
||||
};
|
||||
|
||||
// SAFETY: tmp_ptr was allocated by alloc_zeroed with layout
|
||||
unsafe { dealloc(tmp_ptr, layout) };
|
||||
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
let written: usize = ret.try_into().unwrap();
|
||||
if written < file_offset {
|
||||
Ok(0)
|
||||
} else {
|
||||
let mut to_seek = written - file_offset;
|
||||
if to_seek > buf_len {
|
||||
to_seek = buf_len;
|
||||
}
|
||||
|
||||
self.seek(SeekFrom::Current(to_seek.try_into().unwrap()))
|
||||
.unwrap();
|
||||
Ok(to_seek)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn flush(&mut self) -> std::io::Result<()> {
|
||||
self.file.sync_all()
|
||||
}
|
||||
}
|
||||
|
||||
impl Seek for RawFile {
|
||||
fn seek(&mut self, newpos: SeekFrom) -> std::io::Result<u64> {
|
||||
match self.file.seek(newpos) {
|
||||
Ok(pos) => {
|
||||
self.position = pos;
|
||||
Ok(pos)
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl WriteZeroesAt for RawFile {
|
||||
fn write_zeroes_at(&mut self, offset: u64, length: usize) -> std::io::Result<usize> {
|
||||
self.file.write_zeroes_at(offset, length)
|
||||
}
|
||||
}
|
||||
|
||||
impl PunchHole for RawFile {
|
||||
fn punch_hole(&mut self, offset: u64, length: u64) -> std::io::Result<()> {
|
||||
self.file.punch_hole(offset, length)
|
||||
}
|
||||
}
|
||||
|
||||
impl FileSync for RawFile {
|
||||
fn fsync(&mut self) -> std::io::Result<()> {
|
||||
self.file.fsync()
|
||||
}
|
||||
}
|
||||
|
||||
impl SeekHole for RawFile {
|
||||
fn seek_hole(&mut self, offset: u64) -> std::io::Result<Option<u64>> {
|
||||
match self.file.seek_hole(offset) {
|
||||
Ok(pos) => {
|
||||
if let Some(p) = pos {
|
||||
self.position = p;
|
||||
}
|
||||
Ok(pos)
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
fn seek_data(&mut self, offset: u64) -> std::io::Result<Option<u64>> {
|
||||
match self.file.seek_data(offset) {
|
||||
Ok(pos) => {
|
||||
if let Some(p) = pos {
|
||||
self.position = p;
|
||||
}
|
||||
Ok(pos)
|
||||
}
|
||||
Err(e) => Err(e),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl BlockBackend for RawFile {
|
||||
fn logical_size(&self) -> std::result::Result<u64, crate::Error> {
|
||||
Ok(query_device_size(&self.file)
|
||||
.map_err(crate::Error::RawFileError)?
|
||||
.0)
|
||||
}
|
||||
|
||||
fn physical_size(&self) -> std::result::Result<u64, crate::Error> {
|
||||
Ok(query_device_size(&self.file)
|
||||
.map_err(crate::Error::RawFileError)?
|
||||
.1)
|
||||
}
|
||||
}
|
||||
|
||||
impl Clone for RawFile {
|
||||
fn clone(&self) -> Self {
|
||||
RawFile {
|
||||
file: self.file.try_clone().expect("RawFile cloning failed"),
|
||||
alignment: self.alignment,
|
||||
position: self.position,
|
||||
direct_io: self.direct_io,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl AsRawFd for RawFile {
|
||||
fn as_raw_fd(&self) -> RawFd {
|
||||
self.file.as_raw_fd()
|
||||
}
|
||||
}
|
||||
|
||||
impl AsFd for RawFile {
|
||||
fn as_fd(&self) -> BorrowedFd<'_> {
|
||||
self.file.as_fd()
|
||||
}
|
||||
}
|
||||
268
block/src/formats/qcow/internal/refcount.rs
Normal file
268
block/src/formats/qcow/internal/refcount.rs
Normal file
@@ -0,0 +1,268 @@
|
||||
// Copyright 2018 The Chromium OS Authors. All rights reserved.
|
||||
// Use of this source code is governed by a BSD-style license that can be
|
||||
// found in the LICENSE-BSD-3-Clause file.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
use std::io;
|
||||
|
||||
use libc::EINVAL;
|
||||
use thiserror::Error;
|
||||
|
||||
use super::qcow_raw_file::QcowRawFile;
|
||||
use super::vec_cache::{CacheMap, Cacheable, VecCache};
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
pub enum Error {
|
||||
/// `EvictingCache` - Error writing a refblock from the cache to disk.
|
||||
#[error("Failed to write a refblock from the cache to disk")]
|
||||
EvictingRefCounts(#[source] io::Error),
|
||||
/// `InvalidIndex` - Address requested isn't within the range of the disk.
|
||||
#[error("Address requested is not within the range of the disk")]
|
||||
InvalidIndex,
|
||||
/// `RefblockUnaligned` - Refcount block offset is not cluster aligned.
|
||||
#[error("Refcount block offset {0:#x} is not cluster aligned")]
|
||||
RefblockUnaligned(u64),
|
||||
/// `NeedCluster` - Handle this error by reading the cluster and calling the function again.
|
||||
#[error("Cluster with addr={0} needs to be read")]
|
||||
NeedCluster(u64),
|
||||
/// `NeedNewCluster` - Handle this error by allocating a cluster and calling the function again.
|
||||
#[error("New cluster needs to be allocated for refcounts")]
|
||||
NeedNewCluster,
|
||||
/// `ReadingRefCounts` - Error reading the file into the refcount cache.
|
||||
#[error("Failed to read the file into the refcount cache")]
|
||||
ReadingRefCounts(#[source] io::Error),
|
||||
/// `RefcountOverflow` - Refcount value exceeds maximum for the refcount width.
|
||||
#[error("Refcount value {value} exceeds {refcount_bits}-bit max ({max})")]
|
||||
RefcountOverflow {
|
||||
value: u64,
|
||||
max: u64,
|
||||
refcount_bits: u64,
|
||||
},
|
||||
}
|
||||
|
||||
pub type Result<T> = std::result::Result<T, Error>;
|
||||
|
||||
/// Represents the refcount entries for an open qcow file.
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct RefCount {
|
||||
ref_table: VecCache<u64>,
|
||||
refcount_table_offset: u64,
|
||||
refblock_cache: CacheMap<VecCache<u64>>,
|
||||
refcount_block_entries: u64, // number of refcounts in a cluster.
|
||||
cluster_size: u64,
|
||||
max_valid_cluster_offset: u64,
|
||||
max_refcount: u64, // maximum refcount value for this image's refcount_order
|
||||
refcount_bits: u64, // number of bits per refcount entry
|
||||
}
|
||||
|
||||
impl RefCount {
|
||||
/// Creates a `RefCount` from `file`, reading the refcount table from `refcount_table_offset`.
|
||||
/// `refcount_table_entries` specifies the number of refcount blocks used by this image.
|
||||
/// `refcount_block_entries` indicates the number of refcounts in each refcount block.
|
||||
/// `refcount_bits` is the number of bits per refcount (1, 2, 4, 8, 16, 32, or 64).
|
||||
/// Each refcount table entry points to a refcount block.
|
||||
pub fn new(
|
||||
raw_file: &mut QcowRawFile,
|
||||
refcount_table_offset: u64,
|
||||
refcount_table_entries: u64,
|
||||
refcount_block_entries: u64,
|
||||
cluster_size: u64,
|
||||
refcount_bits: u64,
|
||||
) -> io::Result<RefCount> {
|
||||
let ref_table = VecCache::from_vec(raw_file.read_pointer_table(
|
||||
refcount_table_offset,
|
||||
refcount_table_entries,
|
||||
None,
|
||||
)?);
|
||||
let max_valid_cluster_index = (ref_table.len() as u64) * refcount_block_entries - 1;
|
||||
let max_valid_cluster_offset = max_valid_cluster_index * cluster_size;
|
||||
let max_refcount = if refcount_bits >= 64 {
|
||||
u64::MAX
|
||||
} else {
|
||||
(1u64 << refcount_bits) - 1
|
||||
};
|
||||
Ok(RefCount {
|
||||
ref_table,
|
||||
refcount_table_offset,
|
||||
refblock_cache: CacheMap::new(50),
|
||||
refcount_block_entries,
|
||||
cluster_size,
|
||||
max_valid_cluster_offset,
|
||||
max_refcount,
|
||||
refcount_bits,
|
||||
})
|
||||
}
|
||||
|
||||
/// Returns the number of refcounts per block.
|
||||
pub fn refcounts_per_block(&self) -> u64 {
|
||||
self.refcount_block_entries
|
||||
}
|
||||
|
||||
/// Returns the maximum valid cluster offset in the raw file for this refcount table.
|
||||
pub fn max_valid_cluster_offset(&self) -> u64 {
|
||||
self.max_valid_cluster_offset
|
||||
}
|
||||
|
||||
/// Returns `NeedNewCluster` if a new cluster needs to be allocated for refcounts. If an
|
||||
/// existing cluster needs to be read, `NeedCluster(addr)` is returned. The Caller should
|
||||
/// allocate a cluster or read the required one and call this function again with the cluster.
|
||||
/// On success, an optional address of a dropped cluster is returned. The dropped cluster can
|
||||
/// be reused for other purposes.
|
||||
pub fn set_cluster_refcount(
|
||||
&mut self,
|
||||
raw_file: &mut QcowRawFile,
|
||||
cluster_address: u64,
|
||||
refcount: u64,
|
||||
mut new_cluster: Option<(u64, VecCache<u64>)>,
|
||||
) -> Result<Option<u64>> {
|
||||
if refcount > self.max_refcount {
|
||||
return Err(Error::RefcountOverflow {
|
||||
value: refcount,
|
||||
max: self.max_refcount,
|
||||
refcount_bits: self.refcount_bits,
|
||||
});
|
||||
}
|
||||
|
||||
let (table_index, block_index) = self.get_refcount_index(cluster_address);
|
||||
|
||||
let block_addr_disk = *self.ref_table.get(table_index).ok_or(Error::InvalidIndex)?;
|
||||
|
||||
// Fill the cache if this block isn't yet there.
|
||||
if !self.refblock_cache.contains_key(table_index) {
|
||||
// Need a new cluster
|
||||
if let Some((addr, table)) = new_cluster.take() {
|
||||
self.ref_table[table_index] = addr;
|
||||
let ref_table = &self.ref_table;
|
||||
self.refblock_cache
|
||||
.insert(table_index, table, |index, evicted| {
|
||||
raw_file.write_refcount_block(ref_table[index], evicted.get_values())
|
||||
})
|
||||
.map_err(Error::EvictingRefCounts)?;
|
||||
} else {
|
||||
if block_addr_disk == 0 {
|
||||
return Err(Error::NeedNewCluster);
|
||||
}
|
||||
return Err(Error::NeedCluster(block_addr_disk));
|
||||
}
|
||||
}
|
||||
|
||||
// Unwrap is safe here as the entry was filled directly above.
|
||||
let dropped_cluster = if self.refblock_cache.get(table_index).unwrap().dirty() {
|
||||
None
|
||||
} else {
|
||||
// Free the previously used block and use a new one. Writing modified counts to new
|
||||
// blocks keeps the on-disk state consistent even if it's out of date.
|
||||
if let Some((addr, _)) = new_cluster.take() {
|
||||
self.ref_table[table_index] = addr;
|
||||
Some(block_addr_disk)
|
||||
} else {
|
||||
return Err(Error::NeedNewCluster);
|
||||
}
|
||||
};
|
||||
|
||||
self.refblock_cache.get_mut(table_index).unwrap()[block_index] = refcount;
|
||||
Ok(dropped_cluster)
|
||||
}
|
||||
|
||||
/// Flush the dirty refcount blocks. This must be done before flushing the table that points to
|
||||
/// the blocks.
|
||||
pub fn flush_blocks(&mut self, raw_file: &mut QcowRawFile) -> io::Result<()> {
|
||||
// Write out all dirty L2 tables.
|
||||
for (table_index, block) in self.refblock_cache.iter_mut().filter(|(_k, v)| v.dirty()) {
|
||||
let addr = self.ref_table[*table_index];
|
||||
if addr != 0 {
|
||||
raw_file.write_refcount_block(addr, block.get_values())?;
|
||||
} else {
|
||||
return Err(std::io::Error::from_raw_os_error(EINVAL));
|
||||
}
|
||||
block.mark_clean();
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Flush the refcount table that keeps the address of the refcounts blocks.
|
||||
/// Returns true if the table changed since the previous `flush_table()` call.
|
||||
pub fn flush_table(&mut self, raw_file: &mut QcowRawFile) -> io::Result<bool> {
|
||||
if self.ref_table.dirty() {
|
||||
raw_file
|
||||
.write_pointer_table_direct(self.refcount_table_offset, self.ref_table.iter())?;
|
||||
self.ref_table.mark_clean();
|
||||
Ok(true)
|
||||
} else {
|
||||
Ok(false)
|
||||
}
|
||||
}
|
||||
|
||||
/// Gets the refcount for a cluster with the given address.
|
||||
pub fn get_cluster_refcount(
|
||||
&mut self,
|
||||
raw_file: &mut QcowRawFile,
|
||||
address: u64,
|
||||
) -> Result<u64> {
|
||||
let (table_index, block_index) = self.get_refcount_index(address);
|
||||
let block_addr_disk = *self.ref_table.get(table_index).ok_or(Error::InvalidIndex)?;
|
||||
if block_addr_disk == 0 {
|
||||
return Ok(0);
|
||||
}
|
||||
if block_addr_disk & (self.cluster_size - 1) != 0 {
|
||||
return Err(Error::RefblockUnaligned(block_addr_disk));
|
||||
}
|
||||
if !self.refblock_cache.contains_key(table_index) {
|
||||
let table = VecCache::from_vec(
|
||||
raw_file
|
||||
.read_refcount_block(block_addr_disk)
|
||||
.map_err(Error::ReadingRefCounts)?,
|
||||
);
|
||||
let ref_table = &self.ref_table;
|
||||
self.refblock_cache
|
||||
.insert(table_index, table, |index, evicted| {
|
||||
raw_file.write_refcount_block(ref_table[index], evicted.get_values())
|
||||
})
|
||||
.map_err(Error::EvictingRefCounts)?;
|
||||
}
|
||||
Ok(self.refblock_cache.get(table_index).unwrap()[block_index])
|
||||
}
|
||||
|
||||
/// Returns the refcount table for this file. This is only useful for debugging.
|
||||
pub fn ref_table(&self) -> &[u64] {
|
||||
self.ref_table.get_values()
|
||||
}
|
||||
|
||||
/// Returns the refcounts stored in the given block.
|
||||
pub fn refcount_block(
|
||||
&mut self,
|
||||
raw_file: &mut QcowRawFile,
|
||||
table_index: usize,
|
||||
) -> Result<Option<&[u64]>> {
|
||||
let block_addr_disk = *self.ref_table.get(table_index).ok_or(Error::InvalidIndex)?;
|
||||
if block_addr_disk == 0 {
|
||||
return Ok(None);
|
||||
}
|
||||
if !self.refblock_cache.contains_key(table_index) {
|
||||
let table = VecCache::from_vec(
|
||||
raw_file
|
||||
.read_refcount_block(block_addr_disk)
|
||||
.map_err(Error::ReadingRefCounts)?,
|
||||
);
|
||||
// TODO(dgreid) - closure needs to return an error.
|
||||
let ref_table = &self.ref_table;
|
||||
self.refblock_cache
|
||||
.insert(table_index, table, |index, evicted| {
|
||||
raw_file.write_refcount_block(ref_table[index], evicted.get_values())
|
||||
})
|
||||
.map_err(Error::EvictingRefCounts)?;
|
||||
}
|
||||
// The index must exist as it was just inserted if it didn't already.
|
||||
Ok(Some(
|
||||
self.refblock_cache.get(table_index).unwrap().get_values(),
|
||||
))
|
||||
}
|
||||
|
||||
// Gets the address of the refcount block and the index into the block for the given address.
|
||||
fn get_refcount_index(&self, address: u64) -> (usize, usize) {
|
||||
let block_index = (address / self.cluster_size) % self.refcount_block_entries;
|
||||
let refcount_table_index = (address / self.cluster_size) / self.refcount_block_entries;
|
||||
(refcount_table_index as usize, block_index as usize)
|
||||
}
|
||||
}
|
||||
84
block/src/formats/qcow/internal/util.rs
Normal file
84
block/src/formats/qcow/internal/util.rs
Normal file
@@ -0,0 +1,84 @@
|
||||
// Copyright 2018 The Chromium OS Authors. All rights reserved.
|
||||
// Use of this source code is governed by a BSD-style license that can be
|
||||
// found in the LICENSE-BSD-3-Clause file.
|
||||
//
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
//! Pure helper functions and constants for QCOW2 L1/L2 table entry
|
||||
//! manipulation and integer arithmetic. Shared across the `qcow` submodules.
|
||||
|
||||
/// Nesting depth limit for disk formats that can open other disk files.
|
||||
pub(crate) const MAX_NESTING_DEPTH: u32 = 10;
|
||||
|
||||
// bits 0-8 and 56-63 are reserved.
|
||||
pub(super) const L1_TABLE_OFFSET_MASK: u64 = 0x00ff_ffff_ffff_fe00;
|
||||
pub(super) const L2_TABLE_OFFSET_MASK: u64 = 0x00ff_ffff_ffff_fe00;
|
||||
// Flags
|
||||
pub(super) const ZERO_FLAG: u64 = 1 << 0;
|
||||
pub(super) const COMPRESSED_FLAG: u64 = 1 << 62;
|
||||
pub(super) const COMPRESSED_SECTOR_SIZE: u64 = 512;
|
||||
pub(super) const CLUSTER_USED_FLAG: u64 = 1 << 63;
|
||||
|
||||
/// Check if L2 entry is empty (unallocated).
|
||||
pub(super) fn l2_entry_is_empty(l2_entry: u64) -> bool {
|
||||
l2_entry == 0
|
||||
}
|
||||
|
||||
/// Check bit 0 - only valid for standard clusters.
|
||||
pub(super) fn l2_entry_is_zero(l2_entry: u64) -> bool {
|
||||
l2_entry & ZERO_FLAG != 0
|
||||
}
|
||||
|
||||
/// Check if L2 entry refers to a compressed cluster.
|
||||
pub(super) fn l2_entry_is_compressed(l2_entry: u64) -> bool {
|
||||
l2_entry & COMPRESSED_FLAG != 0
|
||||
}
|
||||
|
||||
/// Get file offset and size of compressed cluster data.
|
||||
pub(super) fn l2_entry_compressed_cluster_layout(l2_entry: u64, cluster_bits: u32) -> (u64, usize) {
|
||||
let compressed_size_shift = 62 - (cluster_bits - 8);
|
||||
let compressed_size_mask = (1 << (cluster_bits - 8)) - 1;
|
||||
let compressed_cluster_addr = l2_entry & ((1 << compressed_size_shift) - 1);
|
||||
let nsectors = (l2_entry >> compressed_size_shift & compressed_size_mask) + 1;
|
||||
let compressed_cluster_size = ((nsectors * COMPRESSED_SECTOR_SIZE)
|
||||
- (compressed_cluster_addr & (COMPRESSED_SECTOR_SIZE - 1)))
|
||||
as usize;
|
||||
(compressed_cluster_addr, compressed_cluster_size)
|
||||
}
|
||||
|
||||
/// Get file offset of standard (non-compressed) cluster.
|
||||
pub(super) fn l2_entry_std_cluster_addr(l2_entry: u64) -> u64 {
|
||||
l2_entry & L2_TABLE_OFFSET_MASK
|
||||
}
|
||||
|
||||
/// Make L2 entry for standard (non-compressed) cluster.
|
||||
pub(super) fn l2_entry_make_std(cluster_addr: u64) -> u64 {
|
||||
(cluster_addr & L2_TABLE_OFFSET_MASK) | CLUSTER_USED_FLAG
|
||||
}
|
||||
|
||||
/// Make L2 entry for preallocated zero cluster.
|
||||
pub(super) fn l2_entry_make_zero(cluster_addr: u64) -> u64 {
|
||||
(cluster_addr & L2_TABLE_OFFSET_MASK) | CLUSTER_USED_FLAG | ZERO_FLAG
|
||||
}
|
||||
|
||||
/// Make L2 entry for an unallocated cluster that reads as logical zeros.
|
||||
pub(super) fn l2_entry_make_zero_plain() -> u64 {
|
||||
ZERO_FLAG
|
||||
}
|
||||
|
||||
/// Make L1 entry with optional flags.
|
||||
pub(super) fn l1_entry_make(cluster_addr: u64, refcount_is_one: bool) -> u64 {
|
||||
(cluster_addr & L1_TABLE_OFFSET_MASK) | (refcount_is_one as u64 * CLUSTER_USED_FLAG)
|
||||
}
|
||||
|
||||
/// Ceiling of the division of `dividend`/`divisor`.
|
||||
pub(super) fn div_round_up_u32(dividend: u32, divisor: u32) -> u32 {
|
||||
dividend / divisor + u32::from(!dividend.is_multiple_of(divisor))
|
||||
}
|
||||
|
||||
/// Ceiling of the division of `dividend`/`divisor`.
|
||||
pub(super) fn div_round_up_u64(dividend: u64, divisor: u64) -> u64 {
|
||||
dividend / divisor + u64::from(!dividend.is_multiple_of(divisor))
|
||||
}
|
||||
210
block/src/formats/qcow/internal/vec_cache.rs
Normal file
210
block/src/formats/qcow/internal/vec_cache.rs
Normal file
@@ -0,0 +1,210 @@
|
||||
// Copyright 2018 The Chromium OS Authors. All rights reserved.
|
||||
// Use of this source code is governed by a BSD-style license that can be
|
||||
// found in the LICENSE-BSD-3-Clause file.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0 AND BSD-3-Clause
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::collections::hash_map::IterMut;
|
||||
use std::io;
|
||||
use std::ops::{Deref, Index, IndexMut};
|
||||
use std::slice::SliceIndex;
|
||||
|
||||
/// Trait that allows for checking if an implementor is dirty. Useful for types that are cached so
|
||||
/// it can be checked if they need to be committed to disk.
|
||||
pub trait Cacheable {
|
||||
/// Used to check if the item needs to be written out or if it can be discarded.
|
||||
fn dirty(&self) -> bool;
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
/// Represents a vector that implements the `Cacheable` trait so it can be held in a cache.
|
||||
pub struct VecCache<T: 'static + Copy + Default> {
|
||||
vec: Box<[T]>,
|
||||
dirty: bool,
|
||||
}
|
||||
|
||||
impl<T: 'static + Copy + Default> VecCache<T> {
|
||||
/// Creates a `VecCache` that can hold `count` elements.
|
||||
pub fn new(count: usize) -> VecCache<T> {
|
||||
VecCache {
|
||||
vec: vec![Default::default(); count].into_boxed_slice(),
|
||||
dirty: true,
|
||||
}
|
||||
}
|
||||
|
||||
/// Creates a `VecCache` from the passed in `vec`.
|
||||
pub fn from_vec(vec: Vec<T>) -> VecCache<T> {
|
||||
VecCache {
|
||||
vec: vec.into_boxed_slice(),
|
||||
dirty: false,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn get<I>(&self, index: I) -> Option<&<I as SliceIndex<[T]>>::Output>
|
||||
where
|
||||
I: SliceIndex<[T]>,
|
||||
{
|
||||
self.vec.get(index)
|
||||
}
|
||||
|
||||
/// Gets a reference to the underlying vector.
|
||||
pub fn get_values(&self) -> &[T] {
|
||||
&self.vec
|
||||
}
|
||||
|
||||
/// Mark this cache element as clean.
|
||||
pub fn mark_clean(&mut self) {
|
||||
self.dirty = false;
|
||||
}
|
||||
|
||||
/// Returns the number of elements in the vector.
|
||||
pub fn len(&self) -> usize {
|
||||
self.vec.len()
|
||||
}
|
||||
|
||||
/// Extends the cache capacity to `new_len` elements.
|
||||
///
|
||||
/// No-op if `new_len <= self.len()`. Allocates a new buffer, copies
|
||||
/// existing data, and fills new elements with default values.
|
||||
/// Marks the cache as dirty.
|
||||
pub fn extend(&mut self, new_len: usize) {
|
||||
if new_len <= self.vec.len() {
|
||||
return;
|
||||
}
|
||||
let mut new_vec = vec![Default::default(); new_len];
|
||||
new_vec[..self.vec.len()].copy_from_slice(&self.vec);
|
||||
self.vec = new_vec.into_boxed_slice();
|
||||
self.dirty = true;
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: 'static + Copy + Default> Cacheable for VecCache<T> {
|
||||
fn dirty(&self) -> bool {
|
||||
self.dirty
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: 'static + Copy + Default> Index<usize> for VecCache<T> {
|
||||
type Output = T;
|
||||
|
||||
fn index(&self, index: usize) -> &T {
|
||||
self.vec.index(index)
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: 'static + Copy + Default> IndexMut<usize> for VecCache<T> {
|
||||
fn index_mut(&mut self, index: usize) -> &mut T {
|
||||
self.dirty = true;
|
||||
self.vec.index_mut(index)
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: 'static + Copy + Default> Deref for VecCache<T> {
|
||||
type Target = [T];
|
||||
|
||||
fn deref(&self) -> &[T] {
|
||||
&self.vec
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct CacheMap<T: Cacheable> {
|
||||
capacity: usize,
|
||||
map: HashMap<usize, T>,
|
||||
}
|
||||
|
||||
impl<T: Cacheable> CacheMap<T> {
|
||||
pub fn new(capacity: usize) -> Self {
|
||||
CacheMap {
|
||||
capacity,
|
||||
map: HashMap::with_capacity(capacity),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn contains_key(&self, key: usize) -> bool {
|
||||
self.map.contains_key(&key)
|
||||
}
|
||||
|
||||
pub fn get(&self, index: usize) -> Option<&T> {
|
||||
self.map.get(&index)
|
||||
}
|
||||
|
||||
pub fn get_mut(&mut self, index: usize) -> Option<&mut T> {
|
||||
self.map.get_mut(&index)
|
||||
}
|
||||
|
||||
pub fn iter_mut(&mut self) -> IterMut<'_, usize, T> {
|
||||
self.map.iter_mut()
|
||||
}
|
||||
|
||||
// Check if the refblock cache is full and we need to evict.
|
||||
pub fn insert<F>(&mut self, index: usize, block: T, write_callback: F) -> io::Result<()>
|
||||
where
|
||||
F: FnOnce(usize, T) -> io::Result<()>,
|
||||
{
|
||||
if self.map.len() == self.capacity {
|
||||
// TODO(dgreid) - smarter eviction strategy.
|
||||
let to_evict = *self.map.iter().next().unwrap().0;
|
||||
if let Some(evicted) = self.map.remove(&to_evict)
|
||||
&& evicted.dirty()
|
||||
{
|
||||
write_callback(to_evict, evicted)?;
|
||||
}
|
||||
}
|
||||
self.map.insert(index, block);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod unit_tests {
|
||||
use super::*;
|
||||
|
||||
struct NumCache(());
|
||||
impl Cacheable for NumCache {
|
||||
fn dirty(&self) -> bool {
|
||||
true
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn evicts_when_full() {
|
||||
let mut cache = CacheMap::<NumCache>::new(3);
|
||||
let mut evicted = None;
|
||||
cache
|
||||
.insert(0, NumCache(()), |index, _| {
|
||||
evicted = Some(index);
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(evicted, None);
|
||||
cache
|
||||
.insert(1, NumCache(()), |index, _| {
|
||||
evicted = Some(index);
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(evicted, None);
|
||||
cache
|
||||
.insert(2, NumCache(()), |index, _| {
|
||||
evicted = Some(index);
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(evicted, None);
|
||||
cache
|
||||
.insert(3, NumCache(()), |index, _| {
|
||||
evicted = Some(index);
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
assert!(evicted.is_some());
|
||||
|
||||
// Check that three of the four items inserted are still there and that the most recently
|
||||
// inserted is one of them.
|
||||
let num_items = (0..=3).filter(|k| cache.contains_key(*k)).count();
|
||||
assert_eq!(num_items, 3);
|
||||
assert!(cache.contains_key(3));
|
||||
}
|
||||
}
|
||||
275
block/src/formats/qcow/mod.rs
Normal file
275
block/src/formats/qcow/mod.rs
Normal file
@@ -0,0 +1,275 @@
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
//! QCOW2 disk image format.
|
||||
//!
|
||||
//! Provides [`QcowDisk`], the `DiskFile` wrapper for QCOW2 images
|
||||
//! with backing file and compression support.
|
||||
|
||||
pub(crate) mod common;
|
||||
pub mod internal;
|
||||
pub mod worker;
|
||||
|
||||
use std::fs::File;
|
||||
use std::os::unix::io::AsRawFd;
|
||||
use std::sync::Arc;
|
||||
use std::{fmt, io};
|
||||
|
||||
use self::internal::backing::shared_backing_from;
|
||||
use self::internal::metadata::{BackingRead, QcowMetadata};
|
||||
use self::internal::qcow_raw_file::QcowRawFile;
|
||||
use self::internal::{MAX_NESTING_DEPTH, RawFile, parse_qcow};
|
||||
#[cfg(feature = "io_uring")]
|
||||
use self::worker::async_uring::QcowAsync;
|
||||
use self::worker::sync::QcowSync;
|
||||
use crate::async_io::{AsyncIo, BorrowedDiskFd, DiskFileError};
|
||||
use crate::disk_file;
|
||||
use crate::error::{BlockError, BlockErrorKind, BlockResult, ErrorOp};
|
||||
|
||||
/// Unified DiskFile wrapper for QCOW2 disk images.
|
||||
///
|
||||
/// Holds the in memory QCOW2 metadata, the data file, and an optional
|
||||
/// backing file. The metadata is wrapped in an `Arc` because
|
||||
/// [`QcowSync`] and [`QcowAsync`] I/O workers receive a clone when
|
||||
/// they are created via [`create_async_io`](DiskFile::create_async_io).
|
||||
/// The backing file is likewise shared with workers through an `Arc`.
|
||||
///
|
||||
/// The `sparse` flag controls whether the image advertises discard
|
||||
/// support to the guest. The `use_io_uring` flag selects between the
|
||||
/// [`QcowSync`] and [`QcowAsync`] I/O backends. Both are recorded at
|
||||
/// construction time and propagated through [`try_clone`](DiskFile::try_clone).
|
||||
pub struct QcowDisk {
|
||||
metadata: Arc<QcowMetadata>,
|
||||
backing_file: Option<Arc<dyn BackingRead>>,
|
||||
sparse: bool,
|
||||
data_raw_file: QcowRawFile,
|
||||
use_io_uring: bool,
|
||||
}
|
||||
|
||||
impl fmt::Debug for QcowDisk {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
f.debug_struct("QcowDisk")
|
||||
.field("sparse", &self.sparse)
|
||||
.field("has_backing", &self.backing_file.is_some())
|
||||
.field("use_io_uring", &self.use_io_uring)
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
impl QcowDisk {
|
||||
pub fn new(
|
||||
file: File,
|
||||
direct_io: bool,
|
||||
backing_files: bool,
|
||||
sparse: bool,
|
||||
use_io_uring: bool,
|
||||
) -> BlockResult<Self> {
|
||||
#[cfg(not(feature = "io_uring"))]
|
||||
if use_io_uring {
|
||||
return Err(BlockError::new(
|
||||
BlockErrorKind::UnsupportedFeature,
|
||||
DiskFileError::NewAsyncIo(io::Error::other(
|
||||
"io_uring requested but feature is not enabled",
|
||||
)),
|
||||
));
|
||||
}
|
||||
|
||||
let max_nesting_depth = if backing_files { MAX_NESTING_DEPTH } else { 0 };
|
||||
let raw_file = RawFile::new(file, direct_io);
|
||||
let (inner, backing_file, sparse) = parse_qcow(raw_file, max_nesting_depth, sparse)
|
||||
.map_err(|e| {
|
||||
let e = if !backing_files && matches!(e.kind(), BlockErrorKind::Overflow) {
|
||||
e.with_kind(BlockErrorKind::UnsupportedFeature)
|
||||
} else {
|
||||
e
|
||||
};
|
||||
e.with_op(ErrorOp::Open)
|
||||
})?;
|
||||
let data_raw_file = inner.raw_file.clone();
|
||||
Ok(QcowDisk {
|
||||
metadata: Arc::new(QcowMetadata::new(inner)),
|
||||
backing_file: backing_file.map(shared_backing_from).transpose()?,
|
||||
sparse,
|
||||
data_raw_file,
|
||||
use_io_uring,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for QcowDisk {
|
||||
fn drop(&mut self) {
|
||||
self.metadata.shutdown();
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::DiskSize for QcowDisk {
|
||||
fn logical_size(&self) -> BlockResult<u64> {
|
||||
Ok(self.metadata.virtual_size())
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::PhysicalSize for QcowDisk {
|
||||
fn physical_size(&self) -> BlockResult<u64> {
|
||||
Ok(self.data_raw_file.physical_size()?)
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::DiskFd for QcowDisk {
|
||||
fn fd(&self) -> BorrowedDiskFd<'_> {
|
||||
BorrowedDiskFd::new(self.data_raw_file.as_raw_fd())
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::Geometry for QcowDisk {}
|
||||
|
||||
impl disk_file::SparseCapable for QcowDisk {
|
||||
fn supports_sparse_operations(&self) -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
fn supports_zero_flag(&self) -> bool {
|
||||
true
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::Resizable for QcowDisk {
|
||||
fn resize(&mut self, size: u64) -> BlockResult<()> {
|
||||
if self.backing_file.is_some() {
|
||||
return Err(BlockError::new(
|
||||
BlockErrorKind::UnsupportedFeature,
|
||||
DiskFileError::ResizeError(io::Error::other(
|
||||
"resize not supported with backing files",
|
||||
)),
|
||||
)
|
||||
.with_op(ErrorOp::Resize));
|
||||
}
|
||||
self.metadata.resize(size).map_err(|e| {
|
||||
BlockError::new(BlockErrorKind::Io, DiskFileError::ResizeError(e))
|
||||
.with_op(ErrorOp::Resize)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl disk_file::DiskFile for QcowDisk {}
|
||||
|
||||
impl disk_file::AsyncDiskFile for QcowDisk {
|
||||
fn try_clone(&self) -> BlockResult<Box<dyn disk_file::AsyncDiskFile>> {
|
||||
Ok(Box::new(QcowDisk {
|
||||
metadata: Arc::clone(&self.metadata),
|
||||
backing_file: self.backing_file.as_ref().map(Arc::clone),
|
||||
sparse: self.sparse,
|
||||
data_raw_file: self.data_raw_file.clone(),
|
||||
use_io_uring: self.use_io_uring,
|
||||
}))
|
||||
}
|
||||
|
||||
fn create_async_io(&self, ring_depth: u32) -> BlockResult<Box<dyn AsyncIo>> {
|
||||
if self.use_io_uring {
|
||||
#[cfg(feature = "io_uring")]
|
||||
{
|
||||
return Ok(Box::new(
|
||||
QcowAsync::new(
|
||||
Arc::clone(&self.metadata),
|
||||
self.data_raw_file.clone(),
|
||||
self.backing_file.as_ref().map(Arc::clone),
|
||||
self.sparse,
|
||||
ring_depth,
|
||||
)
|
||||
.map_err(|e| {
|
||||
BlockError::new(BlockErrorKind::Io, DiskFileError::NewAsyncIo(e))
|
||||
})?,
|
||||
));
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "io_uring"))]
|
||||
unreachable!("use_io_uring is set but io_uring feature is not enabled");
|
||||
}
|
||||
|
||||
let _ = ring_depth;
|
||||
Ok(Box::new(QcowSync::new(
|
||||
Arc::clone(&self.metadata),
|
||||
self.data_raw_file.clone(),
|
||||
self.backing_file.as_ref().map(Arc::clone),
|
||||
self.sparse,
|
||||
)))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod unit_tests {
|
||||
use vmm_sys_util::tempfile::TempFile;
|
||||
|
||||
use self::internal::{QcowFile, RawFile};
|
||||
use super::*;
|
||||
use crate::async_io::AsyncIo;
|
||||
use crate::disk_file::{AsyncDiskFile, DiskSize, PhysicalSize};
|
||||
|
||||
const TEST_SIZE: u64 = 0x5566_7788;
|
||||
|
||||
fn make_qcow_file() -> File {
|
||||
let temp_file = TempFile::new().unwrap();
|
||||
{
|
||||
let raw = RawFile::new(temp_file.as_file().try_clone().unwrap(), false);
|
||||
QcowFile::new(raw, 3, TEST_SIZE, true).unwrap();
|
||||
}
|
||||
temp_file.into_file()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_sync_returns_correct_size() {
|
||||
let file = make_qcow_file();
|
||||
let disk = QcowDisk::new(file, false, false, true, false).unwrap();
|
||||
assert_eq!(disk.logical_size().unwrap(), TEST_SIZE);
|
||||
}
|
||||
|
||||
fn assert_async_io_from_dyn(disk: &dyn AsyncDiskFile, expect_batch: bool) {
|
||||
let io: Box<dyn AsyncIo> = disk.create_async_io(128).unwrap();
|
||||
assert_eq!(io.batch_requests_enabled(), expect_batch);
|
||||
}
|
||||
|
||||
fn assert_async_io(disk: &QcowDisk, expect_batch: bool) {
|
||||
assert_async_io_from_dyn(disk, expect_batch);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sync_backend_disables_batch_requests() {
|
||||
let file = make_qcow_file();
|
||||
let disk = QcowDisk::new(file, false, false, true, false).unwrap();
|
||||
assert_async_io(&disk, false);
|
||||
}
|
||||
|
||||
#[cfg(feature = "io_uring")]
|
||||
#[test]
|
||||
fn io_uring_backend_enables_batch_requests() {
|
||||
let file = make_qcow_file();
|
||||
let disk = QcowDisk::new(file, false, false, true, true).unwrap();
|
||||
assert_async_io(&disk, true);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_clone_preserves_sync_dispatch() {
|
||||
let file = make_qcow_file();
|
||||
let disk = QcowDisk::new(file, false, false, true, false).unwrap();
|
||||
let cloned = disk.try_clone().unwrap();
|
||||
assert_async_io_from_dyn(cloned.as_ref(), false);
|
||||
}
|
||||
|
||||
#[cfg(feature = "io_uring")]
|
||||
#[test]
|
||||
fn try_clone_preserves_io_uring_dispatch() {
|
||||
let file = make_qcow_file();
|
||||
let disk = QcowDisk::new(file, false, false, true, true).unwrap();
|
||||
let cloned = disk.try_clone().unwrap();
|
||||
assert_async_io_from_dyn(cloned.as_ref(), true);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn physical_size_less_than_logical() {
|
||||
// make_qcow_file() writes no guest data, so the file on disk
|
||||
// only contains QCOW2 headers and metadata tables.
|
||||
let file = make_qcow_file();
|
||||
let disk = QcowDisk::new(file, false, false, true, false).unwrap();
|
||||
assert!(disk.physical_size().unwrap() < disk.logical_size().unwrap());
|
||||
}
|
||||
}
|
||||
1183
block/src/formats/qcow/worker/async_uring.rs
Normal file
1183
block/src/formats/qcow/worker/async_uring.rs
Normal file
File diff suppressed because it is too large
Load Diff
9
block/src/formats/qcow/worker/mod.rs
Normal file
9
block/src/formats/qcow/worker/mod.rs
Normal file
@@ -0,0 +1,9 @@
|
||||
// Copyright 2026 The Cloud Hypervisor Authors. All rights reserved.
|
||||
//
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
#[cfg(feature = "io_uring")]
|
||||
pub(crate) mod async_uring;
|
||||
pub(crate) mod sync;
|
||||
|
||||
pub(crate) use super::{common, internal};
|
||||
2201
block/src/formats/qcow/worker/sync.rs
Normal file
2201
block/src/formats/qcow/worker/sync.rs
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user