From faf6f7b6353aa5c07e09a2bf710b7ed91b4050fd Mon Sep 17 00:00:00 2001 From: Anatol Belski Date: Thu, 26 Mar 2026 23:03:38 +0100 Subject: [PATCH] performance-metrics: Add QCOW2 L2 cache cold miss micro benchmark Add micro_bench_qcow_l2_cache_miss which reads one cluster from each of num_ops distinct L2 tables in a sparsely allocated image. Clusters are spaced L2_ENTRIES_PER_TABLE apart so every read touches a different L2 table, forcing eviction when num_ops exceeds the cache capacity. Workloads: 128 and 256 L2 tables. Signed-off-by: Anatol Belski --- performance-metrics/src/main.rs | 26 +++++++++++++++- performance-metrics/src/micro_bench_block.rs | 31 ++++++++++++++++++-- 2 files changed, 54 insertions(+), 3 deletions(-) diff --git a/performance-metrics/src/main.rs b/performance-metrics/src/main.rs index ffb42edfd..2c33bf9a8 100644 --- a/performance-metrics/src/main.rs +++ b/performance-metrics/src/main.rs @@ -378,7 +378,7 @@ mod adjuster { } } -const TEST_LIST: [PerformanceTest; 80] = [ +const TEST_LIST: [PerformanceTest; 82] = [ PerformanceTest { name: "boot_time_ms", func_ptr: performance_boot_time, @@ -1469,6 +1469,30 @@ const TEST_LIST: [PerformanceTest; 80] = [ }, unit_adjuster: adjuster::s_to_us, }, + PerformanceTest { + name: "micro_block_qcow_l2_cache_miss_128_us", + func_ptr: micro_bench_block::micro_bench_qcow_l2_cache_miss, + control: PerformanceTestControl { + test_timeout: 10, + test_iterations: 20, + warmup_iterations: 5, + num_ops: Some(128), + ..PerformanceTestControl::default() + }, + unit_adjuster: adjuster::s_to_us, + }, + PerformanceTest { + name: "micro_block_qcow_l2_cache_miss_256_us", + func_ptr: micro_bench_block::micro_bench_qcow_l2_cache_miss, + control: PerformanceTestControl { + test_timeout: 10, + test_iterations: 20, + warmup_iterations: 5, + num_ops: Some(256), + ..PerformanceTestControl::default() + }, + unit_adjuster: adjuster::s_to_us, + }, ]; fn run_test_with_timeout( diff --git a/performance-metrics/src/micro_bench_block.rs b/performance-metrics/src/micro_bench_block.rs index 1b9ff3b8e..2b83ab495 100644 --- a/performance-metrics/src/micro_bench_block.rs +++ b/performance-metrics/src/micro_bench_block.rs @@ -16,8 +16,8 @@ use block::raw_async_aio::RawFileAsyncAio; use crate::PerformanceTestControl; use crate::util::{ - self, BLOCK_SIZE, QCOW_CLUSTER_SIZE, deterministic_permutation, drain_completions, read_iovec, - submit_reads, submit_writes, write_iovec, + self, BLOCK_SIZE, L2_ENTRIES_PER_TABLE, QCOW_CLUSTER_SIZE, deterministic_permutation, + drain_completions, read_iovec, submit_reads, submit_writes, write_iovec, }; /// Submit num_ops AIO writes, wait for them all to land, then time @@ -301,3 +301,30 @@ pub fn micro_bench_qcow_multi_cluster_read(control: &PerformanceTestControl) -> elapsed } + +/// Read one cluster from each of num_ops distinct L2 tables in a +/// sparsely allocated QCOW2 image. +/// +/// The clusters are spaced L2_ENTRIES_PER_TABLE apart so every read +/// touches a different L2 table. With num_ops exceeding the L2 cache +/// capacity (100 entries), this forces eviction on nearly every read +/// and measures the cold L2 cache miss overhead. +/// +/// Returns the total read wall clock time in seconds. +pub fn micro_bench_qcow_l2_cache_miss(control: &PerformanceTestControl) -> f64 { + let num_ops = control.num_ops.expect("num_ops required") as usize; + let (_tmp, disk) = util::sparse_qcow_tempfile(num_ops); + let mut async_io = disk.new_async_io(1).expect("new_async_io failed"); + + let mut buf = vec![0u8; QCOW_CLUSTER_SIZE as usize]; + let iovec = read_iovec(&mut buf); + + let stride = L2_ENTRIES_PER_TABLE as u64 * QCOW_CLUSTER_SIZE; + let start = Instant::now(); + submit_reads(async_io.as_mut(), num_ops, stride, &[iovec]); + let elapsed = start.elapsed().as_secs_f64(); + + drain_completions(async_io.as_mut(), num_ops); + + elapsed +}