virtio-devices: net: handle corrupted requests with NEEDS_RESET

A buggy or malicious guest may write an inappropriate value into
virtqueue's next_avail field. This will result in an error
when iterating over the queue:

863837ef86/virtio-queue/src/queue.rs (L708)

but this error is (logged and) ignored if pop_descriptor_chain()
is used:

863837ef86/virtio-queue/src/queue.rs (L583)

A reasonable approach, implemented here, is to mark the device as
NEEDS_RESET and ignore further queue events until the guest
reinitializes the device.

How this patch was tested:

Linux kernel was patched to trigger a bad next_avail when the
virtqueue queue counter reaches 5000:

--------------- START OF LINUX KERNEL PATCH ----------
$ git diff
diff --git a/drivers/virtio/virtio_ring.c b/drivers/virtio/virtio_ring.c
index b784aab668670..989f2a0c64a77 100644
--- a/drivers/virtio/virtio_ring.c
+++ b/drivers/virtio/virtio_ring.c
@@ -15,6 +15,9 @@
 #include <linux/spinlock.h>
 #include <xen/xen.h>

+
+void virtqueue_kick_always(struct virtqueue *vq);
+
 #ifdef DEBUG
 /* For development, we want to crash whenever the ring is screwed. */
 #define BAD_RING(_vq, fmt, args...)                            \
@@ -677,6 +680,12 @@ static inline int virtqueue_add_split(
                   struct virtqueue *_vq,
         * new available array entries. */
        virtio_wmb(vq->weak_barriers);
        vq->split.avail_idx_shadow++;
+       {
+        if ((vq->split.avail_idx_shadow % 100) == 0)
+            printk(KERN_ERR "avail idx: %d",
+                  (int)vq->split.avail_idx_shadow);
+               if (vq->split.avail_idx_shadow == 5000)
+               vq->split.avail_idx_shadow = 0;
+       }
        vq->split.vring.avail->idx = cpu_to_virtio16(_vq->vdev,
                                      vq->split.avail_idx_shadow);
        vq->num_added++;
@@ -689,6 +698,11 @@ static inline int virtqueue_add_split(
                  struct virtqueue *_vq,
        if (unlikely(vq->num_added == (1 << 16) - 1))
                virtqueue_kick(_vq);

+       {
+               if (unlikely(vq->split.avail_idx_shadow == 0))
+                       virtqueue_kick_always(_vq);
+       }
+
        return 0;

 unmap_release:
@@ -2515,6 +2529,11 @@ bool virtqueue_kick(struct virtqueue *vq)
 }
 EXPORT_SYMBOL_GPL(virtqueue_kick);

+void virtqueue_kick_always(struct virtqueue *vq)
+{
+       virtqueue_kick_prepare(vq);
+       virtqueue_notify(vq);
+}
 /**
  * virtqueue_get_buf_ctx - get the next used buffer
  * @_vq: the struct virtqueue we're talking about.
--------------- END OF LINUX KERNEL PATCH ----------

Then the kernel was booted, and the host pinged until the
nic became unresponsive:

ping -i 0.002 192.168.4.1

Device status was confirmed using

cat /sys/class/net/eth0/device/status

(it was 0x4f).

Then the device was re-initialized:

DEV_NAME=$(basename $(readlink -f /sys/class/net/eth0/device))
echo $DEV_NAME | tee /sys/bus/virtio/drivers/virtio_net/unbind
echo $DEV_NAME | tee /sys/bus/virtio/drivers/virtio_net/bind
ip link set eth0 up

At this point networking became healthly again.

Signed-off-by: Peter Oskolkov <posk@google.com>
This commit is contained in:
Peter Oskolkov
2026-03-12 09:56:49 -07:00
committed by Bo Chen
parent b5053ae4de
commit 563303b50a
3 changed files with 70 additions and 14 deletions

View File

@@ -51,7 +51,13 @@ impl TxVirtio {
let mut retry_write = false;
let mut rate_limit_reached = false;
while let Some(mut desc_chain) = queue.pop_descriptor_chain(mem) {
loop {
let mut iter = queue
.iter(mem)
.map_err(NetQueuePairError::QueueIteratorFailed)?;
let Some(mut desc_chain) = iter.next() else {
break;
};
if rate_limit_reached {
queue.go_to_previous_position();
break;
@@ -180,7 +186,13 @@ impl RxVirtio {
let mut exhausted_descs = true;
let mut rate_limit_reached = false;
while let Some(mut desc_chain) = queue.pop_descriptor_chain(mem) {
loop {
let mut iter = queue
.iter(mem)
.map_err(NetQueuePairError::QueueIteratorFailed)?;
let Some(mut desc_chain) = iter.next() else {
break;
};
if rate_limit_reached {
exhausted_descs = false;
queue.go_to_previous_position();

View File

@@ -66,6 +66,7 @@ const DEVICE_ACKNOWLEDGE: u32 = 0x01;
const DEVICE_DRIVER: u32 = 0x02;
const DEVICE_DRIVER_OK: u32 = 0x04;
const DEVICE_FEATURES_OK: u32 = 0x08;
const DEVICE_NEEDS_RESET: u32 = 0x40;
const DEVICE_FAILED: u32 = 0x80;
const VIRTIO_F_RING_INDIRECT_DESC: u32 = 28;

View File

@@ -16,7 +16,7 @@ use std::{result, thread};
use anyhow::anyhow;
use event_monitor::event;
use log::{debug, error, info};
use log::{debug, error, info, warn};
#[cfg(not(fuzzing))]
use net_util::virtio_features_to_tap_offload;
use net_util::{
@@ -179,7 +179,6 @@ struct NetEpollHandler {
// a restore as the vCPU thread isn't ready to handle the interrupt. This causes
// issues when combined with VIRTIO_RING_F_EVENT_IDX interrupt suppression.
driver_awake: bool,
#[allow(unused)]
device_status: Arc<AtomicU8>,
}
@@ -194,6 +193,9 @@ impl NetEpollHandler {
}
fn handle_rx_event(&mut self) -> result::Result<(), DeviceError> {
if self.needs_reset() {
return Ok(());
}
let queue_evt = &self.queue_evt_pair.0;
if let Err(e) = queue_evt.read() {
error!("Failed to get rx queue event: {e:?}");
@@ -222,13 +224,43 @@ impl NetEpollHandler {
Ok(())
}
fn handle_queue_iterator_error(&mut self, err: &virtio_queue::Error) {
// The guest submitted a corrupted VirtQ request, and the error
// was logged during queue processing. We cannot just ignore the
// error, as the guest could continue spamming the VMM with bad
// requests, triggering excessive error logging. So we mark
// the device "NEEDS_RESET", effectively stopping all request
// processing (see self.needs_reset() usage) until the guest
// resets and reactivates the device.
warn!(
"Corrupted request detected (virtqueue error: {err:?}). \
Setting device status to 'NEEDS_RESET' and stopping processing queues until reset."
);
self.device_status
.fetch_or(crate::DEVICE_NEEDS_RESET as u8, Ordering::SeqCst);
// Let the guest know that the device status has changed.
if let Err(e) = self.interrupt_cb.trigger(VirtioInterruptType::Config) {
error!("Failed to signal config interrupt: {e:?}");
}
}
fn process_tx(&mut self) -> result::Result<(), DeviceError> {
if self
if self.needs_reset() {
return Ok(());
}
let res = self
.net
.process_tx(&self.mem.memory(), &mut self.queue_pair.1)
.map_err(DeviceError::NetQueuePair)?
|| !self.driver_awake
{
.process_tx(&self.mem.memory(), &mut self.queue_pair.1);
if let Err(net_util::NetQueuePairError::QueueIteratorFailed(err)) = res {
self.handle_queue_iterator_error(&err);
return Ok(());
}
if res.map_err(DeviceError::NetQueuePair)? || !self.driver_awake {
self.signal_used_queue(self.queue_index_base + 1)?;
debug!("Signalling TX queue");
} else {
@@ -252,12 +284,19 @@ impl NetEpollHandler {
}
fn handle_rx_tap_event(&mut self) -> result::Result<(), DeviceError> {
if self
if self.needs_reset() {
return Ok(());
}
let res = self
.net
.process_rx(&self.mem.memory(), &mut self.queue_pair.0)
.map_err(DeviceError::NetQueuePair)?
|| !self.driver_awake
{
.process_rx(&self.mem.memory(), &mut self.queue_pair.0);
if let Err(net_util::NetQueuePairError::QueueIteratorFailed(err)) = res {
self.handle_queue_iterator_error(&err);
return Ok(());
}
if res.map_err(DeviceError::NetQueuePair)? || !self.driver_awake {
self.signal_used_queue(self.queue_index_base)?;
debug!("Signalling RX queue");
} else {
@@ -307,6 +346,10 @@ impl NetEpollHandler {
Ok(())
}
fn needs_reset(&self) -> bool {
(self.device_status.load(Ordering::Acquire) & crate::DEVICE_NEEDS_RESET as u8) != 0
}
}
impl EpollHelperHandler for NetEpollHandler {