* [PATCH v9 1/5] vfio: selftests: igb: Add driver for Intel 82576 device
2026-07-30 23:31 [PATCH v9 0/5] vfio: selftests: Add driver for Intel Ethernet Gigabit Controller (IGB) Josh Hilke
@ 2026-07-30 23:31 ` Josh Hilke
2026-07-30 23:46 ` sashiko-bot
2026-07-30 23:31 ` [PATCH v9 2/5] vfio: selftests: Add helpers to re-enable interrupts Josh Hilke
` (3 subsequent siblings)
4 siblings, 1 reply; 7+ messages in thread
From: Josh Hilke @ 2026-07-30 23:31 UTC (permalink / raw)
To: David Matlack, Alex Williamson
Cc: Shuah Khan, linux-kernel, kvm, linux-kselftest, Vipin Sharma,
Josh Hilke, Alex Williamson
Add a VFIO selftest driver for the Intel Gigabit Ethernet controller
(IGB), specifically targeting the 82576 device. IGB is fully virtualized
in QEMU which makes it easy to run VFIO selftests without needing any
specific hardware.
All VFIO selftest drivers must implement DMA/memcpy operations, but IGB
doesn't have a native memcpy feature, so the loopback feature (described
in section 3.5.6.3 of IGB specification) is used to implement it. To
support testing on both QEMU and physical hardware, the driver uses PHY
internal loopback with some QEMU-specific fallbacks. The driver also
supports MSI-X routing and interrupt management, and disables PCIe
completion timeout retries to ensure clean recovery during invalid-DMA
tests.
Users can verify the driver works in QEMU by building the kernel,
building VFIO selftests, and then running the vfio_pci_driver_test using
this command:
vng \
--run arch/x86/boot/bzImage \
--user root \
--disable-microvm \
--memory 32G \
--cpus 8 \
--qemu-opts="-M q35,accel=kvm,kernel-irqchip=split" \
--qemu-opts="-device intel-iommu,intremap=on,caching-mode=on,device-iotlb=on" \
--qemu-opts="-netdev user,id=net0 -device igb,netdev=net0,addr=09.0" \
--append "console=ttyS0 earlyprintk=ttyS0 intel_iommu=on iommu=pt" \
--exec "modprobe vfio-pci && \
./tools/testing/selftests/vfio/scripts/setup.sh 0000:00:09.0 && \
./tools/testing/selftests/vfio/scripts/run.sh ./tools/testing/selftests/vfio/vfio_pci_driver_test"
Assisted-by: Claude:claude-opus-4-7
Assisted-by: Gemini:gemini-3.1-pro-preview
Co-developed-by: Alex Williamson <alex.williamson@nvidia.com>
Signed-off-by: Alex Williamson <alex.williamson@nvidia.com>
Signed-off-by: Josh Hilke <jrhilke@google.com>
---
.../selftests/vfio/lib/drivers/igb/e1000_82575.h | 1 +
.../selftests/vfio/lib/drivers/igb/e1000_defines.h | 1 +
.../selftests/vfio/lib/drivers/igb/e1000_regs.h | 1 +
tools/testing/selftests/vfio/lib/drivers/igb/igb.c | 520 +++++++++++++++++++++
tools/testing/selftests/vfio/lib/libvfio.mk | 1 +
tools/testing/selftests/vfio/lib/vfio_pci_driver.c | 3 +-
6 files changed, 526 insertions(+), 1 deletion(-)
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h
new file mode 120000
index 000000000000..b84affdec559
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_82575.h
\ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h
new file mode 120000
index 000000000000..9f97f4330086
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_defines.h
\ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h
new file mode 120000
index 000000000000..c733634171bb
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_regs.h
\ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/igb.c b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
new file mode 100644
index 000000000000..2f7e5cb26271
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
@@ -0,0 +1,520 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include <unistd.h>
+#include <errno.h>
+#include <stdint.h>
+#include <linux/io.h>
+#include <linux/pci_regs.h>
+#include <linux/pci_ids.h>
+#include <linux/kernel.h>
+#include <linux/compiler.h>
+#include <asm/barrier.h>
+#include <linux/mii.h>
+#include <libvfio/vfio_pci_device.h>
+
+#include "e1000_regs.h"
+#include "e1000_defines.h"
+#include "e1000_82575.h"
+
+#define PCI_DEVICE_ID_INTEL_82576 0x10C9
+#define IGB_MAX_CHUNK_SIZE 1024
+#define MSIX_VECTOR 0
+#define MSIX_VECTOR_MASK (1 << MSIX_VECTOR)
+#define RING_SIZE 4096 /* Number of descriptors in ring */
+
+struct igb_tx_desc {
+ union {
+ struct {
+ u64 buffer_addr; /* Address of descriptor's data buffer */
+ u32 cmd_type_len; /* Command/Type/Length */
+ u32 olinfo_status; /* Context/Buffer info */
+ } read;
+
+ struct {
+ u64 rsvd; /* Reserved */
+ u32 nxtseq_seed; /* Next sequence seed */
+ u32 status; /* Descriptor status */
+ } wb;
+ };
+};
+
+struct igb_rx_desc {
+ union {
+ struct {
+ u64 pkt_addr; /* Packet buffer address */
+ u64 hdr_addr; /* Header buffer address */
+ } read;
+ struct {
+ u16 pkt_info; /* RSS type, Packet type */
+ u16 hdr_info; /* Split Head, buf len */
+ u32 rss; /* RSS Hash */
+ u32 status_error; /* ext status/error */
+ u16 length; /* Packet length */
+ u16 vlan; /* VLAN tag */
+ } wb; /* writeback */
+ };
+};
+
+struct igb {
+ void *bar0;
+ u32 tx_tail;
+ u32 rx_tail;
+ struct igb_tx_desc tx_ring[RING_SIZE] __attribute__((aligned(128)));
+ struct igb_rx_desc rx_ring[RING_SIZE] __attribute__((aligned(128)));
+};
+
+static inline struct igb *to_igb_state(struct vfio_pci_device *device)
+{
+ return (struct igb *)device->driver.region.vaddr;
+}
+
+static inline void igb_write32(struct igb *igb, u32 reg, u32 val)
+{
+ writel(val, igb->bar0 + reg);
+}
+
+static inline u32 igb_read32(struct igb *igb, u32 reg)
+{
+ return readl(igb->bar0 + reg);
+}
+
+static int igb_write_phy(struct igb *igb, u32 offset, u16 data)
+{
+ u32 mdic;
+ int i;
+
+ /*
+ * Write a PHY register over MDIO.
+ *
+ * A production driver would hold the SW/FW semaphore (SWSM.SWESMBI + the
+ * SW_FW_SYNC PHY bit) across the MDIO transaction to serialize against the
+ * device's management firmware. The selftest owns the assigned function
+ * exclusively on a dedicated test device with no active manageability
+ * contending for the PHY, so the sync is omitted; it should be added here
+ * if this ever needs to run on a manageability-enabled NIC.
+ */
+ mdic = (((u32)data) |
+ (offset << E1000_MDIC_REG_SHIFT) |
+ (1 << E1000_MDIC_PHY_SHIFT) |
+ E1000_MDIC_OP_WRITE);
+
+ igb_write32(igb, E1000_MDIC, mdic);
+
+ for (i = 0; i < 1000; i++) {
+ usleep(50);
+ mdic = igb_read32(igb, E1000_MDIC);
+ if (mdic & E1000_MDIC_READY)
+ break;
+ }
+
+ if (!(mdic & E1000_MDIC_READY))
+ return -1;
+
+ if (mdic & E1000_MDIC_ERROR)
+ return -1;
+
+ return 0;
+}
+
+/*
+ * Configure the device for PHY internal loopback per 82576 datasheet
+ * section 3.5.6.3.1. Force the PHY to 1Gb/s full duplex with loopback
+ * enabled, then force the MAC link state to match. Internal loopback
+ * wraps data at the end of the PHY datapath (section 3.5.6.3), so the
+ * physical link state is irrelevant.
+ *
+ * Section 3.5.6.1 directs to "Use PHY Loopback instead of MAC Loopback
+ * on the 82576", and section 3.5.6.2 states "MAC Loopback is not used
+ * on this device." RCTL.LBM_MAC is still set elsewhere as a QEMU-only
+ * accommodation; see the RCTL programming in the caller for the
+ * rationale.
+ */
+static void igb_setup_loopback(struct igb *igb)
+{
+ u32 ctrl;
+ int ret;
+
+ /*
+ * Kick the autoneg machinery solely to bring STATUS.LU up under
+ * QEMU's igb emulation: QEMU only updates STATUS.LU via its
+ * autoneg-done timer, and without LU set its receive path
+ * (e1000x_hw_rx_enabled) drops every loopback frame. On real
+ * hardware autoneg cannot complete before the next PHY write
+ * below clears the autoneg-enable bit, so this is effectively a
+ * no-op there.
+ */
+ (void)igb_write_phy(igb, MII_BMCR,
+ BMCR_ANENABLE | BMCR_ANRESTART);
+
+ /* PHY control: loopback + 1Gb/s full duplex, autoneg disabled. */
+ ret = igb_write_phy(igb, MII_BMCR,
+ BMCR_LOOPBACK |
+ BMCR_SPEED1000 |
+ BMCR_FULLDPLX);
+ VFIO_ASSERT_EQ(ret, 0, "Failed to write PHY control register");
+
+ /*
+ * Brief delay before forcing the MAC, mirroring the kernel ethtool
+ * selftest in igb_integrated_phy_loopback(). Not specified by the
+ * datasheet, but empirically required by the kernel driver.
+ */
+ usleep(50000);
+
+ /*
+ * Force the MAC to 1Gb/s full duplex with link up. Without forcing
+ * the link state the descriptor engine does not run, since the chip
+ * normally waits for a real negotiated link.
+ */
+ ctrl = igb_read32(igb, E1000_CTRL);
+ ctrl &= ~E1000_CTRL_SPD_SEL;
+ ctrl |= E1000_CTRL_FRCSPD |
+ E1000_CTRL_FRCDPX |
+ E1000_CTRL_SPD_1000 |
+ E1000_CTRL_FD |
+ E1000_CTRL_SLU;
+ igb_write32(igb, E1000_CTRL, ctrl);
+
+ /*
+ * Settling delay matching the kernel ethtool selftest's msleep(500)
+ * at the tail of igb_integrated_phy_loopback(). Not specified by
+ * the datasheet; empirical, and inherited from the kernel driver.
+ */
+ usleep(500000);
+}
+
+static int igb_probe(struct vfio_pci_device *device)
+{
+ if (!vfio_pci_device_match(device, PCI_VENDOR_ID_INTEL, PCI_DEVICE_ID_INTEL_82576))
+ return -EINVAL;
+
+ return 0;
+}
+
+static void igb_reset(struct igb *igb)
+{
+ int retries = 20;
+
+ igb_write32(igb, E1000_CTRL, igb_read32(igb, E1000_CTRL) | E1000_CTRL_RST);
+ /*
+ * Must wait at least 1 millisecond after setting the reset bit before
+ * checking if this device is ready to be used (82576 datasheet section
+ * 4.2.1.6.1). The delay also ensures the reset has taken effect and
+ * cleared EECD.AUTO_RD before it is polled below.
+ */
+ usleep(1000);
+
+ /*
+ * Poll NVM Auto Read Done rather than CTRL.RST, matching
+ * igb_get_auto_rd_done() in the igb driver: AUTO_RD implies both that
+ * the reset completed and that the device finished re-reading its
+ * configuration from NVM, which is what actually makes it usable.
+ */
+ while (retries-- > 0 && !(igb_read32(igb, E1000_EECD) & E1000_EECD_AUTO_RD))
+ usleep(1000);
+
+ /*
+ * QEMU's igb emulation does not set E1000_EECD_AUTO_RD. If we timed out,
+ * check if CTRL.RST is cleared, which is what QEMU uses to signal reset
+ * completion.
+ */
+ if (retries < 0) {
+ VFIO_ASSERT_EQ(igb_read32(igb, E1000_CTRL) & E1000_CTRL_RST, 0,
+ "Device reset did not complete (CTRL.RST not cleared)");
+ }
+
+ igb_write32(igb, E1000_IMC, 0xFFFFFFFF);
+}
+
+static void igb_init(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+ u64 iova_tx, iova_rx;
+ u32 ctrl, rctl;
+ u16 cmd_reg;
+ int retries;
+
+ VFIO_ASSERT_GE(device->driver.region.size, sizeof(struct igb));
+
+ /* Set up rings and calculate IOVAs */
+ igb->bar0 = device->bars[0].vaddr;
+
+ iova_tx = to_iova(device, igb->tx_ring);
+ iova_rx = to_iova(device, igb->rx_ring);
+
+ igb_reset(igb);
+
+ /* Signal that the driver is loaded */
+ ctrl = igb_read32(igb, E1000_CTRL_EXT);
+ ctrl |= E1000_CTRL_EXT_DRV_LOAD;
+ ctrl &= ~E1000_CTRL_EXT_LINK_MODE_MASK;
+ igb_write32(igb, E1000_CTRL_EXT, ctrl);
+
+ /* Enable PCI Bus Master. */
+ cmd_reg = vfio_pci_config_readw(device, PCI_COMMAND);
+ if ((cmd_reg & (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) !=
+ (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) {
+ cmd_reg |= (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY);
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd_reg);
+ }
+
+ /* Configure PHY internal loopback for testing. */
+ igb_setup_loopback(igb);
+
+ /*
+ * Disable DMA re-send on PCIe completion timeout (82576 datasheet
+ * section 8.6.1, GCR.Completion_Timeout_Resend, bit 16). The
+ * mix_and_match test intentionally submits descriptors targeting
+ * unmapped IOVAs; with the default (set) value, the device keeps
+ * retrying the failed read indefinitely, which keeps PCIe AER and
+ * IOMMU error handling busy and interferes with reset recovery.
+ */
+ ctrl = igb_read32(igb, E1000_GCR);
+ ctrl &= ~E1000_GCR_CMPL_TMOUT_RESEND;
+ igb_write32(igb, E1000_GCR, ctrl);
+
+ /* Configure TX and RX descriptor rings */
+ igb_write32(igb, E1000_TDBAL(0), (u32)iova_tx);
+ igb_write32(igb, E1000_TDBAH(0), (u32)(iova_tx >> 32));
+ igb_write32(igb, E1000_TDLEN(0), RING_SIZE * sizeof(struct igb_tx_desc));
+ igb_write32(igb, E1000_TDH(0), 0);
+ igb_write32(igb, E1000_TDT(0), 0);
+ igb_write32(igb, E1000_TXDCTL(0), E1000_TXDCTL_QUEUE_ENABLE);
+
+ igb_write32(igb, E1000_RDBAL(0), (u32)iova_rx);
+ igb_write32(igb, E1000_RDBAH(0), (u32)(iova_rx >> 32));
+ igb_write32(igb, E1000_RDLEN(0), RING_SIZE * sizeof(struct igb_rx_desc));
+ igb_write32(igb, E1000_RDH(0), 0);
+ igb_write32(igb, E1000_RDT(0), 0);
+
+ /*
+ * Select the advanced one-buffer descriptor format. Per 82576
+ * datasheet section 7.1.5.2: "SRRCTL[n].DESCTYPE must be set to a
+ * value other than 000b for the 82576 to write back the special
+ * descriptors." struct igb_rx_desc matches the advanced one-buffer
+ * writeback layout (section 7.1.5.2), so polling rx.wb.status_error
+ * requires this format. Section 8.10.2 specifies DESCTYPE[27:25].
+ *
+ * The direct write also zeroes SRRCTL.BSIZEPACKET, which is
+ * intentional: per section 7.1.3.1 a zero BSIZEPACKET falls back to
+ * the RCTL.BSIZE buffer size, whose reset default (00b) is 2048
+ * bytes -- ample for the loopback frames here.
+ */
+ igb_write32(igb, E1000_SRRCTL(0), E1000_SRRCTL_DESCTYPE_ADV_ONEBUF);
+
+ igb_write32(igb, E1000_RXDCTL(0), E1000_RXDCTL_QUEUE_ENABLE);
+
+ /* Wait for TX and RX queues to be enabled */
+ retries = 2000;
+ while (retries-- > 0) {
+ if ((igb_read32(igb, E1000_TXDCTL(0)) & E1000_TXDCTL_QUEUE_ENABLE) &&
+ (igb_read32(igb, E1000_RXDCTL(0)) & E1000_RXDCTL_QUEUE_ENABLE))
+ break;
+ usleep(10);
+ }
+ VFIO_ASSERT_GE(retries, 0);
+
+ /*
+ * Enable Receiver and Transmitter. RCTL.LBM_MAC is set in addition
+ * to PHY loopback as a QEMU-only accommodation: QEMU's emulated igb
+ * does not honor PHY register 0 bit 14 (PHY internal loopback) and
+ * relies on RCTL.LBM_MAC to wrap TX descriptors back to the RX
+ * queue. Datasheet 8.10.1 (RCTL register) advises "When using the
+ * internal PHY, LBM should remain set to 00b", so setting LBM_MAC
+ * here deviates from datasheet guidance; empirically the bit has
+ * no observable effect on real 82576 hardware because MAC loopback
+ * is not implemented (datasheet 3.5.6.2). Setting both lets the
+ * selftest work on both real hardware and QEMU without conditional
+ * code paths.
+ */
+ rctl = E1000_RCTL_EN | /* Receiver Enable */
+ E1000_RCTL_UPE | /* Unicast Promiscuous (for dummy MAC) */
+ E1000_RCTL_MPE | /* Multicast Promiscuous */
+ E1000_RCTL_BAM | /* Broadcast Accept Mode */
+ E1000_RCTL_LBM_MAC | /* MAC Loopback - for QEMU emulation only */
+ E1000_RCTL_SECRC; /* Strip CRC (needed for memcmp) */
+ igb_write32(igb, E1000_RCTL, rctl);
+ igb_write32(igb, E1000_TCTL, E1000_TCTL_EN | E1000_TCTL_PSP);
+
+ /* Enable MSI-X with 1 vector for the test */
+ vfio_pci_msix_enable(device, MSIX_VECTOR, 1);
+
+ /*
+ * Program MSI-X interrupt routing per 82576 datasheet:
+ *
+ * GPIE (section 7.3.2.11, Table 7-47): set Multiple_MSIX (bit 4) to
+ * route interrupt causes through IVAR mapping, and EIAME (bit 30)
+ * to apply EIAM on MSI-X assertion (without EIAME, EIAM only
+ * applies on EICR read/write).
+ *
+ * EIAC (section 8.8.5): enable auto-clear of EICR for vector 0.
+ * Without auto-clear the cause stays set after delivery and the
+ * test can see spurious interrupts on the next memcpy batch.
+ *
+ * EIAM (section 8.8.6): enable auto-mask of EIMS for vector 0 on
+ * MSI-X assertion (effective because EIAME is set).
+ *
+ * IVAR (section 7.3.1.2, register definition in 8.8.13): map RX
+ * cause 0 to MSI-X vector 0 and mark the entry valid.
+ */
+ igb_write32(igb, E1000_GPIE, E1000_GPIE_MSIX_MODE | E1000_GPIE_EIAME);
+ igb_write32(igb, E1000_EIAC, MSIX_VECTOR_MASK);
+ igb_write32(igb, E1000_EIAM, MSIX_VECTOR_MASK);
+
+ /* Map vector 0 to interrupt cause 0 and mark it valid */
+ igb_write32(igb, E1000_IVAR0, E1000_IVAR_VALID);
+
+ /* Enable interrupts on vector 0 */
+ igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK);
+
+ /* Initialize driver state and capability limits */
+ igb->tx_tail = 0;
+ igb->rx_tail = 0;
+
+ device->driver.max_memcpy_size = IGB_MAX_CHUNK_SIZE;
+ device->driver.max_memcpy_count = RING_SIZE - 1;
+ device->driver.msi = MSIX_VECTOR;
+}
+
+static void igb_remove(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ igb_write32(igb, E1000_RCTL, 0);
+ igb_write32(igb, E1000_TCTL, 0);
+ igb_reset(igb);
+
+ vfio_pci_msix_disable(device);
+}
+
+static void igb_irq_disable(struct igb *igb)
+{
+ igb_write32(igb, E1000_EIMC, MSIX_VECTOR_MASK);
+}
+
+static void igb_irq_enable(struct igb *igb)
+{
+ igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK);
+}
+
+static void igb_irq_clear(struct igb *igb)
+{
+ /*
+ * Use write-to-clear (datasheet 7.3.4.2). In MSI-X mode with EIAC
+ * programmed, section 8.8.5 explicitly states "If any bits are set
+ * in EIAC, the EICR register should not be read", which rules out
+ * the read-to-clear path in 7.3.4.3. Bits not in EIAC are still
+ * cleared by writing 1.
+ */
+ igb_write32(igb, E1000_EICR, 0xFFFFFFFF);
+}
+
+static void igb_memcpy_start(struct vfio_pci_device *device, iova_t src,
+ iova_t dst, u64 size, u64 count)
+{
+ struct igb *igb = to_igb_state(device);
+ struct igb_rx_desc *rx;
+ struct igb_tx_desc *tx;
+ u32 i;
+
+ igb_irq_disable(igb);
+
+ for (i = 0; i < count; i++) {
+ tx = &igb->tx_ring[igb->tx_tail];
+ rx = &igb->rx_ring[igb->rx_tail];
+
+ memset(tx, 0, sizeof(struct igb_tx_desc));
+ memset(rx, 0, sizeof(struct igb_rx_desc));
+
+ rx->read.pkt_addr = cpu_to_le64(dst);
+ rx->read.hdr_addr = cpu_to_le64(0);
+
+ tx->read.buffer_addr = cpu_to_le64(src);
+ /*
+ * Build an advanced data descriptor per 82576 datasheet
+ * section 7.2.2.3. DEXT marks the descriptor as advanced
+ * (required by hardware); DTYP=data selects the data
+ * descriptor; IFCS asks the MAC to append the Ethernet
+ * FCS (without it the frame is dropped as malformed);
+ * EOP marks end of packet. DTALEN is the buffer length
+ * in bits 15:0 of cmd_type_len.
+ */
+ tx->read.cmd_type_len = cpu_to_le32((uint32_t)size |
+ E1000_ADVTXD_DTYP_DATA |
+ E1000_ADVTXD_DCMD_DEXT |
+ E1000_ADVTXD_DCMD_IFCS |
+ E1000_ADVTXD_DCMD_EOP);
+ /*
+ * PAYLEN (section 7.2.2.3.11) is the total payload size
+ * in olinfo_status[31:14].
+ */
+ tx->read.olinfo_status =
+ cpu_to_le32((uint32_t)size << E1000_ADVTXD_PAYLEN_SHIFT);
+
+ igb->tx_tail = (igb->tx_tail + 1) % RING_SIZE;
+ igb->rx_tail = (igb->rx_tail + 1) % RING_SIZE;
+ }
+
+ igb_write32(igb, E1000_RDT(0), igb->rx_tail);
+ igb_write32(igb, E1000_TDT(0), igb->tx_tail);
+}
+
+static int igb_memcpy_wait(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+ struct igb_rx_desc *rx;
+ u32 status = 0;
+ u32 prev_tail;
+ int retries;
+
+ prev_tail = (igb->rx_tail + RING_SIZE - 1) % RING_SIZE;
+ rx = &igb->rx_ring[prev_tail];
+
+ /*
+ * Real 82576 hardware processes the descriptor ring at line rate.
+ * max_memcpy_size = (RING_SIZE - 1) * IGB_MAX_CHUNK_SIZE ~= 4 MB,
+ * split into 4095 1 KB frames. At 1 Gb/s (~125 MB/s) the worst
+ * valid memcpy takes ~32 ms on the wire, plus per-frame preamble,
+ * SFD, IFG and FCS overhead (~3%) and descriptor fetch/writeback
+ * latency. Wait up to ~200 ms before declaring the device hung;
+ * ~6x the line-rate floor leaves comfortable headroom for host
+ * scheduling jitter while keeping the intentional invalid-DMA
+ * tests bounded.
+ */
+ retries = 200;
+ while (retries-- > 0) {
+ status = le32_to_cpu(READ_ONCE(rx->wb.status_error));
+ if (status & 1)
+ break;
+ usleep(1000);
+ }
+
+ if (status & 1)
+ /*
+ * Ensure the test code doesn't speculatively read the DMA
+ * destination buffer before we have verified that the
+ * descriptor writeback is complete.
+ */
+ rmb();
+
+ igb_irq_clear(igb);
+
+ igb_irq_enable(igb);
+
+ return (status & 1) ? 0 : -ETIMEDOUT;
+}
+
+static void igb_send_msi(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ igb_write32(igb, E1000_EICS, MSIX_VECTOR_MASK);
+}
+
+const struct vfio_pci_driver_ops igb_ops = {
+ .name = "igb",
+ .probe = igb_probe,
+ .init = igb_init,
+ .remove = igb_remove,
+ .memcpy_start = igb_memcpy_start,
+ .memcpy_wait = igb_memcpy_wait,
+ .send_msi = igb_send_msi,
+};
diff --git a/tools/testing/selftests/vfio/lib/libvfio.mk b/tools/testing/selftests/vfio/lib/libvfio.mk
index 82132b648744..bcfa74ae040e 100644
--- a/tools/testing/selftests/vfio/lib/libvfio.mk
+++ b/tools/testing/selftests/vfio/lib/libvfio.mk
@@ -16,6 +16,7 @@ LIBVFIO_C += drivers/dsa/dsa.c
endif
LIBVFIO_C += drivers/nv_falcon/nv_falcon.c
+LIBVFIO_C += drivers/igb/igb.c
LIBVFIO_OUTPUT := $(OUTPUT)/libvfio
diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
index 153bf4a7a19f..5e65434d2318 100644
--- a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
+++ b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
@@ -6,8 +6,8 @@
extern struct vfio_pci_driver_ops dsa_ops;
extern struct vfio_pci_driver_ops ioat_ops;
#endif
-
extern struct vfio_pci_driver_ops nv_falcon_ops;
+extern struct vfio_pci_driver_ops igb_ops;
static struct vfio_pci_driver_ops *driver_ops[] = {
#ifdef __x86_64__
@@ -15,6 +15,7 @@ static struct vfio_pci_driver_ops *driver_ops[] = {
&ioat_ops,
#endif
&nv_falcon_ops,
+ &igb_ops,
};
void vfio_pci_driver_probe(struct vfio_pci_device *device)
--
2.55.0.508.g3f0d502094-goog
^ permalink raw reply related [flat|nested] 7+ messages in thread* [PATCH v9 2/5] vfio: selftests: Add helpers to re-enable interrupts
2026-07-30 23:31 [PATCH v9 0/5] vfio: selftests: Add driver for Intel Ethernet Gigabit Controller (IGB) Josh Hilke
2026-07-30 23:31 ` [PATCH v9 1/5] vfio: selftests: igb: Add driver for Intel 82576 device Josh Hilke
@ 2026-07-30 23:31 ` Josh Hilke
2026-07-30 23:31 ` [PATCH v9 3/5] vfio: selftests: igb: Factor hardware programming into igb_hw_init() Josh Hilke
` (2 subsequent siblings)
4 siblings, 0 replies; 7+ messages in thread
From: Josh Hilke @ 2026-07-30 23:31 UTC (permalink / raw)
To: David Matlack, Alex Williamson
Cc: Shuah Khan, linux-kernel, kvm, linux-kselftest, Vipin Sharma,
Josh Hilke, Alex Williamson
From: Alex Williamson <alex.williamson@nvidia.com>
Selftest drivers that recover from a fault by issuing VFIO_DEVICE_RESET
need to re-arm device interrupts afterwards. VFIO_DEVICE_RESET tears
down the kernel-side IRQ trigger so a subsequent VFIO_DEVICE_SET_IRQS
is required, but the user-side eventfds (and any fd cached in a test
fixture) are still valid and must be preserved.
vfio_pci_irq_enable() refuses to be called for vectors that already
have an eventfd (VFIO_ASSERT_LT), and vfio_pci_irq_disable() closes
all eventfds before resetting the trigger, so neither is suitable.
Add vfio_pci_irq_reenable(device, index, vector, count) which asserts
that the requested range has existing eventfds and re-issues
VFIO_DEVICE_SET_IRQS using them. Signature mirrors vfio_pci_irq_enable().
Add vfio_pci_msi{,x}_reenable() wrappers around vfio_pci_irq_reenable()
for additional ease of use and readability.
Assisted-by: Claude:claude-opus-4-7
Signed-off-by: Alex Williamson <alex.williamson@nvidia.com>
Reviewed-by: David Matlack <dmatlack@google.com>
---
.../vfio/lib/include/libvfio/vfio_pci_device.h | 14 ++++++++++++++
tools/testing/selftests/vfio/lib/vfio_pci_device.c | 22 ++++++++++++++++++++++
2 files changed, 36 insertions(+)
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
index 59acef381d05..89a039ab3075 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
@@ -84,6 +84,8 @@ static inline void vfio_pci_cmd_clear(struct vfio_pci_device *device, u16 bits)
void vfio_pci_irq_enable(struct vfio_pci_device *device, u32 index,
u32 vector, int count);
void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index);
+void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index,
+ u32 vector, int count);
void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector);
static inline void fcntl_set_nonblock(int fd)
@@ -108,6 +110,12 @@ static inline void vfio_pci_msi_disable(struct vfio_pci_device *device)
vfio_pci_irq_disable(device, VFIO_PCI_MSI_IRQ_INDEX);
}
+static inline void vfio_pci_msi_reenable(struct vfio_pci_device *device,
+ u32 vector, int count)
+{
+ vfio_pci_irq_reenable(device, VFIO_PCI_MSI_IRQ_INDEX, vector, count);
+}
+
static inline void vfio_pci_msix_enable(struct vfio_pci_device *device,
u32 vector, int count)
{
@@ -119,6 +127,12 @@ static inline void vfio_pci_msix_disable(struct vfio_pci_device *device)
vfio_pci_irq_disable(device, VFIO_PCI_MSIX_IRQ_INDEX);
}
+static inline void vfio_pci_msix_reenable(struct vfio_pci_device *device,
+ u32 vector, int count)
+{
+ vfio_pci_irq_reenable(device, VFIO_PCI_MSIX_IRQ_INDEX, vector, count);
+}
+
static inline int __to_iova(struct vfio_pci_device *device, void *vaddr, iova_t *iova)
{
return __iommu_hva2iova(device->iommu, vaddr, iova);
diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_device.c b/tools/testing/selftests/vfio/lib/vfio_pci_device.c
index 28868343f2ab..65a4fffb480c 100644
--- a/tools/testing/selftests/vfio/lib/vfio_pci_device.c
+++ b/tools/testing/selftests/vfio/lib/vfio_pci_device.c
@@ -105,6 +105,28 @@ void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index)
vfio_pci_irq_set(device, index, 0, 0, NULL);
}
+/*
+ * Re-issue VFIO_DEVICE_SET_IRQS for an already-enabled vector range using
+ * the existing eventfds. Intended for drivers that need to re-arm device
+ * interrupts after a VFIO_DEVICE_RESET, which tears down the kernel-side
+ * IRQ trigger but leaves user-side eventfds intact. Recreating the
+ * eventfds would invalidate any test-fixture cache of the fd, so this
+ * helper deliberately preserves them.
+ */
+void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index,
+ u32 vector, int count)
+{
+ int i;
+
+ check_supported_irq_index(index);
+
+ for (i = vector; i < vector + count; i++)
+ VFIO_ASSERT_GE(device->msi_eventfds[i], 0,
+ "vector %d eventfd not allocated\n", i);
+
+ vfio_pci_irq_set(device, index, vector, count, device->msi_eventfds + vector);
+}
+
static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index,
struct vfio_irq_info *irq_info)
{
--
2.55.0.508.g3f0d502094-goog
^ permalink raw reply related [flat|nested] 7+ messages in thread* [PATCH v9 3/5] vfio: selftests: igb: Factor hardware programming into igb_hw_init()
2026-07-30 23:31 [PATCH v9 0/5] vfio: selftests: Add driver for Intel Ethernet Gigabit Controller (IGB) Josh Hilke
2026-07-30 23:31 ` [PATCH v9 1/5] vfio: selftests: igb: Add driver for Intel 82576 device Josh Hilke
2026-07-30 23:31 ` [PATCH v9 2/5] vfio: selftests: Add helpers to re-enable interrupts Josh Hilke
@ 2026-07-30 23:31 ` Josh Hilke
2026-07-30 23:31 ` [PATCH v9 4/5] vfio: selftests: Retry on EAGAIN during device reset Josh Hilke
2026-07-30 23:31 ` [PATCH v9 5/5] vfio: selftests: igb: Recover after DMA-read faults Josh Hilke
4 siblings, 0 replies; 7+ messages in thread
From: Josh Hilke @ 2026-07-30 23:31 UTC (permalink / raw)
To: David Matlack, Alex Williamson
Cc: Shuah Khan, linux-kernel, kvm, linux-kselftest, Vipin Sharma,
Josh Hilke, Alex Williamson
From: Alex Williamson <alex.williamson@nvidia.com>
Split the device register programming out of igb_init() into a new
igb_hw_init() helper so that the same sequence can be re-run after a
VFIO_DEVICE_RESET to restore the registers that CTRL.RST clears. No
functional change for the initial path.
igb_init() now performs the one-shot setup: region size assertion, BAR
mapping, CTRL.RST + IMC mask-all to put the device into a known state,
and vfio_pci_msix_enable() to set up the kernel-side IRQ trigger.
igb_hw_init() does the rest: ring pointer setup and IOVA calc,
CTRL_EXT, PCI bus master, GCR, PHY loopback, descriptor rings, RCTL,
TCTL, GPIE/EIAC/EIAM/EIMS/IVAR, and driver-state initialization.
vfio_pci_msix_enable() moves from after RCTL/TCTL to before all
device-side programming. Its only side effects are the VFIO kernel
IRQ trigger setup and the PCI MSI-X capability bits in config space;
neither has any ordering dependency on the 82576 device register
writes performed in igb_hw_init(). Performing it once in igb_init()
keeps igb_hw_init() reusable from the reset recovery path (which uses
vfio_pci_irq_reenable() to re-arm the existing trigger).
Assisted-by: Claude:claude-opus-4-7
Signed-off-by: Alex Williamson <alex.williamson@nvidia.com>
Reviewed-by: David Matlack <dmatlack@google.com>
---
tools/testing/selftests/vfio/lib/drivers/igb/igb.c | 41 ++++++++++++++++------
1 file changed, 31 insertions(+), 10 deletions(-)
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/igb.c b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
index 2f7e5cb26271..ac7b52d89757 100644
--- a/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
@@ -224,7 +224,13 @@ static void igb_reset(struct igb *igb)
igb_write32(igb, E1000_IMC, 0xFFFFFFFF);
}
-static void igb_init(struct vfio_pci_device *device)
+/*
+ * Program the device into a usable state. Split out of igb_init() so it
+ * can be reused after a device reset to re-program the registers that
+ * CTRL.RST clears. Expects bar0 to be mapped and MSI-X already enabled
+ * via VFIO.
+ */
+static void igb_hw_init(struct vfio_pci_device *device)
{
struct igb *igb = to_igb_state(device);
u64 iova_tx, iova_rx;
@@ -232,15 +238,10 @@ static void igb_init(struct vfio_pci_device *device)
u16 cmd_reg;
int retries;
- VFIO_ASSERT_GE(device->driver.region.size, sizeof(struct igb));
-
- /* Set up rings and calculate IOVAs */
- igb->bar0 = device->bars[0].vaddr;
-
iova_tx = to_iova(device, igb->tx_ring);
iova_rx = to_iova(device, igb->rx_ring);
- igb_reset(igb);
+
/* Signal that the driver is loaded */
ctrl = igb_read32(igb, E1000_CTRL_EXT);
@@ -334,9 +335,6 @@ static void igb_init(struct vfio_pci_device *device)
igb_write32(igb, E1000_RCTL, rctl);
igb_write32(igb, E1000_TCTL, E1000_TCTL_EN | E1000_TCTL_PSP);
- /* Enable MSI-X with 1 vector for the test */
- vfio_pci_msix_enable(device, MSIX_VECTOR, 1);
-
/*
* Program MSI-X interrupt routing per 82576 datasheet:
*
@@ -374,6 +372,29 @@ static void igb_init(struct vfio_pci_device *device)
device->driver.msi = MSIX_VECTOR;
}
+static void igb_init(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ VFIO_ASSERT_GE(device->driver.region.size, sizeof(struct igb));
+
+ igb->bar0 = device->bars[0].vaddr;
+
+ igb_reset(igb);
+
+ /*
+ * Enable MSI-X via VFIO before device-side register programming.
+ * vfio_pci_msix_enable() only touches the VFIO IRQ machinery and the
+ * PCI MSI-X capability via config space; it has no ordering
+ * dependency on the device-side writes performed by igb_hw_init().
+ * Placing it here keeps igb_hw_init() reusable from the reset
+ * recovery path (which calls vfio_pci_irq_reenable() instead).
+ */
+ vfio_pci_msix_enable(device, MSIX_VECTOR, 1);
+
+ igb_hw_init(device);
+}
+
static void igb_remove(struct vfio_pci_device *device)
{
struct igb *igb = to_igb_state(device);
--
2.55.0.508.g3f0d502094-goog
^ permalink raw reply related [flat|nested] 7+ messages in thread