summaryrefslogtreecommitdiff
path: root/tools/testing/selftests/vfio
diff options
context:
space:
mode:
Diffstat (limited to 'tools/testing/selftests/vfio')
-rw-r--r--tools/testing/selftests/vfio/Makefile17
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c15
l---------tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h1
l---------tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h1
l---------tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h1
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/igb/igb.c585
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h352
-rw-r--r--tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c783
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio.h10
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/assert.h23
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/iommu.h6
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h1
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h12
-rw-r--r--tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h40
-rw-r--r--tools/testing/selftests/vfio/lib/iommu.c25
-rw-r--r--tools/testing/selftests/vfio/lib/iova_allocator.c5
-rw-r--r--tools/testing/selftests/vfio/lib/libvfio.c25
-rw-r--r--tools/testing/selftests/vfio/lib/libvfio.mk12
-rw-r--r--tools/testing/selftests/vfio/lib/sysfs.c149
-rw-r--r--tools/testing/selftests/vfio/lib/vfio_pci_device.c238
-rw-r--r--tools/testing/selftests/vfio/lib/vfio_pci_driver.c22
-rw-r--r--tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c142
-rw-r--r--tools/testing/selftests/vfio/vfio_dma_mapping_test.c15
-rw-r--r--tools/testing/selftests/vfio/vfio_iommufd_setup_test.c1
-rw-r--r--tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c4
-rw-r--r--tools/testing/selftests/vfio/vfio_pci_device_test.c21
-rw-r--r--tools/testing/selftests/vfio/vfio_pci_driver_test.c57
-rw-r--r--tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c217
28 files changed, 2663 insertions, 117 deletions
diff --git a/tools/testing/selftests/vfio/Makefile b/tools/testing/selftests/vfio/Makefile
index 3c796ca99a50..2c32c48db509 100644
--- a/tools/testing/selftests/vfio/Makefile
+++ b/tools/testing/selftests/vfio/Makefile
@@ -1,9 +1,18 @@
+ARCH ?= $(shell uname -m)
+
+ifeq (,$(filter $(ARCH),aarch64 arm64 x86 x86_64))
+# Do nothing on unsupported architectures
+include ../lib.mk
+else
+
CFLAGS = $(KHDR_INCLUDES)
TEST_GEN_PROGS += vfio_dma_mapping_test
+TEST_GEN_PROGS += vfio_dma_mapping_mmio_test
TEST_GEN_PROGS += vfio_iommufd_setup_test
TEST_GEN_PROGS += vfio_pci_device_test
TEST_GEN_PROGS += vfio_pci_device_init_perf_test
TEST_GEN_PROGS += vfio_pci_driver_test
+TEST_GEN_PROGS += vfio_pci_sriov_uapi_test
TEST_FILES += scripts/cleanup.sh
TEST_FILES += scripts/lib.sh
@@ -15,15 +24,21 @@ include lib/libvfio.mk
CFLAGS += -I$(top_srcdir)/tools/include
CFLAGS += -MD
+CFLAGS += -Wall -Werror
CFLAGS += $(EXTRA_CFLAGS)
LDFLAGS += -pthread
-$(TEST_GEN_PROGS): %: %.o $(LIBVFIO_O)
+$(TEST_GEN_PROGS): $(OUTPUT)/%: $(OUTPUT)/%.o $(LIBVFIO_O)
$(CC) $(CFLAGS) $(CPPFLAGS) $(LDFLAGS) $< $(LIBVFIO_O) $(LDLIBS) -o $@
TEST_GEN_PROGS_O = $(patsubst %, %.o, $(TEST_GEN_PROGS))
+$(TEST_GEN_PROGS_O): $(OUTPUT)/%.o: %.c
+ $(CC) $(CFLAGS) $(CPPFLAGS) $(TARGET_ARCH) -c $< -o $@
+
TEST_DEP_FILES = $(patsubst %.o, %.d, $(TEST_GEN_PROGS_O) $(LIBVFIO_O))
-include $(TEST_DEP_FILES)
EXTRA_CLEAN += $(TEST_GEN_PROGS_O) $(TEST_DEP_FILES)
+
+endif
diff --git a/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c b/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c
index c75045bcab79..19d9630b24c2 100644
--- a/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c
+++ b/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c
@@ -65,9 +65,20 @@ static bool dsa_int_handle_request_required(struct vfio_pci_device *device)
static int dsa_probe(struct vfio_pci_device *device)
{
- if (!vfio_pci_device_match(device, PCI_VENDOR_ID_INTEL,
- PCI_DEVICE_ID_INTEL_DSA_SPR0))
+ const u16 vendor_id = vfio_pci_config_readw(device, PCI_VENDOR_ID);
+ const u16 device_id = vfio_pci_config_readw(device, PCI_DEVICE_ID);
+
+ if (vendor_id != PCI_VENDOR_ID_INTEL)
+ return -EINVAL;
+
+ switch (device_id) {
+ case PCI_DEVICE_ID_INTEL_DSA_SPR0:
+ case PCI_DEVICE_ID_INTEL_DSA_DMR:
+ case PCI_DEVICE_ID_INTEL_DSA_GNRD:
+ break;
+ default:
return -EINVAL;
+ }
if (dsa_int_handle_request_required(device)) {
dev_err(device, "Device requires requesting interrupt handles\n");
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h
new file mode 120000
index 000000000000..b84affdec559
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_82575.h \ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h
new file mode 120000
index 000000000000..9f97f4330086
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_defines.h \ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h
new file mode 120000
index 000000000000..c733634171bb
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h
@@ -0,0 +1 @@
+../../../../../../../drivers/net/ethernet/intel/igb/e1000_regs.h \ No newline at end of file
diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/igb.c b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
new file mode 100644
index 000000000000..fd9e05d77ea4
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c
@@ -0,0 +1,585 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include <unistd.h>
+#include <errno.h>
+#include <stdint.h>
+#include <linux/io.h>
+#include <linux/pci_regs.h>
+#include <linux/pci_ids.h>
+#include <linux/kernel.h>
+#include <linux/compiler.h>
+#include <asm/barrier.h>
+#include <linux/mii.h>
+#include <libvfio/vfio_pci_device.h>
+
+#include "e1000_regs.h"
+#include "e1000_defines.h"
+#include "e1000_82575.h"
+
+#define PCI_DEVICE_ID_INTEL_82576 0x10C9
+#define IGB_MAX_CHUNK_SIZE 1024
+#define MSIX_VECTOR 0
+#define MSIX_VECTOR_MASK (1 << MSIX_VECTOR)
+#define RING_SIZE 4096 /* Number of descriptors in ring */
+
+struct igb_tx_desc {
+ union {
+ struct {
+ u64 buffer_addr; /* Address of descriptor's data buffer */
+ u32 cmd_type_len; /* Command/Type/Length */
+ u32 olinfo_status; /* Context/Buffer info */
+ } read;
+
+ struct {
+ u64 rsvd; /* Reserved */
+ u32 nxtseq_seed; /* Next sequence seed */
+ u32 status; /* Descriptor status */
+ } wb;
+ };
+};
+
+struct igb_rx_desc {
+ union {
+ struct {
+ u64 pkt_addr; /* Packet buffer address */
+ u64 hdr_addr; /* Header buffer address */
+ } read;
+ struct {
+ u16 pkt_info; /* RSS type, Packet type */
+ u16 hdr_info; /* Split Head, buf len */
+ u32 rss; /* RSS Hash */
+ u32 status_error; /* ext status/error */
+ u16 length; /* Packet length */
+ u16 vlan; /* VLAN tag */
+ } wb; /* writeback */
+ };
+};
+
+struct igb {
+ void *bar0;
+ u32 tx_tail;
+ u32 rx_tail;
+ struct igb_tx_desc tx_ring[RING_SIZE] __attribute__((aligned(128)));
+ struct igb_rx_desc rx_ring[RING_SIZE] __attribute__((aligned(128)));
+};
+
+static inline struct igb *to_igb_state(struct vfio_pci_device *device)
+{
+ return (struct igb *)device->driver.region.vaddr;
+}
+
+static inline void igb_write32(struct igb *igb, u32 reg, u32 val)
+{
+ writel(val, igb->bar0 + reg);
+}
+
+static inline u32 igb_read32(struct igb *igb, u32 reg)
+{
+ return readl(igb->bar0 + reg);
+}
+
+static int igb_write_phy(struct igb *igb, u32 offset, u16 data)
+{
+ u32 mdic;
+ int i;
+
+ /*
+ * Write a PHY register over MDIO.
+ *
+ * A production driver would hold the SW/FW semaphore (SWSM.SWESMBI + the
+ * SW_FW_SYNC PHY bit) across the MDIO transaction to serialize against the
+ * device's management firmware. The selftest owns the assigned function
+ * exclusively on a dedicated test device with no active manageability
+ * contending for the PHY, so the sync is omitted; it should be added here
+ * if this ever needs to run on a manageability-enabled NIC.
+ */
+ mdic = (((u32)data) |
+ (offset << E1000_MDIC_REG_SHIFT) |
+ (1 << E1000_MDIC_PHY_SHIFT) |
+ E1000_MDIC_OP_WRITE);
+
+ igb_write32(igb, E1000_MDIC, mdic);
+
+ for (i = 0; i < 1000; i++) {
+ usleep(50);
+ mdic = igb_read32(igb, E1000_MDIC);
+ if (mdic & E1000_MDIC_READY)
+ break;
+ }
+
+ if (!(mdic & E1000_MDIC_READY))
+ return -1;
+
+ if (mdic & E1000_MDIC_ERROR)
+ return -1;
+
+ return 0;
+}
+
+/*
+ * Configure the device for PHY internal loopback per 82576 datasheet
+ * section 3.5.6.3.1. Force the PHY to 1Gb/s full duplex with loopback
+ * enabled, then force the MAC link state to match. Internal loopback
+ * wraps data at the end of the PHY datapath (section 3.5.6.3), so the
+ * physical link state is irrelevant.
+ *
+ * Section 3.5.6.1 directs to "Use PHY Loopback instead of MAC Loopback
+ * on the 82576", and section 3.5.6.2 states "MAC Loopback is not used
+ * on this device." RCTL.LBM_MAC is still set elsewhere as a QEMU-only
+ * accommodation; see the RCTL programming in the caller for the
+ * rationale.
+ */
+static void igb_setup_loopback(struct igb *igb)
+{
+ u32 ctrl;
+ int ret;
+
+ /*
+ * Kick the autoneg machinery solely to bring STATUS.LU up under
+ * QEMU's igb emulation: QEMU only updates STATUS.LU via its
+ * autoneg-done timer, and without LU set its receive path
+ * (e1000x_hw_rx_enabled) drops every loopback frame. On real
+ * hardware autoneg cannot complete before the next PHY write
+ * below clears the autoneg-enable bit, so this is effectively a
+ * no-op there.
+ */
+ (void)igb_write_phy(igb, MII_BMCR,
+ BMCR_ANENABLE | BMCR_ANRESTART);
+
+ /* PHY control: loopback + 1Gb/s full duplex, autoneg disabled. */
+ ret = igb_write_phy(igb, MII_BMCR,
+ BMCR_LOOPBACK |
+ BMCR_SPEED1000 |
+ BMCR_FULLDPLX);
+ VFIO_ASSERT_EQ(ret, 0, "Failed to write PHY control register");
+
+ /*
+ * Brief delay before forcing the MAC, mirroring the kernel ethtool
+ * selftest in igb_integrated_phy_loopback(). Not specified by the
+ * datasheet, but empirically required by the kernel driver.
+ */
+ usleep(50000);
+
+ /*
+ * Force the MAC to 1Gb/s full duplex with link up. Without forcing
+ * the link state the descriptor engine does not run, since the chip
+ * normally waits for a real negotiated link.
+ */
+ ctrl = igb_read32(igb, E1000_CTRL);
+ ctrl &= ~E1000_CTRL_SPD_SEL;
+ ctrl |= E1000_CTRL_FRCSPD |
+ E1000_CTRL_FRCDPX |
+ E1000_CTRL_SPD_1000 |
+ E1000_CTRL_FD |
+ E1000_CTRL_SLU;
+ igb_write32(igb, E1000_CTRL, ctrl);
+
+ /*
+ * Settling delay matching the kernel ethtool selftest's msleep(500)
+ * at the tail of igb_integrated_phy_loopback(). Not specified by
+ * the datasheet; empirical, and inherited from the kernel driver.
+ */
+ usleep(500000);
+}
+
+static int igb_probe(struct vfio_pci_device *device)
+{
+ if (!vfio_pci_device_match(device, PCI_VENDOR_ID_INTEL, PCI_DEVICE_ID_INTEL_82576))
+ return -EINVAL;
+
+ return 0;
+}
+
+static void igb_reset(struct igb *igb)
+{
+ int retries = 20;
+
+ igb_write32(igb, E1000_CTRL, igb_read32(igb, E1000_CTRL) | E1000_CTRL_RST);
+ /*
+ * Must wait at least 1 millisecond after setting the reset bit before
+ * checking if this device is ready to be used (82576 datasheet section
+ * 4.2.1.6.1). The delay also ensures the reset has taken effect and
+ * cleared EECD.AUTO_RD before it is polled below.
+ */
+ usleep(1000);
+
+ /*
+ * Poll NVM Auto Read Done rather than CTRL.RST, matching
+ * igb_get_auto_rd_done() in the igb driver: AUTO_RD implies both that
+ * the reset completed and that the device finished re-reading its
+ * configuration from NVM, which is what actually makes it usable.
+ */
+ while (retries-- > 0 && !(igb_read32(igb, E1000_EECD) & E1000_EECD_AUTO_RD))
+ usleep(1000);
+
+ /*
+ * QEMU's igb emulation does not set E1000_EECD_AUTO_RD. If we timed out,
+ * check if CTRL.RST is cleared, which is what QEMU uses to signal reset
+ * completion.
+ */
+ if (retries < 0) {
+ VFIO_ASSERT_EQ(igb_read32(igb, E1000_CTRL) & E1000_CTRL_RST, 0,
+ "Device reset did not complete (CTRL.RST not cleared)");
+ }
+
+ igb_write32(igb, E1000_IMC, 0xFFFFFFFF);
+}
+
+/*
+ * Program the device into a usable state. Split out of igb_init() so it
+ * can be reused after a device reset to re-program the registers that
+ * CTRL.RST clears. Expects bar0 to be mapped and MSI-X already enabled
+ * via VFIO.
+ */
+static void igb_hw_init(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+ u64 iova_tx, iova_rx;
+ u32 ctrl, rctl;
+ u16 cmd_reg;
+ int retries;
+
+ iova_tx = to_iova(device, igb->tx_ring);
+ iova_rx = to_iova(device, igb->rx_ring);
+
+
+
+ /* Signal that the driver is loaded */
+ ctrl = igb_read32(igb, E1000_CTRL_EXT);
+ ctrl |= E1000_CTRL_EXT_DRV_LOAD;
+ ctrl &= ~E1000_CTRL_EXT_LINK_MODE_MASK;
+ igb_write32(igb, E1000_CTRL_EXT, ctrl);
+
+ /* Enable PCI Bus Master. */
+ cmd_reg = vfio_pci_config_readw(device, PCI_COMMAND);
+ if ((cmd_reg & (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) !=
+ (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) {
+ cmd_reg |= (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY);
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd_reg);
+ }
+
+ /* Configure PHY internal loopback for testing. */
+ igb_setup_loopback(igb);
+
+ /*
+ * Disable DMA re-send on PCIe completion timeout (82576 datasheet
+ * section 8.6.1, GCR.Completion_Timeout_Resend, bit 16). The
+ * mix_and_match test intentionally submits descriptors targeting
+ * unmapped IOVAs; with the default (set) value, the device keeps
+ * retrying the failed read indefinitely, which keeps PCIe AER and
+ * IOMMU error handling busy and interferes with reset recovery.
+ */
+ ctrl = igb_read32(igb, E1000_GCR);
+ ctrl &= ~E1000_GCR_CMPL_TMOUT_RESEND;
+ igb_write32(igb, E1000_GCR, ctrl);
+
+ /* Configure TX and RX descriptor rings */
+ igb_write32(igb, E1000_TDBAL(0), (u32)iova_tx);
+ igb_write32(igb, E1000_TDBAH(0), (u32)(iova_tx >> 32));
+ igb_write32(igb, E1000_TDLEN(0), RING_SIZE * sizeof(struct igb_tx_desc));
+ igb_write32(igb, E1000_TDH(0), 0);
+ igb_write32(igb, E1000_TDT(0), 0);
+ igb_write32(igb, E1000_TXDCTL(0), E1000_TXDCTL_QUEUE_ENABLE);
+
+ igb_write32(igb, E1000_RDBAL(0), (u32)iova_rx);
+ igb_write32(igb, E1000_RDBAH(0), (u32)(iova_rx >> 32));
+ igb_write32(igb, E1000_RDLEN(0), RING_SIZE * sizeof(struct igb_rx_desc));
+ igb_write32(igb, E1000_RDH(0), 0);
+ igb_write32(igb, E1000_RDT(0), 0);
+
+ /*
+ * Select the advanced one-buffer descriptor format. Per 82576
+ * datasheet section 7.1.5.2: "SRRCTL[n].DESCTYPE must be set to a
+ * value other than 000b for the 82576 to write back the special
+ * descriptors." struct igb_rx_desc matches the advanced one-buffer
+ * writeback layout (section 7.1.5.2), so polling rx.wb.status_error
+ * requires this format. Section 8.10.2 specifies DESCTYPE[27:25].
+ *
+ * The direct write also zeroes SRRCTL.BSIZEPACKET, which is
+ * intentional: per section 7.1.3.1 a zero BSIZEPACKET falls back to
+ * the RCTL.BSIZE buffer size, whose reset default (00b) is 2048
+ * bytes -- ample for the loopback frames here.
+ */
+ igb_write32(igb, E1000_SRRCTL(0), E1000_SRRCTL_DESCTYPE_ADV_ONEBUF);
+
+ igb_write32(igb, E1000_RXDCTL(0), E1000_RXDCTL_QUEUE_ENABLE);
+
+ /*
+ * Enable Receiver and Transmitter. RCTL.LBM_MAC is set in addition
+ * to PHY loopback as a QEMU-only accommodation: QEMU's emulated igb
+ * does not honor PHY register 0 bit 14 (PHY internal loopback) and
+ * relies on RCTL.LBM_MAC to wrap TX descriptors back to the RX
+ * queue. Datasheet 8.10.1 (RCTL register) advises "When using the
+ * internal PHY, LBM should remain set to 00b", so setting LBM_MAC
+ * here deviates from datasheet guidance; empirically the bit has
+ * no observable effect on real 82576 hardware because MAC loopback
+ * is not implemented (datasheet 3.5.6.2). Setting both lets the
+ * selftest work on both real hardware and QEMU without conditional
+ * code paths.
+ */
+ rctl = E1000_RCTL_EN | /* Receiver Enable */
+ E1000_RCTL_UPE | /* Unicast Promiscuous (for dummy MAC) */
+ E1000_RCTL_MPE | /* Multicast Promiscuous */
+ E1000_RCTL_BAM | /* Broadcast Accept Mode */
+ E1000_RCTL_LBM_MAC | /* MAC Loopback - for QEMU emulation only */
+ E1000_RCTL_SECRC; /* Strip CRC (needed for memcmp) */
+ igb_write32(igb, E1000_RCTL, rctl);
+ igb_write32(igb, E1000_TCTL, E1000_TCTL_EN | E1000_TCTL_PSP);
+
+ /*
+ * Wait for TX and RX queues to be enabled. Per the RXDCTL/TXDCTL
+ * register definitions (8.10.10/8.12.13), the per-queue enable bit
+ * "remains zero" until the global RCTL.RXEN/TCTL.TXEN are set, so
+ * E1000_RCTL_EN and E1000_TCTL_EN must already be written above.
+ */
+ retries = 2000;
+ while (retries-- > 0) {
+ if ((igb_read32(igb, E1000_TXDCTL(0)) & E1000_TXDCTL_QUEUE_ENABLE) &&
+ (igb_read32(igb, E1000_RXDCTL(0)) & E1000_RXDCTL_QUEUE_ENABLE))
+ break;
+ usleep(10);
+ }
+ VFIO_ASSERT_GE(retries, 0);
+
+ /*
+ * Program MSI-X interrupt routing per 82576 datasheet:
+ *
+ * GPIE (section 7.3.2.11, Table 7-47): set Multiple_MSIX (bit 4) to
+ * route interrupt causes through IVAR mapping, and EIAME (bit 30)
+ * to apply EIAM on MSI-X assertion (without EIAME, EIAM only
+ * applies on EICR read/write).
+ *
+ * EIAC (section 8.8.5): enable auto-clear of EICR for vector 0.
+ * Without auto-clear the cause stays set after delivery and the
+ * test can see spurious interrupts on the next memcpy batch.
+ *
+ * EIAM (section 8.8.6): enable auto-mask of EIMS for vector 0 on
+ * MSI-X assertion (effective because EIAME is set).
+ *
+ * IVAR (section 7.3.1.2, register definition in 8.8.13): map RX
+ * cause 0 to MSI-X vector 0 and mark the entry valid.
+ */
+ igb_write32(igb, E1000_GPIE, E1000_GPIE_MSIX_MODE | E1000_GPIE_EIAME);
+ igb_write32(igb, E1000_EIAC, MSIX_VECTOR_MASK);
+ igb_write32(igb, E1000_EIAM, MSIX_VECTOR_MASK);
+
+ /* Map vector 0 to interrupt cause 0 and mark it valid */
+ igb_write32(igb, E1000_IVAR0, E1000_IVAR_VALID);
+
+ /* Enable interrupts on vector 0 */
+ igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK);
+
+ /* Initialize driver state and capability limits */
+ igb->tx_tail = 0;
+ igb->rx_tail = 0;
+
+ device->driver.max_memcpy_size = IGB_MAX_CHUNK_SIZE;
+ device->driver.max_memcpy_count = RING_SIZE - 1;
+ device->driver.msi = MSIX_VECTOR;
+}
+
+static void igb_init(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ VFIO_ASSERT_GE(device->driver.region.size, sizeof(struct igb));
+
+ igb->bar0 = device->bars[0].vaddr;
+
+ igb_reset(igb);
+
+ /*
+ * Enable MSI-X via VFIO before device-side register programming.
+ * vfio_pci_msix_enable() only touches the VFIO IRQ machinery and the
+ * PCI MSI-X capability via config space; it has no ordering
+ * dependency on the device-side writes performed by igb_hw_init().
+ * Placing it here keeps igb_hw_init() reusable from the reset
+ * recovery path (which calls vfio_pci_irq_reenable() instead).
+ */
+ vfio_pci_msix_enable(device, MSIX_VECTOR, 1);
+
+ igb_hw_init(device);
+}
+
+static void igb_remove(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ igb_write32(igb, E1000_RCTL, 0);
+ igb_write32(igb, E1000_TCTL, 0);
+ igb_reset(igb);
+
+ vfio_pci_msix_disable(device);
+}
+
+static void igb_irq_disable(struct igb *igb)
+{
+ igb_write32(igb, E1000_EIMC, MSIX_VECTOR_MASK);
+}
+
+static void igb_irq_enable(struct igb *igb)
+{
+ igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK);
+}
+
+static void igb_irq_clear(struct igb *igb)
+{
+ /*
+ * Use write-to-clear (datasheet 7.3.4.2). In MSI-X mode with EIAC
+ * programmed, section 8.8.5 explicitly states "If any bits are set
+ * in EIAC, the EICR register should not be read", which rules out
+ * the read-to-clear path in 7.3.4.3. Bits not in EIAC are still
+ * cleared by writing 1.
+ */
+ igb_write32(igb, E1000_EICR, 0xFFFFFFFF);
+}
+
+static void igb_memcpy_start(struct vfio_pci_device *device, iova_t src,
+ iova_t dst, u64 size, u64 count)
+{
+ struct igb *igb = to_igb_state(device);
+ struct igb_rx_desc *rx;
+ struct igb_tx_desc *tx;
+ u32 i;
+
+ VFIO_ASSERT_GE(size, 60,
+ "IGB driver requires memcpy size to be at least 60 bytes (Ethernet minimum payload size)");
+
+ igb_irq_disable(igb);
+
+ for (i = 0; i < count; i++) {
+ tx = &igb->tx_ring[igb->tx_tail];
+ rx = &igb->rx_ring[igb->rx_tail];
+
+ memset(tx, 0, sizeof(struct igb_tx_desc));
+ memset(rx, 0, sizeof(struct igb_rx_desc));
+
+ rx->read.pkt_addr = cpu_to_le64(dst);
+ rx->read.hdr_addr = cpu_to_le64(0);
+
+ tx->read.buffer_addr = cpu_to_le64(src);
+ /*
+ * Build an advanced data descriptor per 82576 datasheet
+ * section 7.2.2.3. DEXT marks the descriptor as advanced
+ * (required by hardware); DTYP=data selects the data
+ * descriptor; IFCS asks the MAC to append the Ethernet
+ * FCS (without it the frame is dropped as malformed);
+ * EOP marks end of packet. DTALEN is the buffer length
+ * in bits 15:0 of cmd_type_len.
+ */
+ tx->read.cmd_type_len = cpu_to_le32((uint32_t)size |
+ E1000_ADVTXD_DTYP_DATA |
+ E1000_ADVTXD_DCMD_DEXT |
+ E1000_ADVTXD_DCMD_IFCS |
+ E1000_ADVTXD_DCMD_EOP);
+ /*
+ * PAYLEN (section 7.2.2.3.11) is the total payload size
+ * in olinfo_status[31:14].
+ */
+ tx->read.olinfo_status =
+ cpu_to_le32((uint32_t)size << E1000_ADVTXD_PAYLEN_SHIFT);
+
+ igb->tx_tail = (igb->tx_tail + 1) % RING_SIZE;
+ igb->rx_tail = (igb->rx_tail + 1) % RING_SIZE;
+ }
+
+ igb_write32(igb, E1000_RDT(0), igb->rx_tail);
+ igb_write32(igb, E1000_TDT(0), igb->tx_tail);
+}
+
+/*
+ * Reset the device via VFIO_DEVICE_RESET (PCIe FLR on the 82576) and
+ * re-program it. VFIO_DEVICE_RESET tears down the kernel-side MSI-X
+ * trigger but leaves user-side eventfds intact, so re-arm the trigger
+ * via vfio_pci_irq_reenable() before reprogramming so any caller-cached
+ * eventfd remains valid.
+ *
+ * FLR clears device-side state to power-on reset values (datasheet
+ * 4.2.1.5.1: a PF FLR is "equivalent to a D0->D3->D0 transition"), so
+ * EIMS and EICR come back as 0 from their register-defined initial
+ * values, and igb_hw_init() resets tx_tail/rx_tail to 0. The next
+ * igb_memcpy_start() will memset each descriptor it touches before
+ * submission, so no explicit IMC/EICR writes or ring memsets are
+ * needed here.
+ */
+static void igb_error_reset_and_reinit(struct vfio_pci_device *device)
+{
+ vfio_pci_device_reset(device);
+ vfio_pci_msix_reenable(device, MSIX_VECTOR, 1);
+ igb_hw_init(device);
+}
+
+static int igb_memcpy_wait(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+ struct igb_rx_desc *rx;
+ u32 status = 0;
+ u32 prev_tail;
+ int retries;
+
+ prev_tail = (igb->rx_tail + RING_SIZE - 1) % RING_SIZE;
+ rx = &igb->rx_ring[prev_tail];
+
+ /*
+ * Real 82576 hardware processes the descriptor ring at line rate.
+ * max_memcpy_size = (RING_SIZE - 1) * IGB_MAX_CHUNK_SIZE ~= 4 MB,
+ * split into 4095 1 KB frames. At 1 Gb/s (~125 MB/s) the worst
+ * valid memcpy takes ~32 ms on the wire, plus per-frame preamble,
+ * SFD, IFG and FCS overhead (~3%) and descriptor fetch/writeback
+ * latency. Wait up to ~200 ms before declaring the device hung;
+ * ~6x the line-rate floor leaves comfortable headroom for host
+ * scheduling jitter while keeping the intentional invalid-DMA
+ * tests bounded.
+ */
+ retries = 200;
+ while (retries-- > 0) {
+ status = le32_to_cpu(READ_ONCE(rx->wb.status_error));
+ if (status & 1)
+ break;
+ usleep(1000);
+ }
+
+ if (status & 1)
+ /*
+ * Ensure the test code doesn't speculatively read the DMA
+ * destination buffer before we have verified that the
+ * descriptor writeback is complete.
+ */
+ rmb();
+
+ igb_irq_clear(igb);
+
+ igb_irq_enable(igb);
+
+ if (status & 1)
+ return 0;
+
+ /*
+ * The descriptor never completed. On real 82576 hardware this
+ * typically follows a DMA-read fault from one of the intentional
+ * unmapped-IOVA tests; the fault leaves the descriptor engine
+ * unable to service subsequent valid descriptors. CTRL.RST alone
+ * reinitializes the queue registers but leaves the engine wedged
+ * for the current process, so a broader VFIO_DEVICE_RESET (FLR)
+ * is required.
+ */
+ igb_error_reset_and_reinit(device);
+
+ return -ETIMEDOUT;
+}
+
+static void igb_send_msi(struct vfio_pci_device *device)
+{
+ struct igb *igb = to_igb_state(device);
+
+ igb_write32(igb, E1000_EICS, MSIX_VECTOR_MASK);
+}
+
+const struct vfio_pci_driver_ops igb_ops = {
+ .name = "igb",
+ .probe = igb_probe,
+ .init = igb_init,
+ .remove = igb_remove,
+ .memcpy_start = igb_memcpy_start,
+ .memcpy_wait = igb_memcpy_wait,
+ .send_msi = igb_send_msi,
+};
diff --git a/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h
new file mode 100644
index 000000000000..edce130fd008
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h
@@ -0,0 +1,352 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+ */
+#ifndef _NV_FALCON_HW_H_
+#define _NV_FALCON_HW_H_
+
+#include <linux/types.h>
+
+/* PMC (Power Management Controller) Registers */
+#define NV_PMC_BOOT_0 0x00000000
+#define NV_PMC_ENABLE 0x00000200
+#define NV_PMC_ENABLE_PWR 0x00002000
+#define NV_PMC_ENABLE_HUB 0x20000000
+
+/* Falcon Base Pages for Different Engines */
+#define NV_PPWR_FALCON_BASE 0x10a000
+#define NV_PGSP_FALCON_BASE 0x110000
+
+/* Falcon Common Register Offsets (relative to base_page) */
+#define NV_FALCON_DMACTL_OFFSET 0x010c
+#define NV_FALCON_ENGINE_RESET_OFFSET 0x03c0
+
+/* DMEM Control Register Flags */
+#define NV_PPWR_FALCON_DMEMC_AINCR_TRUE 0x01000000
+#define NV_PPWR_FALCON_DMEMC_AINCW_TRUE 0x02000000
+
+/* Falcon DMEM port offsets (for port 0) */
+#define NV_FALCON_DMEMC_OFFSET 0x1c0
+#define NV_FALCON_DMEMD_OFFSET 0x1c4
+
+/* DMA Register Offsets (relative to base_page) */
+#define NV_FALCON_DMA_ADDR_LOW_OFFSET 0x110
+#define NV_FALCON_DMA_MEM_OFFSET 0x114
+#define NV_FALCON_DMA_CMD_OFFSET 0x118
+#define NV_FALCON_DMA_BLOCK_OFFSET 0x11c
+#define NV_FALCON_DMA_ADDR_HIGH_OFFSET 0x128
+
+/* DMA Global Address Top Bits Register */
+#define NV_GPU_DMA_ADDR_TOP_BITS_REG 0x100f04
+
+/* DMA Command Register Bit Definitions */
+#define NV_FALCON_DMA_CMD_WRITE_BIT 0x20
+#define NV_FALCON_DMA_CMD_SIZE_SHIFT 8
+#define NV_FALCON_DMA_CMD_DONE_BIT 0x2
+
+/*
+ * Falcon DMA is synchronous, so a transfer size and count larger than
+ * its per-operation maximum adds no value.
+ */
+
+/* DMA block size and alignment */
+#define NV_FALCON_DMA_MIN_TRANSFER_SIZE 4
+#define NV_FALCON_DMA_MAX_TRANSFER_SIZE 256
+#define NV_FALCON_DMA_BLOCK_SIZE 256
+#define NV_FALCON_DMA_MAX_TRANSFER_COUNT 1
+
+/* DMACTL register bits */
+#define NV_FALCON_DMACTL_DMEM_SCRUBBING 0x1
+#define NV_FALCON_DMACTL_READY_MASK 0x6
+
+/* Falcon Core Selection Register */
+#define NV_FALCON_CORE_SELECT_OFFSET 0x1668
+#define NV_FALCON_CORE_SELECT_MASK 0x30
+
+/* Falcon mailbox register (for Ada+ reset check) */
+#define NV_FALCON_MAILBOX_TEST_OFFSET 0x40c
+#define NV_FALCON_MAILBOX_RESET_MAGIC 0xbadf5620
+
+/* Falcon Message Queue Register Offsets (relative to base_page) */
+#define NV_FALCON_QUEUE_HEAD_BASE_OFFSET 0x2c00
+#define NV_FALCON_QUEUE_TAIL_BASE_OFFSET 0x2c04
+#define NV_FALCON_QUEUE_STRIDE 0x8
+#define NV_FALCON_MSG_QUEUE_HEAD_BASE_OFFSET 0x2c80
+#define NV_FALCON_MSG_QUEUE_TAIL_BASE_OFFSET 0x2c84
+
+/* FSP Falcon Base Pages */
+#define NV_FSP_FALCON_BASE 0x8f0100
+/* base_page = cpuctl & ~0xfff */
+#define NV_FSP_FALCON_BASE_PAGE 0x8f0000
+#define NV_FSP_EMEM_BASE 0x8f2000
+
+/* FSP EMEM Port Offsets (relative to FSP EMEM base) */
+#define NV_FSP_EMEMC_OFFSET 0xac0
+#define NV_FSP_EMEMD_OFFSET 0xac4
+#define NV_FSP_EMEM_PORT_STRIDE 0x8
+
+/* EMEM Control Register Flags (same as DMEM) */
+#define NV_FALCON_EMEMC_AINCR 0x01000000
+#define NV_FALCON_EMEMC_AINCW 0x02000000
+
+/* FSP RPC channel configuration */
+#define NV_FSP_RPC_CHANNEL_SIZE 1024
+#define NV_FSP_RPC_MAX_PACKET_SIZE 1024
+#define NV_FSP_RPC_CHANNEL_HOPPER 2
+#define NV_FSP_RPC_EMEM_BASE \
+ (NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_RPC_CHANNEL_SIZE)
+
+/* FSP EMEM port 2 registers (pre-computed for Hopper channel 2) */
+#define NV_FSP_EMEM_PORT2_CTRL (NV_FSP_EMEM_BASE + NV_FSP_EMEMC_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_EMEM_PORT_STRIDE)
+#define NV_FSP_EMEM_PORT2_DATA (NV_FSP_EMEM_BASE + NV_FSP_EMEMD_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_EMEM_PORT_STRIDE)
+
+/* FSP queue register offsets (pre-computed for Hopper channel 2) */
+#define NV_FSP_QUEUE_HEAD \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_QUEUE_HEAD_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+#define NV_FSP_QUEUE_TAIL \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_QUEUE_TAIL_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+#define NV_FSP_MSG_QUEUE_HEAD \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_MSG_QUEUE_HEAD_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+#define NV_FSP_MSG_QUEUE_TAIL \
+ (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_MSG_QUEUE_TAIL_BASE_OFFSET + \
+ NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE)
+
+/* MCTP Header */
+#define NV_MCTP_HDR_SEID_SHIFT 16
+#define NV_MCTP_HDR_SEID_MASK 0xff
+#define NV_MCTP_HDR_SEQ_SHIFT 28
+#define NV_MCTP_HDR_SEQ_MASK 0x3
+#define NV_MCTP_HDR_EOM_BIT 0x40000000
+#define NV_MCTP_HDR_SOM_BIT 0x80000000
+
+/* MCTP Message Header */
+#define NV_MCTP_MSG_TYPE_SHIFT 0
+#define NV_MCTP_MSG_TYPE_MASK 0x7f
+#define NV_MCTP_MSG_TYPE_VENDOR_DEFINED 0x7e
+#define NV_MCTP_MSG_VENDOR_ID_SHIFT 8
+#define NV_MCTP_MSG_VENDOR_ID_MASK 0xffff
+#define NV_MCTP_MSG_VENDOR_ID_NVIDIA 0x10de
+#define NV_MCTP_MSG_NVDM_TYPE_SHIFT 24
+#define NV_MCTP_MSG_NVDM_TYPE_MASK 0xff
+
+/* NVDM response type */
+#define NV_NVDM_TYPE_RESPONSE 0x15
+
+/* Minimum response size: mctp_hdr + msg_hdr + status_hdr + type + status */
+#define NV_FSP_RPC_MIN_RESPONSE_WORDS 5
+
+/* FBIF (Frame Buffer Interface) Registers */
+/* Legacy PMU FBIF offsets (Kepler, Maxwell Gen1) */
+#define NV_PMU_LEGACY_FBIF_CTL_OFFSET 0x624
+#define NV_PMU_LEGACY_FBIF_TRANSCFG_OFFSET 0x600
+
+/* PMU FBIF offsets */
+#define NV_PMU_FBIF_CTL_OFFSET 0xe24
+#define NV_PMU_FBIF_TRANSCFG_OFFSET 0xe00
+
+/* GSP FBIF offsets */
+#define NV_GSP_FBIF_CTL_OFFSET 0x624
+#define NV_GSP_FBIF_TRANSCFG_OFFSET 0x600
+
+/* OFA Falcon Base Page and FBIF offsets (used for Hopper+ DMA) */
+#define NV_OFA_FALCON_BASE 0x844000
+#define NV_OFA_FBIF_CTL_OFFSET 0x424
+#define NV_OFA_FBIF_TRANSCFG_OFFSET 0x400
+
+/* OFA DMA support check register (Hopper+) */
+#define NV_OFA_DMA_SUPPORT_CHECK_REG 0x8443c0
+
+/* FSP NVDM command types */
+#define NV_NVDM_TYPE_FBDMA 0x22
+#define NV_FBDMA_SUBCMD_ENABLE 0x1
+
+/* FBIF CTL2 offset (relative to fbif_ctl) */
+#define NV_FBIF_CTL2_OFFSET 0x60
+
+/* FBIF TRANSCFG register bits */
+#define NV_FBIF_TRANSCFG_TARGET_MASK 0x3
+#define NV_FBIF_TRANSCFG_SYSMEM_DEFAULT 0x5
+
+/* FBIF CTL register bits */
+#define NV_FBIF_CTL_ALLOW_PHYS_MODE 0x10
+#define NV_FBIF_CTL_ALLOW_FULL_PHYS_MODE 0x80
+
+/* Memory clear register offsets */
+#define NV_MEM_CLEAR_OFFSET 0x100b20
+#define NV_BOOT_COMPLETE_OFFSET 0x118234
+#define NV_BOOT_COMPLETE_SUCCESS 0x3ff
+
+/* FSP boot complete register (Hopper+) */
+#define NV_FSP_BOOT_COMPLETE_OFFSET 0x200bc
+#define NV_FSP_BOOT_COMPLETE_SUCCESS 0xff
+
+enum gpu_arch {
+ GPU_ARCH_UNKNOWN = -1,
+ GPU_ARCH_KEPLER = 0,
+ GPU_ARCH_MAXWELL_GEN1,
+ GPU_ARCH_MAXWELL_GEN2,
+ GPU_ARCH_PASCAL,
+ GPU_ARCH_PASCAL_10X,
+ GPU_ARCH_VOLTA,
+ GPU_ARCH_TURING,
+ GPU_ARCH_AMPERE,
+ GPU_ARCH_ADA,
+ GPU_ARCH_HOPPER,
+};
+
+enum falcon_type {
+ FALCON_TYPE_PMU_LEGACY = 0,
+ FALCON_TYPE_PMU,
+ FALCON_TYPE_GSP,
+ FALCON_TYPE_OFA,
+};
+
+struct falcon {
+ u32 base_page;
+ u32 dmactl;
+ u32 engine_reset;
+ u32 fbif_ctl;
+ u32 fbif_ctl2;
+ u32 fbif_transcfg;
+ u32 dmem_control_reg;
+ u32 dmem_data_reg;
+ bool no_outside_reset;
+};
+
+struct gpu_properties {
+ u32 pmc_enable_mask;
+ bool memory_clear_supported;
+ enum falcon_type falcon_type;
+};
+
+static const u32 verified_gpu_map[] = {
+ 0x0e40a0a2, /* K520 */
+ 0x0e6000a1, /* GTX660 */
+ 0x0e63a0a1, /* K4000 */
+ 0x0f22d0a1, /* K80 */
+ 0x108000a1, /* GT635 */
+ 0x117010a2, /* GTX750 */
+ 0x117020a2, /* GTX745 */
+ 0x124320a1, /* M60 */
+ 0x130000a1, /* P100 */
+ 0x134000a1, /* P4 */
+ 0x132000a1, /* P40 */
+ 0x140000a1, /* V100 */
+ 0x164000a1, /* T4 */
+ 0xb77000a1, /* A16 */
+ 0x170000a1, /* A100 */
+ 0xb72000a1, /* A10 */
+ 0x180000a1, /* H100 */
+ 0x194000a1, /* L4 */
+ 0x192000a1, /* L40S */
+};
+
+#define VERIFIED_GPU_MAP_SIZE ARRAY_SIZE(verified_gpu_map)
+
+static const struct gpu_properties gpu_properties_map[] = {
+ [GPU_ARCH_KEPLER] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR | NV_PMC_ENABLE_HUB,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU_LEGACY,
+ },
+ [GPU_ARCH_MAXWELL_GEN1] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR | NV_PMC_ENABLE_HUB,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU_LEGACY,
+ },
+ [GPU_ARCH_MAXWELL_GEN2] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_PASCAL] = {
+ .pmc_enable_mask = NV_PMC_ENABLE_PWR,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_PASCAL_10X] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_VOLTA] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = false,
+ .falcon_type = FALCON_TYPE_GSP,
+ },
+ [GPU_ARCH_TURING] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_GSP,
+ },
+ [GPU_ARCH_AMPERE] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_GSP,
+ },
+ [GPU_ARCH_ADA] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_PMU,
+ },
+ [GPU_ARCH_HOPPER] = {
+ .pmc_enable_mask = 0,
+ .memory_clear_supported = true,
+ .falcon_type = FALCON_TYPE_OFA,
+ },
+};
+
+static const struct falcon falcon_map[] = {
+ [FALCON_TYPE_PMU_LEGACY] = {
+ .base_page = NV_PPWR_FALCON_BASE,
+ .dmactl = NV_PPWR_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_PPWR_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_PPWR_FALCON_BASE + NV_PMU_LEGACY_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_PPWR_FALCON_BASE +
+ NV_PMU_LEGACY_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_PPWR_FALCON_BASE + NV_PMU_LEGACY_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = false,
+ },
+ [FALCON_TYPE_PMU] = {
+ .base_page = NV_PPWR_FALCON_BASE,
+ .dmactl = NV_PPWR_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_PPWR_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = false,
+ },
+ [FALCON_TYPE_GSP] = {
+ .base_page = NV_PGSP_FALCON_BASE,
+ .dmactl = NV_PGSP_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_PGSP_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_PGSP_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_PGSP_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = false,
+ },
+ [FALCON_TYPE_OFA] = {
+ .base_page = NV_OFA_FALCON_BASE,
+ .dmactl = NV_OFA_FALCON_BASE + NV_FALCON_DMACTL_OFFSET,
+ .engine_reset = NV_OFA_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET,
+ .fbif_ctl = NV_OFA_FALCON_BASE + NV_OFA_FBIF_CTL_OFFSET,
+ .fbif_ctl2 = NV_OFA_FALCON_BASE + NV_OFA_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET,
+ .fbif_transcfg = NV_OFA_FALCON_BASE + NV_OFA_FBIF_TRANSCFG_OFFSET,
+ .dmem_control_reg = NV_OFA_FALCON_BASE + NV_FALCON_DMEMC_OFFSET,
+ .dmem_data_reg = NV_OFA_FALCON_BASE + NV_FALCON_DMEMD_OFFSET,
+ .no_outside_reset = true,
+ },
+};
+
+#endif /* _NV_FALCON_HW_H_ */
diff --git a/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c
new file mode 100644
index 000000000000..c08aa81c44f4
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c
@@ -0,0 +1,783 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+ */
+#include <stdint.h>
+#include <strings.h>
+#include <unistd.h>
+#include <stdbool.h>
+#include <string.h>
+#include <time.h>
+
+#include <linux/errno.h>
+#include <linux/io.h>
+#include <linux/pci_ids.h>
+
+#include <libvfio.h>
+
+#include "hw.h"
+
+struct gpu_device {
+ enum gpu_arch arch;
+ void *bar0;
+ bool is_memory_clear_supported;
+ const struct falcon *falcon;
+ u32 pmc_enable_mask;
+ bool fsp_dma_enabled;
+
+ /* Pending memcpy parameters, set by memcpy_start() */
+ u64 memcpy_src;
+ u64 memcpy_dst;
+ u64 memcpy_size;
+};
+
+static inline struct gpu_device *to_gpu_device(struct vfio_pci_device *device)
+{
+ return device->driver.region.vaddr;
+}
+
+static enum gpu_arch nv_gpu_arch_lookup(u32 pmc_boot_0)
+{
+ u32 arch = (pmc_boot_0 >> 24) & 0x1f;
+
+ switch (arch) {
+ case 0x0e:
+ case 0x0f:
+ case 0x10:
+ return GPU_ARCH_KEPLER;
+ case 0x11:
+ return GPU_ARCH_MAXWELL_GEN1;
+ case 0x12:
+ return GPU_ARCH_MAXWELL_GEN2;
+ case 0x13:
+ /* P100 (impl 0) uses PMC reset; P4/P40 use engine reset */
+ if (((pmc_boot_0 >> 20) & 0xf) == 0)
+ return GPU_ARCH_PASCAL;
+ return GPU_ARCH_PASCAL_10X;
+ case 0x14:
+ return GPU_ARCH_VOLTA;
+ case 0x16:
+ return GPU_ARCH_TURING;
+ case 0x17:
+ return GPU_ARCH_AMPERE;
+ case 0x18:
+ return GPU_ARCH_HOPPER;
+ case 0x19:
+ return GPU_ARCH_ADA;
+ default:
+ return GPU_ARCH_UNKNOWN;
+ }
+}
+
+static inline u32 gpu_read32(struct gpu_device *gpu, u32 offset)
+{
+ return readl(gpu->bar0 + offset);
+}
+
+static inline void gpu_write32(struct gpu_device *gpu, u32 offset, u32 value)
+{
+ writel(value, gpu->bar0 + offset);
+}
+
+static u64 get_elapsed_ms(struct timespec *start)
+{
+ struct timespec now;
+
+ clock_gettime(CLOCK_MONOTONIC, &now);
+
+ return (now.tv_sec - start->tv_sec) * 1000
+ + (now.tv_nsec - start->tv_nsec) / 1000000;
+}
+
+static int gpu_poll_register(struct vfio_pci_device *device,
+ const char *name, u32 offset,
+ u32 expected, u32 mask, u32 timeout_ms)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ struct timespec start;
+ u64 elapsed_ms;
+ u32 value;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+
+ for (;;) {
+ value = gpu_read32(gpu, offset);
+ if ((value & mask) == expected)
+ return 0;
+
+ elapsed_ms = get_elapsed_ms(&start);
+
+ if (elapsed_ms >= timeout_ms)
+ break;
+
+ usleep(1000);
+ }
+
+ dev_err(device,
+ "Timeout polling %s (0x%x): value=0x%x expected=0x%x mask=0x%x after %lu ms\n",
+ name, offset, value, expected, mask, elapsed_ms);
+ return -ETIMEDOUT;
+}
+
+static int fsp_poll_queue(struct vfio_pci_device *device, const char *name,
+ u32 head_reg, u32 tail_reg, bool wait_empty,
+ u32 timeout_ms)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ struct timespec start;
+ u64 elapsed_ms;
+ u32 head, tail;
+
+ clock_gettime(CLOCK_MONOTONIC, &start);
+
+ for (;;) {
+ head = gpu_read32(gpu, head_reg);
+ tail = gpu_read32(gpu, tail_reg);
+ if (wait_empty ? (head == tail) : (head != tail))
+ return 0;
+
+ elapsed_ms = get_elapsed_ms(&start);
+
+ if (elapsed_ms >= timeout_ms)
+ break;
+
+ usleep(1000);
+ }
+
+ dev_err(device,
+ "Timeout polling %s: head=0x%x tail=0x%x wait_empty=%d after %lu ms\n",
+ name, head, tail, wait_empty, elapsed_ms);
+ return -ETIMEDOUT;
+}
+
+static void fsp_emem_write(struct vfio_pci_device *device, u32 offset,
+ const u32 *data, u32 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 i;
+
+ /* Configure port with auto-increment for read and write */
+ gpu_write32(gpu, NV_FSP_EMEM_PORT2_CTRL,
+ offset | NV_FALCON_EMEMC_AINCR | NV_FALCON_EMEMC_AINCW);
+
+ for (i = 0; i < count; i++)
+ gpu_write32(gpu, NV_FSP_EMEM_PORT2_DATA, data[i]);
+}
+
+static void fsp_emem_read(struct vfio_pci_device *device, u32 offset,
+ u32 *data, u32 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 i;
+
+ /* Configure port with auto-increment for read and write */
+ gpu_write32(gpu, NV_FSP_EMEM_PORT2_CTRL,
+ offset | NV_FALCON_EMEMC_AINCR | NV_FALCON_EMEMC_AINCW);
+
+ for (i = 0; i < count; i++)
+ data[i] = gpu_read32(gpu, NV_FSP_EMEM_PORT2_DATA);
+}
+
+static int fsp_rpc_send_data(struct vfio_pci_device *device, const u32 *data,
+ u32 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ int ret;
+
+ ret = fsp_poll_queue(device, "fsp_cmd_queue_empty",
+ NV_FSP_QUEUE_HEAD, NV_FSP_QUEUE_TAIL, true, 1000);
+ if (ret)
+ return ret;
+
+ fsp_emem_write(device, NV_FSP_RPC_EMEM_BASE, data, count);
+
+ /* Update queue head/tail to signal data is ready */
+ gpu_write32(gpu, NV_FSP_QUEUE_TAIL,
+ NV_FSP_RPC_EMEM_BASE + (count - 1) * 4);
+ gpu_write32(gpu, NV_FSP_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE);
+
+ return ret;
+}
+
+static int fsp_rpc_receive_data(struct vfio_pci_device *device, u32 *data,
+ u32 max_count, u32 timeout_ms)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 head, tail;
+ u32 msg_size_words;
+ int ret;
+
+ ret = fsp_poll_queue(device, "fsp_msg_queue_ready",
+ NV_FSP_MSG_QUEUE_HEAD, NV_FSP_MSG_QUEUE_TAIL,
+ false, timeout_ms);
+ if (ret)
+ return ret;
+
+ head = gpu_read32(gpu, NV_FSP_MSG_QUEUE_HEAD);
+ tail = gpu_read32(gpu, NV_FSP_MSG_QUEUE_TAIL);
+
+ msg_size_words = (tail - head + 4) / 4;
+ if (msg_size_words > max_count)
+ msg_size_words = max_count;
+
+ fsp_emem_read(device, NV_FSP_RPC_EMEM_BASE, data, msg_size_words);
+
+ /* Reset message queue tail to acknowledge receipt */
+ gpu_write32(gpu, NV_FSP_MSG_QUEUE_TAIL, head);
+
+ return msg_size_words;
+}
+
+static void fsp_reset_rpc_state(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 head, tail;
+
+ head = gpu_read32(gpu, NV_FSP_QUEUE_HEAD);
+ tail = gpu_read32(gpu, NV_FSP_QUEUE_TAIL);
+
+ if (head == tail) {
+ head = gpu_read32(gpu, NV_FSP_MSG_QUEUE_HEAD);
+ tail = gpu_read32(gpu, NV_FSP_MSG_QUEUE_TAIL);
+ if (head == tail)
+ return;
+ }
+
+ /* Best-effort drain; timeout is expected if no pending message. */
+ fsp_poll_queue(device, "fsp_msg_queue_drain",
+ NV_FSP_MSG_QUEUE_HEAD, NV_FSP_MSG_QUEUE_TAIL,
+ false, 5000);
+
+ gpu_write32(gpu, NV_FSP_QUEUE_TAIL, NV_FSP_RPC_EMEM_BASE);
+ gpu_write32(gpu, NV_FSP_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE);
+ gpu_write32(gpu, NV_FSP_MSG_QUEUE_TAIL, NV_FSP_RPC_EMEM_BASE);
+ gpu_write32(gpu, NV_FSP_MSG_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE);
+}
+
+static inline u32 mctp_header_build(u8 seid, u8 seq, bool som, bool eom)
+{
+ u32 hdr = 0;
+
+ hdr |= (seid & NV_MCTP_HDR_SEID_MASK) << NV_MCTP_HDR_SEID_SHIFT;
+ hdr |= (seq & NV_MCTP_HDR_SEQ_MASK) << NV_MCTP_HDR_SEQ_SHIFT;
+ if (som)
+ hdr |= NV_MCTP_HDR_SOM_BIT;
+ if (eom)
+ hdr |= NV_MCTP_HDR_EOM_BIT;
+
+ return hdr;
+}
+
+static inline u32 mctp_msg_header_build(u8 nvdm_type)
+{
+ u32 hdr = 0;
+
+ hdr |= (NV_MCTP_MSG_TYPE_VENDOR_DEFINED & NV_MCTP_MSG_TYPE_MASK)
+ << NV_MCTP_MSG_TYPE_SHIFT;
+ hdr |= (NV_MCTP_MSG_VENDOR_ID_NVIDIA & NV_MCTP_MSG_VENDOR_ID_MASK)
+ << NV_MCTP_MSG_VENDOR_ID_SHIFT;
+ hdr |= (nvdm_type & NV_MCTP_MSG_NVDM_TYPE_MASK)
+ << NV_MCTP_MSG_NVDM_TYPE_SHIFT;
+
+ return hdr;
+}
+
+static inline u8 mctp_msg_header_get_nvdm_type(u32 hdr)
+{
+ return (hdr >> NV_MCTP_MSG_NVDM_TYPE_SHIFT) &
+ NV_MCTP_MSG_NVDM_TYPE_MASK;
+}
+
+static int fsp_rpc_send_cmd(struct vfio_pci_device *device, u8 nvdm_type,
+ const u32 *data, u32 data_count, u32 timeout_ms)
+{
+ u32 max_packet_words = NV_FSP_RPC_MAX_PACKET_SIZE / 4;
+ u32 packet[256];
+ u32 resp_buf[256];
+ u32 total_words;
+ int resp_words;
+ u8 resp_nvdm_type;
+ int ret;
+
+ total_words = 2 + data_count;
+ if (total_words > max_packet_words)
+ return -EINVAL;
+
+ packet[0] = mctp_header_build(0, 0, true, true);
+ packet[1] = mctp_msg_header_build(nvdm_type);
+
+ if (data_count > 0)
+ memcpy(&packet[2], data, data_count * sizeof(u32));
+
+ ret = fsp_rpc_send_data(device, packet, total_words);
+ if (ret)
+ return ret;
+
+ resp_words = fsp_rpc_receive_data(device, resp_buf, 256, timeout_ms);
+ if (resp_words < 0)
+ return resp_words;
+
+ if (resp_words < NV_FSP_RPC_MIN_RESPONSE_WORDS)
+ return -EPROTO;
+
+ resp_nvdm_type = mctp_msg_header_get_nvdm_type(resp_buf[1]);
+ if (resp_nvdm_type != NV_NVDM_TYPE_RESPONSE)
+ return -EPROTO;
+
+ if (resp_buf[3] != nvdm_type)
+ return -EPROTO;
+
+ if (resp_buf[4] != 0)
+ return -resp_buf[4];
+
+ return 0;
+}
+
+static int fsp_init(struct vfio_pci_device *device)
+{
+ int ret;
+
+ ret = gpu_poll_register(device, "fsp_boot_complete",
+ NV_FSP_BOOT_COMPLETE_OFFSET,
+ NV_FSP_BOOT_COMPLETE_SUCCESS, 0xffffffff, 5000);
+ if (ret)
+ return ret;
+
+ fsp_reset_rpc_state(device);
+ return ret;
+}
+
+static int fsp_fbdma_enable(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 cmd_data = NV_FBDMA_SUBCMD_ENABLE;
+ int ret = 0;
+
+ if (gpu->fsp_dma_enabled)
+ return ret;
+
+ ret = fsp_rpc_send_cmd(device, NV_NVDM_TYPE_FBDMA, &cmd_data, 1, 5000);
+ if (ret)
+ return ret;
+
+ gpu->fsp_dma_enabled = true;
+ return ret;
+}
+
+static bool fsp_check_ofa_dma_support(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ u32 val = gpu_read32(gpu, NV_OFA_DMA_SUPPORT_CHECK_REG);
+
+ return (val >> 16) != 0xbadf;
+}
+
+static u32 size_to_dma_encoding(u64 size)
+{
+ VFIO_ASSERT_LE(size, NV_FALCON_DMA_MAX_TRANSFER_SIZE);
+ VFIO_ASSERT_GE(size, NV_FALCON_DMA_MIN_TRANSFER_SIZE);
+ VFIO_ASSERT_EQ(size & (size - 1), 0, "size must be power-of-2\n");
+
+ return ffs(size) - 3;
+}
+
+static void falcon_dmem_port_configure(struct vfio_pci_device *device,
+ u32 offset, bool auto_inc_read,
+ bool auto_inc_write)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 memc_value = offset;
+
+ /* Set auto-increment flags */
+ if (auto_inc_read)
+ memc_value |= NV_PPWR_FALCON_DMEMC_AINCR_TRUE;
+ if (auto_inc_write)
+ memc_value |= NV_PPWR_FALCON_DMEMC_AINCW_TRUE;
+
+ gpu_write32(gpu, falcon->dmem_control_reg, memc_value);
+}
+
+static void falcon_select_core_falcon(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 core_select_reg = falcon->base_page + NV_FALCON_CORE_SELECT_OFFSET;
+ u32 core_select;
+
+ core_select = gpu_read32(gpu, core_select_reg);
+
+ /* Clear bits 4:5 to select falcon core (not RISCV) */
+ core_select &= ~NV_FALCON_CORE_SELECT_MASK;
+
+ gpu_write32(gpu, core_select_reg, core_select);
+}
+
+static int falcon_enable(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 mailbox_test_reg;
+ u32 mailbox_val;
+
+ if (falcon->no_outside_reset)
+ return 0;
+
+ /* Ada-specific: Check if falcon needs reset before enable */
+ if (gpu->arch == GPU_ARCH_ADA) {
+ mailbox_test_reg = falcon->base_page +
+ NV_FALCON_MAILBOX_TEST_OFFSET;
+ mailbox_val = gpu_read32(gpu, mailbox_test_reg);
+ if (mailbox_val == NV_FALCON_MAILBOX_RESET_MAGIC)
+ gpu_write32(gpu, falcon->engine_reset, 1);
+ }
+
+ /* Enable the falcon based on control method */
+ if (gpu->pmc_enable_mask != 0) {
+ u32 pmc_enable;
+
+ /* Enable via PMC_ENABLE register */
+ pmc_enable = gpu_read32(gpu, NV_PMC_ENABLE);
+ gpu_write32(gpu, NV_PMC_ENABLE,
+ pmc_enable | gpu->pmc_enable_mask);
+ } else {
+ /* Enable by deasserting engine reset */
+ gpu_write32(gpu, falcon->engine_reset, 0);
+ }
+
+ if (gpu->arch < GPU_ARCH_HOPPER) {
+ falcon_select_core_falcon(device);
+
+ /* Wait for DMACTL to be ready (bits 1:2 should be 0) */
+ return gpu_poll_register(device, "falcon_dmactl",
+ falcon->dmactl, 0,
+ NV_FALCON_DMACTL_READY_MASK, 1000);
+ }
+
+ return 0;
+}
+
+static void falcon_disable(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 pmc_enable;
+
+ if (falcon->no_outside_reset)
+ return;
+
+ if (gpu->pmc_enable_mask != 0) {
+ /* Disable via PMC_ENABLE */
+ pmc_enable = gpu_read32(gpu, NV_PMC_ENABLE);
+ gpu_write32(gpu, NV_PMC_ENABLE,
+ pmc_enable & ~gpu->pmc_enable_mask);
+ } else {
+ /* Disable by asserting engine reset */
+ gpu_write32(gpu, falcon->engine_reset, 1);
+ }
+}
+
+static int falcon_reset(struct vfio_pci_device *device)
+{
+ falcon_disable(device);
+
+ return falcon_enable(device);
+}
+
+static int nv_falcon_dma_init(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon;
+ u32 transcfg;
+ u32 dmactl;
+ u32 ctl;
+ int ret = 0;
+
+ falcon = gpu->falcon;
+
+ vfio_pci_cmd_set(device, PCI_COMMAND_MASTER);
+
+ if (gpu->arch >= GPU_ARCH_HOPPER) {
+ ret = fsp_init(device);
+ if (ret) {
+ dev_err(device, "Failed to init FSP: %d\n", ret);
+ return ret;
+ }
+
+ ret = fsp_fbdma_enable(device);
+ if (ret) {
+ dev_err(device,
+ "Failed to enable FSP FBDMA: %d\n", ret);
+ return ret;
+ }
+
+ if (!fsp_check_ofa_dma_support(device)) {
+ dev_err(device,
+ "OFA DMA not supported with current firmware\n");
+ return -EOPNOTSUPP;
+ }
+ }
+
+ if (gpu->is_memory_clear_supported) {
+ /* For Turing+, wait for boot to complete first */
+ if (gpu->arch >= GPU_ARCH_TURING) {
+ /* Wait for boot complete - Hopper+ uses FSP register */
+ if (gpu->arch >= GPU_ARCH_HOPPER) {
+ ret = gpu_poll_register(device,
+ "fsp_boot_complete",
+ NV_FSP_BOOT_COMPLETE_OFFSET,
+ NV_FSP_BOOT_COMPLETE_SUCCESS,
+ 0xffffffff, 5000);
+ } else {
+ ret = gpu_poll_register(device,
+ "boot_complete",
+ NV_BOOT_COMPLETE_OFFSET,
+ NV_BOOT_COMPLETE_SUCCESS,
+ 0xffffffff, 5000);
+ }
+ if (ret)
+ return ret;
+
+ ret = gpu_poll_register(device,
+ "memory_clear_finished",
+ NV_MEM_CLEAR_OFFSET, 0x1, 0xffffffff, 5000);
+ if (ret)
+ return ret;
+ }
+ }
+
+ ret = falcon_reset(device);
+ if (ret)
+ return ret;
+
+ falcon_dmem_port_configure(device, 0, false, false);
+
+ transcfg = gpu_read32(gpu, falcon->fbif_transcfg);
+ transcfg &= ~NV_FBIF_TRANSCFG_TARGET_MASK;
+ transcfg |= NV_FBIF_TRANSCFG_SYSMEM_DEFAULT;
+ gpu_write32(gpu, falcon->fbif_transcfg, transcfg);
+
+ gpu_write32(gpu, falcon->fbif_ctl2, 0x1);
+
+ ctl = gpu_read32(gpu, falcon->fbif_ctl);
+ ctl |= NV_FBIF_CTL_ALLOW_PHYS_MODE | NV_FBIF_CTL_ALLOW_FULL_PHYS_MODE;
+ gpu_write32(gpu, falcon->fbif_ctl, ctl);
+
+ dmactl = gpu_read32(gpu, falcon->dmactl);
+ dmactl &= ~NV_FALCON_DMACTL_DMEM_SCRUBBING;
+ gpu_write32(gpu, falcon->dmactl, dmactl);
+
+ return ret;
+}
+
+static int nv_falcon_dma(struct vfio_pci_device *device,
+ u64 address, u64 size,
+ bool write)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct falcon *falcon = gpu->falcon;
+ u32 dma_cmd;
+ int ret;
+
+ gpu_write32(gpu, NV_GPU_DMA_ADDR_TOP_BITS_REG,
+ (address >> 47) & 0x1ffff);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_ADDR_HIGH_OFFSET,
+ (address >> 40) & 0x7f);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_ADDR_LOW_OFFSET,
+ (address >> 8) & 0xffffffff);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_BLOCK_OFFSET,
+ address & 0xff);
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_MEM_OFFSET, 0);
+
+ dma_cmd = size_to_dma_encoding(size) << NV_FALCON_DMA_CMD_SIZE_SHIFT;
+
+ /* Set direction: write (DMEM->mem) or read (mem->DMEM) */
+ if (write)
+ dma_cmd |= NV_FALCON_DMA_CMD_WRITE_BIT;
+
+ gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_CMD_OFFSET, dma_cmd);
+
+ ret = gpu_poll_register(device, "dma_done",
+ falcon->base_page + NV_FALCON_DMA_CMD_OFFSET,
+ NV_FALCON_DMA_CMD_DONE_BIT,
+ NV_FALCON_DMA_CMD_DONE_BIT, 1000);
+ if (ret)
+ dev_err(device, "Failed DMA %s (addr=0x%lx, size=%lu)\n",
+ write ? "write" : "read", address, size);
+
+ return ret;
+}
+
+static int nv_falcon_memcpy_chunk(struct vfio_pci_device *device,
+ iova_t src, iova_t dst, u64 size)
+{
+ int ret;
+
+ ret = nv_falcon_dma(device, src, size, false);
+ if (ret)
+ return ret;
+
+ return nv_falcon_dma(device, dst, size, true);
+}
+
+static int nv_falcon_probe(struct vfio_pci_device *device)
+{
+ enum gpu_arch gpu_arch;
+ u32 pmc_boot_0;
+ void *bar0;
+ int i;
+
+ if (vfio_pci_config_readw(device, PCI_VENDOR_ID) !=
+ PCI_VENDOR_ID_NVIDIA)
+ return -ENODEV;
+
+ if (vfio_pci_config_readw(device, PCI_CLASS_DEVICE) >> 8 !=
+ PCI_BASE_CLASS_DISPLAY)
+ return -ENODEV;
+
+ /* Get BAR0 pointer for reading GPU registers */
+ bar0 = device->bars[0].vaddr;
+ if (!bar0)
+ return -ENODEV;
+
+ /* Read PMC_BOOT_0 register from BAR0 to identify GPU */
+ pmc_boot_0 = readl(bar0 + NV_PMC_BOOT_0);
+
+ /* Look up GPU architecture to verify this is a supported GPU */
+ gpu_arch = nv_gpu_arch_lookup(pmc_boot_0);
+ if (gpu_arch == GPU_ARCH_UNKNOWN) {
+ dev_err(device,
+ "Unsupported GPU architecture for PMC_BOOT_0: 0x%x\n",
+ pmc_boot_0);
+ return -ENODEV;
+ }
+
+ /* Check verified GPU map */
+ for (i = 0; i < VERIFIED_GPU_MAP_SIZE; i++) {
+ if (verified_gpu_map[i] == pmc_boot_0)
+ return 0;
+ }
+
+ dev_info(device,
+ "Unvalidated GPU: PMC_BOOT_0: 0x%x, possibly not supported\n",
+ pmc_boot_0);
+
+ return 0;
+}
+
+static void nv_falcon_init(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ const struct gpu_properties *props;
+ u32 pmc_boot_0;
+ int ret;
+
+ VFIO_ASSERT_GE(device->driver.region.size, sizeof(*gpu));
+
+ /* Read PMC_BOOT_0 register from BAR0 to identify GPU */
+ pmc_boot_0 = readl(device->bars[0].vaddr + NV_PMC_BOOT_0);
+
+ /* Look up GPU architecture */
+ gpu->arch = nv_gpu_arch_lookup(pmc_boot_0);
+
+ props = &gpu_properties_map[gpu->arch];
+
+ /* Populate GPU structure */
+ gpu->bar0 = device->bars[0].vaddr;
+ gpu->is_memory_clear_supported = props->memory_clear_supported;
+ gpu->falcon = &falcon_map[props->falcon_type];
+ gpu->pmc_enable_mask = props->pmc_enable_mask;
+
+ /* Initialize falcon for DMA */
+ ret = nv_falcon_dma_init(device);
+ VFIO_ASSERT_EQ(ret, 0, "Failed to initialize falcon DMA: %d\n", ret);
+
+ device->driver.max_memcpy_size = NV_FALCON_DMA_MAX_TRANSFER_SIZE;
+ device->driver.max_memcpy_count = NV_FALCON_DMA_MAX_TRANSFER_COUNT;
+}
+
+static void nv_falcon_remove(struct vfio_pci_device *device)
+{
+ falcon_disable(device);
+ vfio_pci_cmd_clear(device, PCI_COMMAND_MASTER);
+}
+
+/*
+ * Falcon DMA can only process one transfer at a time,
+ * so the actual work is deferred to memcpy_wait() to conform to the
+ * memcpy_start()/memcpy_wait() contract.
+ */
+static void nv_falcon_memcpy_start(struct vfio_pci_device *device,
+ iova_t src, iova_t dst, u64 size, u64 count)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+
+ VFIO_ASSERT_EQ(count, 1);
+ VFIO_ASSERT_EQ(size & (NV_FALCON_DMA_MIN_TRANSFER_SIZE - 1), 0,
+ "size 0x%lx must be %u-byte aligned\n",
+ (unsigned long)size, NV_FALCON_DMA_MIN_TRANSFER_SIZE);
+
+ gpu->memcpy_src = src;
+ gpu->memcpy_dst = dst;
+ gpu->memcpy_size = size;
+}
+
+/*
+ * Return the largest power-of-2 bytes we can transfer from @addr
+ * without crossing a DMA block boundary.
+ */
+static u64 dma_block_remain(u64 addr)
+{
+ u64 offset = addr & (NV_FALCON_DMA_BLOCK_SIZE - 1);
+
+ if (!offset)
+ return NV_FALCON_DMA_BLOCK_SIZE;
+
+ /* Lowest set bit of the offset is the largest aligned chunk */
+ return 1ULL << (ffs(offset) - 1);
+}
+
+static u64 rounddown_pow_of_two(u64 x)
+{
+ return 1ULL << (63 - __builtin_clzll(x));
+}
+
+static int nv_falcon_memcpy_wait(struct vfio_pci_device *device)
+{
+ struct gpu_device *gpu = to_gpu_device(device);
+ iova_t src = gpu->memcpy_src;
+ iova_t dst = gpu->memcpy_dst;
+ u64 remaining = gpu->memcpy_size;
+ int ret = 0;
+
+ /*
+ * Falcon DMA supports power-of-2 transfer sizes in [4, 256] and
+ * cannot cross 256-byte block boundaries. Decompose the request
+ * into the largest valid chunk at each step.
+ */
+ while (remaining) {
+ u64 chunk = rounddown_pow_of_two(remaining);
+
+ chunk = min(chunk, dma_block_remain(src));
+ chunk = min(chunk, dma_block_remain(dst));
+
+ ret = nv_falcon_memcpy_chunk(device, src, dst, chunk);
+ if (ret)
+ break;
+
+ src += chunk;
+ dst += chunk;
+ remaining -= chunk;
+ }
+
+ return ret;
+}
+
+const struct vfio_pci_driver_ops nv_falcon_ops = {
+ .name = "nv_falcon",
+ .probe = nv_falcon_probe,
+ .init = nv_falcon_init,
+ .remove = nv_falcon_remove,
+ .memcpy_start = nv_falcon_memcpy_start,
+ .memcpy_wait = nv_falcon_memcpy_wait,
+};
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio.h b/tools/testing/selftests/vfio/lib/include/libvfio.h
index 279ddcd70194..07862b470777 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio.h
@@ -5,6 +5,7 @@
#include <libvfio/assert.h>
#include <libvfio/iommu.h>
#include <libvfio/iova_allocator.h>
+#include <libvfio/sysfs.h>
#include <libvfio/vfio_pci_device.h>
#include <libvfio/vfio_pci_driver.h>
@@ -23,4 +24,13 @@
const char *vfio_selftests_get_bdf(int *argc, char *argv[]);
char **vfio_selftests_get_bdfs(int *argc, char *argv[], int *nr_bdfs);
+/*
+ * Reserve virtual address space of size at an address satisfying
+ * (vaddr % align) == offset.
+ *
+ * Returns the reserved vaddr. The caller is responsible for unmapping
+ * the returned region.
+ */
+void *mmap_reserve(size_t size, size_t align, size_t offset);
+
#endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_H */
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/assert.h b/tools/testing/selftests/vfio/lib/include/libvfio/assert.h
index f4ebd122d9b6..9fff88f6e4e1 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/assert.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/assert.h
@@ -3,6 +3,7 @@
#define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_ASSERT_H
#include <stdio.h>
+#include <stdlib.h>
#include <string.h>
#include <sys/ioctl.h>
@@ -45,10 +46,32 @@
VFIO_LOG_AND_EXIT(_fmt, ##__VA_ARGS__); \
} while (0)
+#define malloc_assert(_size) ({ \
+ size_t __size = (_size); \
+ void *__ptr = malloc(__size); \
+ VFIO_ASSERT_NOT_NULL(__ptr, "malloc(%zu) failed", \
+ __size); \
+ __ptr; \
+})
+
+#define calloc_assert(_nmemb, _size) ({ \
+ size_t __nmemb = (_nmemb); \
+ size_t __size = (_size); \
+ void *__ptr = calloc(__nmemb, __size); \
+ VFIO_ASSERT_NOT_NULL(__ptr, "calloc(%zu, %zu) failed", \
+ __nmemb, __size); \
+ __ptr; \
+})
+
#define ioctl_assert(_fd, _op, _arg) do { \
void *__arg = (_arg); \
int __ret = ioctl((_fd), (_op), (__arg)); \
VFIO_ASSERT_EQ(__ret, 0, "ioctl(%s, %s, %s) returned %d\n", #_fd, #_op, #_arg, __ret); \
} while (0)
+#define snprintf_assert(_s, _size, _fmt, ...) do { \
+ int __ret = snprintf(_s, _size, _fmt, ##__VA_ARGS__); \
+ VFIO_ASSERT_LT(__ret, _size); \
+} while (0)
+
#endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_ASSERT_H */
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h b/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h
index 5c9b9dc6d993..e9a3386a4719 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h
@@ -61,6 +61,12 @@ iova_t iommu_hva2iova(struct iommu *iommu, void *vaddr);
struct iommu_iova_range *iommu_iova_ranges(struct iommu *iommu, u32 *nranges);
+#define MODE_VFIO_TYPE1_IOMMU "vfio_type1_iommu"
+#define MODE_VFIO_TYPE1V2_IOMMU "vfio_type1v2_iommu"
+#define MODE_IOMMUFD_COMPAT_TYPE1 "iommufd_compat_type1"
+#define MODE_IOMMUFD_COMPAT_TYPE1V2 "iommufd_compat_type1v2"
+#define MODE_IOMMUFD "iommufd"
+
/*
* Generator for VFIO selftests fixture variants that replicate across all
* possible IOMMU modes. Tests must define FIXTURE_VARIANT_ADD_IOMMU_MODE()
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h b/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h
index 8f1d994e9ea2..c7c0796a757f 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h
@@ -2,7 +2,6 @@
#ifndef SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_IOVA_ALLOCATOR_H
#define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_IOVA_ALLOCATOR_H
-#include <uapi/linux/types.h>
#include <linux/list.h>
#include <linux/types.h>
#include <linux/iommufd.h>
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h b/tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h
new file mode 100644
index 000000000000..c9ab1ea8f5a9
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h
@@ -0,0 +1,12 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+#ifndef SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_SYSFS_H
+#define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_SYSFS_H
+
+int sysfs_sriov_totalvfs_get(const char *bdf);
+int sysfs_sriov_numvfs_get(const char *bdf);
+void sysfs_sriov_numvfs_set(const char *bdf, int numvfs);
+char *sysfs_sriov_vf_bdf_get(const char *pf_bdf, int i);
+int sysfs_iommu_group_get(const char *bdf);
+char *sysfs_driver_get(const char *bdf);
+
+#endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_SYSFS_H */
diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
index 2858885a89bb..e19bd94b8dd2 100644
--- a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
+++ b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h
@@ -38,9 +38,12 @@ struct vfio_pci_device {
#define dev_info(_dev, _fmt, ...) printf("%s: " _fmt, (_dev)->bdf, ##__VA_ARGS__)
#define dev_err(_dev, _fmt, ...) fprintf(stderr, "%s: " _fmt, (_dev)->bdf, ##__VA_ARGS__)
+struct vfio_pci_device *vfio_pci_device_alloc(const char *bdf, struct iommu *iommu);
+void vfio_pci_device_free(struct vfio_pci_device *device);
struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu);
void vfio_pci_device_cleanup(struct vfio_pci_device *device);
+int __vfio_pci_device_reset(struct vfio_pci_device *device);
void vfio_pci_device_reset(struct vfio_pci_device *device);
void vfio_pci_config_access(struct vfio_pci_device *device, bool write,
@@ -65,9 +68,25 @@ void vfio_pci_config_access(struct vfio_pci_device *device, bool write,
#define vfio_pci_config_writew(_d, _o, _v) vfio_pci_config_write(_d, _o, _v, u16)
#define vfio_pci_config_writel(_d, _o, _v) vfio_pci_config_write(_d, _o, _v, u32)
+static inline void vfio_pci_cmd_set(struct vfio_pci_device *device, u16 bits)
+{
+ u16 cmd = vfio_pci_config_readw(device, PCI_COMMAND);
+
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd | bits);
+}
+
+static inline void vfio_pci_cmd_clear(struct vfio_pci_device *device, u16 bits)
+{
+ u16 cmd = vfio_pci_config_readw(device, PCI_COMMAND);
+
+ vfio_pci_config_writew(device, PCI_COMMAND, cmd & ~bits);
+}
+
void vfio_pci_irq_enable(struct vfio_pci_device *device, u32 index,
u32 vector, int count);
void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index);
+void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index,
+ u32 vector, int count);
void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector);
static inline void fcntl_set_nonblock(int fd)
@@ -92,6 +111,12 @@ static inline void vfio_pci_msi_disable(struct vfio_pci_device *device)
vfio_pci_irq_disable(device, VFIO_PCI_MSI_IRQ_INDEX);
}
+static inline void vfio_pci_msi_reenable(struct vfio_pci_device *device,
+ u32 vector, int count)
+{
+ vfio_pci_irq_reenable(device, VFIO_PCI_MSI_IRQ_INDEX, vector, count);
+}
+
static inline void vfio_pci_msix_enable(struct vfio_pci_device *device,
u32 vector, int count)
{
@@ -103,6 +128,12 @@ static inline void vfio_pci_msix_disable(struct vfio_pci_device *device)
vfio_pci_irq_disable(device, VFIO_PCI_MSIX_IRQ_INDEX);
}
+static inline void vfio_pci_msix_reenable(struct vfio_pci_device *device,
+ u32 vector, int count)
+{
+ vfio_pci_irq_reenable(device, VFIO_PCI_MSIX_IRQ_INDEX, vector, count);
+}
+
static inline int __to_iova(struct vfio_pci_device *device, void *vaddr, iova_t *iova)
{
return __iommu_hva2iova(device->iommu, vaddr, iova);
@@ -122,4 +153,13 @@ static inline bool vfio_pci_device_match(struct vfio_pci_device *device,
const char *vfio_pci_get_cdev_path(const char *bdf);
+void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf);
+void __vfio_pci_group_get_device_fd(struct vfio_pci_device *device,
+ const char *bdf, const char *vf_token);
+void vfio_container_set_iommu(struct vfio_pci_device *device);
+void vfio_pci_cdev_open(struct vfio_pci_device *device, const char *bdf);
+int __vfio_device_bind_iommufd(int device_fd, int iommufd, const char *vf_token);
+
+void vfio_device_set_vf_token(int fd, const char *vf_token);
+
#endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_VFIO_PCI_DEVICE_H */
diff --git a/tools/testing/selftests/vfio/lib/iommu.c b/tools/testing/selftests/vfio/lib/iommu.c
index 8079d43523f3..b6f3c5c84e01 100644
--- a/tools/testing/selftests/vfio/lib/iommu.c
+++ b/tools/testing/selftests/vfio/lib/iommu.c
@@ -11,7 +11,6 @@
#include <sys/ioctl.h>
#include <sys/mman.h>
-#include <uapi/linux/types.h>
#include <linux/limits.h>
#include <linux/mman.h>
#include <linux/types.h>
@@ -21,32 +20,32 @@
#include "../../../kselftest.h"
#include <libvfio.h>
-const char *default_iommu_mode = "iommufd";
+const char *default_iommu_mode = MODE_IOMMUFD;
/* Reminder: Keep in sync with FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(). */
static const struct iommu_mode iommu_modes[] = {
{
- .name = "vfio_type1_iommu",
+ .name = MODE_VFIO_TYPE1_IOMMU,
.container_path = "/dev/vfio/vfio",
.iommu_type = VFIO_TYPE1_IOMMU,
},
{
- .name = "vfio_type1v2_iommu",
+ .name = MODE_VFIO_TYPE1V2_IOMMU,
.container_path = "/dev/vfio/vfio",
.iommu_type = VFIO_TYPE1v2_IOMMU,
},
{
- .name = "iommufd_compat_type1",
+ .name = MODE_IOMMUFD_COMPAT_TYPE1,
.container_path = "/dev/iommu",
.iommu_type = VFIO_TYPE1_IOMMU,
},
{
- .name = "iommufd_compat_type1v2",
+ .name = MODE_IOMMUFD_COMPAT_TYPE1V2,
.container_path = "/dev/iommu",
.iommu_type = VFIO_TYPE1v2_IOMMU,
},
{
- .name = "iommufd",
+ .name = MODE_IOMMUFD,
},
};
@@ -287,8 +286,7 @@ static struct vfio_iommu_type1_info *vfio_iommu_get_info(int container_fd)
{
struct vfio_iommu_type1_info *info;
- info = malloc(sizeof(*info));
- VFIO_ASSERT_NOT_NULL(info);
+ info = malloc_assert(sizeof(*info));
*info = (struct vfio_iommu_type1_info) {
.argsz = sizeof(*info),
@@ -325,8 +323,7 @@ static struct iommu_iova_range *vfio_iommu_iova_ranges(struct iommu *iommu,
cap_range = container_of(hdr, struct vfio_iommu_type1_info_cap_iova_range, header);
VFIO_ASSERT_GT(cap_range->nr_iovas, 0);
- ranges = calloc(cap_range->nr_iovas, sizeof(*ranges));
- VFIO_ASSERT_NOT_NULL(ranges);
+ ranges = calloc_assert(cap_range->nr_iovas, sizeof(*ranges));
for (u32 i = 0; i < cap_range->nr_iovas; i++) {
ranges[i] = (struct iommu_iova_range){
@@ -358,8 +355,7 @@ static struct iommu_iova_range *iommufd_iova_ranges(struct iommu *iommu,
VFIO_ASSERT_EQ(errno, EMSGSIZE);
VFIO_ASSERT_GT(query.num_iovas, 0);
- ranges = calloc(query.num_iovas, sizeof(*ranges));
- VFIO_ASSERT_NOT_NULL(ranges);
+ ranges = calloc_assert(query.num_iovas, sizeof(*ranges));
query.allowed_iovas = (uintptr_t)ranges;
@@ -425,8 +421,7 @@ struct iommu *iommu_init(const char *iommu_mode)
struct iommu *iommu;
int version;
- iommu = calloc(1, sizeof(*iommu));
- VFIO_ASSERT_NOT_NULL(iommu);
+ iommu = calloc_assert(1, sizeof(*iommu));
INIT_LIST_HEAD(&iommu->dma_regions);
diff --git a/tools/testing/selftests/vfio/lib/iova_allocator.c b/tools/testing/selftests/vfio/lib/iova_allocator.c
index a12b0a51e9e6..4a660f636f49 100644
--- a/tools/testing/selftests/vfio/lib/iova_allocator.c
+++ b/tools/testing/selftests/vfio/lib/iova_allocator.c
@@ -11,7 +11,6 @@
#include <sys/ioctl.h>
#include <sys/mman.h>
-#include <uapi/linux/types.h>
#include <linux/iommufd.h>
#include <linux/limits.h>
#include <linux/mman.h>
@@ -30,8 +29,7 @@ struct iova_allocator *iova_allocator_init(struct iommu *iommu)
ranges = iommu_iova_ranges(iommu, &nranges);
VFIO_ASSERT_NOT_NULL(ranges);
- allocator = malloc(sizeof(*allocator));
- VFIO_ASSERT_NOT_NULL(allocator);
+ allocator = malloc_assert(sizeof(*allocator));
*allocator = (struct iova_allocator){
.ranges = ranges,
@@ -91,4 +89,3 @@ next_range:
allocator->range_offset = 0;
}
}
-
diff --git a/tools/testing/selftests/vfio/lib/libvfio.c b/tools/testing/selftests/vfio/lib/libvfio.c
index a23a3cc5be69..3a3d1ed635c1 100644
--- a/tools/testing/selftests/vfio/lib/libvfio.c
+++ b/tools/testing/selftests/vfio/lib/libvfio.c
@@ -2,6 +2,9 @@
#include <stdio.h>
#include <stdlib.h>
+#include <sys/mman.h>
+
+#include <linux/align.h>
#include "../../../kselftest.h"
#include <libvfio.h>
@@ -76,3 +79,25 @@ const char *vfio_selftests_get_bdf(int *argc, char *argv[])
return vfio_selftests_get_bdfs(argc, argv, &nr_bdfs)[0];
}
+
+void *mmap_reserve(size_t size, size_t align, size_t offset)
+{
+ void *map_base, *map_align;
+ size_t delta;
+
+ VFIO_ASSERT_GT(align, offset);
+ delta = align - offset;
+
+ map_base = mmap(NULL, size + align, PROT_NONE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ VFIO_ASSERT_NE(map_base, MAP_FAILED);
+
+ map_align = (void *)(ALIGN((uintptr_t)map_base + delta, align) - delta);
+
+ if (map_align > map_base)
+ VFIO_ASSERT_EQ(munmap(map_base, map_align - map_base), 0);
+
+ VFIO_ASSERT_EQ(munmap(map_align + size, map_base + align - map_align), 0);
+
+ return map_align;
+}
diff --git a/tools/testing/selftests/vfio/lib/libvfio.mk b/tools/testing/selftests/vfio/lib/libvfio.mk
index 9f47bceed16f..bcfa74ae040e 100644
--- a/tools/testing/selftests/vfio/lib/libvfio.mk
+++ b/tools/testing/selftests/vfio/lib/libvfio.mk
@@ -6,6 +6,7 @@ LIBVFIO_SRCDIR := $(selfdir)/vfio/lib
LIBVFIO_C := iommu.c
LIBVFIO_C += iova_allocator.c
LIBVFIO_C += libvfio.c
+LIBVFIO_C += sysfs.c
LIBVFIO_C += vfio_pci_device.c
LIBVFIO_C += vfio_pci_driver.c
@@ -14,16 +15,23 @@ LIBVFIO_C += drivers/ioat/ioat.c
LIBVFIO_C += drivers/dsa/dsa.c
endif
+LIBVFIO_C += drivers/nv_falcon/nv_falcon.c
+LIBVFIO_C += drivers/igb/igb.c
+
LIBVFIO_OUTPUT := $(OUTPUT)/libvfio
LIBVFIO_O := $(patsubst %.c, $(LIBVFIO_OUTPUT)/%.o, $(LIBVFIO_C))
LIBVFIO_O_DIRS := $(shell dirname $(LIBVFIO_O) | uniq)
-$(shell mkdir -p $(LIBVFIO_O_DIRS))
+
+$(LIBVFIO_O_DIRS):
+ mkdir -p $@
CFLAGS += -I$(LIBVFIO_SRCDIR)/include
-$(LIBVFIO_O): $(LIBVFIO_OUTPUT)/%.o : $(LIBVFIO_SRCDIR)/%.c
+LDLIBS += -luuid
+
+$(LIBVFIO_O): $(LIBVFIO_OUTPUT)/%.o : $(LIBVFIO_SRCDIR)/%.c | $(LIBVFIO_O_DIRS)
$(CC) $(CFLAGS) $(CPPFLAGS) $(TARGET_ARCH) -c $< -o $@
EXTRA_CLEAN += $(LIBVFIO_OUTPUT)
diff --git a/tools/testing/selftests/vfio/lib/sysfs.c b/tools/testing/selftests/vfio/lib/sysfs.c
new file mode 100644
index 000000000000..98a46a2543cd
--- /dev/null
+++ b/tools/testing/selftests/vfio/lib/sysfs.c
@@ -0,0 +1,149 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include <fcntl.h>
+#include <unistd.h>
+#include <stdlib.h>
+#include <string.h>
+#include <linux/limits.h>
+
+#include <libvfio.h>
+
+#define readlink_safe(_path, _buf) ({ \
+ int __ret; \
+ \
+ _Static_assert(!__builtin_types_compatible_p( \
+ __typeof__(_buf), char *), \
+ "readlink_safe: _buf must be an array, not a pointer"); \
+ \
+ __ret = readlink(_path, _buf, sizeof(_buf) - 1); \
+ if (__ret != -1) \
+ _buf[__ret] = 0; \
+ __ret; \
+})
+
+static void readlink_base(const char *path, const char *data_fmt, void *out_data)
+{
+ char rl_path[PATH_MAX];
+ int ret;
+
+ ret = readlink_safe(path, rl_path);
+ VFIO_ASSERT_NE(ret, -1);
+
+ ret = sscanf(basename(rl_path), data_fmt, out_data);
+ VFIO_ASSERT_EQ(ret, 1);
+}
+
+static int sysfs_val_get_int(const char *component, const char *name,
+ const char *file)
+{
+ char path[PATH_MAX];
+ char buf[32];
+ int ret;
+ int fd;
+
+ snprintf_assert(path, PATH_MAX, "/sys/bus/pci/%s/%s/%s", component, name, file);
+ fd = open(path, O_RDONLY);
+ if (fd < 0)
+ return fd;
+
+ VFIO_ASSERT_GT(read(fd, buf, ARRAY_SIZE(buf)), 0);
+ VFIO_ASSERT_EQ(close(fd), 0);
+
+ errno = 0;
+ ret = strtol(buf, NULL, 0);
+ VFIO_ASSERT_EQ(errno, 0, "sysfs path \"%s\" is not an integer: \"%s\"\n", path, buf);
+
+ return ret;
+}
+
+static void sysfs_val_set(const char *component, const char *name,
+ const char *file, const char *val)
+{
+ char path[PATH_MAX];
+ int fd;
+
+ snprintf_assert(path, PATH_MAX, "/sys/bus/pci/%s/%s/%s", component, name, file);
+ VFIO_ASSERT_GT(fd = open(path, O_WRONLY), 0);
+
+ VFIO_ASSERT_EQ(write(fd, val, strlen(val)), strlen(val));
+ VFIO_ASSERT_EQ(close(fd), 0);
+}
+
+static int sysfs_device_val_get(const char *bdf, const char *file)
+{
+ return sysfs_val_get_int("devices", bdf, file);
+}
+
+static void sysfs_device_val_set(const char *bdf, const char *file, const char *val)
+{
+ sysfs_val_set("devices", bdf, file, val);
+}
+
+static void sysfs_device_val_set_int(const char *bdf, const char *file, int val)
+{
+ char val_str[32];
+
+ snprintf_assert(val_str, sizeof(val_str), "%d", val);
+ sysfs_device_val_set(bdf, file, val_str);
+}
+
+int sysfs_sriov_totalvfs_get(const char *bdf)
+{
+ return sysfs_device_val_get(bdf, "sriov_totalvfs");
+}
+
+int sysfs_sriov_numvfs_get(const char *bdf)
+{
+ return sysfs_device_val_get(bdf, "sriov_numvfs");
+}
+
+void sysfs_sriov_numvfs_set(const char *bdf, int numvfs)
+{
+ sysfs_device_val_set_int(bdf, "sriov_numvfs", numvfs);
+}
+
+char *sysfs_sriov_vf_bdf_get(const char *pf_bdf, int i)
+{
+ char path[PATH_MAX];
+ char *out_vf_bdf;
+
+ /* Fit "0000:00:00.0" */
+ out_vf_bdf = calloc_assert(16, sizeof(char));
+
+ snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/virtfn%d", pf_bdf, i);
+ readlink_base(path, "%s", out_vf_bdf);
+
+ return out_vf_bdf;
+}
+
+int sysfs_iommu_group_get(const char *bdf)
+{
+ char path[PATH_MAX];
+ int group;
+
+ snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/iommu_group", bdf);
+ readlink_base(path, "%d", &group);
+
+ return group;
+}
+
+char *sysfs_driver_get(const char *bdf)
+{
+ char driver_path[PATH_MAX];
+ char path[PATH_MAX];
+ char *out_driver;
+ int ret;
+
+ snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/driver", bdf);
+ ret = readlink_safe(path, driver_path);
+ if (ret == -1) {
+ if (errno == ENOENT)
+ return NULL;
+
+ VFIO_FAIL("Failed to read %s\n", path);
+ }
+
+ out_driver = strdup(basename(driver_path));
+ VFIO_ASSERT_NOT_NULL(out_driver);
+
+ return out_driver;
+}
diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_device.c b/tools/testing/selftests/vfio/lib/vfio_pci_device.c
index 8e34b9bfc96b..4063a0e2b3df 100644
--- a/tools/testing/selftests/vfio/lib/vfio_pci_device.c
+++ b/tools/testing/selftests/vfio/lib/vfio_pci_device.c
@@ -1,5 +1,6 @@
// SPDX-License-Identifier: GPL-2.0-only
#include <dirent.h>
+#include <errno.h>
#include <fcntl.h>
#include <libgen.h>
#include <stdint.h>
@@ -11,27 +12,30 @@
#include <sys/ioctl.h>
#include <sys/mman.h>
-#include <uapi/linux/types.h>
+#include <linux/align.h>
#include <linux/iommufd.h>
+#include <linux/kernel.h>
#include <linux/limits.h>
+#include <linux/log2.h>
#include <linux/mman.h>
#include <linux/overflow.h>
+#include <linux/sizes.h>
#include <linux/types.h>
#include <linux/vfio.h>
+#include <uuid/uuid.h>
+
#include "kselftest.h"
#include <libvfio.h>
-#define PCI_SYSFS_PATH "/sys/bus/pci/devices"
-
static void vfio_pci_irq_set(struct vfio_pci_device *device,
u32 index, u32 vector, u32 count, int *fds)
{
- u8 buf[sizeof(struct vfio_irq_set) + sizeof(int) * count] = {};
- struct vfio_irq_set *irq = (void *)&buf;
- int *irq_fds = (void *)&irq->data;
+ size_t argsz = sizeof(struct vfio_irq_set) + sizeof(int) * count;
+ struct vfio_irq_set *irq;
- irq->argsz = sizeof(buf);
+ irq = calloc_assert(1, argsz);
+ irq->argsz = argsz;
irq->flags = VFIO_IRQ_SET_ACTION_TRIGGER;
irq->index = index;
irq->start = vector;
@@ -39,12 +43,13 @@ static void vfio_pci_irq_set(struct vfio_pci_device *device,
if (count) {
irq->flags |= VFIO_IRQ_SET_DATA_EVENTFD;
- memcpy(irq_fds, fds, sizeof(int) * count);
+ memcpy(irq->data, fds, sizeof(int) * count);
} else {
irq->flags |= VFIO_IRQ_SET_DATA_NONE;
}
ioctl_assert(device->fd, VFIO_DEVICE_SET_IRQS, irq);
+ free(irq);
}
void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector)
@@ -101,6 +106,28 @@ void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index)
vfio_pci_irq_set(device, index, 0, 0, NULL);
}
+/*
+ * Re-issue VFIO_DEVICE_SET_IRQS for an already-enabled vector range using
+ * the existing eventfds. Intended for drivers that need to re-arm device
+ * interrupts after a VFIO_DEVICE_RESET, which tears down the kernel-side
+ * IRQ trigger but leaves user-side eventfds intact. Recreating the
+ * eventfds would invalidate any test-fixture cache of the fd, so this
+ * helper deliberately preserves them.
+ */
+void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index,
+ u32 vector, int count)
+{
+ int i;
+
+ check_supported_irq_index(index);
+
+ for (i = vector; i < vector + count; i++)
+ VFIO_ASSERT_GE(device->msi_eventfds[i], 0,
+ "vector %d eventfd not allocated\n", i);
+
+ vfio_pci_irq_set(device, index, vector, count, device->msi_eventfds + vector);
+}
+
static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index,
struct vfio_irq_info *irq_info)
{
@@ -110,6 +137,45 @@ static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index,
ioctl_assert(device->fd, VFIO_DEVICE_GET_IRQ_INFO, irq_info);
}
+static int vfio_device_feature_ioctl(int fd, u32 flags, void *data,
+ size_t data_size)
+{
+ size_t argsz = sizeof(struct vfio_device_feature) + data_size;
+ struct vfio_device_feature *feature;
+ int ret;
+
+ feature = calloc_assert(1, argsz);
+ memcpy(feature->data, data, data_size);
+
+ feature->argsz = argsz;
+ feature->flags = flags;
+
+ ret = ioctl(fd, VFIO_DEVICE_FEATURE, feature);
+ free(feature);
+
+ return ret;
+}
+
+static void vfio_device_feature_set(int fd, u16 feature, void *data, size_t data_size)
+{
+ u32 flags = VFIO_DEVICE_FEATURE_SET | feature;
+ int ret;
+
+ ret = vfio_device_feature_ioctl(fd, flags, data, data_size);
+ VFIO_ASSERT_EQ(ret, 0, "Failed to set feature %u\n", feature);
+}
+
+void vfio_device_set_vf_token(int fd, const char *vf_token)
+{
+ uuid_t token_uuid = {0};
+
+ VFIO_ASSERT_NOT_NULL(vf_token, "vf_token is NULL");
+ VFIO_ASSERT_EQ(uuid_parse(vf_token, token_uuid), 0);
+
+ vfio_device_feature_set(fd, VFIO_DEVICE_FEATURE_PCI_VF_TOKEN,
+ token_uuid, sizeof(uuid_t));
+}
+
static void vfio_pci_region_get(struct vfio_pci_device *device, int index,
struct vfio_region_info *info)
{
@@ -124,20 +190,38 @@ static void vfio_pci_region_get(struct vfio_pci_device *device, int index,
static void vfio_pci_bar_map(struct vfio_pci_device *device, int index)
{
struct vfio_pci_bar *bar = &device->bars[index];
+ size_t align, size;
int prot = 0;
+ void *vaddr;
VFIO_ASSERT_LT(index, PCI_STD_NUM_BARS);
VFIO_ASSERT_NULL(bar->vaddr);
VFIO_ASSERT_TRUE(bar->info.flags & VFIO_REGION_INFO_FLAG_MMAP);
+ VFIO_ASSERT_TRUE(is_power_of_2(bar->info.size));
if (bar->info.flags & VFIO_REGION_INFO_FLAG_READ)
prot |= PROT_READ;
if (bar->info.flags & VFIO_REGION_INFO_FLAG_WRITE)
prot |= PROT_WRITE;
- bar->vaddr = mmap(NULL, bar->info.size, prot, MAP_FILE | MAP_SHARED,
+ size = bar->info.size;
+
+ /*
+ * Align BAR mmaps to improve page fault granularity during potential
+ * subsequent IOMMU mapping of these BAR vaddr. 1G for x86 is the
+ * largest hugepage size across any architecture, so no benefit from
+ * larger alignment. BARs smaller than 1G will be aligned by their
+ * power-of-two size, guaranteeing sufficient alignment for smaller
+ * hugepages, if present.
+ */
+ align = min_t(size_t, size, SZ_1G);
+
+ vaddr = mmap_reserve(size, align, 0);
+ bar->vaddr = mmap(vaddr, size, prot, MAP_SHARED | MAP_FIXED,
device->fd, bar->info.offset);
VFIO_ASSERT_NE(bar->vaddr, MAP_FAILED);
+
+ madvise(bar->vaddr, size, MADV_HUGEPAGE);
}
static void vfio_pci_bar_unmap(struct vfio_pci_device *device, int index)
@@ -176,30 +260,29 @@ void vfio_pci_config_access(struct vfio_pci_device *device, bool write,
write ? "write to" : "read from", config);
}
-void vfio_pci_device_reset(struct vfio_pci_device *device)
+int __vfio_pci_device_reset(struct vfio_pci_device *device)
{
- ioctl_assert(device->fd, VFIO_DEVICE_RESET, NULL);
+ if (ioctl(device->fd, VFIO_DEVICE_RESET, NULL))
+ return -errno;
+
+ return 0;
}
-static unsigned int vfio_pci_get_group_from_dev(const char *bdf)
+void vfio_pci_device_reset(struct vfio_pci_device *device)
{
- char dev_iommu_group_path[PATH_MAX] = {0};
- char sysfs_path[PATH_MAX] = {0};
- unsigned int group;
- int ret;
-
- snprintf(sysfs_path, PATH_MAX, "%s/%s/iommu_group", PCI_SYSFS_PATH, bdf);
+ int retries = 20;
+ int r;
- ret = readlink(sysfs_path, dev_iommu_group_path, sizeof(dev_iommu_group_path));
- VFIO_ASSERT_NE(ret, -1, "Failed to get the IOMMU group for device: %s\n", bdf);
+ do {
+ r = __vfio_pci_device_reset(device);
+ if (r == -EAGAIN)
+ usleep(10000);
+ } while (r == -EAGAIN && retries-- > 0);
- ret = sscanf(basename(dev_iommu_group_path), "%u", &group);
- VFIO_ASSERT_EQ(ret, 1, "Failed to get the IOMMU group for device: %s\n", bdf);
-
- return group;
+ VFIO_ASSERT_EQ(r, 0, "ioctl(device->fd, VFIO_DEVICE_RESET) failed\n");
}
-static void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf)
+void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf)
{
struct vfio_group_status group_status = {
.argsz = sizeof(group_status),
@@ -207,8 +290,8 @@ static void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf
char group_path[32];
int group;
- group = vfio_pci_get_group_from_dev(bdf);
- snprintf(group_path, sizeof(group_path), "/dev/vfio/%d", group);
+ group = sysfs_iommu_group_get(bdf);
+ snprintf_assert(group_path, sizeof(group_path), "/dev/vfio/%d", group);
device->group_fd = open(group_path, O_RDWR);
VFIO_ASSERT_GE(device->group_fd, 0, "open(%s) failed\n", group_path);
@@ -219,14 +302,37 @@ static void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf
ioctl_assert(device->group_fd, VFIO_GROUP_SET_CONTAINER, &device->iommu->container_fd);
}
-static void vfio_pci_container_setup(struct vfio_pci_device *device, const char *bdf)
+void __vfio_pci_group_get_device_fd(struct vfio_pci_device *device,
+ const char *bdf, const char *vf_token)
+{
+ char arg[64];
+
+ /*
+ * If a vf_token exists, argument to VFIO_GROUP_GET_DEVICE_FD
+ * will be in the form of the following example:
+ * "0000:04:10.0 vf_token=bd8d9d2b-5a5f-4f5a-a211-f591514ba1f3"
+ */
+ if (vf_token)
+ snprintf_assert(arg, ARRAY_SIZE(arg), "%s vf_token=%s", bdf, vf_token);
+ else
+ snprintf_assert(arg, ARRAY_SIZE(arg), "%s", bdf);
+
+ device->fd = ioctl(device->group_fd, VFIO_GROUP_GET_DEVICE_FD, arg);
+}
+
+static void vfio_pci_group_get_device_fd(struct vfio_pci_device *device,
+ const char *bdf, const char *vf_token)
+{
+ __vfio_pci_group_get_device_fd(device, bdf, vf_token);
+ VFIO_ASSERT_GE(device->fd, 0);
+}
+
+void vfio_container_set_iommu(struct vfio_pci_device *device)
{
struct iommu *iommu = device->iommu;
unsigned long iommu_type = iommu->mode->iommu_type;
int ret;
- vfio_pci_group_setup(device, bdf);
-
ret = ioctl(iommu->container_fd, VFIO_CHECK_EXTENSION, iommu_type);
VFIO_ASSERT_GT(ret, 0, "VFIO IOMMU type %lu not supported\n", iommu_type);
@@ -236,9 +342,14 @@ static void vfio_pci_container_setup(struct vfio_pci_device *device, const char
* because the IOMMU type is already set.
*/
(void)ioctl(iommu->container_fd, VFIO_SET_IOMMU, (void *)iommu_type);
+}
- device->fd = ioctl(device->group_fd, VFIO_GROUP_GET_DEVICE_FD, bdf);
- VFIO_ASSERT_GE(device->fd, 0);
+static void vfio_pci_container_setup(struct vfio_pci_device *device,
+ const char *bdf, const char *vf_token)
+{
+ vfio_pci_group_setup(device, bdf);
+ vfio_container_set_iommu(device);
+ vfio_pci_group_get_device_fd(device, bdf, vf_token);
}
static void vfio_pci_device_setup(struct vfio_pci_device *device)
@@ -276,10 +387,9 @@ const char *vfio_pci_get_cdev_path(const char *bdf)
char *cdev_path;
DIR *dir;
- cdev_path = calloc(PATH_MAX, 1);
- VFIO_ASSERT_NOT_NULL(cdev_path);
+ cdev_path = calloc_assert(PATH_MAX, 1);
- snprintf(dir_path, sizeof(dir_path), "/sys/bus/pci/devices/%s/vfio-dev/", bdf);
+ snprintf_assert(dir_path, sizeof(dir_path), "/sys/bus/pci/devices/%s/vfio-dev/", bdf);
dir = opendir(dir_path);
VFIO_ASSERT_NOT_NULL(dir, "Failed to open directory %s\n", dir_path);
@@ -289,7 +399,7 @@ const char *vfio_pci_get_cdev_path(const char *bdf)
if (strncmp("vfio", entry->d_name, 4))
continue;
- snprintf(cdev_path, PATH_MAX, "/dev/vfio/devices/%s", entry->d_name);
+ snprintf_assert(cdev_path, PATH_MAX, "/dev/vfio/devices/%s", entry->d_name);
break;
}
@@ -299,14 +409,32 @@ const char *vfio_pci_get_cdev_path(const char *bdf)
return cdev_path;
}
-static void vfio_device_bind_iommufd(int device_fd, int iommufd)
+int __vfio_device_bind_iommufd(int device_fd, int iommufd, const char *vf_token)
{
struct vfio_device_bind_iommufd args = {
.argsz = sizeof(args),
.iommufd = iommufd,
};
+ uuid_t token_uuid;
+
+ if (vf_token) {
+ VFIO_ASSERT_EQ(uuid_parse(vf_token, token_uuid), 0);
+ args.flags |= VFIO_DEVICE_BIND_FLAG_TOKEN;
+ args.token_uuid_ptr = (u64)token_uuid;
+ }
+
+ if (ioctl(device_fd, VFIO_DEVICE_BIND_IOMMUFD, &args))
+ return -errno;
- ioctl_assert(device_fd, VFIO_DEVICE_BIND_IOMMUFD, &args);
+ return 0;
+}
+
+static void vfio_device_bind_iommufd(int device_fd, int iommufd,
+ const char *vf_token)
+{
+ int ret = __vfio_device_bind_iommufd(device_fd, iommufd, vf_token);
+
+ VFIO_ASSERT_EQ(ret, 0, "Failed VFIO_DEVICE_BIND_IOMMUFD ioctl\n");
}
static void vfio_device_attach_iommufd_pt(int device_fd, u32 pt_id)
@@ -319,33 +447,51 @@ static void vfio_device_attach_iommufd_pt(int device_fd, u32 pt_id)
ioctl_assert(device_fd, VFIO_DEVICE_ATTACH_IOMMUFD_PT, &args);
}
-static void vfio_pci_iommufd_setup(struct vfio_pci_device *device, const char *bdf)
+void vfio_pci_cdev_open(struct vfio_pci_device *device, const char *bdf)
{
const char *cdev_path = vfio_pci_get_cdev_path(bdf);
device->fd = open(cdev_path, O_RDWR);
VFIO_ASSERT_GE(device->fd, 0);
free((void *)cdev_path);
+}
- vfio_device_bind_iommufd(device->fd, device->iommu->iommufd);
+static void vfio_pci_iommufd_setup(struct vfio_pci_device *device,
+ const char *bdf, const char *vf_token)
+{
+ vfio_pci_cdev_open(device, bdf);
+ vfio_device_bind_iommufd(device->fd, device->iommu->iommufd, vf_token);
vfio_device_attach_iommufd_pt(device->fd, device->iommu->ioas_id);
}
-struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu)
+struct vfio_pci_device *vfio_pci_device_alloc(const char *bdf, struct iommu *iommu)
{
struct vfio_pci_device *device;
- device = calloc(1, sizeof(*device));
- VFIO_ASSERT_NOT_NULL(device);
+ device = calloc_assert(1, sizeof(*device));
VFIO_ASSERT_NOT_NULL(iommu);
device->iommu = iommu;
device->bdf = bdf;
+ return device;
+}
+
+void vfio_pci_device_free(struct vfio_pci_device *device)
+{
+ free(device);
+}
+
+struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu)
+{
+ struct vfio_pci_device *device;
+
+ device = vfio_pci_device_alloc(bdf, iommu);
+
if (iommu->mode->container_path)
- vfio_pci_container_setup(device, bdf);
+ vfio_pci_container_setup(device, bdf, NULL);
else
- vfio_pci_iommufd_setup(device, bdf);
+ vfio_pci_iommufd_setup(device, bdf, NULL);
vfio_pci_device_setup(device);
vfio_pci_driver_probe(device);
@@ -374,5 +520,5 @@ void vfio_pci_device_cleanup(struct vfio_pci_device *device)
if (device->group_fd)
VFIO_ASSERT_EQ(close(device->group_fd), 0);
- free(device);
+ vfio_pci_device_free(device);
}
diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
index 6827f4a6febe..5e65434d2318 100644
--- a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
+++ b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c
@@ -6,12 +6,16 @@
extern struct vfio_pci_driver_ops dsa_ops;
extern struct vfio_pci_driver_ops ioat_ops;
#endif
+extern struct vfio_pci_driver_ops nv_falcon_ops;
+extern struct vfio_pci_driver_ops igb_ops;
static struct vfio_pci_driver_ops *driver_ops[] = {
#ifdef __x86_64__
&dsa_ops,
&ioat_ops,
#endif
+ &nv_falcon_ops,
+ &igb_ops,
};
void vfio_pci_driver_probe(struct vfio_pci_device *device)
@@ -106,7 +110,21 @@ int vfio_pci_driver_memcpy_wait(struct vfio_pci_device *device)
int vfio_pci_driver_memcpy(struct vfio_pci_device *device,
iova_t src, iova_t dst, u64 size)
{
- vfio_pci_driver_memcpy_start(device, src, dst, size, 1);
+ struct vfio_pci_driver *driver = &device->driver;
+ u64 offset = 0;
+
+ while (offset < size) {
+ u64 chunk = min(size - offset, driver->max_memcpy_size);
+ int ret;
+
+ vfio_pci_driver_memcpy_start(device, src + offset,
+ dst + offset, chunk, 1);
+ ret = vfio_pci_driver_memcpy_wait(device);
+ if (ret)
+ return ret;
+
+ offset += chunk;
+ }
- return vfio_pci_driver_memcpy_wait(device);
+ return 0;
}
diff --git a/tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c b/tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c
new file mode 100644
index 000000000000..d7f25ef77671
--- /dev/null
+++ b/tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c
@@ -0,0 +1,142 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include <stdio.h>
+#include <sys/mman.h>
+#include <unistd.h>
+
+#include <uapi/linux/types.h>
+#include <linux/pci_regs.h>
+#include <linux/sizes.h>
+#include <linux/vfio.h>
+
+#include <libvfio.h>
+
+#include "../kselftest_harness.h"
+
+static const char *device_bdf;
+
+static struct vfio_pci_bar *largest_mapped_bar(struct vfio_pci_device *device)
+{
+ u32 flags = VFIO_REGION_INFO_FLAG_READ | VFIO_REGION_INFO_FLAG_WRITE;
+ struct vfio_pci_bar *largest = NULL;
+ u64 bar_size = 0;
+
+ for (int i = 0; i < PCI_STD_NUM_BARS; i++) {
+ struct vfio_pci_bar *bar = &device->bars[i];
+
+ if (!bar->vaddr)
+ continue;
+
+ /*
+ * iommu_map() maps with READ|WRITE, so require the same
+ * abilities for the underlying VFIO region.
+ */
+ if ((bar->info.flags & flags) != flags)
+ continue;
+
+ if (bar->info.size > bar_size) {
+ bar_size = bar->info.size;
+ largest = bar;
+ }
+ }
+
+ return largest;
+}
+
+FIXTURE(vfio_dma_mapping_mmio_test) {
+ struct iommu *iommu;
+ struct vfio_pci_device *device;
+ struct iova_allocator *iova_allocator;
+ struct vfio_pci_bar *bar;
+};
+
+FIXTURE_VARIANT(vfio_dma_mapping_mmio_test) {
+ const char *iommu_mode;
+};
+
+#define FIXTURE_VARIANT_ADD_IOMMU_MODE(_iommu_mode) \
+FIXTURE_VARIANT_ADD(vfio_dma_mapping_mmio_test, _iommu_mode) { \
+ .iommu_mode = #_iommu_mode, \
+}
+
+FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES();
+
+#undef FIXTURE_VARIANT_ADD_IOMMU_MODE
+
+FIXTURE_SETUP(vfio_dma_mapping_mmio_test)
+{
+ self->iommu = iommu_init(variant->iommu_mode);
+ self->device = vfio_pci_device_init(device_bdf, self->iommu);
+ self->iova_allocator = iova_allocator_init(self->iommu);
+ self->bar = largest_mapped_bar(self->device);
+
+ if (!self->bar)
+ SKIP(return, "No mappable BAR found on device %s", device_bdf);
+}
+
+FIXTURE_TEARDOWN(vfio_dma_mapping_mmio_test)
+{
+ iova_allocator_cleanup(self->iova_allocator);
+ vfio_pci_device_cleanup(self->device);
+ iommu_cleanup(self->iommu);
+}
+
+static void do_mmio_map_test(struct iommu *iommu,
+ struct iova_allocator *iova_allocator,
+ void *vaddr, size_t size)
+{
+ struct dma_region region = {
+ .vaddr = vaddr,
+ .size = size,
+ .iova = iova_allocator_alloc(iova_allocator, size),
+ };
+
+ /*
+ * NOTE: Check for iommufd compat success once it lands. Native iommufd
+ * will never support this.
+ */
+ if (!strcmp(iommu->mode->name, MODE_VFIO_TYPE1V2_IOMMU) ||
+ !strcmp(iommu->mode->name, MODE_VFIO_TYPE1_IOMMU)) {
+ iommu_map(iommu, &region);
+ iommu_unmap(iommu, &region);
+ } else {
+ VFIO_ASSERT_NE(__iommu_map(iommu, &region), 0);
+ }
+}
+
+TEST_F(vfio_dma_mapping_mmio_test, map_full_bar)
+{
+ do_mmio_map_test(self->iommu, self->iova_allocator,
+ self->bar->vaddr, self->bar->info.size);
+}
+
+TEST_F(vfio_dma_mapping_mmio_test, map_partial_bar)
+{
+ if (self->bar->info.size < 2 * getpagesize())
+ SKIP(return, "BAR too small (size=0x%llx)", self->bar->info.size);
+
+ do_mmio_map_test(self->iommu, self->iova_allocator,
+ self->bar->vaddr, getpagesize());
+}
+
+/* Test IOMMU mapping of BAR mmap with intentionally poor vaddr alignment. */
+TEST_F(vfio_dma_mapping_mmio_test, map_bar_misaligned)
+{
+ /* Limit size to bound test time for large BARs */
+ size_t size = min_t(size_t, self->bar->info.size, SZ_1G);
+ void *vaddr;
+
+ vaddr = mmap_reserve(size, SZ_1G, getpagesize());
+ vaddr = mmap(vaddr, size, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_FIXED,
+ self->device->fd, self->bar->info.offset);
+ VFIO_ASSERT_NE(vaddr, MAP_FAILED);
+
+ do_mmio_map_test(self->iommu, self->iova_allocator, vaddr, size);
+
+ VFIO_ASSERT_EQ(munmap(vaddr, size), 0);
+}
+
+int main(int argc, char *argv[])
+{
+ device_bdf = vfio_selftests_get_bdf(&argc, argv);
+ return test_harness_run(argc, argv);
+}
diff --git a/tools/testing/selftests/vfio/vfio_dma_mapping_test.c b/tools/testing/selftests/vfio/vfio_dma_mapping_test.c
index 16eba2ecca47..7d0de8c79de1 100644
--- a/tools/testing/selftests/vfio/vfio_dma_mapping_test.c
+++ b/tools/testing/selftests/vfio/vfio_dma_mapping_test.c
@@ -3,7 +3,6 @@
#include <sys/mman.h>
#include <unistd.h>
-#include <uapi/linux/types.h>
#include <linux/iommufd.h>
#include <linux/limits.h>
#include <linux/mman.h>
@@ -45,9 +44,9 @@ static int intel_iommu_mapping_get(const char *bdf, u64 iova,
FILE *file;
char *rest;
- snprintf(iommu_mapping_path, sizeof(iommu_mapping_path),
- "/sys/kernel/debug/iommu/intel/%s/domain_translation_struct",
- bdf);
+ snprintf_assert(iommu_mapping_path, sizeof(iommu_mapping_path),
+ "/sys/kernel/debug/iommu/intel/%s/domain_translation_struct",
+ bdf);
printf("Searching for IOVA 0x%lx in %s\n", iova, iommu_mapping_path);
@@ -162,12 +161,8 @@ TEST_F(vfio_dma_mapping_test, dma_map_unmap)
if (rc == -EOPNOTSUPP)
goto unmap;
- /*
- * IOMMUFD compatibility-mode does not support huge mappings when
- * using VFIO_TYPE1_IOMMU.
- */
- if (!strcmp(variant->iommu_mode, "iommufd_compat_type1"))
- mapping_size = SZ_4K;
+ if (self->iommu->mode->iommu_type == VFIO_TYPE1_IOMMU)
+ goto unmap;
ASSERT_EQ(0, rc);
printf("Found IOMMU mappings for IOVA 0x%lx:\n", region.iova);
diff --git a/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c b/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c
index 17017ed3beac..ec1e5633e080 100644
--- a/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c
+++ b/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c
@@ -1,5 +1,4 @@
// SPDX-License-Identifier: GPL-2.0
-#include <uapi/linux/types.h>
#include <linux/limits.h>
#include <linux/sizes.h>
#include <linux/vfio.h>
diff --git a/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c b/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c
index 33b0c31fe2ed..e1a54e153cd3 100644
--- a/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c
+++ b/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c
@@ -45,8 +45,8 @@ FIXTURE_SETUP(vfio_pci_device_init_perf_test)
int i;
self->iommu = iommu_init(variant->iommu_mode);
- self->threads = calloc(nr_devices, sizeof(self->threads[0]));
- self->thread_args = calloc(nr_devices, sizeof(self->thread_args[0]));
+ self->threads = calloc_assert(nr_devices, sizeof(self->threads[0]));
+ self->thread_args = calloc_assert(nr_devices, sizeof(self->thread_args[0]));
pthread_barrier_init(&self->barrier, NULL, nr_devices);
diff --git a/tools/testing/selftests/vfio/vfio_pci_device_test.c b/tools/testing/selftests/vfio/vfio_pci_device_test.c
index 7c0fe8ce3a61..93c11fd5e081 100644
--- a/tools/testing/selftests/vfio/vfio_pci_device_test.c
+++ b/tools/testing/selftests/vfio/vfio_pci_device_test.c
@@ -39,16 +39,17 @@ FIXTURE_TEARDOWN(vfio_pci_device_test)
iommu_cleanup(self->iommu);
}
-#define read_pci_id_from_sysfs(_file) ({ \
- char __sysfs_path[PATH_MAX]; \
- char __buf[32]; \
- int __fd; \
- \
- snprintf(__sysfs_path, PATH_MAX, "/sys/bus/pci/devices/%s/%s", device_bdf, _file); \
- ASSERT_GT((__fd = open(__sysfs_path, O_RDONLY)), 0); \
- ASSERT_GT(read(__fd, __buf, ARRAY_SIZE(__buf)), 0); \
- ASSERT_EQ(0, close(__fd)); \
- (u16)strtoul(__buf, NULL, 0); \
+#define read_pci_id_from_sysfs(_file) ({ \
+ char __sysfs_path[PATH_MAX]; \
+ char __buf[32]; \
+ int __fd; \
+ \
+ snprintf_assert(__sysfs_path, PATH_MAX, "/sys/bus/pci/devices/%s/%s", \
+ device_bdf, _file); \
+ ASSERT_GT((__fd = open(__sysfs_path, O_RDONLY)), 0); \
+ ASSERT_GT(read(__fd, __buf, ARRAY_SIZE(__buf)), 0); \
+ ASSERT_EQ(0, close(__fd)); \
+ (u16)strtoul(__buf, NULL, 0); \
})
TEST_F(vfio_pci_device_test, config_space_read_write)
diff --git a/tools/testing/selftests/vfio/vfio_pci_driver_test.c b/tools/testing/selftests/vfio/vfio_pci_driver_test.c
index afa0480ddd9b..761bf117d624 100644
--- a/tools/testing/selftests/vfio/vfio_pci_driver_test.c
+++ b/tools/testing/selftests/vfio/vfio_pci_driver_test.c
@@ -11,11 +11,18 @@
static const char *device_bdf;
-#define ASSERT_NO_MSI(_eventfd) do { \
- u64 __value; \
- \
- ASSERT_EQ(-1, read(_eventfd, &__value, 8)); \
- ASSERT_EQ(EAGAIN, errno); \
+#define fcntl_set_msi_nonblock(_self) do { \
+ if (_self->device->driver.ops->send_msi) \
+ fcntl_set_nonblock(_self->msi_fd); \
+} while (0)
+
+#define ASSERT_NO_MSI(_self) do { \
+ u64 __value; \
+ \
+ if (!_self->device->driver.ops->send_msi) \
+ break; \
+ ASSERT_EQ(-1, read(_self->msi_fd, &__value, 8)); \
+ ASSERT_EQ(EAGAIN, errno); \
} while (0)
static void region_setup(struct iommu *iommu,
@@ -89,12 +96,12 @@ FIXTURE_SETUP(vfio_pci_driver_test)
self->msi_fd = self->device->msi_eventfds[driver->msi];
/*
- * Use the maximum size supported by the device for memcpy operations,
- * slimmed down to fit into the memcpy region (divided by 2 so src and
- * dst regions do not overlap).
+ * Use 4x the driver's max_memcpy_size to exercise the chunking
+ * logic in vfio_pci_driver_memcpy(). Cap to half the memcpy
+ * region so src and dst do not overlap.
*/
- self->size = self->device->driver.max_memcpy_size;
- self->size = min(self->size, self->memcpy_region.size / 2);
+ self->size = min_t(u64, driver->max_memcpy_size * 4,
+ self->memcpy_region.size / 2);
self->src = self->memcpy_region.vaddr;
self->dst = self->src + self->size;
@@ -129,7 +136,7 @@ TEST_F(vfio_pci_driver_test, init_remove)
TEST_F(vfio_pci_driver_test, memcpy_success)
{
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
memset(self->src, 'x', self->size);
memset(self->dst, 'y', self->size);
@@ -140,12 +147,12 @@ TEST_F(vfio_pci_driver_test, memcpy_success)
self->size));
ASSERT_EQ(0, memcmp(self->src, self->dst, self->size));
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
TEST_F(vfio_pci_driver_test, memcpy_from_unmapped_iova)
{
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
/*
* Ignore the return value since not all devices will detect and report
@@ -154,12 +161,12 @@ TEST_F(vfio_pci_driver_test, memcpy_from_unmapped_iova)
vfio_pci_driver_memcpy(self->device, self->unmapped_iova,
self->dst_iova, self->size);
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
TEST_F(vfio_pci_driver_test, memcpy_to_unmapped_iova)
{
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
/*
* Ignore the return value since not all devices will detect and report
@@ -168,13 +175,16 @@ TEST_F(vfio_pci_driver_test, memcpy_to_unmapped_iova)
vfio_pci_driver_memcpy(self->device, self->src_iova,
self->unmapped_iova, self->size);
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
TEST_F(vfio_pci_driver_test, send_msi)
{
u64 value;
+ if (!self->device->driver.ops->send_msi)
+ SKIP(return, "Driver does not support send_msi()\n");
+
vfio_pci_driver_send_msi(self->device);
ASSERT_EQ(8, read(self->msi_fd, &value, 8));
ASSERT_EQ(1, value);
@@ -201,6 +211,9 @@ TEST_F(vfio_pci_driver_test, mix_and_match)
self->dst_iova,
self->size);
+ if (!self->device->driver.ops->send_msi)
+ continue;
+
vfio_pci_driver_send_msi(self->device);
ASSERT_EQ(8, read(self->msi_fd, &value, 8));
ASSERT_EQ(1, value);
@@ -211,9 +224,10 @@ TEST_F_TIMEOUT(vfio_pci_driver_test, memcpy_storm, 60)
{
struct vfio_pci_driver *driver = &self->device->driver;
u64 total_size;
+ u64 size;
u64 count;
- fcntl_set_nonblock(self->msi_fd);
+ fcntl_set_msi_nonblock(self);
/*
* Perform up to 250GiB worth of DMA reads and writes across several
@@ -221,16 +235,17 @@ TEST_F_TIMEOUT(vfio_pci_driver_test, memcpy_storm, 60)
* will take too long.
*/
total_size = 250UL * SZ_1G;
- count = min(total_size / self->size, driver->max_memcpy_count);
+ size = min(driver->max_memcpy_size, self->memcpy_region.size / 2);
+ count = min(total_size / size, driver->max_memcpy_count);
- printf("Kicking off %lu memcpys of size 0x%lx\n", count, self->size);
+ printf("Kicking off %lu memcpys of size 0x%lx\n", count, size);
vfio_pci_driver_memcpy_start(self->device,
self->src_iova,
self->dst_iova,
- self->size, count);
+ size, count);
ASSERT_EQ(0, vfio_pci_driver_memcpy_wait(self->device));
- ASSERT_NO_MSI(self->msi_fd);
+ ASSERT_NO_MSI(self);
}
static bool device_has_selftests_driver(const char *bdf)
diff --git a/tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c b/tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c
new file mode 100644
index 000000000000..19d657d00b75
--- /dev/null
+++ b/tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c
@@ -0,0 +1,217 @@
+// SPDX-License-Identifier: GPL-2.0-only
+#include "lib/include/libvfio/assert.h"
+#include <fcntl.h>
+#include <unistd.h>
+#include <stdlib.h>
+#include <sys/ioctl.h>
+#include <linux/limits.h>
+
+#include <libvfio.h>
+
+#include "../kselftest_harness.h"
+
+#define UUID_1 "52ac9bff-3a88-4fbd-901a-0d767c3b6c97"
+#define UUID_2 "88594674-90a0-47a9-aea8-9d9b352ac08a"
+
+static const char *pf_bdf;
+static char *vf_bdf;
+
+static pid_t main_pid;
+
+static int container_setup(struct vfio_pci_device *device, const char *bdf,
+ const char *vf_token)
+{
+ vfio_pci_group_setup(device, bdf);
+ vfio_container_set_iommu(device);
+ __vfio_pci_group_get_device_fd(device, bdf, vf_token);
+
+ /* The device fd will be -1 in case of mismatched tokens */
+ return (device->fd < 0);
+}
+
+static int iommufd_setup(struct vfio_pci_device *device, const char *bdf,
+ const char *vf_token)
+{
+ vfio_pci_cdev_open(device, bdf);
+ return __vfio_device_bind_iommufd(device->fd,
+ device->iommu->iommufd, vf_token);
+}
+
+static int device_init(const char *bdf, struct iommu *iommu,
+ const char *vf_token, struct vfio_pci_device **out_dev)
+{
+ struct vfio_pci_device *device = vfio_pci_device_alloc(bdf, iommu);
+ int ret;
+
+ if (iommu->mode->container_path)
+ ret = container_setup(device, bdf, vf_token);
+ else
+ ret = iommufd_setup(device, bdf, vf_token);
+
+ *out_dev = device;
+ return ret;
+}
+
+static void device_cleanup(struct vfio_pci_device *device)
+{
+ if (!device)
+ return;
+
+ if (device->fd > 0)
+ VFIO_ASSERT_EQ(close(device->fd), 0);
+
+ if (device->group_fd)
+ VFIO_ASSERT_EQ(close(device->group_fd), 0);
+
+ vfio_pci_device_free(device);
+}
+
+FIXTURE(vfio_pci_sriov_uapi_test) {
+ struct vfio_pci_device *pf;
+ struct vfio_pci_device *vf;
+ struct iommu *iommu;
+ char *pf_token;
+};
+
+FIXTURE_VARIANT(vfio_pci_sriov_uapi_test) {
+ const char *iommu_mode;
+ char *vf_token;
+};
+
+#define FIXTURE_VARIANT_ADD_IOMMU_MODE(_iommu_mode, _name, _vf_token) \
+FIXTURE_VARIANT_ADD(vfio_pci_sriov_uapi_test, _iommu_mode ## _ ## _name) { \
+ .iommu_mode = #_iommu_mode, \
+ .vf_token = (_vf_token), \
+}
+
+FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(same_uuid, UUID_1);
+FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(diff_uuid, UUID_2);
+FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(null_uuid, NULL);
+
+FIXTURE_SETUP(vfio_pci_sriov_uapi_test)
+{
+ self->iommu = iommu_init(variant->iommu_mode);
+
+ self->pf_token = UUID_1;
+ ASSERT_EQ(device_init(pf_bdf, self->iommu, self->pf_token, &self->pf), 0);
+}
+
+FIXTURE_TEARDOWN(vfio_pci_sriov_uapi_test)
+{
+ device_cleanup(self->vf);
+ device_cleanup(self->pf);
+ iommu_cleanup(self->iommu);
+}
+
+/*
+ * This asserts if the VF device is successfully created if its token matches
+ * with the token used to create/override the PF or fails during a mismatch.
+ */
+#define ASSERT_COND_VF_CREATION(_ret) do { \
+ if (!variant->vf_token || strcmp(self->pf_token, variant->vf_token)) { \
+ ASSERT_NE((_ret), 0); \
+ } else { \
+ ASSERT_EQ((_ret), 0); \
+ } \
+} while (0)
+
+/*
+ * Validate if the UAPI handles correctly and incorrectly set token on the VF.
+ */
+TEST_F(vfio_pci_sriov_uapi_test, init_token_match)
+{
+ int ret;
+
+ ret = device_init(vf_bdf, self->iommu, variant->vf_token, &self->vf);
+ ASSERT_COND_VF_CREATION(ret);
+}
+
+/*
+ * After closing the PF, validate if the VF access still needs the right token.
+ */
+TEST_F(vfio_pci_sriov_uapi_test, pf_early_close)
+{
+ int ret;
+
+ device_cleanup(self->pf);
+
+ /* Clean the 'pf' to avoid calling device_cleanup() again. */
+ self->pf = NULL;
+
+ ret = device_init(vf_bdf, self->iommu, variant->vf_token, &self->vf);
+ ASSERT_COND_VF_CREATION(ret);
+}
+
+/*
+ * After PF device init, override the existing token and validate if the newly
+ * set token is the one that's active.
+ */
+TEST_F(vfio_pci_sriov_uapi_test, override_token)
+{
+ int ret;
+
+ self->pf_token = UUID_2;
+ vfio_device_set_vf_token(self->pf->fd, self->pf_token);
+
+ ret = device_init(vf_bdf, self->iommu, variant->vf_token, &self->vf);
+ ASSERT_COND_VF_CREATION(ret);
+}
+
+static void vf_teardown(void)
+{
+ /*
+ * The child processes, created by TEST_F()s, inherits this atexit()
+ * handler. Hence, check and destroy the VF only when the main/parent
+ * process exits.
+ */
+ if (getpid() != main_pid)
+ return;
+
+ free(vf_bdf);
+ sysfs_sriov_numvfs_set(pf_bdf, 0);
+}
+
+static void vf_setup(void)
+{
+ char *vf_driver;
+ int nr_vfs;
+
+ nr_vfs = sysfs_sriov_totalvfs_get(pf_bdf);
+ if (nr_vfs <= 0)
+ ksft_exit_skip("SR-IOV may not be supported by the PF: %s\n", pf_bdf);
+
+ nr_vfs = sysfs_sriov_numvfs_get(pf_bdf);
+ if (nr_vfs != 0)
+ ksft_exit_skip("SR-IOV already configured for the PF: %s\n", pf_bdf);
+
+ /* Create only one VF for testing */
+ sysfs_sriov_numvfs_set(pf_bdf, 1);
+
+ /*
+ * Setup an exit handler to destroy the VF in case of failures
+ * during further setup at the end of the test run.
+ */
+ main_pid = getpid();
+ VFIO_ASSERT_EQ(atexit(vf_teardown), 0);
+
+ vf_bdf = sysfs_sriov_vf_bdf_get(pf_bdf, 0);
+
+ /*
+ * The VF inherits the driver from the PF.
+ * Ensure this is 'vfio-pci' before proceeding.
+ */
+ vf_driver = sysfs_driver_get(vf_bdf);
+ VFIO_ASSERT_NE(vf_driver, NULL);
+ VFIO_ASSERT_EQ(strcmp(vf_driver, "vfio-pci"), 0);
+ free(vf_driver);
+
+ printf("Created 1 VF (%s) under the PF: %s\n", vf_bdf, pf_bdf);
+}
+
+int main(int argc, char *argv[])
+{
+ pf_bdf = vfio_selftests_get_bdf(&argc, argv);
+ vf_setup();
+
+ return test_harness_run(argc, argv);
+}