diff options
Diffstat (limited to 'tools/testing/selftests/vfio')
28 files changed, 2663 insertions, 117 deletions
diff --git a/tools/testing/selftests/vfio/Makefile b/tools/testing/selftests/vfio/Makefile index 3c796ca99a50..2c32c48db509 100644 --- a/tools/testing/selftests/vfio/Makefile +++ b/tools/testing/selftests/vfio/Makefile @@ -1,9 +1,18 @@ +ARCH ?= $(shell uname -m) + +ifeq (,$(filter $(ARCH),aarch64 arm64 x86 x86_64)) +# Do nothing on unsupported architectures +include ../lib.mk +else + CFLAGS = $(KHDR_INCLUDES) TEST_GEN_PROGS += vfio_dma_mapping_test +TEST_GEN_PROGS += vfio_dma_mapping_mmio_test TEST_GEN_PROGS += vfio_iommufd_setup_test TEST_GEN_PROGS += vfio_pci_device_test TEST_GEN_PROGS += vfio_pci_device_init_perf_test TEST_GEN_PROGS += vfio_pci_driver_test +TEST_GEN_PROGS += vfio_pci_sriov_uapi_test TEST_FILES += scripts/cleanup.sh TEST_FILES += scripts/lib.sh @@ -15,15 +24,21 @@ include lib/libvfio.mk CFLAGS += -I$(top_srcdir)/tools/include CFLAGS += -MD +CFLAGS += -Wall -Werror CFLAGS += $(EXTRA_CFLAGS) LDFLAGS += -pthread -$(TEST_GEN_PROGS): %: %.o $(LIBVFIO_O) +$(TEST_GEN_PROGS): $(OUTPUT)/%: $(OUTPUT)/%.o $(LIBVFIO_O) $(CC) $(CFLAGS) $(CPPFLAGS) $(LDFLAGS) $< $(LIBVFIO_O) $(LDLIBS) -o $@ TEST_GEN_PROGS_O = $(patsubst %, %.o, $(TEST_GEN_PROGS)) +$(TEST_GEN_PROGS_O): $(OUTPUT)/%.o: %.c + $(CC) $(CFLAGS) $(CPPFLAGS) $(TARGET_ARCH) -c $< -o $@ + TEST_DEP_FILES = $(patsubst %.o, %.d, $(TEST_GEN_PROGS_O) $(LIBVFIO_O)) -include $(TEST_DEP_FILES) EXTRA_CLEAN += $(TEST_GEN_PROGS_O) $(TEST_DEP_FILES) + +endif diff --git a/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c b/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c index c75045bcab79..19d9630b24c2 100644 --- a/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c +++ b/tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c @@ -65,9 +65,20 @@ static bool dsa_int_handle_request_required(struct vfio_pci_device *device) static int dsa_probe(struct vfio_pci_device *device) { - if (!vfio_pci_device_match(device, PCI_VENDOR_ID_INTEL, - PCI_DEVICE_ID_INTEL_DSA_SPR0)) + const u16 vendor_id = vfio_pci_config_readw(device, PCI_VENDOR_ID); + const u16 device_id = vfio_pci_config_readw(device, PCI_DEVICE_ID); + + if (vendor_id != PCI_VENDOR_ID_INTEL) + return -EINVAL; + + switch (device_id) { + case PCI_DEVICE_ID_INTEL_DSA_SPR0: + case PCI_DEVICE_ID_INTEL_DSA_DMR: + case PCI_DEVICE_ID_INTEL_DSA_GNRD: + break; + default: return -EINVAL; + } if (dsa_int_handle_request_required(device)) { dev_err(device, "Device requires requesting interrupt handles\n"); diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h new file mode 120000 index 000000000000..b84affdec559 --- /dev/null +++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_82575.h @@ -0,0 +1 @@ +../../../../../../../drivers/net/ethernet/intel/igb/e1000_82575.h
\ No newline at end of file diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h new file mode 120000 index 000000000000..9f97f4330086 --- /dev/null +++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_defines.h @@ -0,0 +1 @@ +../../../../../../../drivers/net/ethernet/intel/igb/e1000_defines.h
\ No newline at end of file diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h new file mode 120000 index 000000000000..c733634171bb --- /dev/null +++ b/tools/testing/selftests/vfio/lib/drivers/igb/e1000_regs.h @@ -0,0 +1 @@ +../../../../../../../drivers/net/ethernet/intel/igb/e1000_regs.h
\ No newline at end of file diff --git a/tools/testing/selftests/vfio/lib/drivers/igb/igb.c b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c new file mode 100644 index 000000000000..fd9e05d77ea4 --- /dev/null +++ b/tools/testing/selftests/vfio/lib/drivers/igb/igb.c @@ -0,0 +1,585 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include <unistd.h> +#include <errno.h> +#include <stdint.h> +#include <linux/io.h> +#include <linux/pci_regs.h> +#include <linux/pci_ids.h> +#include <linux/kernel.h> +#include <linux/compiler.h> +#include <asm/barrier.h> +#include <linux/mii.h> +#include <libvfio/vfio_pci_device.h> + +#include "e1000_regs.h" +#include "e1000_defines.h" +#include "e1000_82575.h" + +#define PCI_DEVICE_ID_INTEL_82576 0x10C9 +#define IGB_MAX_CHUNK_SIZE 1024 +#define MSIX_VECTOR 0 +#define MSIX_VECTOR_MASK (1 << MSIX_VECTOR) +#define RING_SIZE 4096 /* Number of descriptors in ring */ + +struct igb_tx_desc { + union { + struct { + u64 buffer_addr; /* Address of descriptor's data buffer */ + u32 cmd_type_len; /* Command/Type/Length */ + u32 olinfo_status; /* Context/Buffer info */ + } read; + + struct { + u64 rsvd; /* Reserved */ + u32 nxtseq_seed; /* Next sequence seed */ + u32 status; /* Descriptor status */ + } wb; + }; +}; + +struct igb_rx_desc { + union { + struct { + u64 pkt_addr; /* Packet buffer address */ + u64 hdr_addr; /* Header buffer address */ + } read; + struct { + u16 pkt_info; /* RSS type, Packet type */ + u16 hdr_info; /* Split Head, buf len */ + u32 rss; /* RSS Hash */ + u32 status_error; /* ext status/error */ + u16 length; /* Packet length */ + u16 vlan; /* VLAN tag */ + } wb; /* writeback */ + }; +}; + +struct igb { + void *bar0; + u32 tx_tail; + u32 rx_tail; + struct igb_tx_desc tx_ring[RING_SIZE] __attribute__((aligned(128))); + struct igb_rx_desc rx_ring[RING_SIZE] __attribute__((aligned(128))); +}; + +static inline struct igb *to_igb_state(struct vfio_pci_device *device) +{ + return (struct igb *)device->driver.region.vaddr; +} + +static inline void igb_write32(struct igb *igb, u32 reg, u32 val) +{ + writel(val, igb->bar0 + reg); +} + +static inline u32 igb_read32(struct igb *igb, u32 reg) +{ + return readl(igb->bar0 + reg); +} + +static int igb_write_phy(struct igb *igb, u32 offset, u16 data) +{ + u32 mdic; + int i; + + /* + * Write a PHY register over MDIO. + * + * A production driver would hold the SW/FW semaphore (SWSM.SWESMBI + the + * SW_FW_SYNC PHY bit) across the MDIO transaction to serialize against the + * device's management firmware. The selftest owns the assigned function + * exclusively on a dedicated test device with no active manageability + * contending for the PHY, so the sync is omitted; it should be added here + * if this ever needs to run on a manageability-enabled NIC. + */ + mdic = (((u32)data) | + (offset << E1000_MDIC_REG_SHIFT) | + (1 << E1000_MDIC_PHY_SHIFT) | + E1000_MDIC_OP_WRITE); + + igb_write32(igb, E1000_MDIC, mdic); + + for (i = 0; i < 1000; i++) { + usleep(50); + mdic = igb_read32(igb, E1000_MDIC); + if (mdic & E1000_MDIC_READY) + break; + } + + if (!(mdic & E1000_MDIC_READY)) + return -1; + + if (mdic & E1000_MDIC_ERROR) + return -1; + + return 0; +} + +/* + * Configure the device for PHY internal loopback per 82576 datasheet + * section 3.5.6.3.1. Force the PHY to 1Gb/s full duplex with loopback + * enabled, then force the MAC link state to match. Internal loopback + * wraps data at the end of the PHY datapath (section 3.5.6.3), so the + * physical link state is irrelevant. + * + * Section 3.5.6.1 directs to "Use PHY Loopback instead of MAC Loopback + * on the 82576", and section 3.5.6.2 states "MAC Loopback is not used + * on this device." RCTL.LBM_MAC is still set elsewhere as a QEMU-only + * accommodation; see the RCTL programming in the caller for the + * rationale. + */ +static void igb_setup_loopback(struct igb *igb) +{ + u32 ctrl; + int ret; + + /* + * Kick the autoneg machinery solely to bring STATUS.LU up under + * QEMU's igb emulation: QEMU only updates STATUS.LU via its + * autoneg-done timer, and without LU set its receive path + * (e1000x_hw_rx_enabled) drops every loopback frame. On real + * hardware autoneg cannot complete before the next PHY write + * below clears the autoneg-enable bit, so this is effectively a + * no-op there. + */ + (void)igb_write_phy(igb, MII_BMCR, + BMCR_ANENABLE | BMCR_ANRESTART); + + /* PHY control: loopback + 1Gb/s full duplex, autoneg disabled. */ + ret = igb_write_phy(igb, MII_BMCR, + BMCR_LOOPBACK | + BMCR_SPEED1000 | + BMCR_FULLDPLX); + VFIO_ASSERT_EQ(ret, 0, "Failed to write PHY control register"); + + /* + * Brief delay before forcing the MAC, mirroring the kernel ethtool + * selftest in igb_integrated_phy_loopback(). Not specified by the + * datasheet, but empirically required by the kernel driver. + */ + usleep(50000); + + /* + * Force the MAC to 1Gb/s full duplex with link up. Without forcing + * the link state the descriptor engine does not run, since the chip + * normally waits for a real negotiated link. + */ + ctrl = igb_read32(igb, E1000_CTRL); + ctrl &= ~E1000_CTRL_SPD_SEL; + ctrl |= E1000_CTRL_FRCSPD | + E1000_CTRL_FRCDPX | + E1000_CTRL_SPD_1000 | + E1000_CTRL_FD | + E1000_CTRL_SLU; + igb_write32(igb, E1000_CTRL, ctrl); + + /* + * Settling delay matching the kernel ethtool selftest's msleep(500) + * at the tail of igb_integrated_phy_loopback(). Not specified by + * the datasheet; empirical, and inherited from the kernel driver. + */ + usleep(500000); +} + +static int igb_probe(struct vfio_pci_device *device) +{ + if (!vfio_pci_device_match(device, PCI_VENDOR_ID_INTEL, PCI_DEVICE_ID_INTEL_82576)) + return -EINVAL; + + return 0; +} + +static void igb_reset(struct igb *igb) +{ + int retries = 20; + + igb_write32(igb, E1000_CTRL, igb_read32(igb, E1000_CTRL) | E1000_CTRL_RST); + /* + * Must wait at least 1 millisecond after setting the reset bit before + * checking if this device is ready to be used (82576 datasheet section + * 4.2.1.6.1). The delay also ensures the reset has taken effect and + * cleared EECD.AUTO_RD before it is polled below. + */ + usleep(1000); + + /* + * Poll NVM Auto Read Done rather than CTRL.RST, matching + * igb_get_auto_rd_done() in the igb driver: AUTO_RD implies both that + * the reset completed and that the device finished re-reading its + * configuration from NVM, which is what actually makes it usable. + */ + while (retries-- > 0 && !(igb_read32(igb, E1000_EECD) & E1000_EECD_AUTO_RD)) + usleep(1000); + + /* + * QEMU's igb emulation does not set E1000_EECD_AUTO_RD. If we timed out, + * check if CTRL.RST is cleared, which is what QEMU uses to signal reset + * completion. + */ + if (retries < 0) { + VFIO_ASSERT_EQ(igb_read32(igb, E1000_CTRL) & E1000_CTRL_RST, 0, + "Device reset did not complete (CTRL.RST not cleared)"); + } + + igb_write32(igb, E1000_IMC, 0xFFFFFFFF); +} + +/* + * Program the device into a usable state. Split out of igb_init() so it + * can be reused after a device reset to re-program the registers that + * CTRL.RST clears. Expects bar0 to be mapped and MSI-X already enabled + * via VFIO. + */ +static void igb_hw_init(struct vfio_pci_device *device) +{ + struct igb *igb = to_igb_state(device); + u64 iova_tx, iova_rx; + u32 ctrl, rctl; + u16 cmd_reg; + int retries; + + iova_tx = to_iova(device, igb->tx_ring); + iova_rx = to_iova(device, igb->rx_ring); + + + + /* Signal that the driver is loaded */ + ctrl = igb_read32(igb, E1000_CTRL_EXT); + ctrl |= E1000_CTRL_EXT_DRV_LOAD; + ctrl &= ~E1000_CTRL_EXT_LINK_MODE_MASK; + igb_write32(igb, E1000_CTRL_EXT, ctrl); + + /* Enable PCI Bus Master. */ + cmd_reg = vfio_pci_config_readw(device, PCI_COMMAND); + if ((cmd_reg & (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) != + (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY)) { + cmd_reg |= (PCI_COMMAND_MASTER | PCI_COMMAND_MEMORY); + vfio_pci_config_writew(device, PCI_COMMAND, cmd_reg); + } + + /* Configure PHY internal loopback for testing. */ + igb_setup_loopback(igb); + + /* + * Disable DMA re-send on PCIe completion timeout (82576 datasheet + * section 8.6.1, GCR.Completion_Timeout_Resend, bit 16). The + * mix_and_match test intentionally submits descriptors targeting + * unmapped IOVAs; with the default (set) value, the device keeps + * retrying the failed read indefinitely, which keeps PCIe AER and + * IOMMU error handling busy and interferes with reset recovery. + */ + ctrl = igb_read32(igb, E1000_GCR); + ctrl &= ~E1000_GCR_CMPL_TMOUT_RESEND; + igb_write32(igb, E1000_GCR, ctrl); + + /* Configure TX and RX descriptor rings */ + igb_write32(igb, E1000_TDBAL(0), (u32)iova_tx); + igb_write32(igb, E1000_TDBAH(0), (u32)(iova_tx >> 32)); + igb_write32(igb, E1000_TDLEN(0), RING_SIZE * sizeof(struct igb_tx_desc)); + igb_write32(igb, E1000_TDH(0), 0); + igb_write32(igb, E1000_TDT(0), 0); + igb_write32(igb, E1000_TXDCTL(0), E1000_TXDCTL_QUEUE_ENABLE); + + igb_write32(igb, E1000_RDBAL(0), (u32)iova_rx); + igb_write32(igb, E1000_RDBAH(0), (u32)(iova_rx >> 32)); + igb_write32(igb, E1000_RDLEN(0), RING_SIZE * sizeof(struct igb_rx_desc)); + igb_write32(igb, E1000_RDH(0), 0); + igb_write32(igb, E1000_RDT(0), 0); + + /* + * Select the advanced one-buffer descriptor format. Per 82576 + * datasheet section 7.1.5.2: "SRRCTL[n].DESCTYPE must be set to a + * value other than 000b for the 82576 to write back the special + * descriptors." struct igb_rx_desc matches the advanced one-buffer + * writeback layout (section 7.1.5.2), so polling rx.wb.status_error + * requires this format. Section 8.10.2 specifies DESCTYPE[27:25]. + * + * The direct write also zeroes SRRCTL.BSIZEPACKET, which is + * intentional: per section 7.1.3.1 a zero BSIZEPACKET falls back to + * the RCTL.BSIZE buffer size, whose reset default (00b) is 2048 + * bytes -- ample for the loopback frames here. + */ + igb_write32(igb, E1000_SRRCTL(0), E1000_SRRCTL_DESCTYPE_ADV_ONEBUF); + + igb_write32(igb, E1000_RXDCTL(0), E1000_RXDCTL_QUEUE_ENABLE); + + /* + * Enable Receiver and Transmitter. RCTL.LBM_MAC is set in addition + * to PHY loopback as a QEMU-only accommodation: QEMU's emulated igb + * does not honor PHY register 0 bit 14 (PHY internal loopback) and + * relies on RCTL.LBM_MAC to wrap TX descriptors back to the RX + * queue. Datasheet 8.10.1 (RCTL register) advises "When using the + * internal PHY, LBM should remain set to 00b", so setting LBM_MAC + * here deviates from datasheet guidance; empirically the bit has + * no observable effect on real 82576 hardware because MAC loopback + * is not implemented (datasheet 3.5.6.2). Setting both lets the + * selftest work on both real hardware and QEMU without conditional + * code paths. + */ + rctl = E1000_RCTL_EN | /* Receiver Enable */ + E1000_RCTL_UPE | /* Unicast Promiscuous (for dummy MAC) */ + E1000_RCTL_MPE | /* Multicast Promiscuous */ + E1000_RCTL_BAM | /* Broadcast Accept Mode */ + E1000_RCTL_LBM_MAC | /* MAC Loopback - for QEMU emulation only */ + E1000_RCTL_SECRC; /* Strip CRC (needed for memcmp) */ + igb_write32(igb, E1000_RCTL, rctl); + igb_write32(igb, E1000_TCTL, E1000_TCTL_EN | E1000_TCTL_PSP); + + /* + * Wait for TX and RX queues to be enabled. Per the RXDCTL/TXDCTL + * register definitions (8.10.10/8.12.13), the per-queue enable bit + * "remains zero" until the global RCTL.RXEN/TCTL.TXEN are set, so + * E1000_RCTL_EN and E1000_TCTL_EN must already be written above. + */ + retries = 2000; + while (retries-- > 0) { + if ((igb_read32(igb, E1000_TXDCTL(0)) & E1000_TXDCTL_QUEUE_ENABLE) && + (igb_read32(igb, E1000_RXDCTL(0)) & E1000_RXDCTL_QUEUE_ENABLE)) + break; + usleep(10); + } + VFIO_ASSERT_GE(retries, 0); + + /* + * Program MSI-X interrupt routing per 82576 datasheet: + * + * GPIE (section 7.3.2.11, Table 7-47): set Multiple_MSIX (bit 4) to + * route interrupt causes through IVAR mapping, and EIAME (bit 30) + * to apply EIAM on MSI-X assertion (without EIAME, EIAM only + * applies on EICR read/write). + * + * EIAC (section 8.8.5): enable auto-clear of EICR for vector 0. + * Without auto-clear the cause stays set after delivery and the + * test can see spurious interrupts on the next memcpy batch. + * + * EIAM (section 8.8.6): enable auto-mask of EIMS for vector 0 on + * MSI-X assertion (effective because EIAME is set). + * + * IVAR (section 7.3.1.2, register definition in 8.8.13): map RX + * cause 0 to MSI-X vector 0 and mark the entry valid. + */ + igb_write32(igb, E1000_GPIE, E1000_GPIE_MSIX_MODE | E1000_GPIE_EIAME); + igb_write32(igb, E1000_EIAC, MSIX_VECTOR_MASK); + igb_write32(igb, E1000_EIAM, MSIX_VECTOR_MASK); + + /* Map vector 0 to interrupt cause 0 and mark it valid */ + igb_write32(igb, E1000_IVAR0, E1000_IVAR_VALID); + + /* Enable interrupts on vector 0 */ + igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK); + + /* Initialize driver state and capability limits */ + igb->tx_tail = 0; + igb->rx_tail = 0; + + device->driver.max_memcpy_size = IGB_MAX_CHUNK_SIZE; + device->driver.max_memcpy_count = RING_SIZE - 1; + device->driver.msi = MSIX_VECTOR; +} + +static void igb_init(struct vfio_pci_device *device) +{ + struct igb *igb = to_igb_state(device); + + VFIO_ASSERT_GE(device->driver.region.size, sizeof(struct igb)); + + igb->bar0 = device->bars[0].vaddr; + + igb_reset(igb); + + /* + * Enable MSI-X via VFIO before device-side register programming. + * vfio_pci_msix_enable() only touches the VFIO IRQ machinery and the + * PCI MSI-X capability via config space; it has no ordering + * dependency on the device-side writes performed by igb_hw_init(). + * Placing it here keeps igb_hw_init() reusable from the reset + * recovery path (which calls vfio_pci_irq_reenable() instead). + */ + vfio_pci_msix_enable(device, MSIX_VECTOR, 1); + + igb_hw_init(device); +} + +static void igb_remove(struct vfio_pci_device *device) +{ + struct igb *igb = to_igb_state(device); + + igb_write32(igb, E1000_RCTL, 0); + igb_write32(igb, E1000_TCTL, 0); + igb_reset(igb); + + vfio_pci_msix_disable(device); +} + +static void igb_irq_disable(struct igb *igb) +{ + igb_write32(igb, E1000_EIMC, MSIX_VECTOR_MASK); +} + +static void igb_irq_enable(struct igb *igb) +{ + igb_write32(igb, E1000_EIMS, MSIX_VECTOR_MASK); +} + +static void igb_irq_clear(struct igb *igb) +{ + /* + * Use write-to-clear (datasheet 7.3.4.2). In MSI-X mode with EIAC + * programmed, section 8.8.5 explicitly states "If any bits are set + * in EIAC, the EICR register should not be read", which rules out + * the read-to-clear path in 7.3.4.3. Bits not in EIAC are still + * cleared by writing 1. + */ + igb_write32(igb, E1000_EICR, 0xFFFFFFFF); +} + +static void igb_memcpy_start(struct vfio_pci_device *device, iova_t src, + iova_t dst, u64 size, u64 count) +{ + struct igb *igb = to_igb_state(device); + struct igb_rx_desc *rx; + struct igb_tx_desc *tx; + u32 i; + + VFIO_ASSERT_GE(size, 60, + "IGB driver requires memcpy size to be at least 60 bytes (Ethernet minimum payload size)"); + + igb_irq_disable(igb); + + for (i = 0; i < count; i++) { + tx = &igb->tx_ring[igb->tx_tail]; + rx = &igb->rx_ring[igb->rx_tail]; + + memset(tx, 0, sizeof(struct igb_tx_desc)); + memset(rx, 0, sizeof(struct igb_rx_desc)); + + rx->read.pkt_addr = cpu_to_le64(dst); + rx->read.hdr_addr = cpu_to_le64(0); + + tx->read.buffer_addr = cpu_to_le64(src); + /* + * Build an advanced data descriptor per 82576 datasheet + * section 7.2.2.3. DEXT marks the descriptor as advanced + * (required by hardware); DTYP=data selects the data + * descriptor; IFCS asks the MAC to append the Ethernet + * FCS (without it the frame is dropped as malformed); + * EOP marks end of packet. DTALEN is the buffer length + * in bits 15:0 of cmd_type_len. + */ + tx->read.cmd_type_len = cpu_to_le32((uint32_t)size | + E1000_ADVTXD_DTYP_DATA | + E1000_ADVTXD_DCMD_DEXT | + E1000_ADVTXD_DCMD_IFCS | + E1000_ADVTXD_DCMD_EOP); + /* + * PAYLEN (section 7.2.2.3.11) is the total payload size + * in olinfo_status[31:14]. + */ + tx->read.olinfo_status = + cpu_to_le32((uint32_t)size << E1000_ADVTXD_PAYLEN_SHIFT); + + igb->tx_tail = (igb->tx_tail + 1) % RING_SIZE; + igb->rx_tail = (igb->rx_tail + 1) % RING_SIZE; + } + + igb_write32(igb, E1000_RDT(0), igb->rx_tail); + igb_write32(igb, E1000_TDT(0), igb->tx_tail); +} + +/* + * Reset the device via VFIO_DEVICE_RESET (PCIe FLR on the 82576) and + * re-program it. VFIO_DEVICE_RESET tears down the kernel-side MSI-X + * trigger but leaves user-side eventfds intact, so re-arm the trigger + * via vfio_pci_irq_reenable() before reprogramming so any caller-cached + * eventfd remains valid. + * + * FLR clears device-side state to power-on reset values (datasheet + * 4.2.1.5.1: a PF FLR is "equivalent to a D0->D3->D0 transition"), so + * EIMS and EICR come back as 0 from their register-defined initial + * values, and igb_hw_init() resets tx_tail/rx_tail to 0. The next + * igb_memcpy_start() will memset each descriptor it touches before + * submission, so no explicit IMC/EICR writes or ring memsets are + * needed here. + */ +static void igb_error_reset_and_reinit(struct vfio_pci_device *device) +{ + vfio_pci_device_reset(device); + vfio_pci_msix_reenable(device, MSIX_VECTOR, 1); + igb_hw_init(device); +} + +static int igb_memcpy_wait(struct vfio_pci_device *device) +{ + struct igb *igb = to_igb_state(device); + struct igb_rx_desc *rx; + u32 status = 0; + u32 prev_tail; + int retries; + + prev_tail = (igb->rx_tail + RING_SIZE - 1) % RING_SIZE; + rx = &igb->rx_ring[prev_tail]; + + /* + * Real 82576 hardware processes the descriptor ring at line rate. + * max_memcpy_size = (RING_SIZE - 1) * IGB_MAX_CHUNK_SIZE ~= 4 MB, + * split into 4095 1 KB frames. At 1 Gb/s (~125 MB/s) the worst + * valid memcpy takes ~32 ms on the wire, plus per-frame preamble, + * SFD, IFG and FCS overhead (~3%) and descriptor fetch/writeback + * latency. Wait up to ~200 ms before declaring the device hung; + * ~6x the line-rate floor leaves comfortable headroom for host + * scheduling jitter while keeping the intentional invalid-DMA + * tests bounded. + */ + retries = 200; + while (retries-- > 0) { + status = le32_to_cpu(READ_ONCE(rx->wb.status_error)); + if (status & 1) + break; + usleep(1000); + } + + if (status & 1) + /* + * Ensure the test code doesn't speculatively read the DMA + * destination buffer before we have verified that the + * descriptor writeback is complete. + */ + rmb(); + + igb_irq_clear(igb); + + igb_irq_enable(igb); + + if (status & 1) + return 0; + + /* + * The descriptor never completed. On real 82576 hardware this + * typically follows a DMA-read fault from one of the intentional + * unmapped-IOVA tests; the fault leaves the descriptor engine + * unable to service subsequent valid descriptors. CTRL.RST alone + * reinitializes the queue registers but leaves the engine wedged + * for the current process, so a broader VFIO_DEVICE_RESET (FLR) + * is required. + */ + igb_error_reset_and_reinit(device); + + return -ETIMEDOUT; +} + +static void igb_send_msi(struct vfio_pci_device *device) +{ + struct igb *igb = to_igb_state(device); + + igb_write32(igb, E1000_EICS, MSIX_VECTOR_MASK); +} + +const struct vfio_pci_driver_ops igb_ops = { + .name = "igb", + .probe = igb_probe, + .init = igb_init, + .remove = igb_remove, + .memcpy_start = igb_memcpy_start, + .memcpy_wait = igb_memcpy_wait, + .send_msi = igb_send_msi, +}; diff --git a/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h new file mode 100644 index 000000000000..edce130fd008 --- /dev/null +++ b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/hw.h @@ -0,0 +1,352 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + */ +#ifndef _NV_FALCON_HW_H_ +#define _NV_FALCON_HW_H_ + +#include <linux/types.h> + +/* PMC (Power Management Controller) Registers */ +#define NV_PMC_BOOT_0 0x00000000 +#define NV_PMC_ENABLE 0x00000200 +#define NV_PMC_ENABLE_PWR 0x00002000 +#define NV_PMC_ENABLE_HUB 0x20000000 + +/* Falcon Base Pages for Different Engines */ +#define NV_PPWR_FALCON_BASE 0x10a000 +#define NV_PGSP_FALCON_BASE 0x110000 + +/* Falcon Common Register Offsets (relative to base_page) */ +#define NV_FALCON_DMACTL_OFFSET 0x010c +#define NV_FALCON_ENGINE_RESET_OFFSET 0x03c0 + +/* DMEM Control Register Flags */ +#define NV_PPWR_FALCON_DMEMC_AINCR_TRUE 0x01000000 +#define NV_PPWR_FALCON_DMEMC_AINCW_TRUE 0x02000000 + +/* Falcon DMEM port offsets (for port 0) */ +#define NV_FALCON_DMEMC_OFFSET 0x1c0 +#define NV_FALCON_DMEMD_OFFSET 0x1c4 + +/* DMA Register Offsets (relative to base_page) */ +#define NV_FALCON_DMA_ADDR_LOW_OFFSET 0x110 +#define NV_FALCON_DMA_MEM_OFFSET 0x114 +#define NV_FALCON_DMA_CMD_OFFSET 0x118 +#define NV_FALCON_DMA_BLOCK_OFFSET 0x11c +#define NV_FALCON_DMA_ADDR_HIGH_OFFSET 0x128 + +/* DMA Global Address Top Bits Register */ +#define NV_GPU_DMA_ADDR_TOP_BITS_REG 0x100f04 + +/* DMA Command Register Bit Definitions */ +#define NV_FALCON_DMA_CMD_WRITE_BIT 0x20 +#define NV_FALCON_DMA_CMD_SIZE_SHIFT 8 +#define NV_FALCON_DMA_CMD_DONE_BIT 0x2 + +/* + * Falcon DMA is synchronous, so a transfer size and count larger than + * its per-operation maximum adds no value. + */ + +/* DMA block size and alignment */ +#define NV_FALCON_DMA_MIN_TRANSFER_SIZE 4 +#define NV_FALCON_DMA_MAX_TRANSFER_SIZE 256 +#define NV_FALCON_DMA_BLOCK_SIZE 256 +#define NV_FALCON_DMA_MAX_TRANSFER_COUNT 1 + +/* DMACTL register bits */ +#define NV_FALCON_DMACTL_DMEM_SCRUBBING 0x1 +#define NV_FALCON_DMACTL_READY_MASK 0x6 + +/* Falcon Core Selection Register */ +#define NV_FALCON_CORE_SELECT_OFFSET 0x1668 +#define NV_FALCON_CORE_SELECT_MASK 0x30 + +/* Falcon mailbox register (for Ada+ reset check) */ +#define NV_FALCON_MAILBOX_TEST_OFFSET 0x40c +#define NV_FALCON_MAILBOX_RESET_MAGIC 0xbadf5620 + +/* Falcon Message Queue Register Offsets (relative to base_page) */ +#define NV_FALCON_QUEUE_HEAD_BASE_OFFSET 0x2c00 +#define NV_FALCON_QUEUE_TAIL_BASE_OFFSET 0x2c04 +#define NV_FALCON_QUEUE_STRIDE 0x8 +#define NV_FALCON_MSG_QUEUE_HEAD_BASE_OFFSET 0x2c80 +#define NV_FALCON_MSG_QUEUE_TAIL_BASE_OFFSET 0x2c84 + +/* FSP Falcon Base Pages */ +#define NV_FSP_FALCON_BASE 0x8f0100 +/* base_page = cpuctl & ~0xfff */ +#define NV_FSP_FALCON_BASE_PAGE 0x8f0000 +#define NV_FSP_EMEM_BASE 0x8f2000 + +/* FSP EMEM Port Offsets (relative to FSP EMEM base) */ +#define NV_FSP_EMEMC_OFFSET 0xac0 +#define NV_FSP_EMEMD_OFFSET 0xac4 +#define NV_FSP_EMEM_PORT_STRIDE 0x8 + +/* EMEM Control Register Flags (same as DMEM) */ +#define NV_FALCON_EMEMC_AINCR 0x01000000 +#define NV_FALCON_EMEMC_AINCW 0x02000000 + +/* FSP RPC channel configuration */ +#define NV_FSP_RPC_CHANNEL_SIZE 1024 +#define NV_FSP_RPC_MAX_PACKET_SIZE 1024 +#define NV_FSP_RPC_CHANNEL_HOPPER 2 +#define NV_FSP_RPC_EMEM_BASE \ + (NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_RPC_CHANNEL_SIZE) + +/* FSP EMEM port 2 registers (pre-computed for Hopper channel 2) */ +#define NV_FSP_EMEM_PORT2_CTRL (NV_FSP_EMEM_BASE + NV_FSP_EMEMC_OFFSET + \ + NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_EMEM_PORT_STRIDE) +#define NV_FSP_EMEM_PORT2_DATA (NV_FSP_EMEM_BASE + NV_FSP_EMEMD_OFFSET + \ + NV_FSP_RPC_CHANNEL_HOPPER * NV_FSP_EMEM_PORT_STRIDE) + +/* FSP queue register offsets (pre-computed for Hopper channel 2) */ +#define NV_FSP_QUEUE_HEAD \ + (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_QUEUE_HEAD_BASE_OFFSET + \ + NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE) +#define NV_FSP_QUEUE_TAIL \ + (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_QUEUE_TAIL_BASE_OFFSET + \ + NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE) +#define NV_FSP_MSG_QUEUE_HEAD \ + (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_MSG_QUEUE_HEAD_BASE_OFFSET + \ + NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE) +#define NV_FSP_MSG_QUEUE_TAIL \ + (NV_FSP_FALCON_BASE_PAGE + NV_FALCON_MSG_QUEUE_TAIL_BASE_OFFSET + \ + NV_FSP_RPC_CHANNEL_HOPPER * NV_FALCON_QUEUE_STRIDE) + +/* MCTP Header */ +#define NV_MCTP_HDR_SEID_SHIFT 16 +#define NV_MCTP_HDR_SEID_MASK 0xff +#define NV_MCTP_HDR_SEQ_SHIFT 28 +#define NV_MCTP_HDR_SEQ_MASK 0x3 +#define NV_MCTP_HDR_EOM_BIT 0x40000000 +#define NV_MCTP_HDR_SOM_BIT 0x80000000 + +/* MCTP Message Header */ +#define NV_MCTP_MSG_TYPE_SHIFT 0 +#define NV_MCTP_MSG_TYPE_MASK 0x7f +#define NV_MCTP_MSG_TYPE_VENDOR_DEFINED 0x7e +#define NV_MCTP_MSG_VENDOR_ID_SHIFT 8 +#define NV_MCTP_MSG_VENDOR_ID_MASK 0xffff +#define NV_MCTP_MSG_VENDOR_ID_NVIDIA 0x10de +#define NV_MCTP_MSG_NVDM_TYPE_SHIFT 24 +#define NV_MCTP_MSG_NVDM_TYPE_MASK 0xff + +/* NVDM response type */ +#define NV_NVDM_TYPE_RESPONSE 0x15 + +/* Minimum response size: mctp_hdr + msg_hdr + status_hdr + type + status */ +#define NV_FSP_RPC_MIN_RESPONSE_WORDS 5 + +/* FBIF (Frame Buffer Interface) Registers */ +/* Legacy PMU FBIF offsets (Kepler, Maxwell Gen1) */ +#define NV_PMU_LEGACY_FBIF_CTL_OFFSET 0x624 +#define NV_PMU_LEGACY_FBIF_TRANSCFG_OFFSET 0x600 + +/* PMU FBIF offsets */ +#define NV_PMU_FBIF_CTL_OFFSET 0xe24 +#define NV_PMU_FBIF_TRANSCFG_OFFSET 0xe00 + +/* GSP FBIF offsets */ +#define NV_GSP_FBIF_CTL_OFFSET 0x624 +#define NV_GSP_FBIF_TRANSCFG_OFFSET 0x600 + +/* OFA Falcon Base Page and FBIF offsets (used for Hopper+ DMA) */ +#define NV_OFA_FALCON_BASE 0x844000 +#define NV_OFA_FBIF_CTL_OFFSET 0x424 +#define NV_OFA_FBIF_TRANSCFG_OFFSET 0x400 + +/* OFA DMA support check register (Hopper+) */ +#define NV_OFA_DMA_SUPPORT_CHECK_REG 0x8443c0 + +/* FSP NVDM command types */ +#define NV_NVDM_TYPE_FBDMA 0x22 +#define NV_FBDMA_SUBCMD_ENABLE 0x1 + +/* FBIF CTL2 offset (relative to fbif_ctl) */ +#define NV_FBIF_CTL2_OFFSET 0x60 + +/* FBIF TRANSCFG register bits */ +#define NV_FBIF_TRANSCFG_TARGET_MASK 0x3 +#define NV_FBIF_TRANSCFG_SYSMEM_DEFAULT 0x5 + +/* FBIF CTL register bits */ +#define NV_FBIF_CTL_ALLOW_PHYS_MODE 0x10 +#define NV_FBIF_CTL_ALLOW_FULL_PHYS_MODE 0x80 + +/* Memory clear register offsets */ +#define NV_MEM_CLEAR_OFFSET 0x100b20 +#define NV_BOOT_COMPLETE_OFFSET 0x118234 +#define NV_BOOT_COMPLETE_SUCCESS 0x3ff + +/* FSP boot complete register (Hopper+) */ +#define NV_FSP_BOOT_COMPLETE_OFFSET 0x200bc +#define NV_FSP_BOOT_COMPLETE_SUCCESS 0xff + +enum gpu_arch { + GPU_ARCH_UNKNOWN = -1, + GPU_ARCH_KEPLER = 0, + GPU_ARCH_MAXWELL_GEN1, + GPU_ARCH_MAXWELL_GEN2, + GPU_ARCH_PASCAL, + GPU_ARCH_PASCAL_10X, + GPU_ARCH_VOLTA, + GPU_ARCH_TURING, + GPU_ARCH_AMPERE, + GPU_ARCH_ADA, + GPU_ARCH_HOPPER, +}; + +enum falcon_type { + FALCON_TYPE_PMU_LEGACY = 0, + FALCON_TYPE_PMU, + FALCON_TYPE_GSP, + FALCON_TYPE_OFA, +}; + +struct falcon { + u32 base_page; + u32 dmactl; + u32 engine_reset; + u32 fbif_ctl; + u32 fbif_ctl2; + u32 fbif_transcfg; + u32 dmem_control_reg; + u32 dmem_data_reg; + bool no_outside_reset; +}; + +struct gpu_properties { + u32 pmc_enable_mask; + bool memory_clear_supported; + enum falcon_type falcon_type; +}; + +static const u32 verified_gpu_map[] = { + 0x0e40a0a2, /* K520 */ + 0x0e6000a1, /* GTX660 */ + 0x0e63a0a1, /* K4000 */ + 0x0f22d0a1, /* K80 */ + 0x108000a1, /* GT635 */ + 0x117010a2, /* GTX750 */ + 0x117020a2, /* GTX745 */ + 0x124320a1, /* M60 */ + 0x130000a1, /* P100 */ + 0x134000a1, /* P4 */ + 0x132000a1, /* P40 */ + 0x140000a1, /* V100 */ + 0x164000a1, /* T4 */ + 0xb77000a1, /* A16 */ + 0x170000a1, /* A100 */ + 0xb72000a1, /* A10 */ + 0x180000a1, /* H100 */ + 0x194000a1, /* L4 */ + 0x192000a1, /* L40S */ +}; + +#define VERIFIED_GPU_MAP_SIZE ARRAY_SIZE(verified_gpu_map) + +static const struct gpu_properties gpu_properties_map[] = { + [GPU_ARCH_KEPLER] = { + .pmc_enable_mask = NV_PMC_ENABLE_PWR | NV_PMC_ENABLE_HUB, + .memory_clear_supported = false, + .falcon_type = FALCON_TYPE_PMU_LEGACY, + }, + [GPU_ARCH_MAXWELL_GEN1] = { + .pmc_enable_mask = NV_PMC_ENABLE_PWR | NV_PMC_ENABLE_HUB, + .memory_clear_supported = false, + .falcon_type = FALCON_TYPE_PMU_LEGACY, + }, + [GPU_ARCH_MAXWELL_GEN2] = { + .pmc_enable_mask = NV_PMC_ENABLE_PWR, + .memory_clear_supported = false, + .falcon_type = FALCON_TYPE_PMU, + }, + [GPU_ARCH_PASCAL] = { + .pmc_enable_mask = NV_PMC_ENABLE_PWR, + .memory_clear_supported = false, + .falcon_type = FALCON_TYPE_PMU, + }, + [GPU_ARCH_PASCAL_10X] = { + .pmc_enable_mask = 0, + .memory_clear_supported = false, + .falcon_type = FALCON_TYPE_PMU, + }, + [GPU_ARCH_VOLTA] = { + .pmc_enable_mask = 0, + .memory_clear_supported = false, + .falcon_type = FALCON_TYPE_GSP, + }, + [GPU_ARCH_TURING] = { + .pmc_enable_mask = 0, + .memory_clear_supported = true, + .falcon_type = FALCON_TYPE_GSP, + }, + [GPU_ARCH_AMPERE] = { + .pmc_enable_mask = 0, + .memory_clear_supported = true, + .falcon_type = FALCON_TYPE_GSP, + }, + [GPU_ARCH_ADA] = { + .pmc_enable_mask = 0, + .memory_clear_supported = true, + .falcon_type = FALCON_TYPE_PMU, + }, + [GPU_ARCH_HOPPER] = { + .pmc_enable_mask = 0, + .memory_clear_supported = true, + .falcon_type = FALCON_TYPE_OFA, + }, +}; + +static const struct falcon falcon_map[] = { + [FALCON_TYPE_PMU_LEGACY] = { + .base_page = NV_PPWR_FALCON_BASE, + .dmactl = NV_PPWR_FALCON_BASE + NV_FALCON_DMACTL_OFFSET, + .engine_reset = NV_PPWR_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET, + .fbif_ctl = NV_PPWR_FALCON_BASE + NV_PMU_LEGACY_FBIF_CTL_OFFSET, + .fbif_ctl2 = NV_PPWR_FALCON_BASE + + NV_PMU_LEGACY_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET, + .fbif_transcfg = NV_PPWR_FALCON_BASE + NV_PMU_LEGACY_FBIF_TRANSCFG_OFFSET, + .dmem_control_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMC_OFFSET, + .dmem_data_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMD_OFFSET, + .no_outside_reset = false, + }, + [FALCON_TYPE_PMU] = { + .base_page = NV_PPWR_FALCON_BASE, + .dmactl = NV_PPWR_FALCON_BASE + NV_FALCON_DMACTL_OFFSET, + .engine_reset = NV_PPWR_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET, + .fbif_ctl = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_CTL_OFFSET, + .fbif_ctl2 = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET, + .fbif_transcfg = NV_PPWR_FALCON_BASE + NV_PMU_FBIF_TRANSCFG_OFFSET, + .dmem_control_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMC_OFFSET, + .dmem_data_reg = NV_PPWR_FALCON_BASE + NV_FALCON_DMEMD_OFFSET, + .no_outside_reset = false, + }, + [FALCON_TYPE_GSP] = { + .base_page = NV_PGSP_FALCON_BASE, + .dmactl = NV_PGSP_FALCON_BASE + NV_FALCON_DMACTL_OFFSET, + .engine_reset = NV_PGSP_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET, + .fbif_ctl = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_CTL_OFFSET, + .fbif_ctl2 = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET, + .fbif_transcfg = NV_PGSP_FALCON_BASE + NV_GSP_FBIF_TRANSCFG_OFFSET, + .dmem_control_reg = NV_PGSP_FALCON_BASE + NV_FALCON_DMEMC_OFFSET, + .dmem_data_reg = NV_PGSP_FALCON_BASE + NV_FALCON_DMEMD_OFFSET, + .no_outside_reset = false, + }, + [FALCON_TYPE_OFA] = { + .base_page = NV_OFA_FALCON_BASE, + .dmactl = NV_OFA_FALCON_BASE + NV_FALCON_DMACTL_OFFSET, + .engine_reset = NV_OFA_FALCON_BASE + NV_FALCON_ENGINE_RESET_OFFSET, + .fbif_ctl = NV_OFA_FALCON_BASE + NV_OFA_FBIF_CTL_OFFSET, + .fbif_ctl2 = NV_OFA_FALCON_BASE + NV_OFA_FBIF_CTL_OFFSET + NV_FBIF_CTL2_OFFSET, + .fbif_transcfg = NV_OFA_FALCON_BASE + NV_OFA_FBIF_TRANSCFG_OFFSET, + .dmem_control_reg = NV_OFA_FALCON_BASE + NV_FALCON_DMEMC_OFFSET, + .dmem_data_reg = NV_OFA_FALCON_BASE + NV_FALCON_DMEMD_OFFSET, + .no_outside_reset = true, + }, +}; + +#endif /* _NV_FALCON_HW_H_ */ diff --git a/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c new file mode 100644 index 000000000000..c08aa81c44f4 --- /dev/null +++ b/tools/testing/selftests/vfio/lib/drivers/nv_falcon/nv_falcon.c @@ -0,0 +1,783 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. + */ +#include <stdint.h> +#include <strings.h> +#include <unistd.h> +#include <stdbool.h> +#include <string.h> +#include <time.h> + +#include <linux/errno.h> +#include <linux/io.h> +#include <linux/pci_ids.h> + +#include <libvfio.h> + +#include "hw.h" + +struct gpu_device { + enum gpu_arch arch; + void *bar0; + bool is_memory_clear_supported; + const struct falcon *falcon; + u32 pmc_enable_mask; + bool fsp_dma_enabled; + + /* Pending memcpy parameters, set by memcpy_start() */ + u64 memcpy_src; + u64 memcpy_dst; + u64 memcpy_size; +}; + +static inline struct gpu_device *to_gpu_device(struct vfio_pci_device *device) +{ + return device->driver.region.vaddr; +} + +static enum gpu_arch nv_gpu_arch_lookup(u32 pmc_boot_0) +{ + u32 arch = (pmc_boot_0 >> 24) & 0x1f; + + switch (arch) { + case 0x0e: + case 0x0f: + case 0x10: + return GPU_ARCH_KEPLER; + case 0x11: + return GPU_ARCH_MAXWELL_GEN1; + case 0x12: + return GPU_ARCH_MAXWELL_GEN2; + case 0x13: + /* P100 (impl 0) uses PMC reset; P4/P40 use engine reset */ + if (((pmc_boot_0 >> 20) & 0xf) == 0) + return GPU_ARCH_PASCAL; + return GPU_ARCH_PASCAL_10X; + case 0x14: + return GPU_ARCH_VOLTA; + case 0x16: + return GPU_ARCH_TURING; + case 0x17: + return GPU_ARCH_AMPERE; + case 0x18: + return GPU_ARCH_HOPPER; + case 0x19: + return GPU_ARCH_ADA; + default: + return GPU_ARCH_UNKNOWN; + } +} + +static inline u32 gpu_read32(struct gpu_device *gpu, u32 offset) +{ + return readl(gpu->bar0 + offset); +} + +static inline void gpu_write32(struct gpu_device *gpu, u32 offset, u32 value) +{ + writel(value, gpu->bar0 + offset); +} + +static u64 get_elapsed_ms(struct timespec *start) +{ + struct timespec now; + + clock_gettime(CLOCK_MONOTONIC, &now); + + return (now.tv_sec - start->tv_sec) * 1000 + + (now.tv_nsec - start->tv_nsec) / 1000000; +} + +static int gpu_poll_register(struct vfio_pci_device *device, + const char *name, u32 offset, + u32 expected, u32 mask, u32 timeout_ms) +{ + struct gpu_device *gpu = to_gpu_device(device); + struct timespec start; + u64 elapsed_ms; + u32 value; + + clock_gettime(CLOCK_MONOTONIC, &start); + + for (;;) { + value = gpu_read32(gpu, offset); + if ((value & mask) == expected) + return 0; + + elapsed_ms = get_elapsed_ms(&start); + + if (elapsed_ms >= timeout_ms) + break; + + usleep(1000); + } + + dev_err(device, + "Timeout polling %s (0x%x): value=0x%x expected=0x%x mask=0x%x after %lu ms\n", + name, offset, value, expected, mask, elapsed_ms); + return -ETIMEDOUT; +} + +static int fsp_poll_queue(struct vfio_pci_device *device, const char *name, + u32 head_reg, u32 tail_reg, bool wait_empty, + u32 timeout_ms) +{ + struct gpu_device *gpu = to_gpu_device(device); + struct timespec start; + u64 elapsed_ms; + u32 head, tail; + + clock_gettime(CLOCK_MONOTONIC, &start); + + for (;;) { + head = gpu_read32(gpu, head_reg); + tail = gpu_read32(gpu, tail_reg); + if (wait_empty ? (head == tail) : (head != tail)) + return 0; + + elapsed_ms = get_elapsed_ms(&start); + + if (elapsed_ms >= timeout_ms) + break; + + usleep(1000); + } + + dev_err(device, + "Timeout polling %s: head=0x%x tail=0x%x wait_empty=%d after %lu ms\n", + name, head, tail, wait_empty, elapsed_ms); + return -ETIMEDOUT; +} + +static void fsp_emem_write(struct vfio_pci_device *device, u32 offset, + const u32 *data, u32 count) +{ + struct gpu_device *gpu = to_gpu_device(device); + u32 i; + + /* Configure port with auto-increment for read and write */ + gpu_write32(gpu, NV_FSP_EMEM_PORT2_CTRL, + offset | NV_FALCON_EMEMC_AINCR | NV_FALCON_EMEMC_AINCW); + + for (i = 0; i < count; i++) + gpu_write32(gpu, NV_FSP_EMEM_PORT2_DATA, data[i]); +} + +static void fsp_emem_read(struct vfio_pci_device *device, u32 offset, + u32 *data, u32 count) +{ + struct gpu_device *gpu = to_gpu_device(device); + u32 i; + + /* Configure port with auto-increment for read and write */ + gpu_write32(gpu, NV_FSP_EMEM_PORT2_CTRL, + offset | NV_FALCON_EMEMC_AINCR | NV_FALCON_EMEMC_AINCW); + + for (i = 0; i < count; i++) + data[i] = gpu_read32(gpu, NV_FSP_EMEM_PORT2_DATA); +} + +static int fsp_rpc_send_data(struct vfio_pci_device *device, const u32 *data, + u32 count) +{ + struct gpu_device *gpu = to_gpu_device(device); + int ret; + + ret = fsp_poll_queue(device, "fsp_cmd_queue_empty", + NV_FSP_QUEUE_HEAD, NV_FSP_QUEUE_TAIL, true, 1000); + if (ret) + return ret; + + fsp_emem_write(device, NV_FSP_RPC_EMEM_BASE, data, count); + + /* Update queue head/tail to signal data is ready */ + gpu_write32(gpu, NV_FSP_QUEUE_TAIL, + NV_FSP_RPC_EMEM_BASE + (count - 1) * 4); + gpu_write32(gpu, NV_FSP_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE); + + return ret; +} + +static int fsp_rpc_receive_data(struct vfio_pci_device *device, u32 *data, + u32 max_count, u32 timeout_ms) +{ + struct gpu_device *gpu = to_gpu_device(device); + u32 head, tail; + u32 msg_size_words; + int ret; + + ret = fsp_poll_queue(device, "fsp_msg_queue_ready", + NV_FSP_MSG_QUEUE_HEAD, NV_FSP_MSG_QUEUE_TAIL, + false, timeout_ms); + if (ret) + return ret; + + head = gpu_read32(gpu, NV_FSP_MSG_QUEUE_HEAD); + tail = gpu_read32(gpu, NV_FSP_MSG_QUEUE_TAIL); + + msg_size_words = (tail - head + 4) / 4; + if (msg_size_words > max_count) + msg_size_words = max_count; + + fsp_emem_read(device, NV_FSP_RPC_EMEM_BASE, data, msg_size_words); + + /* Reset message queue tail to acknowledge receipt */ + gpu_write32(gpu, NV_FSP_MSG_QUEUE_TAIL, head); + + return msg_size_words; +} + +static void fsp_reset_rpc_state(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + u32 head, tail; + + head = gpu_read32(gpu, NV_FSP_QUEUE_HEAD); + tail = gpu_read32(gpu, NV_FSP_QUEUE_TAIL); + + if (head == tail) { + head = gpu_read32(gpu, NV_FSP_MSG_QUEUE_HEAD); + tail = gpu_read32(gpu, NV_FSP_MSG_QUEUE_TAIL); + if (head == tail) + return; + } + + /* Best-effort drain; timeout is expected if no pending message. */ + fsp_poll_queue(device, "fsp_msg_queue_drain", + NV_FSP_MSG_QUEUE_HEAD, NV_FSP_MSG_QUEUE_TAIL, + false, 5000); + + gpu_write32(gpu, NV_FSP_QUEUE_TAIL, NV_FSP_RPC_EMEM_BASE); + gpu_write32(gpu, NV_FSP_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE); + gpu_write32(gpu, NV_FSP_MSG_QUEUE_TAIL, NV_FSP_RPC_EMEM_BASE); + gpu_write32(gpu, NV_FSP_MSG_QUEUE_HEAD, NV_FSP_RPC_EMEM_BASE); +} + +static inline u32 mctp_header_build(u8 seid, u8 seq, bool som, bool eom) +{ + u32 hdr = 0; + + hdr |= (seid & NV_MCTP_HDR_SEID_MASK) << NV_MCTP_HDR_SEID_SHIFT; + hdr |= (seq & NV_MCTP_HDR_SEQ_MASK) << NV_MCTP_HDR_SEQ_SHIFT; + if (som) + hdr |= NV_MCTP_HDR_SOM_BIT; + if (eom) + hdr |= NV_MCTP_HDR_EOM_BIT; + + return hdr; +} + +static inline u32 mctp_msg_header_build(u8 nvdm_type) +{ + u32 hdr = 0; + + hdr |= (NV_MCTP_MSG_TYPE_VENDOR_DEFINED & NV_MCTP_MSG_TYPE_MASK) + << NV_MCTP_MSG_TYPE_SHIFT; + hdr |= (NV_MCTP_MSG_VENDOR_ID_NVIDIA & NV_MCTP_MSG_VENDOR_ID_MASK) + << NV_MCTP_MSG_VENDOR_ID_SHIFT; + hdr |= (nvdm_type & NV_MCTP_MSG_NVDM_TYPE_MASK) + << NV_MCTP_MSG_NVDM_TYPE_SHIFT; + + return hdr; +} + +static inline u8 mctp_msg_header_get_nvdm_type(u32 hdr) +{ + return (hdr >> NV_MCTP_MSG_NVDM_TYPE_SHIFT) & + NV_MCTP_MSG_NVDM_TYPE_MASK; +} + +static int fsp_rpc_send_cmd(struct vfio_pci_device *device, u8 nvdm_type, + const u32 *data, u32 data_count, u32 timeout_ms) +{ + u32 max_packet_words = NV_FSP_RPC_MAX_PACKET_SIZE / 4; + u32 packet[256]; + u32 resp_buf[256]; + u32 total_words; + int resp_words; + u8 resp_nvdm_type; + int ret; + + total_words = 2 + data_count; + if (total_words > max_packet_words) + return -EINVAL; + + packet[0] = mctp_header_build(0, 0, true, true); + packet[1] = mctp_msg_header_build(nvdm_type); + + if (data_count > 0) + memcpy(&packet[2], data, data_count * sizeof(u32)); + + ret = fsp_rpc_send_data(device, packet, total_words); + if (ret) + return ret; + + resp_words = fsp_rpc_receive_data(device, resp_buf, 256, timeout_ms); + if (resp_words < 0) + return resp_words; + + if (resp_words < NV_FSP_RPC_MIN_RESPONSE_WORDS) + return -EPROTO; + + resp_nvdm_type = mctp_msg_header_get_nvdm_type(resp_buf[1]); + if (resp_nvdm_type != NV_NVDM_TYPE_RESPONSE) + return -EPROTO; + + if (resp_buf[3] != nvdm_type) + return -EPROTO; + + if (resp_buf[4] != 0) + return -resp_buf[4]; + + return 0; +} + +static int fsp_init(struct vfio_pci_device *device) +{ + int ret; + + ret = gpu_poll_register(device, "fsp_boot_complete", + NV_FSP_BOOT_COMPLETE_OFFSET, + NV_FSP_BOOT_COMPLETE_SUCCESS, 0xffffffff, 5000); + if (ret) + return ret; + + fsp_reset_rpc_state(device); + return ret; +} + +static int fsp_fbdma_enable(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + u32 cmd_data = NV_FBDMA_SUBCMD_ENABLE; + int ret = 0; + + if (gpu->fsp_dma_enabled) + return ret; + + ret = fsp_rpc_send_cmd(device, NV_NVDM_TYPE_FBDMA, &cmd_data, 1, 5000); + if (ret) + return ret; + + gpu->fsp_dma_enabled = true; + return ret; +} + +static bool fsp_check_ofa_dma_support(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + u32 val = gpu_read32(gpu, NV_OFA_DMA_SUPPORT_CHECK_REG); + + return (val >> 16) != 0xbadf; +} + +static u32 size_to_dma_encoding(u64 size) +{ + VFIO_ASSERT_LE(size, NV_FALCON_DMA_MAX_TRANSFER_SIZE); + VFIO_ASSERT_GE(size, NV_FALCON_DMA_MIN_TRANSFER_SIZE); + VFIO_ASSERT_EQ(size & (size - 1), 0, "size must be power-of-2\n"); + + return ffs(size) - 3; +} + +static void falcon_dmem_port_configure(struct vfio_pci_device *device, + u32 offset, bool auto_inc_read, + bool auto_inc_write) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct falcon *falcon = gpu->falcon; + u32 memc_value = offset; + + /* Set auto-increment flags */ + if (auto_inc_read) + memc_value |= NV_PPWR_FALCON_DMEMC_AINCR_TRUE; + if (auto_inc_write) + memc_value |= NV_PPWR_FALCON_DMEMC_AINCW_TRUE; + + gpu_write32(gpu, falcon->dmem_control_reg, memc_value); +} + +static void falcon_select_core_falcon(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct falcon *falcon = gpu->falcon; + u32 core_select_reg = falcon->base_page + NV_FALCON_CORE_SELECT_OFFSET; + u32 core_select; + + core_select = gpu_read32(gpu, core_select_reg); + + /* Clear bits 4:5 to select falcon core (not RISCV) */ + core_select &= ~NV_FALCON_CORE_SELECT_MASK; + + gpu_write32(gpu, core_select_reg, core_select); +} + +static int falcon_enable(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct falcon *falcon = gpu->falcon; + u32 mailbox_test_reg; + u32 mailbox_val; + + if (falcon->no_outside_reset) + return 0; + + /* Ada-specific: Check if falcon needs reset before enable */ + if (gpu->arch == GPU_ARCH_ADA) { + mailbox_test_reg = falcon->base_page + + NV_FALCON_MAILBOX_TEST_OFFSET; + mailbox_val = gpu_read32(gpu, mailbox_test_reg); + if (mailbox_val == NV_FALCON_MAILBOX_RESET_MAGIC) + gpu_write32(gpu, falcon->engine_reset, 1); + } + + /* Enable the falcon based on control method */ + if (gpu->pmc_enable_mask != 0) { + u32 pmc_enable; + + /* Enable via PMC_ENABLE register */ + pmc_enable = gpu_read32(gpu, NV_PMC_ENABLE); + gpu_write32(gpu, NV_PMC_ENABLE, + pmc_enable | gpu->pmc_enable_mask); + } else { + /* Enable by deasserting engine reset */ + gpu_write32(gpu, falcon->engine_reset, 0); + } + + if (gpu->arch < GPU_ARCH_HOPPER) { + falcon_select_core_falcon(device); + + /* Wait for DMACTL to be ready (bits 1:2 should be 0) */ + return gpu_poll_register(device, "falcon_dmactl", + falcon->dmactl, 0, + NV_FALCON_DMACTL_READY_MASK, 1000); + } + + return 0; +} + +static void falcon_disable(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct falcon *falcon = gpu->falcon; + u32 pmc_enable; + + if (falcon->no_outside_reset) + return; + + if (gpu->pmc_enable_mask != 0) { + /* Disable via PMC_ENABLE */ + pmc_enable = gpu_read32(gpu, NV_PMC_ENABLE); + gpu_write32(gpu, NV_PMC_ENABLE, + pmc_enable & ~gpu->pmc_enable_mask); + } else { + /* Disable by asserting engine reset */ + gpu_write32(gpu, falcon->engine_reset, 1); + } +} + +static int falcon_reset(struct vfio_pci_device *device) +{ + falcon_disable(device); + + return falcon_enable(device); +} + +static int nv_falcon_dma_init(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct falcon *falcon; + u32 transcfg; + u32 dmactl; + u32 ctl; + int ret = 0; + + falcon = gpu->falcon; + + vfio_pci_cmd_set(device, PCI_COMMAND_MASTER); + + if (gpu->arch >= GPU_ARCH_HOPPER) { + ret = fsp_init(device); + if (ret) { + dev_err(device, "Failed to init FSP: %d\n", ret); + return ret; + } + + ret = fsp_fbdma_enable(device); + if (ret) { + dev_err(device, + "Failed to enable FSP FBDMA: %d\n", ret); + return ret; + } + + if (!fsp_check_ofa_dma_support(device)) { + dev_err(device, + "OFA DMA not supported with current firmware\n"); + return -EOPNOTSUPP; + } + } + + if (gpu->is_memory_clear_supported) { + /* For Turing+, wait for boot to complete first */ + if (gpu->arch >= GPU_ARCH_TURING) { + /* Wait for boot complete - Hopper+ uses FSP register */ + if (gpu->arch >= GPU_ARCH_HOPPER) { + ret = gpu_poll_register(device, + "fsp_boot_complete", + NV_FSP_BOOT_COMPLETE_OFFSET, + NV_FSP_BOOT_COMPLETE_SUCCESS, + 0xffffffff, 5000); + } else { + ret = gpu_poll_register(device, + "boot_complete", + NV_BOOT_COMPLETE_OFFSET, + NV_BOOT_COMPLETE_SUCCESS, + 0xffffffff, 5000); + } + if (ret) + return ret; + + ret = gpu_poll_register(device, + "memory_clear_finished", + NV_MEM_CLEAR_OFFSET, 0x1, 0xffffffff, 5000); + if (ret) + return ret; + } + } + + ret = falcon_reset(device); + if (ret) + return ret; + + falcon_dmem_port_configure(device, 0, false, false); + + transcfg = gpu_read32(gpu, falcon->fbif_transcfg); + transcfg &= ~NV_FBIF_TRANSCFG_TARGET_MASK; + transcfg |= NV_FBIF_TRANSCFG_SYSMEM_DEFAULT; + gpu_write32(gpu, falcon->fbif_transcfg, transcfg); + + gpu_write32(gpu, falcon->fbif_ctl2, 0x1); + + ctl = gpu_read32(gpu, falcon->fbif_ctl); + ctl |= NV_FBIF_CTL_ALLOW_PHYS_MODE | NV_FBIF_CTL_ALLOW_FULL_PHYS_MODE; + gpu_write32(gpu, falcon->fbif_ctl, ctl); + + dmactl = gpu_read32(gpu, falcon->dmactl); + dmactl &= ~NV_FALCON_DMACTL_DMEM_SCRUBBING; + gpu_write32(gpu, falcon->dmactl, dmactl); + + return ret; +} + +static int nv_falcon_dma(struct vfio_pci_device *device, + u64 address, u64 size, + bool write) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct falcon *falcon = gpu->falcon; + u32 dma_cmd; + int ret; + + gpu_write32(gpu, NV_GPU_DMA_ADDR_TOP_BITS_REG, + (address >> 47) & 0x1ffff); + gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_ADDR_HIGH_OFFSET, + (address >> 40) & 0x7f); + gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_ADDR_LOW_OFFSET, + (address >> 8) & 0xffffffff); + gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_BLOCK_OFFSET, + address & 0xff); + gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_MEM_OFFSET, 0); + + dma_cmd = size_to_dma_encoding(size) << NV_FALCON_DMA_CMD_SIZE_SHIFT; + + /* Set direction: write (DMEM->mem) or read (mem->DMEM) */ + if (write) + dma_cmd |= NV_FALCON_DMA_CMD_WRITE_BIT; + + gpu_write32(gpu, falcon->base_page + NV_FALCON_DMA_CMD_OFFSET, dma_cmd); + + ret = gpu_poll_register(device, "dma_done", + falcon->base_page + NV_FALCON_DMA_CMD_OFFSET, + NV_FALCON_DMA_CMD_DONE_BIT, + NV_FALCON_DMA_CMD_DONE_BIT, 1000); + if (ret) + dev_err(device, "Failed DMA %s (addr=0x%lx, size=%lu)\n", + write ? "write" : "read", address, size); + + return ret; +} + +static int nv_falcon_memcpy_chunk(struct vfio_pci_device *device, + iova_t src, iova_t dst, u64 size) +{ + int ret; + + ret = nv_falcon_dma(device, src, size, false); + if (ret) + return ret; + + return nv_falcon_dma(device, dst, size, true); +} + +static int nv_falcon_probe(struct vfio_pci_device *device) +{ + enum gpu_arch gpu_arch; + u32 pmc_boot_0; + void *bar0; + int i; + + if (vfio_pci_config_readw(device, PCI_VENDOR_ID) != + PCI_VENDOR_ID_NVIDIA) + return -ENODEV; + + if (vfio_pci_config_readw(device, PCI_CLASS_DEVICE) >> 8 != + PCI_BASE_CLASS_DISPLAY) + return -ENODEV; + + /* Get BAR0 pointer for reading GPU registers */ + bar0 = device->bars[0].vaddr; + if (!bar0) + return -ENODEV; + + /* Read PMC_BOOT_0 register from BAR0 to identify GPU */ + pmc_boot_0 = readl(bar0 + NV_PMC_BOOT_0); + + /* Look up GPU architecture to verify this is a supported GPU */ + gpu_arch = nv_gpu_arch_lookup(pmc_boot_0); + if (gpu_arch == GPU_ARCH_UNKNOWN) { + dev_err(device, + "Unsupported GPU architecture for PMC_BOOT_0: 0x%x\n", + pmc_boot_0); + return -ENODEV; + } + + /* Check verified GPU map */ + for (i = 0; i < VERIFIED_GPU_MAP_SIZE; i++) { + if (verified_gpu_map[i] == pmc_boot_0) + return 0; + } + + dev_info(device, + "Unvalidated GPU: PMC_BOOT_0: 0x%x, possibly not supported\n", + pmc_boot_0); + + return 0; +} + +static void nv_falcon_init(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + const struct gpu_properties *props; + u32 pmc_boot_0; + int ret; + + VFIO_ASSERT_GE(device->driver.region.size, sizeof(*gpu)); + + /* Read PMC_BOOT_0 register from BAR0 to identify GPU */ + pmc_boot_0 = readl(device->bars[0].vaddr + NV_PMC_BOOT_0); + + /* Look up GPU architecture */ + gpu->arch = nv_gpu_arch_lookup(pmc_boot_0); + + props = &gpu_properties_map[gpu->arch]; + + /* Populate GPU structure */ + gpu->bar0 = device->bars[0].vaddr; + gpu->is_memory_clear_supported = props->memory_clear_supported; + gpu->falcon = &falcon_map[props->falcon_type]; + gpu->pmc_enable_mask = props->pmc_enable_mask; + + /* Initialize falcon for DMA */ + ret = nv_falcon_dma_init(device); + VFIO_ASSERT_EQ(ret, 0, "Failed to initialize falcon DMA: %d\n", ret); + + device->driver.max_memcpy_size = NV_FALCON_DMA_MAX_TRANSFER_SIZE; + device->driver.max_memcpy_count = NV_FALCON_DMA_MAX_TRANSFER_COUNT; +} + +static void nv_falcon_remove(struct vfio_pci_device *device) +{ + falcon_disable(device); + vfio_pci_cmd_clear(device, PCI_COMMAND_MASTER); +} + +/* + * Falcon DMA can only process one transfer at a time, + * so the actual work is deferred to memcpy_wait() to conform to the + * memcpy_start()/memcpy_wait() contract. + */ +static void nv_falcon_memcpy_start(struct vfio_pci_device *device, + iova_t src, iova_t dst, u64 size, u64 count) +{ + struct gpu_device *gpu = to_gpu_device(device); + + VFIO_ASSERT_EQ(count, 1); + VFIO_ASSERT_EQ(size & (NV_FALCON_DMA_MIN_TRANSFER_SIZE - 1), 0, + "size 0x%lx must be %u-byte aligned\n", + (unsigned long)size, NV_FALCON_DMA_MIN_TRANSFER_SIZE); + + gpu->memcpy_src = src; + gpu->memcpy_dst = dst; + gpu->memcpy_size = size; +} + +/* + * Return the largest power-of-2 bytes we can transfer from @addr + * without crossing a DMA block boundary. + */ +static u64 dma_block_remain(u64 addr) +{ + u64 offset = addr & (NV_FALCON_DMA_BLOCK_SIZE - 1); + + if (!offset) + return NV_FALCON_DMA_BLOCK_SIZE; + + /* Lowest set bit of the offset is the largest aligned chunk */ + return 1ULL << (ffs(offset) - 1); +} + +static u64 rounddown_pow_of_two(u64 x) +{ + return 1ULL << (63 - __builtin_clzll(x)); +} + +static int nv_falcon_memcpy_wait(struct vfio_pci_device *device) +{ + struct gpu_device *gpu = to_gpu_device(device); + iova_t src = gpu->memcpy_src; + iova_t dst = gpu->memcpy_dst; + u64 remaining = gpu->memcpy_size; + int ret = 0; + + /* + * Falcon DMA supports power-of-2 transfer sizes in [4, 256] and + * cannot cross 256-byte block boundaries. Decompose the request + * into the largest valid chunk at each step. + */ + while (remaining) { + u64 chunk = rounddown_pow_of_two(remaining); + + chunk = min(chunk, dma_block_remain(src)); + chunk = min(chunk, dma_block_remain(dst)); + + ret = nv_falcon_memcpy_chunk(device, src, dst, chunk); + if (ret) + break; + + src += chunk; + dst += chunk; + remaining -= chunk; + } + + return ret; +} + +const struct vfio_pci_driver_ops nv_falcon_ops = { + .name = "nv_falcon", + .probe = nv_falcon_probe, + .init = nv_falcon_init, + .remove = nv_falcon_remove, + .memcpy_start = nv_falcon_memcpy_start, + .memcpy_wait = nv_falcon_memcpy_wait, +}; diff --git a/tools/testing/selftests/vfio/lib/include/libvfio.h b/tools/testing/selftests/vfio/lib/include/libvfio.h index 279ddcd70194..07862b470777 100644 --- a/tools/testing/selftests/vfio/lib/include/libvfio.h +++ b/tools/testing/selftests/vfio/lib/include/libvfio.h @@ -5,6 +5,7 @@ #include <libvfio/assert.h> #include <libvfio/iommu.h> #include <libvfio/iova_allocator.h> +#include <libvfio/sysfs.h> #include <libvfio/vfio_pci_device.h> #include <libvfio/vfio_pci_driver.h> @@ -23,4 +24,13 @@ const char *vfio_selftests_get_bdf(int *argc, char *argv[]); char **vfio_selftests_get_bdfs(int *argc, char *argv[], int *nr_bdfs); +/* + * Reserve virtual address space of size at an address satisfying + * (vaddr % align) == offset. + * + * Returns the reserved vaddr. The caller is responsible for unmapping + * the returned region. + */ +void *mmap_reserve(size_t size, size_t align, size_t offset); + #endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_H */ diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/assert.h b/tools/testing/selftests/vfio/lib/include/libvfio/assert.h index f4ebd122d9b6..9fff88f6e4e1 100644 --- a/tools/testing/selftests/vfio/lib/include/libvfio/assert.h +++ b/tools/testing/selftests/vfio/lib/include/libvfio/assert.h @@ -3,6 +3,7 @@ #define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_ASSERT_H #include <stdio.h> +#include <stdlib.h> #include <string.h> #include <sys/ioctl.h> @@ -45,10 +46,32 @@ VFIO_LOG_AND_EXIT(_fmt, ##__VA_ARGS__); \ } while (0) +#define malloc_assert(_size) ({ \ + size_t __size = (_size); \ + void *__ptr = malloc(__size); \ + VFIO_ASSERT_NOT_NULL(__ptr, "malloc(%zu) failed", \ + __size); \ + __ptr; \ +}) + +#define calloc_assert(_nmemb, _size) ({ \ + size_t __nmemb = (_nmemb); \ + size_t __size = (_size); \ + void *__ptr = calloc(__nmemb, __size); \ + VFIO_ASSERT_NOT_NULL(__ptr, "calloc(%zu, %zu) failed", \ + __nmemb, __size); \ + __ptr; \ +}) + #define ioctl_assert(_fd, _op, _arg) do { \ void *__arg = (_arg); \ int __ret = ioctl((_fd), (_op), (__arg)); \ VFIO_ASSERT_EQ(__ret, 0, "ioctl(%s, %s, %s) returned %d\n", #_fd, #_op, #_arg, __ret); \ } while (0) +#define snprintf_assert(_s, _size, _fmt, ...) do { \ + int __ret = snprintf(_s, _size, _fmt, ##__VA_ARGS__); \ + VFIO_ASSERT_LT(__ret, _size); \ +} while (0) + #endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_ASSERT_H */ diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h b/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h index 5c9b9dc6d993..e9a3386a4719 100644 --- a/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h +++ b/tools/testing/selftests/vfio/lib/include/libvfio/iommu.h @@ -61,6 +61,12 @@ iova_t iommu_hva2iova(struct iommu *iommu, void *vaddr); struct iommu_iova_range *iommu_iova_ranges(struct iommu *iommu, u32 *nranges); +#define MODE_VFIO_TYPE1_IOMMU "vfio_type1_iommu" +#define MODE_VFIO_TYPE1V2_IOMMU "vfio_type1v2_iommu" +#define MODE_IOMMUFD_COMPAT_TYPE1 "iommufd_compat_type1" +#define MODE_IOMMUFD_COMPAT_TYPE1V2 "iommufd_compat_type1v2" +#define MODE_IOMMUFD "iommufd" + /* * Generator for VFIO selftests fixture variants that replicate across all * possible IOMMU modes. Tests must define FIXTURE_VARIANT_ADD_IOMMU_MODE() diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h b/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h index 8f1d994e9ea2..c7c0796a757f 100644 --- a/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h +++ b/tools/testing/selftests/vfio/lib/include/libvfio/iova_allocator.h @@ -2,7 +2,6 @@ #ifndef SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_IOVA_ALLOCATOR_H #define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_IOVA_ALLOCATOR_H -#include <uapi/linux/types.h> #include <linux/list.h> #include <linux/types.h> #include <linux/iommufd.h> diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h b/tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h new file mode 100644 index 000000000000..c9ab1ea8f5a9 --- /dev/null +++ b/tools/testing/selftests/vfio/lib/include/libvfio/sysfs.h @@ -0,0 +1,12 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +#ifndef SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_SYSFS_H +#define SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_SYSFS_H + +int sysfs_sriov_totalvfs_get(const char *bdf); +int sysfs_sriov_numvfs_get(const char *bdf); +void sysfs_sriov_numvfs_set(const char *bdf, int numvfs); +char *sysfs_sriov_vf_bdf_get(const char *pf_bdf, int i); +int sysfs_iommu_group_get(const char *bdf); +char *sysfs_driver_get(const char *bdf); + +#endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_SYSFS_H */ diff --git a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h index 2858885a89bb..e19bd94b8dd2 100644 --- a/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h +++ b/tools/testing/selftests/vfio/lib/include/libvfio/vfio_pci_device.h @@ -38,9 +38,12 @@ struct vfio_pci_device { #define dev_info(_dev, _fmt, ...) printf("%s: " _fmt, (_dev)->bdf, ##__VA_ARGS__) #define dev_err(_dev, _fmt, ...) fprintf(stderr, "%s: " _fmt, (_dev)->bdf, ##__VA_ARGS__) +struct vfio_pci_device *vfio_pci_device_alloc(const char *bdf, struct iommu *iommu); +void vfio_pci_device_free(struct vfio_pci_device *device); struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu); void vfio_pci_device_cleanup(struct vfio_pci_device *device); +int __vfio_pci_device_reset(struct vfio_pci_device *device); void vfio_pci_device_reset(struct vfio_pci_device *device); void vfio_pci_config_access(struct vfio_pci_device *device, bool write, @@ -65,9 +68,25 @@ void vfio_pci_config_access(struct vfio_pci_device *device, bool write, #define vfio_pci_config_writew(_d, _o, _v) vfio_pci_config_write(_d, _o, _v, u16) #define vfio_pci_config_writel(_d, _o, _v) vfio_pci_config_write(_d, _o, _v, u32) +static inline void vfio_pci_cmd_set(struct vfio_pci_device *device, u16 bits) +{ + u16 cmd = vfio_pci_config_readw(device, PCI_COMMAND); + + vfio_pci_config_writew(device, PCI_COMMAND, cmd | bits); +} + +static inline void vfio_pci_cmd_clear(struct vfio_pci_device *device, u16 bits) +{ + u16 cmd = vfio_pci_config_readw(device, PCI_COMMAND); + + vfio_pci_config_writew(device, PCI_COMMAND, cmd & ~bits); +} + void vfio_pci_irq_enable(struct vfio_pci_device *device, u32 index, u32 vector, int count); void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index); +void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index, + u32 vector, int count); void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector); static inline void fcntl_set_nonblock(int fd) @@ -92,6 +111,12 @@ static inline void vfio_pci_msi_disable(struct vfio_pci_device *device) vfio_pci_irq_disable(device, VFIO_PCI_MSI_IRQ_INDEX); } +static inline void vfio_pci_msi_reenable(struct vfio_pci_device *device, + u32 vector, int count) +{ + vfio_pci_irq_reenable(device, VFIO_PCI_MSI_IRQ_INDEX, vector, count); +} + static inline void vfio_pci_msix_enable(struct vfio_pci_device *device, u32 vector, int count) { @@ -103,6 +128,12 @@ static inline void vfio_pci_msix_disable(struct vfio_pci_device *device) vfio_pci_irq_disable(device, VFIO_PCI_MSIX_IRQ_INDEX); } +static inline void vfio_pci_msix_reenable(struct vfio_pci_device *device, + u32 vector, int count) +{ + vfio_pci_irq_reenable(device, VFIO_PCI_MSIX_IRQ_INDEX, vector, count); +} + static inline int __to_iova(struct vfio_pci_device *device, void *vaddr, iova_t *iova) { return __iommu_hva2iova(device->iommu, vaddr, iova); @@ -122,4 +153,13 @@ static inline bool vfio_pci_device_match(struct vfio_pci_device *device, const char *vfio_pci_get_cdev_path(const char *bdf); +void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf); +void __vfio_pci_group_get_device_fd(struct vfio_pci_device *device, + const char *bdf, const char *vf_token); +void vfio_container_set_iommu(struct vfio_pci_device *device); +void vfio_pci_cdev_open(struct vfio_pci_device *device, const char *bdf); +int __vfio_device_bind_iommufd(int device_fd, int iommufd, const char *vf_token); + +void vfio_device_set_vf_token(int fd, const char *vf_token); + #endif /* SELFTESTS_VFIO_LIB_INCLUDE_LIBVFIO_VFIO_PCI_DEVICE_H */ diff --git a/tools/testing/selftests/vfio/lib/iommu.c b/tools/testing/selftests/vfio/lib/iommu.c index 8079d43523f3..b6f3c5c84e01 100644 --- a/tools/testing/selftests/vfio/lib/iommu.c +++ b/tools/testing/selftests/vfio/lib/iommu.c @@ -11,7 +11,6 @@ #include <sys/ioctl.h> #include <sys/mman.h> -#include <uapi/linux/types.h> #include <linux/limits.h> #include <linux/mman.h> #include <linux/types.h> @@ -21,32 +20,32 @@ #include "../../../kselftest.h" #include <libvfio.h> -const char *default_iommu_mode = "iommufd"; +const char *default_iommu_mode = MODE_IOMMUFD; /* Reminder: Keep in sync with FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(). */ static const struct iommu_mode iommu_modes[] = { { - .name = "vfio_type1_iommu", + .name = MODE_VFIO_TYPE1_IOMMU, .container_path = "/dev/vfio/vfio", .iommu_type = VFIO_TYPE1_IOMMU, }, { - .name = "vfio_type1v2_iommu", + .name = MODE_VFIO_TYPE1V2_IOMMU, .container_path = "/dev/vfio/vfio", .iommu_type = VFIO_TYPE1v2_IOMMU, }, { - .name = "iommufd_compat_type1", + .name = MODE_IOMMUFD_COMPAT_TYPE1, .container_path = "/dev/iommu", .iommu_type = VFIO_TYPE1_IOMMU, }, { - .name = "iommufd_compat_type1v2", + .name = MODE_IOMMUFD_COMPAT_TYPE1V2, .container_path = "/dev/iommu", .iommu_type = VFIO_TYPE1v2_IOMMU, }, { - .name = "iommufd", + .name = MODE_IOMMUFD, }, }; @@ -287,8 +286,7 @@ static struct vfio_iommu_type1_info *vfio_iommu_get_info(int container_fd) { struct vfio_iommu_type1_info *info; - info = malloc(sizeof(*info)); - VFIO_ASSERT_NOT_NULL(info); + info = malloc_assert(sizeof(*info)); *info = (struct vfio_iommu_type1_info) { .argsz = sizeof(*info), @@ -325,8 +323,7 @@ static struct iommu_iova_range *vfio_iommu_iova_ranges(struct iommu *iommu, cap_range = container_of(hdr, struct vfio_iommu_type1_info_cap_iova_range, header); VFIO_ASSERT_GT(cap_range->nr_iovas, 0); - ranges = calloc(cap_range->nr_iovas, sizeof(*ranges)); - VFIO_ASSERT_NOT_NULL(ranges); + ranges = calloc_assert(cap_range->nr_iovas, sizeof(*ranges)); for (u32 i = 0; i < cap_range->nr_iovas; i++) { ranges[i] = (struct iommu_iova_range){ @@ -358,8 +355,7 @@ static struct iommu_iova_range *iommufd_iova_ranges(struct iommu *iommu, VFIO_ASSERT_EQ(errno, EMSGSIZE); VFIO_ASSERT_GT(query.num_iovas, 0); - ranges = calloc(query.num_iovas, sizeof(*ranges)); - VFIO_ASSERT_NOT_NULL(ranges); + ranges = calloc_assert(query.num_iovas, sizeof(*ranges)); query.allowed_iovas = (uintptr_t)ranges; @@ -425,8 +421,7 @@ struct iommu *iommu_init(const char *iommu_mode) struct iommu *iommu; int version; - iommu = calloc(1, sizeof(*iommu)); - VFIO_ASSERT_NOT_NULL(iommu); + iommu = calloc_assert(1, sizeof(*iommu)); INIT_LIST_HEAD(&iommu->dma_regions); diff --git a/tools/testing/selftests/vfio/lib/iova_allocator.c b/tools/testing/selftests/vfio/lib/iova_allocator.c index a12b0a51e9e6..4a660f636f49 100644 --- a/tools/testing/selftests/vfio/lib/iova_allocator.c +++ b/tools/testing/selftests/vfio/lib/iova_allocator.c @@ -11,7 +11,6 @@ #include <sys/ioctl.h> #include <sys/mman.h> -#include <uapi/linux/types.h> #include <linux/iommufd.h> #include <linux/limits.h> #include <linux/mman.h> @@ -30,8 +29,7 @@ struct iova_allocator *iova_allocator_init(struct iommu *iommu) ranges = iommu_iova_ranges(iommu, &nranges); VFIO_ASSERT_NOT_NULL(ranges); - allocator = malloc(sizeof(*allocator)); - VFIO_ASSERT_NOT_NULL(allocator); + allocator = malloc_assert(sizeof(*allocator)); *allocator = (struct iova_allocator){ .ranges = ranges, @@ -91,4 +89,3 @@ next_range: allocator->range_offset = 0; } } - diff --git a/tools/testing/selftests/vfio/lib/libvfio.c b/tools/testing/selftests/vfio/lib/libvfio.c index a23a3cc5be69..3a3d1ed635c1 100644 --- a/tools/testing/selftests/vfio/lib/libvfio.c +++ b/tools/testing/selftests/vfio/lib/libvfio.c @@ -2,6 +2,9 @@ #include <stdio.h> #include <stdlib.h> +#include <sys/mman.h> + +#include <linux/align.h> #include "../../../kselftest.h" #include <libvfio.h> @@ -76,3 +79,25 @@ const char *vfio_selftests_get_bdf(int *argc, char *argv[]) return vfio_selftests_get_bdfs(argc, argv, &nr_bdfs)[0]; } + +void *mmap_reserve(size_t size, size_t align, size_t offset) +{ + void *map_base, *map_align; + size_t delta; + + VFIO_ASSERT_GT(align, offset); + delta = align - offset; + + map_base = mmap(NULL, size + align, PROT_NONE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + VFIO_ASSERT_NE(map_base, MAP_FAILED); + + map_align = (void *)(ALIGN((uintptr_t)map_base + delta, align) - delta); + + if (map_align > map_base) + VFIO_ASSERT_EQ(munmap(map_base, map_align - map_base), 0); + + VFIO_ASSERT_EQ(munmap(map_align + size, map_base + align - map_align), 0); + + return map_align; +} diff --git a/tools/testing/selftests/vfio/lib/libvfio.mk b/tools/testing/selftests/vfio/lib/libvfio.mk index 9f47bceed16f..bcfa74ae040e 100644 --- a/tools/testing/selftests/vfio/lib/libvfio.mk +++ b/tools/testing/selftests/vfio/lib/libvfio.mk @@ -6,6 +6,7 @@ LIBVFIO_SRCDIR := $(selfdir)/vfio/lib LIBVFIO_C := iommu.c LIBVFIO_C += iova_allocator.c LIBVFIO_C += libvfio.c +LIBVFIO_C += sysfs.c LIBVFIO_C += vfio_pci_device.c LIBVFIO_C += vfio_pci_driver.c @@ -14,16 +15,23 @@ LIBVFIO_C += drivers/ioat/ioat.c LIBVFIO_C += drivers/dsa/dsa.c endif +LIBVFIO_C += drivers/nv_falcon/nv_falcon.c +LIBVFIO_C += drivers/igb/igb.c + LIBVFIO_OUTPUT := $(OUTPUT)/libvfio LIBVFIO_O := $(patsubst %.c, $(LIBVFIO_OUTPUT)/%.o, $(LIBVFIO_C)) LIBVFIO_O_DIRS := $(shell dirname $(LIBVFIO_O) | uniq) -$(shell mkdir -p $(LIBVFIO_O_DIRS)) + +$(LIBVFIO_O_DIRS): + mkdir -p $@ CFLAGS += -I$(LIBVFIO_SRCDIR)/include -$(LIBVFIO_O): $(LIBVFIO_OUTPUT)/%.o : $(LIBVFIO_SRCDIR)/%.c +LDLIBS += -luuid + +$(LIBVFIO_O): $(LIBVFIO_OUTPUT)/%.o : $(LIBVFIO_SRCDIR)/%.c | $(LIBVFIO_O_DIRS) $(CC) $(CFLAGS) $(CPPFLAGS) $(TARGET_ARCH) -c $< -o $@ EXTRA_CLEAN += $(LIBVFIO_OUTPUT) diff --git a/tools/testing/selftests/vfio/lib/sysfs.c b/tools/testing/selftests/vfio/lib/sysfs.c new file mode 100644 index 000000000000..98a46a2543cd --- /dev/null +++ b/tools/testing/selftests/vfio/lib/sysfs.c @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include <fcntl.h> +#include <unistd.h> +#include <stdlib.h> +#include <string.h> +#include <linux/limits.h> + +#include <libvfio.h> + +#define readlink_safe(_path, _buf) ({ \ + int __ret; \ + \ + _Static_assert(!__builtin_types_compatible_p( \ + __typeof__(_buf), char *), \ + "readlink_safe: _buf must be an array, not a pointer"); \ + \ + __ret = readlink(_path, _buf, sizeof(_buf) - 1); \ + if (__ret != -1) \ + _buf[__ret] = 0; \ + __ret; \ +}) + +static void readlink_base(const char *path, const char *data_fmt, void *out_data) +{ + char rl_path[PATH_MAX]; + int ret; + + ret = readlink_safe(path, rl_path); + VFIO_ASSERT_NE(ret, -1); + + ret = sscanf(basename(rl_path), data_fmt, out_data); + VFIO_ASSERT_EQ(ret, 1); +} + +static int sysfs_val_get_int(const char *component, const char *name, + const char *file) +{ + char path[PATH_MAX]; + char buf[32]; + int ret; + int fd; + + snprintf_assert(path, PATH_MAX, "/sys/bus/pci/%s/%s/%s", component, name, file); + fd = open(path, O_RDONLY); + if (fd < 0) + return fd; + + VFIO_ASSERT_GT(read(fd, buf, ARRAY_SIZE(buf)), 0); + VFIO_ASSERT_EQ(close(fd), 0); + + errno = 0; + ret = strtol(buf, NULL, 0); + VFIO_ASSERT_EQ(errno, 0, "sysfs path \"%s\" is not an integer: \"%s\"\n", path, buf); + + return ret; +} + +static void sysfs_val_set(const char *component, const char *name, + const char *file, const char *val) +{ + char path[PATH_MAX]; + int fd; + + snprintf_assert(path, PATH_MAX, "/sys/bus/pci/%s/%s/%s", component, name, file); + VFIO_ASSERT_GT(fd = open(path, O_WRONLY), 0); + + VFIO_ASSERT_EQ(write(fd, val, strlen(val)), strlen(val)); + VFIO_ASSERT_EQ(close(fd), 0); +} + +static int sysfs_device_val_get(const char *bdf, const char *file) +{ + return sysfs_val_get_int("devices", bdf, file); +} + +static void sysfs_device_val_set(const char *bdf, const char *file, const char *val) +{ + sysfs_val_set("devices", bdf, file, val); +} + +static void sysfs_device_val_set_int(const char *bdf, const char *file, int val) +{ + char val_str[32]; + + snprintf_assert(val_str, sizeof(val_str), "%d", val); + sysfs_device_val_set(bdf, file, val_str); +} + +int sysfs_sriov_totalvfs_get(const char *bdf) +{ + return sysfs_device_val_get(bdf, "sriov_totalvfs"); +} + +int sysfs_sriov_numvfs_get(const char *bdf) +{ + return sysfs_device_val_get(bdf, "sriov_numvfs"); +} + +void sysfs_sriov_numvfs_set(const char *bdf, int numvfs) +{ + sysfs_device_val_set_int(bdf, "sriov_numvfs", numvfs); +} + +char *sysfs_sriov_vf_bdf_get(const char *pf_bdf, int i) +{ + char path[PATH_MAX]; + char *out_vf_bdf; + + /* Fit "0000:00:00.0" */ + out_vf_bdf = calloc_assert(16, sizeof(char)); + + snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/virtfn%d", pf_bdf, i); + readlink_base(path, "%s", out_vf_bdf); + + return out_vf_bdf; +} + +int sysfs_iommu_group_get(const char *bdf) +{ + char path[PATH_MAX]; + int group; + + snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/iommu_group", bdf); + readlink_base(path, "%d", &group); + + return group; +} + +char *sysfs_driver_get(const char *bdf) +{ + char driver_path[PATH_MAX]; + char path[PATH_MAX]; + char *out_driver; + int ret; + + snprintf_assert(path, PATH_MAX, "/sys/bus/pci/devices/%s/driver", bdf); + ret = readlink_safe(path, driver_path); + if (ret == -1) { + if (errno == ENOENT) + return NULL; + + VFIO_FAIL("Failed to read %s\n", path); + } + + out_driver = strdup(basename(driver_path)); + VFIO_ASSERT_NOT_NULL(out_driver); + + return out_driver; +} diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_device.c b/tools/testing/selftests/vfio/lib/vfio_pci_device.c index 8e34b9bfc96b..4063a0e2b3df 100644 --- a/tools/testing/selftests/vfio/lib/vfio_pci_device.c +++ b/tools/testing/selftests/vfio/lib/vfio_pci_device.c @@ -1,5 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only #include <dirent.h> +#include <errno.h> #include <fcntl.h> #include <libgen.h> #include <stdint.h> @@ -11,27 +12,30 @@ #include <sys/ioctl.h> #include <sys/mman.h> -#include <uapi/linux/types.h> +#include <linux/align.h> #include <linux/iommufd.h> +#include <linux/kernel.h> #include <linux/limits.h> +#include <linux/log2.h> #include <linux/mman.h> #include <linux/overflow.h> +#include <linux/sizes.h> #include <linux/types.h> #include <linux/vfio.h> +#include <uuid/uuid.h> + #include "kselftest.h" #include <libvfio.h> -#define PCI_SYSFS_PATH "/sys/bus/pci/devices" - static void vfio_pci_irq_set(struct vfio_pci_device *device, u32 index, u32 vector, u32 count, int *fds) { - u8 buf[sizeof(struct vfio_irq_set) + sizeof(int) * count] = {}; - struct vfio_irq_set *irq = (void *)&buf; - int *irq_fds = (void *)&irq->data; + size_t argsz = sizeof(struct vfio_irq_set) + sizeof(int) * count; + struct vfio_irq_set *irq; - irq->argsz = sizeof(buf); + irq = calloc_assert(1, argsz); + irq->argsz = argsz; irq->flags = VFIO_IRQ_SET_ACTION_TRIGGER; irq->index = index; irq->start = vector; @@ -39,12 +43,13 @@ static void vfio_pci_irq_set(struct vfio_pci_device *device, if (count) { irq->flags |= VFIO_IRQ_SET_DATA_EVENTFD; - memcpy(irq_fds, fds, sizeof(int) * count); + memcpy(irq->data, fds, sizeof(int) * count); } else { irq->flags |= VFIO_IRQ_SET_DATA_NONE; } ioctl_assert(device->fd, VFIO_DEVICE_SET_IRQS, irq); + free(irq); } void vfio_pci_irq_trigger(struct vfio_pci_device *device, u32 index, u32 vector) @@ -101,6 +106,28 @@ void vfio_pci_irq_disable(struct vfio_pci_device *device, u32 index) vfio_pci_irq_set(device, index, 0, 0, NULL); } +/* + * Re-issue VFIO_DEVICE_SET_IRQS for an already-enabled vector range using + * the existing eventfds. Intended for drivers that need to re-arm device + * interrupts after a VFIO_DEVICE_RESET, which tears down the kernel-side + * IRQ trigger but leaves user-side eventfds intact. Recreating the + * eventfds would invalidate any test-fixture cache of the fd, so this + * helper deliberately preserves them. + */ +void vfio_pci_irq_reenable(struct vfio_pci_device *device, u32 index, + u32 vector, int count) +{ + int i; + + check_supported_irq_index(index); + + for (i = vector; i < vector + count; i++) + VFIO_ASSERT_GE(device->msi_eventfds[i], 0, + "vector %d eventfd not allocated\n", i); + + vfio_pci_irq_set(device, index, vector, count, device->msi_eventfds + vector); +} + static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index, struct vfio_irq_info *irq_info) { @@ -110,6 +137,45 @@ static void vfio_pci_irq_get(struct vfio_pci_device *device, u32 index, ioctl_assert(device->fd, VFIO_DEVICE_GET_IRQ_INFO, irq_info); } +static int vfio_device_feature_ioctl(int fd, u32 flags, void *data, + size_t data_size) +{ + size_t argsz = sizeof(struct vfio_device_feature) + data_size; + struct vfio_device_feature *feature; + int ret; + + feature = calloc_assert(1, argsz); + memcpy(feature->data, data, data_size); + + feature->argsz = argsz; + feature->flags = flags; + + ret = ioctl(fd, VFIO_DEVICE_FEATURE, feature); + free(feature); + + return ret; +} + +static void vfio_device_feature_set(int fd, u16 feature, void *data, size_t data_size) +{ + u32 flags = VFIO_DEVICE_FEATURE_SET | feature; + int ret; + + ret = vfio_device_feature_ioctl(fd, flags, data, data_size); + VFIO_ASSERT_EQ(ret, 0, "Failed to set feature %u\n", feature); +} + +void vfio_device_set_vf_token(int fd, const char *vf_token) +{ + uuid_t token_uuid = {0}; + + VFIO_ASSERT_NOT_NULL(vf_token, "vf_token is NULL"); + VFIO_ASSERT_EQ(uuid_parse(vf_token, token_uuid), 0); + + vfio_device_feature_set(fd, VFIO_DEVICE_FEATURE_PCI_VF_TOKEN, + token_uuid, sizeof(uuid_t)); +} + static void vfio_pci_region_get(struct vfio_pci_device *device, int index, struct vfio_region_info *info) { @@ -124,20 +190,38 @@ static void vfio_pci_region_get(struct vfio_pci_device *device, int index, static void vfio_pci_bar_map(struct vfio_pci_device *device, int index) { struct vfio_pci_bar *bar = &device->bars[index]; + size_t align, size; int prot = 0; + void *vaddr; VFIO_ASSERT_LT(index, PCI_STD_NUM_BARS); VFIO_ASSERT_NULL(bar->vaddr); VFIO_ASSERT_TRUE(bar->info.flags & VFIO_REGION_INFO_FLAG_MMAP); + VFIO_ASSERT_TRUE(is_power_of_2(bar->info.size)); if (bar->info.flags & VFIO_REGION_INFO_FLAG_READ) prot |= PROT_READ; if (bar->info.flags & VFIO_REGION_INFO_FLAG_WRITE) prot |= PROT_WRITE; - bar->vaddr = mmap(NULL, bar->info.size, prot, MAP_FILE | MAP_SHARED, + size = bar->info.size; + + /* + * Align BAR mmaps to improve page fault granularity during potential + * subsequent IOMMU mapping of these BAR vaddr. 1G for x86 is the + * largest hugepage size across any architecture, so no benefit from + * larger alignment. BARs smaller than 1G will be aligned by their + * power-of-two size, guaranteeing sufficient alignment for smaller + * hugepages, if present. + */ + align = min_t(size_t, size, SZ_1G); + + vaddr = mmap_reserve(size, align, 0); + bar->vaddr = mmap(vaddr, size, prot, MAP_SHARED | MAP_FIXED, device->fd, bar->info.offset); VFIO_ASSERT_NE(bar->vaddr, MAP_FAILED); + + madvise(bar->vaddr, size, MADV_HUGEPAGE); } static void vfio_pci_bar_unmap(struct vfio_pci_device *device, int index) @@ -176,30 +260,29 @@ void vfio_pci_config_access(struct vfio_pci_device *device, bool write, write ? "write to" : "read from", config); } -void vfio_pci_device_reset(struct vfio_pci_device *device) +int __vfio_pci_device_reset(struct vfio_pci_device *device) { - ioctl_assert(device->fd, VFIO_DEVICE_RESET, NULL); + if (ioctl(device->fd, VFIO_DEVICE_RESET, NULL)) + return -errno; + + return 0; } -static unsigned int vfio_pci_get_group_from_dev(const char *bdf) +void vfio_pci_device_reset(struct vfio_pci_device *device) { - char dev_iommu_group_path[PATH_MAX] = {0}; - char sysfs_path[PATH_MAX] = {0}; - unsigned int group; - int ret; - - snprintf(sysfs_path, PATH_MAX, "%s/%s/iommu_group", PCI_SYSFS_PATH, bdf); + int retries = 20; + int r; - ret = readlink(sysfs_path, dev_iommu_group_path, sizeof(dev_iommu_group_path)); - VFIO_ASSERT_NE(ret, -1, "Failed to get the IOMMU group for device: %s\n", bdf); + do { + r = __vfio_pci_device_reset(device); + if (r == -EAGAIN) + usleep(10000); + } while (r == -EAGAIN && retries-- > 0); - ret = sscanf(basename(dev_iommu_group_path), "%u", &group); - VFIO_ASSERT_EQ(ret, 1, "Failed to get the IOMMU group for device: %s\n", bdf); - - return group; + VFIO_ASSERT_EQ(r, 0, "ioctl(device->fd, VFIO_DEVICE_RESET) failed\n"); } -static void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf) +void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf) { struct vfio_group_status group_status = { .argsz = sizeof(group_status), @@ -207,8 +290,8 @@ static void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf char group_path[32]; int group; - group = vfio_pci_get_group_from_dev(bdf); - snprintf(group_path, sizeof(group_path), "/dev/vfio/%d", group); + group = sysfs_iommu_group_get(bdf); + snprintf_assert(group_path, sizeof(group_path), "/dev/vfio/%d", group); device->group_fd = open(group_path, O_RDWR); VFIO_ASSERT_GE(device->group_fd, 0, "open(%s) failed\n", group_path); @@ -219,14 +302,37 @@ static void vfio_pci_group_setup(struct vfio_pci_device *device, const char *bdf ioctl_assert(device->group_fd, VFIO_GROUP_SET_CONTAINER, &device->iommu->container_fd); } -static void vfio_pci_container_setup(struct vfio_pci_device *device, const char *bdf) +void __vfio_pci_group_get_device_fd(struct vfio_pci_device *device, + const char *bdf, const char *vf_token) +{ + char arg[64]; + + /* + * If a vf_token exists, argument to VFIO_GROUP_GET_DEVICE_FD + * will be in the form of the following example: + * "0000:04:10.0 vf_token=bd8d9d2b-5a5f-4f5a-a211-f591514ba1f3" + */ + if (vf_token) + snprintf_assert(arg, ARRAY_SIZE(arg), "%s vf_token=%s", bdf, vf_token); + else + snprintf_assert(arg, ARRAY_SIZE(arg), "%s", bdf); + + device->fd = ioctl(device->group_fd, VFIO_GROUP_GET_DEVICE_FD, arg); +} + +static void vfio_pci_group_get_device_fd(struct vfio_pci_device *device, + const char *bdf, const char *vf_token) +{ + __vfio_pci_group_get_device_fd(device, bdf, vf_token); + VFIO_ASSERT_GE(device->fd, 0); +} + +void vfio_container_set_iommu(struct vfio_pci_device *device) { struct iommu *iommu = device->iommu; unsigned long iommu_type = iommu->mode->iommu_type; int ret; - vfio_pci_group_setup(device, bdf); - ret = ioctl(iommu->container_fd, VFIO_CHECK_EXTENSION, iommu_type); VFIO_ASSERT_GT(ret, 0, "VFIO IOMMU type %lu not supported\n", iommu_type); @@ -236,9 +342,14 @@ static void vfio_pci_container_setup(struct vfio_pci_device *device, const char * because the IOMMU type is already set. */ (void)ioctl(iommu->container_fd, VFIO_SET_IOMMU, (void *)iommu_type); +} - device->fd = ioctl(device->group_fd, VFIO_GROUP_GET_DEVICE_FD, bdf); - VFIO_ASSERT_GE(device->fd, 0); +static void vfio_pci_container_setup(struct vfio_pci_device *device, + const char *bdf, const char *vf_token) +{ + vfio_pci_group_setup(device, bdf); + vfio_container_set_iommu(device); + vfio_pci_group_get_device_fd(device, bdf, vf_token); } static void vfio_pci_device_setup(struct vfio_pci_device *device) @@ -276,10 +387,9 @@ const char *vfio_pci_get_cdev_path(const char *bdf) char *cdev_path; DIR *dir; - cdev_path = calloc(PATH_MAX, 1); - VFIO_ASSERT_NOT_NULL(cdev_path); + cdev_path = calloc_assert(PATH_MAX, 1); - snprintf(dir_path, sizeof(dir_path), "/sys/bus/pci/devices/%s/vfio-dev/", bdf); + snprintf_assert(dir_path, sizeof(dir_path), "/sys/bus/pci/devices/%s/vfio-dev/", bdf); dir = opendir(dir_path); VFIO_ASSERT_NOT_NULL(dir, "Failed to open directory %s\n", dir_path); @@ -289,7 +399,7 @@ const char *vfio_pci_get_cdev_path(const char *bdf) if (strncmp("vfio", entry->d_name, 4)) continue; - snprintf(cdev_path, PATH_MAX, "/dev/vfio/devices/%s", entry->d_name); + snprintf_assert(cdev_path, PATH_MAX, "/dev/vfio/devices/%s", entry->d_name); break; } @@ -299,14 +409,32 @@ const char *vfio_pci_get_cdev_path(const char *bdf) return cdev_path; } -static void vfio_device_bind_iommufd(int device_fd, int iommufd) +int __vfio_device_bind_iommufd(int device_fd, int iommufd, const char *vf_token) { struct vfio_device_bind_iommufd args = { .argsz = sizeof(args), .iommufd = iommufd, }; + uuid_t token_uuid; + + if (vf_token) { + VFIO_ASSERT_EQ(uuid_parse(vf_token, token_uuid), 0); + args.flags |= VFIO_DEVICE_BIND_FLAG_TOKEN; + args.token_uuid_ptr = (u64)token_uuid; + } + + if (ioctl(device_fd, VFIO_DEVICE_BIND_IOMMUFD, &args)) + return -errno; - ioctl_assert(device_fd, VFIO_DEVICE_BIND_IOMMUFD, &args); + return 0; +} + +static void vfio_device_bind_iommufd(int device_fd, int iommufd, + const char *vf_token) +{ + int ret = __vfio_device_bind_iommufd(device_fd, iommufd, vf_token); + + VFIO_ASSERT_EQ(ret, 0, "Failed VFIO_DEVICE_BIND_IOMMUFD ioctl\n"); } static void vfio_device_attach_iommufd_pt(int device_fd, u32 pt_id) @@ -319,33 +447,51 @@ static void vfio_device_attach_iommufd_pt(int device_fd, u32 pt_id) ioctl_assert(device_fd, VFIO_DEVICE_ATTACH_IOMMUFD_PT, &args); } -static void vfio_pci_iommufd_setup(struct vfio_pci_device *device, const char *bdf) +void vfio_pci_cdev_open(struct vfio_pci_device *device, const char *bdf) { const char *cdev_path = vfio_pci_get_cdev_path(bdf); device->fd = open(cdev_path, O_RDWR); VFIO_ASSERT_GE(device->fd, 0); free((void *)cdev_path); +} - vfio_device_bind_iommufd(device->fd, device->iommu->iommufd); +static void vfio_pci_iommufd_setup(struct vfio_pci_device *device, + const char *bdf, const char *vf_token) +{ + vfio_pci_cdev_open(device, bdf); + vfio_device_bind_iommufd(device->fd, device->iommu->iommufd, vf_token); vfio_device_attach_iommufd_pt(device->fd, device->iommu->ioas_id); } -struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu) +struct vfio_pci_device *vfio_pci_device_alloc(const char *bdf, struct iommu *iommu) { struct vfio_pci_device *device; - device = calloc(1, sizeof(*device)); - VFIO_ASSERT_NOT_NULL(device); + device = calloc_assert(1, sizeof(*device)); VFIO_ASSERT_NOT_NULL(iommu); device->iommu = iommu; device->bdf = bdf; + return device; +} + +void vfio_pci_device_free(struct vfio_pci_device *device) +{ + free(device); +} + +struct vfio_pci_device *vfio_pci_device_init(const char *bdf, struct iommu *iommu) +{ + struct vfio_pci_device *device; + + device = vfio_pci_device_alloc(bdf, iommu); + if (iommu->mode->container_path) - vfio_pci_container_setup(device, bdf); + vfio_pci_container_setup(device, bdf, NULL); else - vfio_pci_iommufd_setup(device, bdf); + vfio_pci_iommufd_setup(device, bdf, NULL); vfio_pci_device_setup(device); vfio_pci_driver_probe(device); @@ -374,5 +520,5 @@ void vfio_pci_device_cleanup(struct vfio_pci_device *device) if (device->group_fd) VFIO_ASSERT_EQ(close(device->group_fd), 0); - free(device); + vfio_pci_device_free(device); } diff --git a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c index 6827f4a6febe..5e65434d2318 100644 --- a/tools/testing/selftests/vfio/lib/vfio_pci_driver.c +++ b/tools/testing/selftests/vfio/lib/vfio_pci_driver.c @@ -6,12 +6,16 @@ extern struct vfio_pci_driver_ops dsa_ops; extern struct vfio_pci_driver_ops ioat_ops; #endif +extern struct vfio_pci_driver_ops nv_falcon_ops; +extern struct vfio_pci_driver_ops igb_ops; static struct vfio_pci_driver_ops *driver_ops[] = { #ifdef __x86_64__ &dsa_ops, &ioat_ops, #endif + &nv_falcon_ops, + &igb_ops, }; void vfio_pci_driver_probe(struct vfio_pci_device *device) @@ -106,7 +110,21 @@ int vfio_pci_driver_memcpy_wait(struct vfio_pci_device *device) int vfio_pci_driver_memcpy(struct vfio_pci_device *device, iova_t src, iova_t dst, u64 size) { - vfio_pci_driver_memcpy_start(device, src, dst, size, 1); + struct vfio_pci_driver *driver = &device->driver; + u64 offset = 0; + + while (offset < size) { + u64 chunk = min(size - offset, driver->max_memcpy_size); + int ret; + + vfio_pci_driver_memcpy_start(device, src + offset, + dst + offset, chunk, 1); + ret = vfio_pci_driver_memcpy_wait(device); + if (ret) + return ret; + + offset += chunk; + } - return vfio_pci_driver_memcpy_wait(device); + return 0; } diff --git a/tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c b/tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c new file mode 100644 index 000000000000..d7f25ef77671 --- /dev/null +++ b/tools/testing/selftests/vfio/vfio_dma_mapping_mmio_test.c @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include <stdio.h> +#include <sys/mman.h> +#include <unistd.h> + +#include <uapi/linux/types.h> +#include <linux/pci_regs.h> +#include <linux/sizes.h> +#include <linux/vfio.h> + +#include <libvfio.h> + +#include "../kselftest_harness.h" + +static const char *device_bdf; + +static struct vfio_pci_bar *largest_mapped_bar(struct vfio_pci_device *device) +{ + u32 flags = VFIO_REGION_INFO_FLAG_READ | VFIO_REGION_INFO_FLAG_WRITE; + struct vfio_pci_bar *largest = NULL; + u64 bar_size = 0; + + for (int i = 0; i < PCI_STD_NUM_BARS; i++) { + struct vfio_pci_bar *bar = &device->bars[i]; + + if (!bar->vaddr) + continue; + + /* + * iommu_map() maps with READ|WRITE, so require the same + * abilities for the underlying VFIO region. + */ + if ((bar->info.flags & flags) != flags) + continue; + + if (bar->info.size > bar_size) { + bar_size = bar->info.size; + largest = bar; + } + } + + return largest; +} + +FIXTURE(vfio_dma_mapping_mmio_test) { + struct iommu *iommu; + struct vfio_pci_device *device; + struct iova_allocator *iova_allocator; + struct vfio_pci_bar *bar; +}; + +FIXTURE_VARIANT(vfio_dma_mapping_mmio_test) { + const char *iommu_mode; +}; + +#define FIXTURE_VARIANT_ADD_IOMMU_MODE(_iommu_mode) \ +FIXTURE_VARIANT_ADD(vfio_dma_mapping_mmio_test, _iommu_mode) { \ + .iommu_mode = #_iommu_mode, \ +} + +FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(); + +#undef FIXTURE_VARIANT_ADD_IOMMU_MODE + +FIXTURE_SETUP(vfio_dma_mapping_mmio_test) +{ + self->iommu = iommu_init(variant->iommu_mode); + self->device = vfio_pci_device_init(device_bdf, self->iommu); + self->iova_allocator = iova_allocator_init(self->iommu); + self->bar = largest_mapped_bar(self->device); + + if (!self->bar) + SKIP(return, "No mappable BAR found on device %s", device_bdf); +} + +FIXTURE_TEARDOWN(vfio_dma_mapping_mmio_test) +{ + iova_allocator_cleanup(self->iova_allocator); + vfio_pci_device_cleanup(self->device); + iommu_cleanup(self->iommu); +} + +static void do_mmio_map_test(struct iommu *iommu, + struct iova_allocator *iova_allocator, + void *vaddr, size_t size) +{ + struct dma_region region = { + .vaddr = vaddr, + .size = size, + .iova = iova_allocator_alloc(iova_allocator, size), + }; + + /* + * NOTE: Check for iommufd compat success once it lands. Native iommufd + * will never support this. + */ + if (!strcmp(iommu->mode->name, MODE_VFIO_TYPE1V2_IOMMU) || + !strcmp(iommu->mode->name, MODE_VFIO_TYPE1_IOMMU)) { + iommu_map(iommu, ®ion); + iommu_unmap(iommu, ®ion); + } else { + VFIO_ASSERT_NE(__iommu_map(iommu, ®ion), 0); + } +} + +TEST_F(vfio_dma_mapping_mmio_test, map_full_bar) +{ + do_mmio_map_test(self->iommu, self->iova_allocator, + self->bar->vaddr, self->bar->info.size); +} + +TEST_F(vfio_dma_mapping_mmio_test, map_partial_bar) +{ + if (self->bar->info.size < 2 * getpagesize()) + SKIP(return, "BAR too small (size=0x%llx)", self->bar->info.size); + + do_mmio_map_test(self->iommu, self->iova_allocator, + self->bar->vaddr, getpagesize()); +} + +/* Test IOMMU mapping of BAR mmap with intentionally poor vaddr alignment. */ +TEST_F(vfio_dma_mapping_mmio_test, map_bar_misaligned) +{ + /* Limit size to bound test time for large BARs */ + size_t size = min_t(size_t, self->bar->info.size, SZ_1G); + void *vaddr; + + vaddr = mmap_reserve(size, SZ_1G, getpagesize()); + vaddr = mmap(vaddr, size, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_FIXED, + self->device->fd, self->bar->info.offset); + VFIO_ASSERT_NE(vaddr, MAP_FAILED); + + do_mmio_map_test(self->iommu, self->iova_allocator, vaddr, size); + + VFIO_ASSERT_EQ(munmap(vaddr, size), 0); +} + +int main(int argc, char *argv[]) +{ + device_bdf = vfio_selftests_get_bdf(&argc, argv); + return test_harness_run(argc, argv); +} diff --git a/tools/testing/selftests/vfio/vfio_dma_mapping_test.c b/tools/testing/selftests/vfio/vfio_dma_mapping_test.c index 16eba2ecca47..7d0de8c79de1 100644 --- a/tools/testing/selftests/vfio/vfio_dma_mapping_test.c +++ b/tools/testing/selftests/vfio/vfio_dma_mapping_test.c @@ -3,7 +3,6 @@ #include <sys/mman.h> #include <unistd.h> -#include <uapi/linux/types.h> #include <linux/iommufd.h> #include <linux/limits.h> #include <linux/mman.h> @@ -45,9 +44,9 @@ static int intel_iommu_mapping_get(const char *bdf, u64 iova, FILE *file; char *rest; - snprintf(iommu_mapping_path, sizeof(iommu_mapping_path), - "/sys/kernel/debug/iommu/intel/%s/domain_translation_struct", - bdf); + snprintf_assert(iommu_mapping_path, sizeof(iommu_mapping_path), + "/sys/kernel/debug/iommu/intel/%s/domain_translation_struct", + bdf); printf("Searching for IOVA 0x%lx in %s\n", iova, iommu_mapping_path); @@ -162,12 +161,8 @@ TEST_F(vfio_dma_mapping_test, dma_map_unmap) if (rc == -EOPNOTSUPP) goto unmap; - /* - * IOMMUFD compatibility-mode does not support huge mappings when - * using VFIO_TYPE1_IOMMU. - */ - if (!strcmp(variant->iommu_mode, "iommufd_compat_type1")) - mapping_size = SZ_4K; + if (self->iommu->mode->iommu_type == VFIO_TYPE1_IOMMU) + goto unmap; ASSERT_EQ(0, rc); printf("Found IOMMU mappings for IOVA 0x%lx:\n", region.iova); diff --git a/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c b/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c index 17017ed3beac..ec1e5633e080 100644 --- a/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c +++ b/tools/testing/selftests/vfio/vfio_iommufd_setup_test.c @@ -1,5 +1,4 @@ // SPDX-License-Identifier: GPL-2.0 -#include <uapi/linux/types.h> #include <linux/limits.h> #include <linux/sizes.h> #include <linux/vfio.h> diff --git a/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c b/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c index 33b0c31fe2ed..e1a54e153cd3 100644 --- a/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c +++ b/tools/testing/selftests/vfio/vfio_pci_device_init_perf_test.c @@ -45,8 +45,8 @@ FIXTURE_SETUP(vfio_pci_device_init_perf_test) int i; self->iommu = iommu_init(variant->iommu_mode); - self->threads = calloc(nr_devices, sizeof(self->threads[0])); - self->thread_args = calloc(nr_devices, sizeof(self->thread_args[0])); + self->threads = calloc_assert(nr_devices, sizeof(self->threads[0])); + self->thread_args = calloc_assert(nr_devices, sizeof(self->thread_args[0])); pthread_barrier_init(&self->barrier, NULL, nr_devices); diff --git a/tools/testing/selftests/vfio/vfio_pci_device_test.c b/tools/testing/selftests/vfio/vfio_pci_device_test.c index 7c0fe8ce3a61..93c11fd5e081 100644 --- a/tools/testing/selftests/vfio/vfio_pci_device_test.c +++ b/tools/testing/selftests/vfio/vfio_pci_device_test.c @@ -39,16 +39,17 @@ FIXTURE_TEARDOWN(vfio_pci_device_test) iommu_cleanup(self->iommu); } -#define read_pci_id_from_sysfs(_file) ({ \ - char __sysfs_path[PATH_MAX]; \ - char __buf[32]; \ - int __fd; \ - \ - snprintf(__sysfs_path, PATH_MAX, "/sys/bus/pci/devices/%s/%s", device_bdf, _file); \ - ASSERT_GT((__fd = open(__sysfs_path, O_RDONLY)), 0); \ - ASSERT_GT(read(__fd, __buf, ARRAY_SIZE(__buf)), 0); \ - ASSERT_EQ(0, close(__fd)); \ - (u16)strtoul(__buf, NULL, 0); \ +#define read_pci_id_from_sysfs(_file) ({ \ + char __sysfs_path[PATH_MAX]; \ + char __buf[32]; \ + int __fd; \ + \ + snprintf_assert(__sysfs_path, PATH_MAX, "/sys/bus/pci/devices/%s/%s", \ + device_bdf, _file); \ + ASSERT_GT((__fd = open(__sysfs_path, O_RDONLY)), 0); \ + ASSERT_GT(read(__fd, __buf, ARRAY_SIZE(__buf)), 0); \ + ASSERT_EQ(0, close(__fd)); \ + (u16)strtoul(__buf, NULL, 0); \ }) TEST_F(vfio_pci_device_test, config_space_read_write) diff --git a/tools/testing/selftests/vfio/vfio_pci_driver_test.c b/tools/testing/selftests/vfio/vfio_pci_driver_test.c index afa0480ddd9b..761bf117d624 100644 --- a/tools/testing/selftests/vfio/vfio_pci_driver_test.c +++ b/tools/testing/selftests/vfio/vfio_pci_driver_test.c @@ -11,11 +11,18 @@ static const char *device_bdf; -#define ASSERT_NO_MSI(_eventfd) do { \ - u64 __value; \ - \ - ASSERT_EQ(-1, read(_eventfd, &__value, 8)); \ - ASSERT_EQ(EAGAIN, errno); \ +#define fcntl_set_msi_nonblock(_self) do { \ + if (_self->device->driver.ops->send_msi) \ + fcntl_set_nonblock(_self->msi_fd); \ +} while (0) + +#define ASSERT_NO_MSI(_self) do { \ + u64 __value; \ + \ + if (!_self->device->driver.ops->send_msi) \ + break; \ + ASSERT_EQ(-1, read(_self->msi_fd, &__value, 8)); \ + ASSERT_EQ(EAGAIN, errno); \ } while (0) static void region_setup(struct iommu *iommu, @@ -89,12 +96,12 @@ FIXTURE_SETUP(vfio_pci_driver_test) self->msi_fd = self->device->msi_eventfds[driver->msi]; /* - * Use the maximum size supported by the device for memcpy operations, - * slimmed down to fit into the memcpy region (divided by 2 so src and - * dst regions do not overlap). + * Use 4x the driver's max_memcpy_size to exercise the chunking + * logic in vfio_pci_driver_memcpy(). Cap to half the memcpy + * region so src and dst do not overlap. */ - self->size = self->device->driver.max_memcpy_size; - self->size = min(self->size, self->memcpy_region.size / 2); + self->size = min_t(u64, driver->max_memcpy_size * 4, + self->memcpy_region.size / 2); self->src = self->memcpy_region.vaddr; self->dst = self->src + self->size; @@ -129,7 +136,7 @@ TEST_F(vfio_pci_driver_test, init_remove) TEST_F(vfio_pci_driver_test, memcpy_success) { - fcntl_set_nonblock(self->msi_fd); + fcntl_set_msi_nonblock(self); memset(self->src, 'x', self->size); memset(self->dst, 'y', self->size); @@ -140,12 +147,12 @@ TEST_F(vfio_pci_driver_test, memcpy_success) self->size)); ASSERT_EQ(0, memcmp(self->src, self->dst, self->size)); - ASSERT_NO_MSI(self->msi_fd); + ASSERT_NO_MSI(self); } TEST_F(vfio_pci_driver_test, memcpy_from_unmapped_iova) { - fcntl_set_nonblock(self->msi_fd); + fcntl_set_msi_nonblock(self); /* * Ignore the return value since not all devices will detect and report @@ -154,12 +161,12 @@ TEST_F(vfio_pci_driver_test, memcpy_from_unmapped_iova) vfio_pci_driver_memcpy(self->device, self->unmapped_iova, self->dst_iova, self->size); - ASSERT_NO_MSI(self->msi_fd); + ASSERT_NO_MSI(self); } TEST_F(vfio_pci_driver_test, memcpy_to_unmapped_iova) { - fcntl_set_nonblock(self->msi_fd); + fcntl_set_msi_nonblock(self); /* * Ignore the return value since not all devices will detect and report @@ -168,13 +175,16 @@ TEST_F(vfio_pci_driver_test, memcpy_to_unmapped_iova) vfio_pci_driver_memcpy(self->device, self->src_iova, self->unmapped_iova, self->size); - ASSERT_NO_MSI(self->msi_fd); + ASSERT_NO_MSI(self); } TEST_F(vfio_pci_driver_test, send_msi) { u64 value; + if (!self->device->driver.ops->send_msi) + SKIP(return, "Driver does not support send_msi()\n"); + vfio_pci_driver_send_msi(self->device); ASSERT_EQ(8, read(self->msi_fd, &value, 8)); ASSERT_EQ(1, value); @@ -201,6 +211,9 @@ TEST_F(vfio_pci_driver_test, mix_and_match) self->dst_iova, self->size); + if (!self->device->driver.ops->send_msi) + continue; + vfio_pci_driver_send_msi(self->device); ASSERT_EQ(8, read(self->msi_fd, &value, 8)); ASSERT_EQ(1, value); @@ -211,9 +224,10 @@ TEST_F_TIMEOUT(vfio_pci_driver_test, memcpy_storm, 60) { struct vfio_pci_driver *driver = &self->device->driver; u64 total_size; + u64 size; u64 count; - fcntl_set_nonblock(self->msi_fd); + fcntl_set_msi_nonblock(self); /* * Perform up to 250GiB worth of DMA reads and writes across several @@ -221,16 +235,17 @@ TEST_F_TIMEOUT(vfio_pci_driver_test, memcpy_storm, 60) * will take too long. */ total_size = 250UL * SZ_1G; - count = min(total_size / self->size, driver->max_memcpy_count); + size = min(driver->max_memcpy_size, self->memcpy_region.size / 2); + count = min(total_size / size, driver->max_memcpy_count); - printf("Kicking off %lu memcpys of size 0x%lx\n", count, self->size); + printf("Kicking off %lu memcpys of size 0x%lx\n", count, size); vfio_pci_driver_memcpy_start(self->device, self->src_iova, self->dst_iova, - self->size, count); + size, count); ASSERT_EQ(0, vfio_pci_driver_memcpy_wait(self->device)); - ASSERT_NO_MSI(self->msi_fd); + ASSERT_NO_MSI(self); } static bool device_has_selftests_driver(const char *bdf) diff --git a/tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c b/tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c new file mode 100644 index 000000000000..19d657d00b75 --- /dev/null +++ b/tools/testing/selftests/vfio/vfio_pci_sriov_uapi_test.c @@ -0,0 +1,217 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include "lib/include/libvfio/assert.h" +#include <fcntl.h> +#include <unistd.h> +#include <stdlib.h> +#include <sys/ioctl.h> +#include <linux/limits.h> + +#include <libvfio.h> + +#include "../kselftest_harness.h" + +#define UUID_1 "52ac9bff-3a88-4fbd-901a-0d767c3b6c97" +#define UUID_2 "88594674-90a0-47a9-aea8-9d9b352ac08a" + +static const char *pf_bdf; +static char *vf_bdf; + +static pid_t main_pid; + +static int container_setup(struct vfio_pci_device *device, const char *bdf, + const char *vf_token) +{ + vfio_pci_group_setup(device, bdf); + vfio_container_set_iommu(device); + __vfio_pci_group_get_device_fd(device, bdf, vf_token); + + /* The device fd will be -1 in case of mismatched tokens */ + return (device->fd < 0); +} + +static int iommufd_setup(struct vfio_pci_device *device, const char *bdf, + const char *vf_token) +{ + vfio_pci_cdev_open(device, bdf); + return __vfio_device_bind_iommufd(device->fd, + device->iommu->iommufd, vf_token); +} + +static int device_init(const char *bdf, struct iommu *iommu, + const char *vf_token, struct vfio_pci_device **out_dev) +{ + struct vfio_pci_device *device = vfio_pci_device_alloc(bdf, iommu); + int ret; + + if (iommu->mode->container_path) + ret = container_setup(device, bdf, vf_token); + else + ret = iommufd_setup(device, bdf, vf_token); + + *out_dev = device; + return ret; +} + +static void device_cleanup(struct vfio_pci_device *device) +{ + if (!device) + return; + + if (device->fd > 0) + VFIO_ASSERT_EQ(close(device->fd), 0); + + if (device->group_fd) + VFIO_ASSERT_EQ(close(device->group_fd), 0); + + vfio_pci_device_free(device); +} + +FIXTURE(vfio_pci_sriov_uapi_test) { + struct vfio_pci_device *pf; + struct vfio_pci_device *vf; + struct iommu *iommu; + char *pf_token; +}; + +FIXTURE_VARIANT(vfio_pci_sriov_uapi_test) { + const char *iommu_mode; + char *vf_token; +}; + +#define FIXTURE_VARIANT_ADD_IOMMU_MODE(_iommu_mode, _name, _vf_token) \ +FIXTURE_VARIANT_ADD(vfio_pci_sriov_uapi_test, _iommu_mode ## _ ## _name) { \ + .iommu_mode = #_iommu_mode, \ + .vf_token = (_vf_token), \ +} + +FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(same_uuid, UUID_1); +FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(diff_uuid, UUID_2); +FIXTURE_VARIANT_ADD_ALL_IOMMU_MODES(null_uuid, NULL); + +FIXTURE_SETUP(vfio_pci_sriov_uapi_test) +{ + self->iommu = iommu_init(variant->iommu_mode); + + self->pf_token = UUID_1; + ASSERT_EQ(device_init(pf_bdf, self->iommu, self->pf_token, &self->pf), 0); +} + +FIXTURE_TEARDOWN(vfio_pci_sriov_uapi_test) +{ + device_cleanup(self->vf); + device_cleanup(self->pf); + iommu_cleanup(self->iommu); +} + +/* + * This asserts if the VF device is successfully created if its token matches + * with the token used to create/override the PF or fails during a mismatch. + */ +#define ASSERT_COND_VF_CREATION(_ret) do { \ + if (!variant->vf_token || strcmp(self->pf_token, variant->vf_token)) { \ + ASSERT_NE((_ret), 0); \ + } else { \ + ASSERT_EQ((_ret), 0); \ + } \ +} while (0) + +/* + * Validate if the UAPI handles correctly and incorrectly set token on the VF. + */ +TEST_F(vfio_pci_sriov_uapi_test, init_token_match) +{ + int ret; + + ret = device_init(vf_bdf, self->iommu, variant->vf_token, &self->vf); + ASSERT_COND_VF_CREATION(ret); +} + +/* + * After closing the PF, validate if the VF access still needs the right token. + */ +TEST_F(vfio_pci_sriov_uapi_test, pf_early_close) +{ + int ret; + + device_cleanup(self->pf); + + /* Clean the 'pf' to avoid calling device_cleanup() again. */ + self->pf = NULL; + + ret = device_init(vf_bdf, self->iommu, variant->vf_token, &self->vf); + ASSERT_COND_VF_CREATION(ret); +} + +/* + * After PF device init, override the existing token and validate if the newly + * set token is the one that's active. + */ +TEST_F(vfio_pci_sriov_uapi_test, override_token) +{ + int ret; + + self->pf_token = UUID_2; + vfio_device_set_vf_token(self->pf->fd, self->pf_token); + + ret = device_init(vf_bdf, self->iommu, variant->vf_token, &self->vf); + ASSERT_COND_VF_CREATION(ret); +} + +static void vf_teardown(void) +{ + /* + * The child processes, created by TEST_F()s, inherits this atexit() + * handler. Hence, check and destroy the VF only when the main/parent + * process exits. + */ + if (getpid() != main_pid) + return; + + free(vf_bdf); + sysfs_sriov_numvfs_set(pf_bdf, 0); +} + +static void vf_setup(void) +{ + char *vf_driver; + int nr_vfs; + + nr_vfs = sysfs_sriov_totalvfs_get(pf_bdf); + if (nr_vfs <= 0) + ksft_exit_skip("SR-IOV may not be supported by the PF: %s\n", pf_bdf); + + nr_vfs = sysfs_sriov_numvfs_get(pf_bdf); + if (nr_vfs != 0) + ksft_exit_skip("SR-IOV already configured for the PF: %s\n", pf_bdf); + + /* Create only one VF for testing */ + sysfs_sriov_numvfs_set(pf_bdf, 1); + + /* + * Setup an exit handler to destroy the VF in case of failures + * during further setup at the end of the test run. + */ + main_pid = getpid(); + VFIO_ASSERT_EQ(atexit(vf_teardown), 0); + + vf_bdf = sysfs_sriov_vf_bdf_get(pf_bdf, 0); + + /* + * The VF inherits the driver from the PF. + * Ensure this is 'vfio-pci' before proceeding. + */ + vf_driver = sysfs_driver_get(vf_bdf); + VFIO_ASSERT_NE(vf_driver, NULL); + VFIO_ASSERT_EQ(strcmp(vf_driver, "vfio-pci"), 0); + free(vf_driver); + + printf("Created 1 VF (%s) under the PF: %s\n", vf_bdf, pf_bdf); +} + +int main(int argc, char *argv[]) +{ + pf_bdf = vfio_selftests_get_bdf(&argc, argv); + vf_setup(); + + return test_harness_run(argc, argv); +} |
