summaryrefslogtreecommitdiff
path: root/drivers/net/ethernet/broadcom
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-08-20 08:16:04 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-08-20 08:16:04 -0700
commit91ec2035134982b98fab0609a9fd8480e8217dc1 (patch)
treefed1207d0ad121a52ea1da4b4b8a2c52e3f2c15b /drivers/net/ethernet/broadcom
parent5a8cd539ac19f7a68e68e1d25ef9ca2ff55b8500 (diff)
parent61eb236c41c2a4717015dff18016a75a5eb90052 (diff)
Merge tag 'net-next-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next
Pull networking updates from Jakub Kicinski: "One of the 'small improvements all over the place' releases for us. It's hard to draw any direct comparisons because summer vacations disrupted our patch processing (and presumably - generation) quite a bit. Quick and dirty count suggests we (Paolo and I) merged a very similar number of net (632) and net-next (648) patches. This is not telling the full story either because 1/3 to 1/2 of the net-next patches also *seem* like AI-driven low priority fixes, cleanups and clarifications. We are completely overwhelmed, of course. The glimmer of hope is that we secured sufficient LLM budget and access (thank you Meta!) to run reviews with multiple frontier models on each patch. This eliminates some hallucinations. That said, in terms of review, the LLMs can only do so much. The sad truth is that our APIs (especially for rare events like PCIe errors, timeouts etc) have always been racy, and now LLMs don't let us ignore that. I expect our direction for the next release will be to tweak the reviews a little bit more, but start shifting focus to letting the LLMs take care of the busy work - managing patchwork, automating common process complaints, editing commit messages, and maybe applying patches which already got "reviewed-by" tags from people we trust... Core & protocols: - A few steps lowering rtnl_lock dependence: - per-netns netdev unregistration for select SW drivers (e.g. veth, ipvlan, tunnels) - rtnl_lock-less FIB rule changes (RTM_NEWRULE and RTM_DELRULE) - prepare software drivers and TC qdiscs for rtnl_lock-less GET - Support BIG TCP (>64kB TSO) in UDP tunnels (vxlan, geneve) - Support buffers larger than PAGE_SIZE in devmem zero-copy API - Improve MPTCP handling of extreme memory pressure handling, when out-of-order queue had to be pruned - Report the per-group user count via RTM_GETMULTICAST - Expose the route deletion reason in RTM_DELROUTE - Add a SO_RIGHTS_NOTRUNC option to UNIX sockets to enable more useful handling of LSM denials when receiving SCM_RIGHTS messages: instead of truncating the message at the first blocked fd, keep every fd slot and store the LSM errno in the blocked slot - IPv6 Segment Routing - support looking up the post-encap SID (address) in a different/specified routing table - Support PRP RedBox (interlink) creation - Support per-nexthop UDP dst port in VXLAN - Continue converting getsockopt callbacks in a number of protocols to iov_iter Ethernet: - Merge initial CXL support for AMD/Solarflare NICs (shared branch with the CXL tree) - New drivers: - ADIN1140 10BASE-T1S MACPHY - Initial skeleton of Intel iXD and ZTE Dinghai drivers - High-speed NICs: - AMD/Pensando: - support firmware flashing - Cisco (enic): - SR-IOV V2 admin channel and MBOX protocol - Huawei (hns3): - support for ethtool pfc_prevention_tout - nVidia/Mellanox: - support sharing bandwidth control across interfaces of the same device - Marvell (octeontx2-pf): - link RQ page pools to netdev for Netlink stats - Google vNIC: - XDP metadata support for DQ RDA - Microsoft vNIC: - support forcing full-page RX buffers - Other NICs: - Synopsys IP: - eic7700: support for eth1 - Microchip (lan743x): - support for RMII interface - Wangxun: - support for ethtool -G and -C for VFs - add Tx timeout and PCIe error handling - Intel (igb/igc): - RSS key get/set support - support for forcing link speed without auto-negotiation - Switches: - NXP (dpaa2): - support bonding/LAG offload - Mediatek: - mt7530: EN7528 support - initial support for MT7628 - Micrel (ksz8/9): - refactoring work to move towards library model - PTP support for KSZ8463 - nVidia/Mellanox: - support rtnl-lock-less ethtool callbacks - Realtek: - rtl8366rb: use generic RTL83xx code - support SGMII and HSGMII for RTL8367S - PHYs: - Airoha: - EcoNet EN7528 PHY support - DAPU Telecom - DAPU Telecom DAP8211R(I) Gigabit PHY support - Realtek: - support RTL8261C_CG - support RTL8261D Wireless: - nl80211: per-link statistics support for multi-link operation - mac80211: AQL/airtime-fairness support for multicast - Merge Peripheral Authentication Service (PAS) / TEE support for ath12k (shared branch with the firmware/qcom tree) - New drivers: - mm81x for Morse Micro Long-Range S1G devices - nxpwifi for NXP devices (mostly forked off from mwifiex) - Driver changes: - Broadcom (brcmfmac): - DPP support, some Cypress part update - MediaTek (mt76): - mt7928 support - mt7925 NAN support - mt7996 AP powersave improvements - Qualcomm (ath12k): - much kernel infrastructure integration work - AHB platform MultiPD support - Realtek (rt89): - LED support - RTL8922DE support - dual-BT coex for RTL8922D - Intel: - new FW version support Bluetooth: - HCI: add support for Shorter Connection Interval (SCI) feature - af_bluetooth: add minimal context analysis annotations - Driver changes: - Intel: - add Bluetooth SAR revision 2 support - add vendor_reset PCI sysfs for PLDR - Mediatek: - add USB IDs for MT7902 and MT7922 devices - Realtek: - add USB IDs for 8761CU and 8852BE devices - NXP: - add M.2 Bluetooth device support using pwrseq Misc: - DPLL support for manual/numerical oscillator control (NCO) (implement in zl3073x) - MCTP support for MCTP over USB v1.1 (DMTF DSP0283) - Power-over-Ethernet: support Realtek PSE controllers - Remove the IBM EHEA driver - Remove tulip/xircom_cb driver" * tag 'net-next-7.3' of git://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next: (1433 commits) net/mlx5e: do not HW-GRO coalesce small frames net: openvswitch: fix nf_connlabels leak in ovs_ct_init net: add missing ref_tracker_dir_exit() to alloc_netdev_mqs() net: openvswitch: fix flow mask use-after-free on flow deletion sctp: stop processing a packet once its association is deleted dpll: zl3073x: add PTP clock support dpll: zl3073x: add channel ToD, phase step and TIE operations dpll: zl3073x: scale poll interval proportionally to timeout ptp: vmclock: prevent read-only mappings from becoming writable ipv4: reject undersized MTUs in ip_do_fragment() bonding: initialize err for empty target lists net: dsa: initial support for MT7628 embedded switch net: dsa: initial MT7628 tagging driver net: phy: mediatek: add phy driver for MT7628 built-in Fast Ethernet PHYs dt-bindings: net: dsa: add MT7628 ESW net: pse-pd: realtek-pse-mcu: add UART transport net: pse-pd: realtek-pse-mcu: add I2C transport net: pse-pd: add Realtek PSE MCU core dt-bindings: net: pse-pd: add bindings for Realtek PSE MCU vsock: use sock_error() to consume sk_err after a failed connect ...
Diffstat (limited to 'drivers/net/ethernet/broadcom')
-rw-r--r--drivers/net/ethernet/broadcom/bnge/bnge_netdev.c132
-rw-r--r--drivers/net/ethernet/broadcom/bnge/bnge_netdev.h6
-rw-r--r--drivers/net/ethernet/broadcom/bnx2x/bnx2x_cmn.c8
-rw-r--r--drivers/net/ethernet/broadcom/bnx2x/bnx2x_sp.c6
-rw-r--r--drivers/net/ethernet/broadcom/bnxt/bnxt.c138
-rw-r--r--drivers/net/ethernet/broadcom/bnxt/bnxt.h11
6 files changed, 222 insertions, 79 deletions
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index ac4c93e5b634..a4288f0258f8 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -20,6 +20,7 @@
#include <net/page_pool/helpers.h>
#include "bnge.h"
+#include "bnge_hwrm.h"
#include "bnge_hwrm_lib.h"
#include "bnge_ethtool.h"
#include "bnge_rmem.h"
@@ -2144,16 +2145,16 @@ err_del_l2_filter:
return rc;
}
-static bool bnge_mc_list_updated(struct bnge_net *bn, u32 *rx_mask)
+static bool bnge_mc_list_updated(struct bnge_net *bn, u32 *rx_mask,
+ const struct netdev_hw_addr_list *mc)
{
struct bnge_vnic_info *vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
- struct net_device *dev = bn->netdev;
struct netdev_hw_addr *ha;
int mc_count = 0, off = 0;
bool update = false;
u8 *haddr;
- netdev_for_each_mc_addr(ha, dev) {
+ netdev_hw_addr_list_for_each(ha, mc) {
if (mc_count >= BNGE_MAX_MC_ADDRS) {
*rx_mask |= CFA_L2_SET_RX_MASK_REQ_MASK_ALL_MCAST;
vnic->mc_list_count = 0;
@@ -2177,17 +2178,17 @@ static bool bnge_mc_list_updated(struct bnge_net *bn, u32 *rx_mask)
return update;
}
-static bool bnge_uc_list_updated(struct bnge_net *bn)
+static bool bnge_uc_list_updated(struct bnge_net *bn,
+ const struct netdev_hw_addr_list *uc)
{
struct bnge_vnic_info *vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
- struct net_device *dev = bn->netdev;
struct netdev_hw_addr *ha;
int off = 0;
- if (netdev_uc_count(dev) != (vnic->uc_filter_count - 1))
+ if (netdev_hw_addr_list_count(uc) != (vnic->uc_filter_count - 1))
return true;
- netdev_for_each_uc_addr(ha, dev) {
+ netdev_hw_addr_list_for_each(ha, uc) {
if (!ether_addr_equal(ha->addr, vnic->uc_list + off))
return true;
@@ -2201,18 +2202,14 @@ static bool bnge_promisc_ok(struct bnge_net *bn)
return true;
}
-static int bnge_cfg_def_vnic(struct bnge_net *bn)
+static int bnge_cfg_rx_mode(struct bnge_net *bn, struct netdev_hw_addr_list *uc,
+ bool uc_update, bool snapshot)
{
struct bnge_vnic_info *vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
struct net_device *dev = bn->netdev;
struct bnge_dev *bd = bn->bd;
struct netdev_hw_addr *ha;
int i, off = 0, rc;
- bool uc_update;
-
- netif_addr_lock_bh(dev);
- uc_update = bnge_uc_list_updated(bn);
- netif_addr_unlock_bh(dev);
if (!uc_update)
goto skip_uc;
@@ -2226,22 +2223,28 @@ static int bnge_cfg_def_vnic(struct bnge_net *bn)
vnic->uc_filter_count = 1;
- netif_addr_lock_bh(dev);
- if (netdev_uc_count(dev) > (BNGE_MAX_UC_ADDRS - 1)) {
+ if (!snapshot)
+ netif_addr_lock_bh(dev);
+ if (netdev_hw_addr_list_count(uc) > (BNGE_MAX_UC_ADDRS - 1)) {
vnic->rx_mask |= CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS;
} else {
- netdev_for_each_uc_addr(ha, dev) {
+ netdev_hw_addr_list_for_each(ha, uc) {
memcpy(vnic->uc_list + off, ha->addr, ETH_ALEN);
off += ETH_ALEN;
vnic->uc_filter_count++;
}
}
- netif_addr_unlock_bh(dev);
+ if (!snapshot)
+ netif_addr_unlock_bh(dev);
for (i = 1, off = 0; i < vnic->uc_filter_count; i++, off += ETH_ALEN) {
rc = bnge_hwrm_set_vnic_filter(bn, 0, i, vnic->uc_list + off);
if (rc) {
- netdev_err(dev, "HWRM vnic filter failure rc: %d\n", rc);
+ if (rc == -EAGAIN)
+ netdev_warn(dev, "FW busy while setting vnic filter, will retry\n");
+ else
+ netdev_err(dev, "HWRM vnic filter failure rc: %d\n",
+ rc);
vnic->uc_filter_count = i;
return rc;
}
@@ -2260,13 +2263,58 @@ skip_uc:
vnic->mc_list_count = 0;
rc = bnge_hwrm_cfa_l2_set_rx_mask(bd, vnic);
}
- if (rc)
- netdev_err(dev, "HWRM cfa l2 rx mask failure rc: %d\n",
- rc);
+ if (rc) {
+ if (rc == -EAGAIN) {
+ netdev_warn(dev, "FW busy while setting l2 rx mask in CFA, will retry\n");
+ vnic->rx_mask &= ~BNGE_RX_MASK_CFG_FLAGS;
+ } else {
+ netdev_err(dev, "HWRM CFA L2 rx mask failure rc: %d\n",
+ rc);
+ }
+ }
return rc;
}
+static int bnge_set_rx_mode(struct net_device *dev,
+ struct netdev_hw_addr_list *uc,
+ struct netdev_hw_addr_list *mc)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_vnic_info *vnic;
+ bool mc_update = false;
+ bool uc_update;
+ u32 mask;
+
+ if (!test_bit(BNGE_STATE_OPEN, &bn->bd->state))
+ return 0;
+
+ vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
+ mask = vnic->rx_mask;
+ mask &= ~BNGE_RX_MASK_CFG_FLAGS;
+
+ if (dev->flags & IFF_PROMISC)
+ mask |= CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS;
+
+ uc_update = bnge_uc_list_updated(bn, uc);
+
+ if (dev->flags & IFF_BROADCAST)
+ mask |= CFA_L2_SET_RX_MASK_REQ_MASK_BCAST;
+ if (dev->flags & IFF_ALLMULTI) {
+ mask |= CFA_L2_SET_RX_MASK_REQ_MASK_ALL_MCAST;
+ vnic->mc_list_count = 0;
+ } else if (dev->flags & IFF_MULTICAST) {
+ mc_update = bnge_mc_list_updated(bn, &mask, mc);
+ }
+
+ if (mask != vnic->rx_mask || uc_update || mc_update) {
+ vnic->rx_mask = mask;
+ return bnge_cfg_rx_mode(bn, uc, uc_update, true);
+ }
+
+ return 0;
+}
+
static void bnge_disable_int(struct bnge_net *bn)
{
struct bnge_dev *bd = bn->bd;
@@ -2695,13 +2743,17 @@ static int bnge_init_chip(struct bnge_net *bn)
} else if (bn->netdev->flags & IFF_MULTICAST) {
u32 mask = 0;
- bnge_mc_list_updated(bn, &mask);
+ bnge_mc_list_updated(bn, &mask, &bn->netdev->mc);
vnic->rx_mask |= mask;
}
- rc = bnge_cfg_def_vnic(bn);
- if (rc)
+ rc = bnge_cfg_rx_mode(bn, &bn->netdev->uc, true, false);
+ if (rc == -EAGAIN) {
+ netif_rx_mode_schedule_retry(bn->netdev);
+ rc = 0;
+ } else if (rc) {
goto err_out;
+ }
return 0;
err_out:
@@ -2811,6 +2863,24 @@ static void bnge_tx_enable(struct bnge_net *bn)
netif_carrier_on(bn->netdev);
}
+static int bnge_hwrm_if_change(struct bnge_dev *bd, bool up)
+{
+ struct hwrm_func_drv_if_change_input *req;
+ int rc;
+
+ if (!(bd->fw_cap & BNGE_FW_CAP_IF_CHANGE))
+ return 0;
+
+ rc = bnge_hwrm_req_init(bd, req, HWRM_FUNC_DRV_IF_CHANGE);
+ if (rc)
+ return rc;
+
+ if (up)
+ req->flags = cpu_to_le32(FUNC_DRV_IF_CHANGE_REQ_FLAGS_UP);
+
+ return bnge_hwrm_req_send(bd, req);
+}
+
static int bnge_open_core(struct bnge_net *bn)
{
struct bnge_dev *bd = bn->bd;
@@ -2818,16 +2888,22 @@ static int bnge_open_core(struct bnge_net *bn)
netif_carrier_off(bn->netdev);
+ rc = bnge_hwrm_if_change(bd, true);
+ if (rc) {
+ netdev_err(bn->netdev, "bnge_hwrm_if_change err: %d\n", rc);
+ return rc;
+ }
+
rc = bnge_reserve_rings(bd);
if (rc) {
netdev_err(bn->netdev, "bnge_reserve_rings err: %d\n", rc);
- return rc;
+ goto err_if_change;
}
rc = bnge_alloc_core(bn);
if (rc) {
netdev_err(bn->netdev, "bnge_alloc_core err: %d\n", rc);
- return rc;
+ goto err_if_change;
}
bnge_init_napi(bn);
@@ -2874,6 +2950,8 @@ err_free_irq:
err_del_napi:
bnge_del_napi(bn);
bnge_free_core(bn);
+err_if_change:
+ bnge_hwrm_if_change(bd, false);
return rc;
}
@@ -3104,6 +3182,7 @@ static int bnge_close(struct net_device *dev)
bnge_close_core(bn);
bnge_hwrm_shutdown_link(bn->bd);
+ bnge_hwrm_if_change(bn->bd, false);
bn->sp_event = 0;
return 0;
@@ -3193,6 +3272,7 @@ static const struct net_device_ops bnge_netdev_ops = {
.ndo_stop = bnge_close,
.ndo_start_xmit = bnge_start_xmit,
.ndo_get_stats64 = bnge_get_stats64,
+ .ndo_set_rx_mode_async = bnge_set_rx_mode,
.ndo_features_check = bnge_features_check,
};
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
index d177919c2e11..476b5bab96fe 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
@@ -561,6 +561,12 @@ struct bnge_napi {
#define BNGE_VNIC_DEFAULT 0
#define BNGE_MAX_UC_ADDRS 4
+#define BNGE_RX_MASK_CFG_FLAGS \
+ (CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS | \
+ CFA_L2_SET_RX_MASK_REQ_MASK_MCAST | \
+ CFA_L2_SET_RX_MASK_REQ_MASK_ALL_MCAST | \
+ CFA_L2_SET_RX_MASK_REQ_MASK_BCAST)
+
struct bnge_vnic_info {
u16 fw_vnic_id;
#define BNGE_MAX_CTX_PER_VNIC 8
diff --git a/drivers/net/ethernet/broadcom/bnx2x/bnx2x_cmn.c b/drivers/net/ethernet/broadcom/bnx2x/bnx2x_cmn.c
index 5b2640bd31c3..5a9742fd3ddf 100644
--- a/drivers/net/ethernet/broadcom/bnx2x/bnx2x_cmn.c
+++ b/drivers/net/ethernet/broadcom/bnx2x/bnx2x_cmn.c
@@ -4742,13 +4742,13 @@ int bnx2x_alloc_mem_bp(struct bnx2x *bp)
/* fp array: RSS plus CNIC related L2 queues */
fp_array_size = BNX2X_MAX_RSS_COUNT(bp) + CNIC_SUPPORT(bp);
- bp->fp_array_size = fp_array_size;
- BNX2X_DEV_INFO("fp_array_size %d\n", bp->fp_array_size);
-
- fp = kzalloc_objs(*fp, bp->fp_array_size);
+ BNX2X_DEV_INFO("fp_array_size %d\n", fp_array_size);
+ fp = kzalloc_objs(*fp, fp_array_size);
if (!fp)
goto alloc_err;
bp->fp = fp;
+ bp->fp_array_size = fp_array_size;
+
for (i = 0; i < bp->fp_array_size; i++) {
fp[i].tpa_info =
kzalloc_objs(struct bnx2x_agg_info,
diff --git a/drivers/net/ethernet/broadcom/bnx2x/bnx2x_sp.c b/drivers/net/ethernet/broadcom/bnx2x/bnx2x_sp.c
index 07a908a2c72f..d560524d317d 100644
--- a/drivers/net/ethernet/broadcom/bnx2x/bnx2x_sp.c
+++ b/drivers/net/ethernet/broadcom/bnx2x/bnx2x_sp.c
@@ -26,6 +26,7 @@
#include <linux/netdevice.h>
#include <linux/etherdevice.h>
#include <linux/crc32c.h>
+#include <linux/slab.h>
#include "bnx2x.h"
#include "bnx2x_cmn.h"
#include "bnx2x_sp.h"
@@ -2664,7 +2665,7 @@ static void bnx2x_free_groups(struct list_head *mcast_group_list)
struct bnx2x_mcast_elem_group,
mcast_group_link);
list_del(&current_mcast_group->mcast_group_link);
- free_page((unsigned long)current_mcast_group);
+ kfree(current_mcast_group);
}
}
@@ -2713,8 +2714,7 @@ static int bnx2x_mcast_enqueue_cmd(struct bnx2x *bp,
total_elems = BNX2X_MCAST_BINS_NUM;
}
while (total_elems > 0) {
- elem_group = (struct bnx2x_mcast_elem_group *)
- __get_free_page(GFP_ATOMIC | __GFP_ZERO);
+ elem_group = kzalloc(PAGE_SIZE, GFP_ATOMIC);
if (!elem_group) {
bnx2x_free_groups(&new_cmd->group_head);
kfree(new_cmd);
diff --git a/drivers/net/ethernet/broadcom/bnxt/bnxt.c b/drivers/net/ethernet/broadcom/bnxt/bnxt.c
index bc7b37cb74a7..9377bf675981 100644
--- a/drivers/net/ethernet/broadcom/bnxt/bnxt.c
+++ b/drivers/net/ethernet/broadcom/bnxt/bnxt.c
@@ -11592,7 +11592,7 @@ static int bnxt_get_num_msix(struct bnxt *bp)
static int bnxt_init_int_mode(struct bnxt *bp)
{
- int i, total_vecs, max, rc = 0, min = 1, ulp_msix, tx_cp, tbl_size;
+ int i, total_vecs, max, rc, min = 1, ulp_msix, tx_cp, tbl_size;
total_vecs = bnxt_get_num_msix(bp);
max = bnxt_get_max_func_irqs(bp);
@@ -11617,26 +11617,26 @@ static int bnxt_init_int_mode(struct bnxt *bp)
if (pci_msix_can_alloc_dyn(bp->pdev))
tbl_size = max;
bp->irq_tbl = kzalloc_objs(*bp->irq_tbl, tbl_size);
- if (bp->irq_tbl) {
- for (i = 0; i < total_vecs; i++)
- bp->irq_tbl[i].vector = pci_irq_vector(bp->pdev, i);
-
- bp->total_irqs = total_vecs;
- /* Trim rings based upon num of vectors allocated */
- rc = bnxt_trim_rings(bp, &bp->rx_nr_rings, &bp->tx_nr_rings,
- total_vecs - ulp_msix, min == 1);
- if (rc)
- goto msix_setup_exit;
-
- tx_cp = bnxt_num_tx_to_cp(bp, bp->tx_nr_rings);
- bp->cp_nr_rings = (min == 1) ?
- max_t(int, tx_cp, bp->rx_nr_rings) :
- tx_cp + bp->rx_nr_rings;
-
- } else {
+ if (!bp->irq_tbl) {
rc = -ENOMEM;
goto msix_setup_exit;
}
+
+ for (i = 0; i < total_vecs; i++)
+ bp->irq_tbl[i].vector = pci_irq_vector(bp->pdev, i);
+
+ bp->total_irqs = total_vecs;
+ /* Trim rings based upon num of vectors allocated */
+ rc = bnxt_trim_rings(bp, &bp->rx_nr_rings, &bp->tx_nr_rings,
+ total_vecs - ulp_msix, min == 1);
+ if (rc)
+ goto msix_setup_exit;
+
+ tx_cp = bnxt_num_tx_to_cp(bp, bp->tx_nr_rings);
+ bp->cp_nr_rings = (min == 1) ?
+ max_t(int, tx_cp, bp->rx_nr_rings) :
+ tx_cp + bp->rx_nr_rings;
+
return 0;
msix_setup_exit:
@@ -11792,6 +11792,9 @@ static void bnxt_irq_affinity_notify(struct irq_affinity_notify *notify,
irq = container_of(notify, struct bnxt_irq, affinity_notify);
+ cpumask_copy(irq->bp->ring_cpu_mask[irq->ring_nr], mask);
+ set_bit(irq->ring_nr, irq->bp->ring_affinity_set);
+
#ifdef CONFIG_RFS_ACCEL
if (irq->bp->dev->rx_cpu_rmap && irq->ring_nr < irq->bp->rx_nr_rings) {
int err;
@@ -11807,8 +11810,6 @@ static void bnxt_irq_affinity_notify(struct irq_affinity_notify *notify,
if (!irq->bp->tph_mode)
return;
- cpumask_copy(irq->cpu_mask, mask);
-
if (irq->ring_nr >= irq->bp->rx_nr_rings)
return;
@@ -11850,10 +11851,6 @@ static void bnxt_register_irq_notifier(struct bnxt *bp, struct bnxt_irq *irq)
irq->bp = bp;
- /* Nothing to do if TPH is not enabled */
- if (!bp->tph_mode)
- return;
-
/* Register IRQ affinity notifier */
notify = &irq->affinity_notify;
notify->irq = irq->vector;
@@ -11863,6 +11860,41 @@ static void bnxt_register_irq_notifier(struct bnxt *bp, struct bnxt_irq *irq)
irq_set_affinity_notifier(irq->vector, notify);
}
+static int bnxt_alloc_ring_cpu_masks(struct bnxt *bp)
+{
+ int i;
+
+ bp->ring_cpu_mask = kzalloc_objs(*bp->ring_cpu_mask, bp->max_irqs);
+ if (!bp->ring_cpu_mask)
+ return -ENOMEM;
+
+ bp->ring_affinity_set = bitmap_zalloc(bp->max_irqs, GFP_KERNEL);
+ if (!bp->ring_affinity_set)
+ return -ENOMEM;
+
+ for (i = 0; i < bp->max_irqs; i++)
+ if (!zalloc_cpumask_var(&bp->ring_cpu_mask[i], GFP_KERNEL))
+ return -ENOMEM;
+
+ return 0;
+}
+
+static void bnxt_free_ring_cpu_masks(struct bnxt *bp)
+{
+ int i;
+
+ if (!bp->ring_cpu_mask)
+ return;
+
+ for (i = 0; i < bp->max_irqs; i++)
+ free_cpumask_var(bp->ring_cpu_mask[i]);
+
+ bitmap_free(bp->ring_affinity_set);
+ bp->ring_affinity_set = NULL;
+ kfree(bp->ring_cpu_mask);
+ bp->ring_cpu_mask = NULL;
+}
+
static void bnxt_free_irq(struct bnxt *bp)
{
struct bnxt_irq *irq;
@@ -11877,13 +11909,7 @@ static void bnxt_free_irq(struct bnxt *bp)
irq = &bp->irq_tbl[map_idx];
if (irq->requested) {
bnxt_release_irq_notifier(irq);
-
- if (irq->have_cpumask) {
- irq_update_affinity_hint(irq->vector, NULL);
- free_cpumask_var(irq->cpu_mask);
- irq->have_cpumask = 0;
- }
-
+ irq_update_affinity_hint(irq->vector, NULL);
free_irq(irq->vector, bp->bnapi[i]);
}
@@ -11925,9 +11951,9 @@ static int bnxt_request_irq(struct bnxt *bp)
bp->tph_mode = PCI_TPH_ST_IV_MODE;
for (i = 0, j = 0; i < bp->cp_nr_rings; i++) {
+ struct cpumask *cpu_mask = bp->ring_cpu_mask[i];
int map_idx = bnxt_cp_num_to_irq_num(bp, i);
struct bnxt_irq *irq = &bp->irq_tbl[map_idx];
- unsigned int cpu_num;
u16 tag;
if (IS_ENABLED(CONFIG_RFS_ACCEL) &&
@@ -11946,19 +11972,24 @@ static int bnxt_request_irq(struct bnxt *bp)
netif_napi_set_irq_locked(&bp->bnapi[i]->napi, irq->vector);
irq->requested = 1;
-
- if (!zalloc_cpumask_var(&irq->cpu_mask, GFP_KERNEL))
- continue;
-
- irq->have_cpumask = 1;
irq->msix_nr = map_idx;
irq->ring_nr = i;
- cpu_num = cpumask_local_spread(i, numa_node);
- cpumask_set_cpu(cpu_num, irq->cpu_mask);
+
+ /* Reuse the mask recorded before the IRQs were freed. Nothing
+ * was recorded yet on the very first request, and the mask
+ * may have gone stale if the CPUs went offline in between.
+ */
+ if (!test_bit(i, bp->ring_affinity_set) ||
+ !cpumask_intersects(cpu_mask, cpu_online_mask)) {
+ clear_bit(i, bp->ring_affinity_set);
+ cpumask_clear(cpu_mask);
+ cpumask_set_cpu(cpumask_local_spread(i, numa_node),
+ cpu_mask);
+ }
/* Init ST table entry if we can get the mapping */
if (!pcie_tph_get_cpu_st(bp->pdev, TPH_MEM_TYPE_VM,
- cpu_num, &tag)) {
+ cpumask_first(cpu_mask), &tag)) {
pcie_tph_set_st_entry(bp->pdev, irq->msix_nr, tag);
irq->tag = tag;
irq->new_tag = tag;
@@ -11966,10 +11997,19 @@ static int bnxt_request_irq(struct bnxt *bp)
bnxt_register_irq_notifier(bp, irq);
- rc = irq_update_affinity_hint(irq->vector, irq->cpu_mask);
+ /* Only put the IRQ back where it was configured to be, our own
+ * placement is just a hint, the core spreads within
+ * irq_default_affinity which we know nothing about.
+ * Set after installing the notifier, if we race with the user
+ * it's better to overwrite than miss the notification.
+ */
+ if (test_bit(i, bp->ring_affinity_set))
+ rc = irq_set_affinity_and_hint(irq->vector, cpu_mask);
+ else
+ rc = irq_update_affinity_hint(irq->vector, cpu_mask);
if (rc) {
netdev_warn(bp->dev,
- "Update affinity hint failed, IRQ = %d\n",
+ "Setting IRQ affinity failed, IRQ = %d\n",
irq->vector);
break;
}
@@ -15025,6 +15065,7 @@ static void bnxt_unmap_bars(struct bnxt *bp, struct pci_dev *pdev)
static void bnxt_cleanup_pci(struct bnxt *bp)
{
+ pci_disable_ptm(bp->pdev);
bnxt_unmap_bars(bp, bp->pdev);
pci_release_regions(bp->pdev);
if (pci_is_enabled(bp->pdev))
@@ -15593,6 +15634,8 @@ static int bnxt_init_board(struct pci_dev *pdev, struct net_device *dev)
goto init_err_release;
}
+ pci_enable_ptm(pdev);
+
INIT_WORK(&bp->sp_task, bnxt_sp_task);
INIT_DELAYED_WORK(&bp->fw_reset_task, bnxt_fw_reset_task);
@@ -16607,6 +16650,7 @@ static void bnxt_remove_one(struct pci_dev *pdev)
bnxt_shutdown_tc(bp);
bnxt_clear_int_mode(bp);
+ bnxt_free_ring_cpu_masks(bp);
bnxt_hwrm_func_drv_unrgtr(bp);
bnxt_free_hwrm_resources(bp);
bnxt_hwmon_uninit(bp);
@@ -17045,6 +17089,11 @@ static int bnxt_init_one(struct pci_dev *pdev, const struct pci_device_id *ent)
bp->msg_enable = BNXT_DEF_MSG_ENABLE;
bnxt_set_max_func_irqs(bp, max_irqs);
+ bp->max_irqs = max_irqs;
+ rc = bnxt_alloc_ring_cpu_masks(bp);
+ if (rc)
+ goto init_err_free;
+
if (bnxt_vf_pciid(bp->board_idx))
bp->flags |= BNXT_FLAG_VF;
@@ -17099,7 +17148,7 @@ static int bnxt_init_one(struct pci_dev *pdev, const struct pci_device_id *ent)
}
dev->hw_features = NETIF_F_IP_CSUM | NETIF_F_IPV6_CSUM | NETIF_F_SG |
- NETIF_F_TSO | NETIF_F_TSO6 |
+ NETIF_F_TSO | NETIF_F_TSO6 | NETIF_F_TSO_ECN |
NETIF_F_GSO_UDP_TUNNEL | NETIF_F_GSO_GRE |
NETIF_F_GSO_IPXIP4 |
NETIF_F_GSO_UDP_TUNNEL_CSUM | NETIF_F_GSO_GRE_CSUM |
@@ -17112,7 +17161,7 @@ static int bnxt_init_one(struct pci_dev *pdev, const struct pci_device_id *ent)
dev->hw_enc_features =
NETIF_F_IP_CSUM | NETIF_F_IPV6_CSUM | NETIF_F_SG |
- NETIF_F_TSO | NETIF_F_TSO6 |
+ NETIF_F_TSO | NETIF_F_TSO6 | NETIF_F_TSO_ECN |
NETIF_F_GSO_UDP_TUNNEL | NETIF_F_GSO_GRE |
NETIF_F_GSO_UDP_TUNNEL_CSUM | NETIF_F_GSO_GRE_CSUM |
NETIF_F_GSO_IPXIP4 | NETIF_F_GSO_PARTIAL;
@@ -17294,6 +17343,7 @@ init_err_pci_clean:
bp->rss_indir_tbl = NULL;
init_err_free:
+ bnxt_free_ring_cpu_masks(bp);
free_netdev(dev);
return rc;
}
diff --git a/drivers/net/ethernet/broadcom/bnxt/bnxt.h b/drivers/net/ethernet/broadcom/bnxt/bnxt.h
index dc8ec5e5733e..ab894f8addef 100644
--- a/drivers/net/ethernet/broadcom/bnxt/bnxt.h
+++ b/drivers/net/ethernet/broadcom/bnxt/bnxt.h
@@ -1261,9 +1261,7 @@ struct bnxt_irq {
irq_handler_t handler;
unsigned int vector;
u8 requested:1;
- u8 have_cpumask:1;
char name[IFNAMSIZ + BNXT_IRQ_NAME_EXTRA];
- cpumask_var_t cpu_mask;
struct bnxt *bp;
int msix_nr;
@@ -2482,6 +2480,15 @@ struct bnxt {
pci_channel_offline((bp)->pdev))
struct bnxt_irq *irq_tbl;
+ /* IRQ affinity, indexed by completion ring. Kept across IRQ
+ * reallocation, the MSI-X vector index is not stable.
+ */
+ cpumask_var_t *ring_cpu_mask;
+ /* Rings for which the mask above was configured from the outside,
+ * rather than being our own default placement.
+ */
+ unsigned long *ring_affinity_set;
+ int max_irqs;
int total_irqs;
int ulp_num_msix_want;
u8 mac_addr[ETH_ALEN];