From 1b82958f3f035df5ccaab5430a2302f08a5d5351 Mon Sep 17 00:00:00 2001 From: Alexander Duyck Date: Mon, 14 Sep 2026 14:09:57 -0700 Subject: net: ethtool: keep rtnl_lock for the ioctl self test An offline self test that brings the interface down and back up with netif_close() / netif_open() requires rtnl_lock for both. Since the ethtool IOCTL path became rtnl-optional for ops-locked drivers, the ETHTOOL_TEST ioctl runs holding only the netdev instance lock, so on an ops-locked driver the self test now tears the device down without rtnl_lock. With lockdep this reproduces deterministically on every offline self test on such a driver; note the sole lock held is the instance lock, not rtnl: WARNING: suspicious RCU usage net/core/netpoll.c:207 suspicious rcu_dereference_protected() usage! 1 lock held by ethtool/107: #0: (&dev->lock){+.+.}, at: dev_ethtool Call Trace: netpoll_poll_disable __dev_close_many netif_close_many netif_close fbnic_self_test dev_ethtool_locked dev_ethtool dev_ioctl sock_ioctl __x64_sys_ioctl Without lockdep the same condition trips ASSERT_RTNL() in __dev_close_many() / __dev_open(); that check only samples the global rtnl state, so it can be masked by a concurrent rtnl holder, but the device is still being reconfigured without the lock it requires. The ethtool self_test is a legacy ioctl-only command, so an ETHTOOL_TEST case is only needed on the ioctl path. Add an opt-in bit for drivers whose self test needs rtnl_lock and set it on the ops-locked drivers whose offline self test tears the interface down and up: - fbnic (ops-locked via queue_mgmt_ops): fbnic_self_test() offline path uses netif_close() / netif_open(). - bnxt (ops-locked via queue_mgmt_ops): bnxt_self_test() offline path goes through bnxt_close_nic() / bnxt_half_open_nic() / bnxt_half_close_nic() / bnxt_open_nic(), which close and reopen the device. Fixes: f994752b1127 ("net: ethtool: optionally skip rtnl_lock on IOCTL path") Signed-off-by: Alexander Duyck Reviewed-by: Simon Horman Link: https://patch.msgid.link/178942019771.7700.338431553546884773.stgit@ahduyck-xeon-server.home.arpa Signed-off-by: Jakub Kicinski --- include/linux/ethtool.h | 2 ++ 1 file changed, 2 insertions(+) (limited to 'include/linux') diff --git a/include/linux/ethtool.h b/include/linux/ethtool.h index 253600c0eccd..c4c9ce038611 100644 --- a/include/linux/ethtool.h +++ b/include/linux/ethtool.h @@ -944,6 +944,7 @@ struct kernel_ethtool_ts_info { #define ETHTOOL_OP_NEEDS_RTNL_SPAUSEPARAM BIT(6) #define ETHTOOL_OP_NEEDS_RTNL_RSS BIT(7) #define ETHTOOL_OP_NEEDS_RTNL_GLINK BIT(8) +#define ETHTOOL_OP_NEEDS_RTNL_TEST BIT(9) /** * struct ethtool_ops - optional netdev operations @@ -981,6 +982,7 @@ struct kernel_ethtool_ts_info { * - netdev_update_features() * - netif_set_real_num_tx_queues() * - ethtool_op_get_link() (syncs link watch under rtnl_lock) + * - netif_open() / netif_close() (used by @self_test) * * @get_drvinfo: Report driver/device information. Modern drivers no * longer have to implement this callback. Most fields are -- cgit v1.2.3 From ab888242fce4f16f6c4d4c6ec53939ad36aa3b3a Mon Sep 17 00:00:00 2001 From: Xiang Mei Date: Tue, 15 Sep 2026 01:31:52 -0700 Subject: vlan: require the MAC header to be present in __vlan_insert_inner_tag() __vlan_insert_inner_tag() only guarantees head room via skb_cow_head(), never that mac_len bytes of MAC header are present. Its ETH_HLEN wrappers - __vlan_insert_tag() under skb_vlan_push(), and vlan_insert_tag() under validate_xmit_vlan() on the generic transmit path - therefore rewrite the first 16 bytes at skb->data: a 12-byte memmove plus two 2-byte stores at +12 and +14. No caller supplies the bound, while the pop helpers use skb_ensure_writable()/pskb_may_pull(). An IFF_TUN device has hard_header_len == 0, so packet_snd() accepts a one-byte AF_PACKET/SOCK_RAW frame. The first vlan push only sets a hwaccel tag; the next - clsact "action vlan push" or bpf_skb_vlan_push() - enters the helper with skb->len still 1. The head comes from skbuff_small_head without __GFP_ZERO, so each push drags bytes from beyond skb->tail into the frame. After three the one-byte send leaves as 13 bytes carrying 11 bytes of uninitialised slab: 0000: 5a b3 62 12 80 88 ff ff 00 b3 62 12 81 `------------------------------' only 0x5a was sent; the rest is slab, here the top 56 bits of a linear-map address Require the MAC header the helper rewrites to be present, so such a frame is dropped rather than transmitted. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Cc: stable@vger.kernel.org Reported-by: co+0ea1ac045375cf05@bugs.sh Signed-off-by: Xiang Mei Reviewed-by: Simon Horman Link: https://patch.msgid.link/20260915083152.705309-1-xmei5@asu.edu Signed-off-by: Jakub Kicinski --- include/linux/if_vlan.h | 3 +++ 1 file changed, 3 insertions(+) (limited to 'include/linux') diff --git a/include/linux/if_vlan.h b/include/linux/if_vlan.h index 20cc16ea4e5a..4846032bf4ff 100644 --- a/include/linux/if_vlan.h +++ b/include/linux/if_vlan.h @@ -365,6 +365,9 @@ static inline int __vlan_insert_inner_tag(struct sk_buff *skb, const u8 meta_len = mac_len > ETH_TLEN ? skb_metadata_len(skb) : 0; struct vlan_ethhdr *veth; + if (unlikely(!pskb_may_pull(skb, mac_len))) + return -EINVAL; + if (skb_cow_head(skb, meta_len + VLAN_HLEN) < 0) return -ENOMEM; -- cgit v1.2.3 From 9518405613863d0bf0700927a21367f5942cf058 Mon Sep 17 00:00:00 2001 From: Willem de Bruijn Date: Fri, 18 Sep 2026 20:47:29 -0400 Subject: packet: use ubuf_info completion for TX_RING packets tpacket_snd sends skbs with frags pointing into its ring slots. Slots are released when skb->destructor is called. A call to skb_orphan calls skb->destructor before the skb is freed. This can cause the slot to be reused while still linked into the skb. Switch to standard zerocopy completion (ubuf_info) so the slot is only released once all references to the payload are freed or copied. Restore skb->destructor to standard sock_wfree. The ubuf_info completion callback can be called with a NULL skb, but only from net_zcopy_put and related API, used by zerocopy implementations that hold their own reference on the uarg, such as MSG_ZEROCOPY. This uarg is only ever completed from skb_zcopy_clear, so skb is always set. To prevent userspace from aliasing in-flight state on shared ring slots, allocate tpacket_uarg per packet, rather than per slot. This adds a small allocation to the transmit path. Use standard kmalloc to allow backporting to stable kernels. The uarg holds an sk_wmem_alloc reference, rather than an sk_refcnt reference. packet_free_tx_ring waits on sk_wmem_alloc before freeing the ring pages. Always allocate vec->deferred for tx_ring so page-backed rings also wait on sk_wmem_alloc when skb_copy_ubufs drops page refs before calling tpacket_ubuf_complete. Drop the tx_ring.pg_vec test that tpacket_destruct_skb performed before accessing the slot. The sk_wmem_alloc reference now guarantees that the slot is valid. The test is also not sufficient by itself, as it reads pg_vec without pg_vec_lock, so it can race with packet_set_ring. As a result a slot is released when its payload is copied, which can be before transmission (e.g., in skb_orphan_frags_rx). If copied before skb_tx_timestamp() is called, no slot timestamp is recorded, similar to when skb_orphan() was called early in the datapath before this patch. Revert the now unused previous skb_zcopy_.._nouarg infra. Depends on commit 992cc9f94ca9 ("net/packet: defer vmalloc TX_RING free until skbs finish"). Reported-by: Katherine Leaver Reported-by: Bjoern Doebel Closes: https://lore.kernel.org/netdev/20260909085542.3370986-1-doebel@amazon.de/ Fixes: 5cd8d46ea156 ("packet: copy user buffers before orphan or clone") Cc: stable@vger.kernel.org Signed-off-by: Willem de Bruijn Link: https://patch.msgid.link/20260919004748.1463985-3-willemdebruijn.kernel@gmail.com Signed-off-by: Jakub Kicinski --- include/linux/skbuff.h | 19 +------------------ 1 file changed, 1 insertion(+), 18 deletions(-) (limited to 'include/linux') diff --git a/include/linux/skbuff.h b/include/linux/skbuff.h index 421f6fc45451..b14d6be7370b 100644 --- a/include/linux/skbuff.h +++ b/include/linux/skbuff.h @@ -1834,22 +1834,6 @@ static inline void skb_zcopy_set(struct sk_buff *skb, struct ubuf_info *uarg, } } -static inline void skb_zcopy_set_nouarg(struct sk_buff *skb, void *val) -{ - skb_shinfo(skb)->destructor_arg = (void *)((uintptr_t) val | 0x1UL); - skb_shinfo(skb)->flags |= SKBFL_ZEROCOPY_FRAG; -} - -static inline bool skb_zcopy_is_nouarg(struct sk_buff *skb) -{ - return (uintptr_t) skb_shinfo(skb)->destructor_arg & 0x1UL; -} - -static inline void *skb_zcopy_get_nouarg(struct sk_buff *skb) -{ - return (void *)((uintptr_t) skb_shinfo(skb)->destructor_arg & ~0x1UL); -} - static inline void net_zcopy_put(struct ubuf_info *uarg) { if (uarg) @@ -1872,8 +1856,7 @@ static inline void skb_zcopy_clear(struct sk_buff *skb, bool zerocopy_success) struct ubuf_info *uarg = skb_zcopy(skb); if (uarg) { - if (!skb_zcopy_is_nouarg(skb)) - uarg->ops->complete(skb, uarg, zerocopy_success); + uarg->ops->complete(skb, uarg, zerocopy_success); skb_shinfo(skb)->flags &= ~SKBFL_ALL_ZEROCOPY; } -- cgit v1.2.3