[PATCH net] xsk: freeze deferred pool teardown without blocking unregister

From: James Hilliard

Date: Thu Oct 01 2026 - 01:10:09 EST


Deferred pool destruction calls ndo_bpf() under RTNL. system_wq is
not frozen during system sleep, so that callback can run after a
device has suspended and gated its clocks. Use system_freezable_wq
so running destruction finishes before device suspend and new work
waits until process thaw.

Keep assigned pools visible to NETDEV_UNREGISTER independently of
the socket list. A released socket has already left that list, but
its final pool put can queue destruction after workqueues freeze.
A resume-time unregister would then wait for a device reference
whose release cannot run until the resume completes.

Track assigned pools under RTNL and detach remaining pools after the
notifier socket walk, including copy-mode pools. Remove the entry on
assignment failure and normal teardown. The deferred worker still
owns the pool and later observes the cleared device pointer, avoiding
a second driver detach or device put.

Fixes: 1c1efc2af158 ("xsk: Create and free buffer pool independently from umem")
Signed-off-by: James Hilliard <james.hilliard1@xxxxxxxxx>
---
include/net/xsk_buff_pool.h | 3 +++
net/xdp/xsk.c | 4 ++++
net/xdp/xsk_buff_pool.c | 25 ++++++++++++++++++++++++-
3 files changed, 31 insertions(+), 1 deletion(-)

diff --git a/include/net/xsk_buff_pool.h b/include/net/xsk_buff_pool.h
index a7df573784fd..cea6e883f7cd 100644
--- a/include/net/xsk_buff_pool.h
+++ b/include/net/xsk_buff_pool.h
@@ -48,6 +48,8 @@ struct xsk_buff_pool {
struct device *dev;
struct net_device *netdev;
struct list_head xsk_tx_list;
+ /* Assigned pools, including deferred releases; protected by RTNL. */
+ struct list_head dev_list;
/* Protects modifications to the xsk_tx_list */
spinlock_t xsk_tx_list_lock;
refcount_t users;
@@ -119,6 +121,7 @@ bool xp_put_pool(struct xsk_buff_pool *pool);
void xp_clear_dev(struct xsk_buff_pool *pool);
void xp_add_xsk(struct xsk_buff_pool *pool, struct xdp_sock *xs);
void xp_del_xsk(struct xsk_buff_pool *pool, struct xdp_sock *xs);
+void xp_clear_dev_all(struct net_device *dev);

/* AF_XDP, and XDP core. */
void xp_free(struct xdp_buff_xsk *xskb);
diff --git a/net/xdp/xsk.c b/net/xdp/xsk.c
index 33475b180ea6..a2ae0e10d893 100644
--- a/net/xdp/xsk.c
+++ b/net/xdp/xsk.c
@@ -2145,6 +2145,10 @@ static int xsk_notifier(struct notifier_block *this,
mutex_unlock(&xs->mutex);
}
mutex_unlock(&net->xdp.lock);
+ /* A released socket is no longer on xdp.list. Its pool can still
+ * hold a device reference on the frozen release workqueue.
+ */
+ xp_clear_dev_all(dev);
break;
}
return NOTIFY_DONE;
diff --git a/net/xdp/xsk_buff_pool.c b/net/xdp/xsk_buff_pool.c
index 9d2d94f1fb75..b5fd0850db5b 100644
--- a/net/xdp/xsk_buff_pool.c
+++ b/net/xdp/xsk_buff_pool.c
@@ -12,6 +12,11 @@

#define ETH_PAD_LEN (ETH_HLEN + 2 * VLAN_HLEN + ETH_FCS_LEN)

+/* Socket removal precedes the final pool put. Keep assigned pools visible to
+ * NETDEV_UNREGISTER even while their release work is waiting for process thaw.
+ */
+static LIST_HEAD(xsk_dev_pools);
+
void xp_add_xsk(struct xsk_buff_pool *pool, struct xdp_sock *xs)
{
if (!xs->tx)
@@ -204,6 +209,7 @@ int xp_assign_dev(struct xsk_buff_pool *pool,
pool->cached_need_wakeup = XDP_WAKEUP_TX;

dev_hold(netdev);
+ list_add_tail(&pool->dev_list, &xsk_dev_pools);

if (force_copy)
/* For copy-mode, we are done. */
@@ -265,6 +271,8 @@ int xp_assign_dev(struct xsk_buff_pool *pool,
err = 0; /* fallback to copy mode */
if (err) {
xsk_clear_pool_at_qid(netdev, queue_id);
+ list_del(&pool->dev_list);
+ pool->netdev = NULL;
dev_put(netdev);
}
return err;
@@ -291,17 +299,29 @@ void xp_clear_dev(struct xsk_buff_pool *pool)
{
struct net_device *netdev = pool->netdev;

+ ASSERT_RTNL();
if (!pool->netdev)
return;

netdev_lock_ops(netdev);
xp_disable_drv_zc(pool);
xsk_clear_pool_at_qid(pool->netdev, pool->queue_id);
+ list_del(&pool->dev_list);
pool->netdev = NULL;
netdev_unlock_ops(netdev);
dev_put(netdev);
}

+void xp_clear_dev_all(struct net_device *dev)
+{
+ struct xsk_buff_pool *pool, *next;
+
+ ASSERT_RTNL();
+ list_for_each_entry_safe(pool, next, &xsk_dev_pools, dev_list)
+ if (pool->netdev == dev)
+ xp_clear_dev(pool);
+}
+
static void xp_release_deferred(struct work_struct *work)
{
struct xsk_buff_pool *pool = container_of(work, struct xsk_buff_pool,
@@ -337,7 +357,10 @@ bool xp_put_pool(struct xsk_buff_pool *pool)

if (refcount_dec_and_test(&pool->users)) {
INIT_WORK(&pool->work, xp_release_deferred);
- schedule_work(&pool->work);
+ /* Teardown calls ndo_bpf(), which may need powered hardware.
+ * RTNL alone does not exclude the device's system PM callbacks.
+ */
+ queue_work(system_freezable_wq, &pool->work);
return true;
}


---
base-commit: a7bfaba4823e3c165bb2004c74eff7c096672bc7
change-id: 20260930-xsk-suspend-teardown-902f98408c93

Best regards,
--
James Hilliard <james.hilliard1@xxxxxxxxx>