net: Add per-netns netdev unregistration infra.

When we need to unregister a netdev in a different netns, we will
delegate its unregistration to per-netns work.

There are three types of such cross-netns devices:

  1. Paired devices (e.g., netkit, veth, vxcan)
     -> Unregistering one device also deletes its peer, which
        may reside in another netns.

  2. Tunnel devices (e.g., bareudp, geneve, etc)
     -> Destroying a netns removes devices in another netns if
        their backend sockets reside in the dying netns

  3. Stacked devices (e.g., ipvlan, macvlan, etc)
     -> Removing the lower device also removes multiple upper
        devices, each of which may reside in different namespaces.

In these cases, we will use unregister_netdevice_queue_net() to
queue such potential cross-netns devices for destruction.

Each driver must not call both unregister_netdevice_queue_net()
and unregister_netdevice_queue() for the same device.  See the
subsequent veth/bareudp/ipvlan patches for how they avoid double
queueing.

unregister_netdevice_queue_net() takes net and dev.  If dev resides
in the net, it simply calls unregister_netdevice_queue().

If dev_net(dev) is different from the net, it enqueues the device
to dev_net(dev)->dev_unreg_head and schedules the per-netns work.

When __rtnl_net_unlock() is called from the per-netns work (or another
thread already holding the lock), unregister_netdevice_many_net()
collects the queued devices and calls unregister_netdevice_many()
to perform the actual unregistration.

During netns dismantle, rtnl_net_flush_workqueue() is called at the
end of default_device_exit_batch() to ensure that cross-netns
devices in the other alive netns are unregistered.

Once RTNL is removed, a device could be moved to another netns while
being queued to net->dev_unreg_head.

__dev_change_net_namespace() handles this race by acquiring
net->dev_unreg_lock of both the old and new netns after dev_set_net()
and moving the device between their dev_unreg_head lists.

Since dev_set_net() and unregister_netdevice_queue_net() are
synchronised by netdev_lock(), the device is either queued to the
old netns's dev_unreg_head and then moved, or queued directly to
the new netns.

Note that unregister_netdevice_move_net() does not need to call
rtnl_net_queue_work() because __dev_change_net_namespace() is
(supposed to be) called with rtnl_net_lock().  (Not all callers
hold it yet, but the race does not happen until all callers
are converted and RTNL is removed.)

Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
Link: https://patch.msgid.link/20260703001009.1572444-7-kuniyu@google.com
Signed-off-by: Paolo Abeni <pabeni@redhat.com>
This commit is contained in:
Kuniyuki Iwashima
2026-07-11 12:57:49 +02:00
committed by Paolo Abeni
parent 0fa296dd52
commit af3634d4ac
5 changed files with 117 additions and 0 deletions
+18
View File
@@ -1845,6 +1845,8 @@ enum netdev_reg_state {
* @napi_list: List entry used for polling NAPI devices
* @unreg_list: List entry when we are unregistering the
* device; see the function unregister_netdev
* @unreg_list_net:List entry when we are unregistering the cross-netns
* device; see the function unregister_netdevice_queue_net()
* @close_list: List entry used when we are closing the device
* @ptype_all: Device-specific packet handlers for all protocols
* @ptype_specific: Device-specific, protocol-specific packet handlers
@@ -2241,6 +2243,9 @@ struct net_device {
struct list_head dev_list;
struct list_head napi_list;
struct list_head unreg_list;
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
struct list_head unreg_list_net;
#endif
struct list_head close_list;
struct list_head ptype_all;
@@ -3472,6 +3477,19 @@ static inline void unregister_netdevice(struct net_device *dev)
unregister_netdevice_queue(dev, NULL);
}
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
void unregister_netdevice_queue_net(struct net *net, struct net_device *dev,
struct list_head *head);
void unregister_netdevice_many_net(struct net *net);
#else
static inline void unregister_netdevice_queue_net(struct net *net,
struct net_device *dev,
struct list_head *head)
{
unregister_netdevice_queue(dev, head);
}
#endif
int netdev_refcnt_read(const struct net_device *dev);
void free_netdev(struct net_device *dev);
+2
View File
@@ -198,6 +198,8 @@ struct net {
/* Move to a better place when the config guard is removed. */
struct mutex rtnl_mutex;
struct work_struct rtnl_work;
struct list_head dev_unreg_head;
spinlock_t dev_unreg_lock;
#endif
#if IS_ENABLED(CONFIG_VSOCKETS)
struct netns_vsock vsock;
+91
View File
@@ -12097,6 +12097,9 @@ struct net_device *alloc_netdev_mqs(int sizeof_priv, const char *name,
INIT_LIST_HEAD(&dev->napi_list);
INIT_LIST_HEAD(&dev->unreg_list);
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
INIT_LIST_HEAD(&dev->unreg_list_net);
#endif
INIT_LIST_HEAD(&dev->close_list);
INIT_LIST_HEAD(&dev->link_watch_list);
INIT_LIST_HEAD(&dev->adj_list.upper);
@@ -12314,6 +12317,10 @@ void unregister_netdevice_queue(struct net_device *dev, struct list_head *head)
{
ASSERT_RTNL();
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
DEBUG_NET_WARN_ON_ONCE(!list_empty(&dev->unreg_list_net));
#endif
if (head) {
list_move_tail(&dev->unreg_list, head);
} else {
@@ -12490,6 +12497,16 @@ void unregister_netdevice_many_notify(struct list_head *head,
synchronize_net();
list_for_each_entry(dev, head, unreg_list) {
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
struct net *net = dev_net(dev);
/* spin_lock() can be moved outside of the loop
* once the per-netns RTNL conversion completes.
*/
spin_lock(&net->dev_unreg_lock);
list_del(&dev->unreg_list_net);
spin_unlock(&net->dev_unreg_lock);
#endif
netdev_put(dev, &dev->dev_registered_tracker);
net_set_todo(dev);
cnt++;
@@ -12512,6 +12529,74 @@ void unregister_netdevice_many(struct list_head *head)
}
EXPORT_SYMBOL(unregister_netdevice_many);
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
void unregister_netdevice_queue_net(struct net *net, struct net_device *dev,
struct list_head *head)
{
netdev_lock(dev);
if (net_eq(dev_net(dev), net)) {
netdev_unlock(dev);
unregister_netdevice_queue(dev, head);
return;
}
net = dev_net(dev);
spin_lock(&net->dev_unreg_lock);
DEBUG_NET_WARN_ON_ONCE(!list_empty(&dev->unreg_list));
DEBUG_NET_WARN_ON_ONCE(!list_empty(&dev->unreg_list_net));
list_add_tail(&dev->unreg_list_net, &net->dev_unreg_head);
rtnl_net_queue_work(net);
spin_unlock(&net->dev_unreg_lock);
netdev_unlock(dev);
}
EXPORT_SYMBOL(unregister_netdevice_queue_net);
static void unregister_netdevice_move_net(struct net *net_old,
struct net *net,
struct net_device *dev)
{
if (net_old > net) {
spin_lock(&net->dev_unreg_lock);
spin_lock_nested(&net_old->dev_unreg_lock, SINGLE_DEPTH_NESTING);
} else {
spin_lock(&net_old->dev_unreg_lock);
spin_lock_nested(&net->dev_unreg_lock, SINGLE_DEPTH_NESTING);
}
if (!list_empty(&dev->unreg_list_net)) {
list_del(&dev->unreg_list_net);
list_add_tail(&dev->unreg_list_net, &net->dev_unreg_head);
}
spin_unlock(&net_old->dev_unreg_lock);
spin_unlock(&net->dev_unreg_lock);
}
void unregister_netdevice_many_net(struct net *net)
{
struct net_device *dev, *tmp;
LIST_HEAD(unreg_head_net);
LIST_HEAD(unreg_head);
spin_lock(&net->dev_unreg_lock);
list_splice_init(&net->dev_unreg_head, &unreg_head_net);
spin_unlock(&net->dev_unreg_lock);
list_for_each_entry_safe(dev, tmp, &unreg_head_net, unreg_list_net) {
list_del_init(&dev->unreg_list_net);
list_add_tail(&dev->unreg_list, &unreg_head);
}
unregister_netdevice_many(&unreg_head);
}
#endif
/**
* unregister_netdev - remove device from the kernel
* @dev: device
@@ -12668,6 +12753,10 @@ int __dev_change_net_namespace(struct net_device *dev, struct net *net,
netdev_unlock(dev);
dev->ifindex = new_ifindex;
#ifdef CONFIG_DEBUG_NET_SMALL_RTNL
unregister_netdevice_move_net(net_old, net, dev);
#endif
if (new_name[0]) {
/* Rename the netdev to prepared name */
write_seqlock_bh(&netdev_rename_lock);
@@ -13110,6 +13199,8 @@ static void __net_exit default_device_exit_batch(struct list_head *net_list)
}
unregister_netdevice_many(&dev_kill_list);
rtnl_unlock();
rtnl_net_flush_workqueue();
}
static struct pernet_operations __net_initdata default_device_ops = {
+2
View File
@@ -423,6 +423,8 @@ static __net_init int preinit_net(struct net *net, struct user_namespace *user_n
mutex_init(&net->rtnl_mutex);
lock_set_cmp_fn(&net->rtnl_mutex, rtnl_net_lock_cmp_fn, NULL);
INIT_WORK(&net->rtnl_work, rtnl_net_work_func);
INIT_LIST_HEAD(&net->dev_unreg_head);
spin_lock_init(&net->dev_unreg_lock);
#endif
INIT_LIST_HEAD(&net->ptype_all);
+4
View File
@@ -197,6 +197,7 @@ void __rtnl_net_unlock(struct net *net)
{
ASSERT_RTNL();
unregister_netdevice_many_net(net);
mutex_unlock(&net->rtnl_mutex);
}
EXPORT_SYMBOL(__rtnl_net_unlock);
@@ -290,6 +291,9 @@ void rtnl_net_work_func(struct work_struct *work)
{
struct net *net = container_of(work, struct net, rtnl_work);
if (list_empty(&net->dev_unreg_head))
return;
rtnl_net_lock(net);
rtnl_net_unlock(net);
}