Thread (10 messages) flat view 10 messages, 3 authors, 1d ago
WARM1d

Revision v3 of 3 in this series.

Revisions (3)
  1. v1 [diff vs current]
  2. v2 [diff vs current]
  3. v3 current

[PATCH v3 net-next 7/7] ip_tunnel: Support per-netns device unregistration.

From: Kuniyuki Iwashima <kuniyu@google.com>
Date: 2026-09-12 23:00:52
Subsystem: networking [general], networking [ipv4/ipv6], the rest · Maintainers: "David S. Miller", Eric Dumazet, Jakub Kicinski, Paolo Abeni, David Ahern, Ido Schimmel, Linus Torvalds

ip_tunnel_delete_net() iterates ip_tunnel devices whose link_net
is dying and queues them for destruction.

The devices may reside in different netns.

Let's use unregister_netdevice_queue_net() to support per-netns
device unregistration.

Even after ip_tunnel_delete_net() queues a cross-netns ip_tunnel
device, ip_tunnel_changelink(), ip_tunnel_dellink(), and
ip_tunnel_ctl() could be called concurrently for it (once RTNL is
removed).  In such a case, __rtnl_net_unlock() will perform the
unregistration.

Also, ip_tunnel_ctl() needs to check check_net(t->net), otherwise
it could create a new dev in dying netns after ip_tunnel_delete_net().

In the example below, we can see the fallback tunnel device (gre0)
and the cross-netns device (gre1) are unregistered by different
processes:

  # bpftrace -e '#include <linux/netdevice.h>
  kprobe:ip_tunnel_uninit {
      $dev = (struct net_device *)arg0;
      printf("PID: %d | DEV: %s%s\n", pid, $dev->name, kstack());
  }
  kprobe:ipgre_exit_rtnl {
      printf("PID: %d%s\n", pid, kstack());
  }' &

  # ip netns add ns1
  # ip netns add ns2
  # ip -n ns1 link add name gre1 link-netns ns2 \
    type gre local 192.168.0.1 remote 192.168.1.1
  # ip netns del ns2

  PID: 12
          ipgre_exit_rtnl+5
          ops_undo_list+702
          cleanup_net+1122
          process_scheduled_works+2538
  ...
  PID: 12 | DEV: gre0  <------ fallback device (itn->fb_tunnel_dev).
          ip_tunnel_uninit+5
          unregister_netdevice_many_notify+7129
          unregister_netdevice_many_net+1050
          __rtnl_net_unlock+37
          ops_undo_list+754
          cleanup_net+1122
          process_scheduled_works+2538
  ...
  PID: 10 | DEV: gre1
          ip_tunnel_uninit+5
          unregister_netdevice_many_notify+7129
          unregister_netdevice_many_net+1050
          rtnl_net_work_func+136
          process_scheduled_works+2538

Signed-off-by: Kuniyuki Iwashima <kuniyu@google.com>
---
v3: Use check_net() in ip_tunnel_ctl().
v2: Check if (!itn->fb_tunnel_dev) in ip_tunnel_ctl().
---
 net/ipv4/ip_tunnel.c | 32 ++++++++++++++++++++++++++------
 1 file changed, 26 insertions(+), 6 deletions(-)
diff --git a/net/ipv4/ip_tunnel.c b/net/ipv4/ip_tunnel.c
index d44976395c7c..0875474a578a 100644
--- a/net/ipv4/ip_tunnel.c
+++ b/net/ipv4/ip_tunnel.c
@@ -205,6 +205,11 @@ static void ip_tunnel_del(struct ip_tunnel_net *itn, struct ip_tunnel *t)
 	hlist_del_init_rcu(&t->hash_node);
 }
 
+static bool ip_tunnel_unregistering(struct ip_tunnel *t)
+{
+	return hlist_unhashed(&t->hash_node);
+}
+
 static struct ip_tunnel *ip_tunnel_find(struct ip_tunnel_net *itn,
 					struct ip_tunnel_parm_kern *parms,
 					int type)
@@ -893,20 +898,22 @@ static void ip_tunnel_update(struct ip_tunnel_net *itn,
 	netdev_state_change(dev);
 }
 
-static void __ip_tunnel_dellink(struct net_device *dev, struct list_head *head)
+static void __ip_tunnel_dellink(struct net *net, struct net_device *dev,
+				struct list_head *head)
 {
 	struct ip_tunnel *tunnel = netdev_priv(dev);
 	struct ip_tunnel_net *itn;
 
 	itn = net_generic(tunnel->net, tunnel->ip_tnl_net_id);
 	ip_tunnel_del(itn, tunnel);
-	unregister_netdevice_queue(dev, head);
+	unregister_netdevice_queue_net(net, dev, head);
 }
 
 int ip_tunnel_ctl(struct net_device *dev, struct ip_tunnel_parm_kern *p,
 		  int cmd)
 {
 	struct ip_tunnel *t = netdev_priv(dev);
+	struct net *orig_net = dev_net(dev);
 	struct ip_tunnel_net *itn;
 	LIST_HEAD(dev_kill_list);
 	struct net *net = t->net;
@@ -918,6 +925,11 @@ int ip_tunnel_ctl(struct net_device *dev, struct ip_tunnel_parm_kern *p,
 
 	mutex_lock(&itn->tunnels_lock);
 
+	if (!check_net(net)) {
+		err = -EBUSY;
+		goto done;
+	}
+
 	switch (cmd) {
 	case SIOCGETTUNNEL:
 		if (dev == itn->fb_tunnel_dev) {
@@ -977,7 +989,7 @@ int ip_tunnel_ctl(struct net_device *dev, struct ip_tunnel_parm_kern *p,
 			}
 		}
 
-		if (t) {
+		if (t && !ip_tunnel_unregistering(t)) {
 			err = 0;
 			ip_tunnel_update(itn, t, dev, p, true, 0);
 		} else {
@@ -1001,7 +1013,9 @@ int ip_tunnel_ctl(struct net_device *dev, struct ip_tunnel_parm_kern *p,
 			dev = t->dev;
 		}
 
-		__ip_tunnel_dellink(dev, &dev_kill_list);
+		if (!ip_tunnel_unregistering(t))
+			__ip_tunnel_dellink(orig_net, dev, &dev_kill_list);
+
 		err = 0;
 		break;
 
@@ -1109,7 +1123,8 @@ void ip_tunnel_dellink(struct net_device *dev, struct list_head *head)
 
 	if (itn->fb_tunnel_dev != dev) {
 		mutex_lock(&itn->tunnels_lock);
-		__ip_tunnel_dellink(dev, head);
+		if (!ip_tunnel_unregistering(tunnel))
+			__ip_tunnel_dellink(dev_net(dev), dev, head);
 		mutex_unlock(&itn->tunnels_lock);
 	}
 }
@@ -1192,7 +1207,7 @@ void ip_tunnel_delete_net(struct net *net, unsigned int id,
 		struct ip_tunnel *t;
 
 		hlist_for_each_entry_safe(t, n, thead, hash_node)
-			__ip_tunnel_dellink(t->dev, head);
+			__ip_tunnel_dellink(net, t->dev, head);
 	}
 
 	mutex_unlock(&itn->tunnels_lock);
@@ -1302,6 +1317,11 @@ int ip_tunnel_changelink(struct net_device *dev, struct nlattr *tb[],
 		}
 	}
 
+	if (ip_tunnel_unregistering(t)) {
+		err = -ENODEV;
+		goto out;
+	}
+
 	ip_tunnel_update(itn, t, dev, p, !tb[IFLA_MTU], fwmark);
 out:
 	mutex_unlock(&itn->tunnels_lock);
-- 
2.55.0.1007.g17ff1f9808-goog
Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help