IPv6 uncached routes are linked to the global per-cpu lists, rt6_uncached_list. When unregistering a netdev, rt6_uncached_list_flush_dev() iterates over the potentially long lists to find uncached routes tied to the device and swap it with blackhole_netdev. Since it is called for every device in dying netns under RTNL, it adds O(N_dev x (N_cpu + N_route)) costs to any batched device unregistration. Let's call it once per batched device unregistration without RTNL. Because it now runs without RTNL, rt->dst.dev and rt->rt6i_idev->dev may transition to NETREG_UNREGISTERED at different times. Thus, we must check that both devices are blackhole_netdev before unlinking the route; otherwise, the remaining device reference would not be flushed when the other device is unregistered. In addition, since netdev_run_todo() runs even when IPv6 is disabled, rt6_uncached_list_flush_dev() must check ipv6_mod_enabled() before touching rt6_uncached_list. Note also that netdev_run_todo() cannot be called before rt6_uncached_list is initialised because: 1. unregister_netdevice_many_notify() panics when dev_boot_phase == 1 (net_dev_init() is subsys_initcall()). 2. the only device registered before fs_initcall_sync() is loopback and loopback_net_init() panics in case of failure. Reported-by: Chris J Arges Closes: https://lore.kernel.org/netdev/20260917-hash-bucket-route-lists-v3-0-30493a37b6eb@cloudflare.com/ Signed-off-by: Kuniyuki Iwashima --- v3: * Unlink only when both devices are blackhole_netdev * Clarify netdev_run_todo() does not run before rt6_uncached_list is initialised * Fix CONFIG_IPV6=n build --- include/net/ip6_route.h | 7 +++++++ net/core/dev.c | 5 ++++- net/ipv6/route.c | 24 ++++++++++++++++-------- 3 files changed, 27 insertions(+), 9 deletions(-) diff --git a/include/net/ip6_route.h b/include/net/ip6_route.h index 0f9b7a260d25..b9de335b639a 100644 --- a/include/net/ip6_route.h +++ b/include/net/ip6_route.h @@ -231,6 +231,13 @@ void rt6_multipath_rebalance(struct fib6_info *f6i); void rt6_uncached_list_add(struct rt6_info *rt); void rt6_uncached_list_del(struct rt6_info *rt); +#ifdef CONFIG_IPV6 +void rt6_uncached_list_flush_dev(struct net_device *dev); +#else +static inline void rt6_uncached_list_flush_dev(struct net_device *dev) +{ +} +#endif static inline const struct rt6_info *skb_rt6_info(const struct sk_buff *skb) { diff --git a/net/core/dev.c b/net/core/dev.c index 7dba0292f052..12a0e7d6567b 100644 --- a/net/core/dev.c +++ b/net/core/dev.c @@ -131,6 +131,7 @@ #include #include #include +#include #include #include #include @@ -11839,8 +11840,10 @@ void netdev_run_todo(void) linkwatch_sync_dev(dev); } - if (!list_empty(&list)) + if (!list_empty(&list)) { rt_flush_dev(NULL); + rt6_uncached_list_flush_dev(NULL); + } cnt = 0; while (!list_empty(&list)) { diff --git a/net/ipv6/route.c b/net/ipv6/route.c index a76869ff87cd..b1dcb5e5da88 100644 --- a/net/ipv6/route.c +++ b/net/ipv6/route.c @@ -158,10 +158,16 @@ void rt6_uncached_list_del(struct rt6_info *rt) } } -static void rt6_uncached_list_flush_dev(struct net_device *dev) +void rt6_uncached_list_flush_dev(struct net_device *dev) { int cpu; + if (!ipv6_mod_enabled()) + return; + + if (dev && dev->dismantle) + return; + for_each_possible_cpu(cpu) { struct uncached_list *ul = per_cpu_ptr(&rt6_uncached_list, cpu); struct rt6_info *rt, *safe; @@ -173,22 +179,24 @@ static void rt6_uncached_list_flush_dev(struct net_device *dev) list_for_each_entry_safe(rt, safe, &ul->head, dst.rt_uncached) { struct inet6_dev *rt_idev = rt->rt6i_idev; struct net_device *rt_dev = rt->dst.dev; - bool handled = false; - if (rt_idev && rt_idev->dev == dev) { + if (rt_idev && + (dev ? rt_idev->dev == dev : + READ_ONCE(rt_idev->dev->reg_state) == NETREG_UNREGISTERED)) { rt->rt6i_idev = in6_dev_get(blackhole_netdev); in6_dev_put(rt_idev); - handled = true; } - if (rt_dev == dev) { - rt->dst.dev = blackhole_netdev; + if (dev ? rt_dev == dev : + READ_ONCE(rt_dev->reg_state) == NETREG_UNREGISTERED) { + rcu_assign_pointer(rt->dst.dev_rcu, blackhole_netdev); netdev_ref_replace(rt_dev, blackhole_netdev, &rt->dst.dev_tracker, GFP_ATOMIC); - handled = true; } - if (handled) + + if (rt->dst.dev == blackhole_netdev && + (!rt->rt6i_idev || rt->rt6i_idev->dev == blackhole_netdev)) list_del_init(&rt->dst.rt_uncached); } spin_unlock_bh(&ul->lock); -- 2.56.0.rc1.315.gc6ed9934b7-goog