]> git.ipfire.org Git - thirdparty/kernel/linux.git/commitdiff
ipv4: fix use-after-free in fib_nhc_update_mtu()
authorChengfeng Ye <nicoyip.dev@gmail.com>
Fri, 7 Aug 2026 18:17:10 +0000 (02:17 +0800)
committerPaolo Abeni <pabeni@redhat.com>
Tue, 11 Aug 2026 09:25:10 +0000 (11:25 +0200)
fib_nhc_update_mtu() walks the nexthop exception table under RTNL, but
RTNL does not serialize this walk with PMTU exception updates. The walk
uses rcu_dereference_protected() with a constant true condition without
holding fnhe_lock.

The following interleaving can therefore occur:

  CPU 0                              CPU 1
  fib_nhc_update_mtu()               update_or_create_fnhe()
    load fnhe                          spin_lock_bh(&fnhe_lock)
                                       fnhe_remove_oldest()
                                         unlink fnhe
                                         kfree_rcu(fnhe, rcu)
    <quiescent state>
    access fnhe after grace period

KASAN reported:

  BUG: KASAN: slab-use-after-free in fib_nhc_update_mtu+0x3df/0x410
  Read of size 8 at addr ffff888107d49000 by task poc/90
  Call Trace:
   fib_nhc_update_mtu+0x3df/0x410
   fib_sync_mtu+0x7a/0xd0
   fib_netdev_event+0x229/0x3f0
   netif_set_mtu_ext+0x33a/0x570
   dev_set_mtu+0x88/0x120

The same walk updates fnhe_pmtu and fnhe_mtu_locked. These fields form a
pair and other writers serialize them with fnhe_lock. RCU alone prevents
reclamation, but would still allow concurrent writers to leave a mixed
pair.

Walk the table under RCU and acquire fnhe_lock only while updating each
exception. RCU keeps the current entry alive while the short critical
section serializes its paired PMTU fields. This avoids holding the global
lock while scanning all 2048 buckets for every nexthop.

Fixes: af7d6cce5369 ("net: ipv4: update fnhe_pmtu when first hop's MTU changes")
Cc: stable@vger.kernel.org
Suggested-by: Ido Schimmel <idosch@nvidia.com>
Signed-off-by: Chengfeng Ye <nicoyip.dev@gmail.com>
Reviewed-by: Ido Schimmel <idosch@nvidia.com>
Link: https://patch.msgid.link/20260807181710.1178747-1-nicoyip.dev@gmail.com
Signed-off-by: Paolo Abeni <pabeni@redhat.com>
include/net/route.h
net/ipv4/fib_semantics.c
net/ipv4/route.c

index f90106f383c56d5cdaa8d455c840dbcf9a5e9555..45290177a33c4aef57630d994a9a52af138ca1b1 100644 (file)
@@ -276,6 +276,8 @@ int fib_dump_info_fnhe(struct sk_buff *skb, struct netlink_callback *cb,
                       u32 table_id, struct fib_info *fi,
                       int *fa_index, int fa_start, unsigned int flags);
 
+void fnhe_update_pmtu(struct fib_nh_exception *fnhe, u32 new, u32 orig);
+
 static inline void ip_rt_put(struct rtable *rt)
 {
        /* dst_release() accepts a NULL parameter.
index 78f84ae3ee12099aeea78aac18c6bd9030908af2..0483519b7fb0dad4c80b2bb1cd8bcddc320db1df 100644 (file)
@@ -1895,42 +1895,30 @@ static int call_fib_nh_notifiers(struct fib_nh *nh,
        return NOTIFY_DONE;
 }
 
-/* Update the PMTU of exceptions when:
- * - the new MTU of the first hop becomes smaller than the PMTU
- * - the old MTU was the same as the PMTU, and it limited discovery of
- *   larger MTUs on the path. With that limit raised, we can now
- *   discover larger MTUs
- * A special case is locked exceptions, for which the PMTU is smaller
- * than the minimal accepted PMTU:
- * - if the new MTU is greater than the PMTU, don't make any change
- * - otherwise, unlock and set PMTU
+/* Walk the exceptions of a nexthop after its first hop MTU changed. The
+ * chain is RCU protected here, while fnhe_update_pmtu() takes fnhe_lock
+ * for the update of each entry.
  */
 void fib_nhc_update_mtu(struct fib_nh_common *nhc, u32 new, u32 orig)
 {
        struct fnhe_hash_bucket *bucket;
        int i;
 
-       bucket = rcu_dereference_protected(nhc->nhc_exceptions, 1);
+       rcu_read_lock();
+       bucket = rcu_dereference(nhc->nhc_exceptions);
        if (!bucket)
-               return;
+               goto out;
 
        for (i = 0; i < FNHE_HASH_SIZE; i++) {
                struct fib_nh_exception *fnhe;
 
-               for (fnhe = rcu_dereference_protected(bucket[i].chain, 1);
+               for (fnhe = rcu_dereference(bucket[i].chain);
                     fnhe;
-                    fnhe = rcu_dereference_protected(fnhe->fnhe_next, 1)) {
-                       if (fnhe->fnhe_mtu_locked) {
-                               if (new <= fnhe->fnhe_pmtu) {
-                                       fnhe->fnhe_pmtu = new;
-                                       fnhe->fnhe_mtu_locked = false;
-                               }
-                       } else if (new < fnhe->fnhe_pmtu ||
-                                  orig == fnhe->fnhe_pmtu) {
-                               fnhe->fnhe_pmtu = new;
-                       }
-               }
+                    fnhe = rcu_dereference(fnhe->fnhe_next))
+                       fnhe_update_pmtu(fnhe, new, orig);
        }
+out:
+       rcu_read_unlock();
 }
 
 void fib_sync_mtu(struct net_device *dev, u32 orig_mtu)
index 152d8cb28f65aacee378521e16885c9cf1a4870e..b82401a6baed71f6601cf2e9b6593c65aa8f0f15 100644 (file)
@@ -741,6 +741,35 @@ out_unlock:
        spin_unlock_bh(&fnhe_lock);
 }
 
+/* Update the PMTU of an exception when:
+ * - the new MTU of the first hop becomes smaller than the PMTU
+ * - the old MTU was the same as the PMTU, and it limited discovery of
+ *   larger MTUs on the path. With that limit raised, we can now
+ *   discover larger MTUs
+ * A special case is locked exceptions, for which the PMTU is smaller
+ * than the minimal accepted PMTU:
+ * - if the new MTU is greater than the PMTU, don't make any change
+ * - otherwise, unlock and set PMTU
+ *
+ * fnhe_lock keeps fnhe_pmtu and fnhe_mtu_locked consistent against
+ * update_or_create_fnhe(), which sets both under the same lock.
+ */
+void fnhe_update_pmtu(struct fib_nh_exception *fnhe, u32 new, u32 orig)
+{
+       spin_lock_bh(&fnhe_lock);
+
+       if (fnhe->fnhe_mtu_locked) {
+               if (new <= fnhe->fnhe_pmtu) {
+                       fnhe->fnhe_pmtu = new;
+                       fnhe->fnhe_mtu_locked = false;
+               }
+       } else if (new < fnhe->fnhe_pmtu || orig == fnhe->fnhe_pmtu) {
+               fnhe->fnhe_pmtu = new;
+       }
+
+       spin_unlock_bh(&fnhe_lock);
+}
+
 static void __ip_do_redirect(struct rtable *rt, struct sk_buff *skb, struct flowi4 *fl4,
                             bool kill_route)
 {