[PATCH nf-next v6 1/2] netfilter: flowtable: tear down direct xmit flows when the fdb entry moves
From: Julius Bairaktaris
Date: Sun Oct 04 2026 - 13:17:42 EST
A direct xmit flow stores the bridge port the fdb resolved when the flow
was created. When the host moves to another port of the bridge, the flow
keeps sending to the old port, and packets from the other side keep it
from timing out. A hardware offloaded flow behaves the same.
Look the forward path up again in the gc and tear the flow down when the
bridge now resolves the destination to another port. An aged out fdb
entry leaves the flow alone, because offloaded packets do not refresh
it. The tuple stores the route's output device, where the lookup starts,
and the bridge port, which differs from out.ifidx when the port is a
vlan device.
The check adds about 1.4 us per bridged flow and gc run (1800 flows,
x86 guest with lockdep, 3 runs: 2.2 -> 4.7 ms).
Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Julius Bairaktaris <julius@xxxxxxxxxxxxxx>
---
include/net/netfilter/nf_flow_table.h | 4 +++
net/netfilter/nf_flow_table_core.c | 44 +++++++++++++++++++++++++++
net/netfilter/nf_flow_table_path.c | 7 +++++
3 files changed, 55 insertions(+)
diff --git a/include/net/netfilter/nf_flow_table.h b/include/net/netfilter/nf_flow_table.h
index f2e2771f188f..ba0739247d86 100644
--- a/include/net/netfilter/nf_flow_table.h
+++ b/include/net/netfilter/nf_flow_table.h
@@ -164,6 +164,8 @@ struct flow_offload_tuple {
};
struct {
u32 ifidx;
+ u32 path_ifidx;
+ u32 bridge_ifidx;
u8 h_source[ETH_ALEN];
u8 h_dest[ETH_ALEN];
} out;
@@ -232,6 +234,8 @@ struct nf_flow_route {
struct {
u32 ifindex;
u32 hw_ifindex;
+ u32 path_ifindex;
+ u32 bridge_ifindex;
u8 h_source[ETH_ALEN];
u8 h_dest[ETH_ALEN];
u8 needs_gso_segment:1;
diff --git a/net/netfilter/nf_flow_table_core.c b/net/netfilter/nf_flow_table_core.c
index 03241d4bfd5e..0ed379addd41 100644
--- a/net/netfilter/nf_flow_table_core.c
+++ b/net/netfilter/nf_flow_table_core.c
@@ -4,6 +4,7 @@
#include <linux/module.h>
#include <linux/netfilter.h>
#include <linux/rhashtable.h>
+#include <linux/etherdevice.h>
#include <linux/netdevice.h>
#include <net/ip.h>
#include <net/ip6_route.h>
@@ -139,6 +140,9 @@ static int flow_offload_fill_route(struct flow_offload *flow,
memcpy(flow_tuple->out.h_source, route->tuple[dir].out.h_source,
ETH_ALEN);
flow_tuple->out.ifidx = route->tuple[dir].out.ifindex;
+ flow_tuple->out.path_ifidx = route->tuple[dir].out.path_ifindex;
+ flow_tuple->out.bridge_ifidx =
+ route->tuple[dir].out.bridge_ifindex;
break;
case FLOW_OFFLOAD_XMIT_XFRM:
case FLOW_OFFLOAD_XMIT_NEIGH:
@@ -565,15 +569,55 @@ static void nf_flow_table_extend_ct_timeout(struct nf_conn *ct)
nf_ct_put(ct);
}
+/* A direct xmit flow through a bridge is stale once the bridge resolves
+ * its destination to another port. An fdb entry that aged out is not a
+ * move: offloaded packets do not pass the bridge to refresh it.
+ */
+static bool nf_flow_bridge_port_stale(struct net *net,
+ const struct flow_offload_tuple *tuple)
+{
+ struct net_device_path_ctx ctx = {
+ .ether_type = tuple->l3proto == NFPROTO_IPV4 ?
+ htons(ETH_P_IP) : htons(ETH_P_IPV6),
+ };
+ struct net_device_path_stack stack;
+ bool stale = false;
+ int i, port = 0;
+
+ if (tuple->xmit_type != FLOW_OFFLOAD_XMIT_DIRECT ||
+ !tuple->out.bridge_ifidx)
+ return false;
+
+ ether_addr_copy(ctx.daddr, tuple->out.h_dest);
+
+ rcu_read_lock();
+ ctx.dev = dev_get_by_index_rcu(net, tuple->out.path_ifidx);
+ if (ctx.dev && !dev_fill_forward_path(&ctx, &stack)) {
+ /* The last bridge, as in nft_dev_path_info() */
+ for (i = 0; i < stack.num_paths - 1; i++) {
+ if (stack.path[i].type == DEV_PATH_BRIDGE)
+ port = stack.path[i + 1].dev->ifindex;
+ }
+ stale = port && port != tuple->out.bridge_ifidx;
+ dev_fill_forward_path_release(&stack);
+ }
+ rcu_read_unlock();
+
+ return stale;
+}
+
static void nf_flow_offload_gc_step(struct nf_flowtable *flow_table,
struct flow_offload *flow, void *data)
{
+ struct net *net = read_pnet(&flow_table->net);
bool teardown = test_bit(NF_FLOW_TEARDOWN, &flow->flags);
if (nf_flow_has_expired(flow) ||
nf_ct_is_dying(flow->ct) ||
!nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
!nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
+ nf_flow_bridge_port_stale(net, &flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
+ nf_flow_bridge_port_stale(net, &flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
nf_flow_custom_gc(flow_table, flow)) {
flow_offload_teardown(flow);
teardown = true;
diff --git a/net/netfilter/nf_flow_table_path.c b/net/netfilter/nf_flow_table_path.c
index 1e55644f2edb..386d41e6c713 100644
--- a/net/netfilter/nf_flow_table_path.c
+++ b/net/netfilter/nf_flow_table_path.c
@@ -88,6 +88,7 @@ struct nft_forward_info {
__be16 proto;
} encap[NF_FLOW_TABLE_ENCAP_MAX];
u8 num_encaps;
+ u32 bridge_ifidx;
struct flow_offload_tunnel tun;
struct dst_entry *tun_dst;
u8 num_tuns;
@@ -180,6 +181,10 @@ static int nft_dev_path_info(struct net_device_path_stack *stack,
case DEV_PATH_BR_VLAN_KEEP:
break;
}
+ /* dev_fill_forward_path() adds the bridge port after
+ * the bridge.
+ */
+ info->bridge_ifidx = stack->path[i + 1].dev->ifindex;
info->xmit_type = FLOW_OFFLOAD_XMIT_DIRECT;
break;
default:
@@ -257,6 +262,8 @@ static int nft_dev_forward_path(const struct nft_pktinfo *pkt,
if (info.xmit_type == FLOW_OFFLOAD_XMIT_DIRECT) {
memcpy(route->tuple[dir].out.h_source, info.h_source, ETH_ALEN);
memcpy(route->tuple[dir].out.h_dest, info.h_dest, ETH_ALEN);
+ route->tuple[dir].out.path_ifindex = dst->dev->ifindex;
+ route->tuple[dir].out.bridge_ifindex = info.bridge_ifidx;
route->tuple[dir].xmit_type = info.xmit_type;
}
route->tuple[dir].out.needs_gso_segment = info.needs_gso_segment;
--
2.53.0