From: Julius Bairaktaris <julius@bairaktaris.de>
To: netfilter-devel@vger.kernel.org
Cc: pablo@netfilter.org, fw@strlen.de, kadlec@netfilter.org,
phil@nwl.cc, horms@kernel.org, davem@davemloft.net,
edumazet@google.com, kuba@kernel.org, pabeni@redhat.com,
shuah@kernel.org, razor@blackwall.org, ericwouds@gmail.com,
dqfext@gmail.com, netdev@vger.kernel.org, coreteam@netfilter.org,
linux-kselftest@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: [PATCH nf-next v6 1/2] netfilter: flowtable: tear down direct xmit flows when the fdb entry moves
Date: Sun, 4 Oct 2026 19:16:04 +0200 [thread overview]
Message-ID: <20261004171605.3544792-2-julius@bairaktaris.de> (raw)
In-Reply-To: <20261004171605.3544792-1-julius@bairaktaris.de>
A direct xmit flow stores the bridge port the fdb resolved when the flow
was created. When the host moves to another port of the bridge, the flow
keeps sending to the old port, and packets from the other side keep it
from timing out. A hardware offloaded flow behaves the same.
Look the forward path up again in the gc and tear the flow down when the
bridge now resolves the destination to another port. An aged out fdb
entry leaves the flow alone, because offloaded packets do not refresh
it. The tuple stores the route's output device, where the lookup starts,
and the bridge port, which differs from out.ifidx when the port is a
vlan device.
The check adds about 1.4 us per bridged flow and gc run (1800 flows,
x86 guest with lockdep, 3 runs: 2.2 -> 4.7 ms).
Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Julius Bairaktaris <julius@bairaktaris.de>
---
include/net/netfilter/nf_flow_table.h | 4 +++
net/netfilter/nf_flow_table_core.c | 44 +++++++++++++++++++++++++++
net/netfilter/nf_flow_table_path.c | 7 +++++
3 files changed, 55 insertions(+)
diff --git a/include/net/netfilter/nf_flow_table.h b/include/net/netfilter/nf_flow_table.h
index f2e2771f188f..ba0739247d86 100644
--- a/include/net/netfilter/nf_flow_table.h
+++ b/include/net/netfilter/nf_flow_table.h
@@ -164,6 +164,8 @@ struct flow_offload_tuple {
};
struct {
u32 ifidx;
+ u32 path_ifidx;
+ u32 bridge_ifidx;
u8 h_source[ETH_ALEN];
u8 h_dest[ETH_ALEN];
} out;
@@ -232,6 +234,8 @@ struct nf_flow_route {
struct {
u32 ifindex;
u32 hw_ifindex;
+ u32 path_ifindex;
+ u32 bridge_ifindex;
u8 h_source[ETH_ALEN];
u8 h_dest[ETH_ALEN];
u8 needs_gso_segment:1;
diff --git a/net/netfilter/nf_flow_table_core.c b/net/netfilter/nf_flow_table_core.c
index 03241d4bfd5e..0ed379addd41 100644
--- a/net/netfilter/nf_flow_table_core.c
+++ b/net/netfilter/nf_flow_table_core.c
@@ -4,6 +4,7 @@
#include <linux/module.h>
#include <linux/netfilter.h>
#include <linux/rhashtable.h>
+#include <linux/etherdevice.h>
#include <linux/netdevice.h>
#include <net/ip.h>
#include <net/ip6_route.h>
@@ -139,6 +140,9 @@ static int flow_offload_fill_route(struct flow_offload *flow,
memcpy(flow_tuple->out.h_source, route->tuple[dir].out.h_source,
ETH_ALEN);
flow_tuple->out.ifidx = route->tuple[dir].out.ifindex;
+ flow_tuple->out.path_ifidx = route->tuple[dir].out.path_ifindex;
+ flow_tuple->out.bridge_ifidx =
+ route->tuple[dir].out.bridge_ifindex;
break;
case FLOW_OFFLOAD_XMIT_XFRM:
case FLOW_OFFLOAD_XMIT_NEIGH:
@@ -565,15 +569,55 @@ static void nf_flow_table_extend_ct_timeout(struct nf_conn *ct)
nf_ct_put(ct);
}
+/* A direct xmit flow through a bridge is stale once the bridge resolves
+ * its destination to another port. An fdb entry that aged out is not a
+ * move: offloaded packets do not pass the bridge to refresh it.
+ */
+static bool nf_flow_bridge_port_stale(struct net *net,
+ const struct flow_offload_tuple *tuple)
+{
+ struct net_device_path_ctx ctx = {
+ .ether_type = tuple->l3proto == NFPROTO_IPV4 ?
+ htons(ETH_P_IP) : htons(ETH_P_IPV6),
+ };
+ struct net_device_path_stack stack;
+ bool stale = false;
+ int i, port = 0;
+
+ if (tuple->xmit_type != FLOW_OFFLOAD_XMIT_DIRECT ||
+ !tuple->out.bridge_ifidx)
+ return false;
+
+ ether_addr_copy(ctx.daddr, tuple->out.h_dest);
+
+ rcu_read_lock();
+ ctx.dev = dev_get_by_index_rcu(net, tuple->out.path_ifidx);
+ if (ctx.dev && !dev_fill_forward_path(&ctx, &stack)) {
+ /* The last bridge, as in nft_dev_path_info() */
+ for (i = 0; i < stack.num_paths - 1; i++) {
+ if (stack.path[i].type == DEV_PATH_BRIDGE)
+ port = stack.path[i + 1].dev->ifindex;
+ }
+ stale = port && port != tuple->out.bridge_ifidx;
+ dev_fill_forward_path_release(&stack);
+ }
+ rcu_read_unlock();
+
+ return stale;
+}
+
static void nf_flow_offload_gc_step(struct nf_flowtable *flow_table,
struct flow_offload *flow, void *data)
{
+ struct net *net = read_pnet(&flow_table->net);
bool teardown = test_bit(NF_FLOW_TEARDOWN, &flow->flags);
if (nf_flow_has_expired(flow) ||
nf_ct_is_dying(flow->ct) ||
!nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
!nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
+ nf_flow_bridge_port_stale(net, &flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
+ nf_flow_bridge_port_stale(net, &flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
nf_flow_custom_gc(flow_table, flow)) {
flow_offload_teardown(flow);
teardown = true;
diff --git a/net/netfilter/nf_flow_table_path.c b/net/netfilter/nf_flow_table_path.c
index 1e55644f2edb..386d41e6c713 100644
--- a/net/netfilter/nf_flow_table_path.c
+++ b/net/netfilter/nf_flow_table_path.c
@@ -88,6 +88,7 @@ struct nft_forward_info {
__be16 proto;
} encap[NF_FLOW_TABLE_ENCAP_MAX];
u8 num_encaps;
+ u32 bridge_ifidx;
struct flow_offload_tunnel tun;
struct dst_entry *tun_dst;
u8 num_tuns;
@@ -180,6 +181,10 @@ static int nft_dev_path_info(struct net_device_path_stack *stack,
case DEV_PATH_BR_VLAN_KEEP:
break;
}
+ /* dev_fill_forward_path() adds the bridge port after
+ * the bridge.
+ */
+ info->bridge_ifidx = stack->path[i + 1].dev->ifindex;
info->xmit_type = FLOW_OFFLOAD_XMIT_DIRECT;
break;
default:
@@ -257,6 +262,8 @@ static int nft_dev_forward_path(const struct nft_pktinfo *pkt,
if (info.xmit_type == FLOW_OFFLOAD_XMIT_DIRECT) {
memcpy(route->tuple[dir].out.h_source, info.h_source, ETH_ALEN);
memcpy(route->tuple[dir].out.h_dest, info.h_dest, ETH_ALEN);
+ route->tuple[dir].out.path_ifindex = dst->dev->ifindex;
+ route->tuple[dir].out.bridge_ifindex = info.bridge_ifidx;
route->tuple[dir].xmit_type = info.xmit_type;
}
route->tuple[dir].out.needs_gso_segment = info.needs_gso_segment;
--
2.53.0
next prev parent reply other threads:[~2026-10-04 17:16 UTC|newest]
Thread overview: 3+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-04 17:16 [PATCH nf-next v6 0/2] netfilter: flowtable: tear down bridged flows on layer 2 roaming Julius Bairaktaris
2026-10-04 17:16 ` Julius Bairaktaris [this message]
2026-10-04 17:16 ` [PATCH nf-next v6 2/2] selftests: netfilter: nft_flowtable.sh: roam a host between two bridge ports Julius Bairaktaris
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261004171605.3544792-2-julius@bairaktaris.de \
--to=julius@bairaktaris.de \
--cc=coreteam@netfilter.org \
--cc=davem@davemloft.net \
--cc=dqfext@gmail.com \
--cc=edumazet@google.com \
--cc=ericwouds@gmail.com \
--cc=fw@strlen.de \
--cc=horms@kernel.org \
--cc=kadlec@netfilter.org \
--cc=kuba@kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-kselftest@vger.kernel.org \
--cc=netdev@vger.kernel.org \
--cc=netfilter-devel@vger.kernel.org \
--cc=pabeni@redhat.com \
--cc=pablo@netfilter.org \
--cc=phil@nwl.cc \
--cc=razor@blackwall.org \
--cc=shuah@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox