* [PATCH nf-next v6 1/2] netfilter: flowtable: tear down direct xmit flows when the fdb entry moves
2026-10-04 17:16 [PATCH nf-next v6 0/2] netfilter: flowtable: tear down bridged flows on layer 2 roaming Julius Bairaktaris
@ 2026-10-04 17:16 ` Julius Bairaktaris
2026-10-04 17:16 ` [PATCH nf-next v6 2/2] selftests: netfilter: nft_flowtable.sh: roam a host between two bridge ports Julius Bairaktaris
1 sibling, 0 replies; 3+ messages in thread
From: Julius Bairaktaris @ 2026-10-04 17:16 UTC (permalink / raw)
To: netfilter-devel
Cc: pablo, fw, kadlec, phil, horms, davem, edumazet, kuba, pabeni,
shuah, razor, ericwouds, dqfext, netdev, coreteam,
linux-kselftest, linux-kernel
A direct xmit flow stores the bridge port the fdb resolved when the flow
was created. When the host moves to another port of the bridge, the flow
keeps sending to the old port, and packets from the other side keep it
from timing out. A hardware offloaded flow behaves the same.
Look the forward path up again in the gc and tear the flow down when the
bridge now resolves the destination to another port. An aged out fdb
entry leaves the flow alone, because offloaded packets do not refresh
it. The tuple stores the route's output device, where the lookup starts,
and the bridge port, which differs from out.ifidx when the port is a
vlan device.
The check adds about 1.4 us per bridged flow and gc run (1800 flows,
x86 guest with lockdep, 3 runs: 2.2 -> 4.7 ms).
Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Julius Bairaktaris <julius@bairaktaris.de>
---
include/net/netfilter/nf_flow_table.h | 4 +++
net/netfilter/nf_flow_table_core.c | 44 +++++++++++++++++++++++++++
net/netfilter/nf_flow_table_path.c | 7 +++++
3 files changed, 55 insertions(+)
diff --git a/include/net/netfilter/nf_flow_table.h b/include/net/netfilter/nf_flow_table.h
index f2e2771f188f..ba0739247d86 100644
--- a/include/net/netfilter/nf_flow_table.h
+++ b/include/net/netfilter/nf_flow_table.h
@@ -164,6 +164,8 @@ struct flow_offload_tuple {
};
struct {
u32 ifidx;
+ u32 path_ifidx;
+ u32 bridge_ifidx;
u8 h_source[ETH_ALEN];
u8 h_dest[ETH_ALEN];
} out;
@@ -232,6 +234,8 @@ struct nf_flow_route {
struct {
u32 ifindex;
u32 hw_ifindex;
+ u32 path_ifindex;
+ u32 bridge_ifindex;
u8 h_source[ETH_ALEN];
u8 h_dest[ETH_ALEN];
u8 needs_gso_segment:1;
diff --git a/net/netfilter/nf_flow_table_core.c b/net/netfilter/nf_flow_table_core.c
index 03241d4bfd5e..0ed379addd41 100644
--- a/net/netfilter/nf_flow_table_core.c
+++ b/net/netfilter/nf_flow_table_core.c
@@ -4,6 +4,7 @@
#include <linux/module.h>
#include <linux/netfilter.h>
#include <linux/rhashtable.h>
+#include <linux/etherdevice.h>
#include <linux/netdevice.h>
#include <net/ip.h>
#include <net/ip6_route.h>
@@ -139,6 +140,9 @@ static int flow_offload_fill_route(struct flow_offload *flow,
memcpy(flow_tuple->out.h_source, route->tuple[dir].out.h_source,
ETH_ALEN);
flow_tuple->out.ifidx = route->tuple[dir].out.ifindex;
+ flow_tuple->out.path_ifidx = route->tuple[dir].out.path_ifindex;
+ flow_tuple->out.bridge_ifidx =
+ route->tuple[dir].out.bridge_ifindex;
break;
case FLOW_OFFLOAD_XMIT_XFRM:
case FLOW_OFFLOAD_XMIT_NEIGH:
@@ -565,15 +569,55 @@ static void nf_flow_table_extend_ct_timeout(struct nf_conn *ct)
nf_ct_put(ct);
}
+/* A direct xmit flow through a bridge is stale once the bridge resolves
+ * its destination to another port. An fdb entry that aged out is not a
+ * move: offloaded packets do not pass the bridge to refresh it.
+ */
+static bool nf_flow_bridge_port_stale(struct net *net,
+ const struct flow_offload_tuple *tuple)
+{
+ struct net_device_path_ctx ctx = {
+ .ether_type = tuple->l3proto == NFPROTO_IPV4 ?
+ htons(ETH_P_IP) : htons(ETH_P_IPV6),
+ };
+ struct net_device_path_stack stack;
+ bool stale = false;
+ int i, port = 0;
+
+ if (tuple->xmit_type != FLOW_OFFLOAD_XMIT_DIRECT ||
+ !tuple->out.bridge_ifidx)
+ return false;
+
+ ether_addr_copy(ctx.daddr, tuple->out.h_dest);
+
+ rcu_read_lock();
+ ctx.dev = dev_get_by_index_rcu(net, tuple->out.path_ifidx);
+ if (ctx.dev && !dev_fill_forward_path(&ctx, &stack)) {
+ /* The last bridge, as in nft_dev_path_info() */
+ for (i = 0; i < stack.num_paths - 1; i++) {
+ if (stack.path[i].type == DEV_PATH_BRIDGE)
+ port = stack.path[i + 1].dev->ifindex;
+ }
+ stale = port && port != tuple->out.bridge_ifidx;
+ dev_fill_forward_path_release(&stack);
+ }
+ rcu_read_unlock();
+
+ return stale;
+}
+
static void nf_flow_offload_gc_step(struct nf_flowtable *flow_table,
struct flow_offload *flow, void *data)
{
+ struct net *net = read_pnet(&flow_table->net);
bool teardown = test_bit(NF_FLOW_TEARDOWN, &flow->flags);
if (nf_flow_has_expired(flow) ||
nf_ct_is_dying(flow->ct) ||
!nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
!nf_flow_dst_check(&flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
+ nf_flow_bridge_port_stale(net, &flow->tuplehash[FLOW_OFFLOAD_DIR_ORIGINAL].tuple) ||
+ nf_flow_bridge_port_stale(net, &flow->tuplehash[FLOW_OFFLOAD_DIR_REPLY].tuple) ||
nf_flow_custom_gc(flow_table, flow)) {
flow_offload_teardown(flow);
teardown = true;
diff --git a/net/netfilter/nf_flow_table_path.c b/net/netfilter/nf_flow_table_path.c
index 1e55644f2edb..386d41e6c713 100644
--- a/net/netfilter/nf_flow_table_path.c
+++ b/net/netfilter/nf_flow_table_path.c
@@ -88,6 +88,7 @@ struct nft_forward_info {
__be16 proto;
} encap[NF_FLOW_TABLE_ENCAP_MAX];
u8 num_encaps;
+ u32 bridge_ifidx;
struct flow_offload_tunnel tun;
struct dst_entry *tun_dst;
u8 num_tuns;
@@ -180,6 +181,10 @@ static int nft_dev_path_info(struct net_device_path_stack *stack,
case DEV_PATH_BR_VLAN_KEEP:
break;
}
+ /* dev_fill_forward_path() adds the bridge port after
+ * the bridge.
+ */
+ info->bridge_ifidx = stack->path[i + 1].dev->ifindex;
info->xmit_type = FLOW_OFFLOAD_XMIT_DIRECT;
break;
default:
@@ -257,6 +262,8 @@ static int nft_dev_forward_path(const struct nft_pktinfo *pkt,
if (info.xmit_type == FLOW_OFFLOAD_XMIT_DIRECT) {
memcpy(route->tuple[dir].out.h_source, info.h_source, ETH_ALEN);
memcpy(route->tuple[dir].out.h_dest, info.h_dest, ETH_ALEN);
+ route->tuple[dir].out.path_ifindex = dst->dev->ifindex;
+ route->tuple[dir].out.bridge_ifindex = info.bridge_ifidx;
route->tuple[dir].xmit_type = info.xmit_type;
}
route->tuple[dir].out.needs_gso_segment = info.needs_gso_segment;
--
2.53.0
^ permalink raw reply related [flat|nested] 3+ messages in thread* [PATCH nf-next v6 2/2] selftests: netfilter: nft_flowtable.sh: roam a host between two bridge ports
2026-10-04 17:16 [PATCH nf-next v6 0/2] netfilter: flowtable: tear down bridged flows on layer 2 roaming Julius Bairaktaris
2026-10-04 17:16 ` [PATCH nf-next v6 1/2] netfilter: flowtable: tear down direct xmit flows when the fdb entry moves Julius Bairaktaris
@ 2026-10-04 17:16 ` Julius Bairaktaris
1 sibling, 0 replies; 3+ messages in thread
From: Julius Bairaktaris @ 2026-10-04 17:16 UTC (permalink / raw)
To: netfilter-devel
Cc: pablo, fw, kadlec, phil, horms, davem, edumazet, kuba, pabeni,
shuah, razor, ericwouds, dqfext, netdev, coreteam,
linux-kselftest, linux-kernel
Move a host to another port of a bridge on the router while its flow is
offloaded, and check that the transfer continues and is offloaded again.
The cases are plain and vlan bridge ports, "bridge fdb replace", a
pinned and then deleted fdb entry, an aged out entry (the flow must
stay), and the plain case over IPv6.
Without the gc check the transfer stalls when the host moves.
Assisted-by: Claude:claude-fable-5-1
Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Julius Bairaktaris <julius@bairaktaris.de>
---
.../selftests/net/netfilter/nft_flowtable.sh | 201 ++++++++++++++++++
1 file changed, 201 insertions(+)
diff --git a/tools/testing/selftests/net/netfilter/nft_flowtable.sh b/tools/testing/selftests/net/netfilter/nft_flowtable.sh
index 449c518bd947..51e17ffce114 100755
--- a/tools/testing/selftests/net/netfilter/nft_flowtable.sh
+++ b/tools/testing/selftests/net/netfilter/nft_flowtable.sh
@@ -851,6 +851,207 @@ test_ipip
test_bridge
+# Roam a host between two ports of a bridge on the router while its flow is
+# offloaded, as a station moving between two access points attached to ports
+# of one bridge. The host keeps its address and mac on a bridge of its own
+# and roams by moving that bridge's port, so only the fdb changes. With a vid
+# the bridge ports are vlan devices, whose ifindex differs from the device
+# the flow transmits on. With "replace" the router's fdb entry is sticky and
+# the move is a "bridge fdb replace". With "del" the host stays, its entry is
+# pinned to the other port and then deleted, and a ping relearns it. With
+# "age" the host stays and its entry ages out, which must not tear the flow
+# down. A flow that stays on the fast path leaves no more than a few packets
+# in the forward chain, the counter tells offloaded from slow path. The
+# plain roam also runs over IPv6.
+test_bridge_roam()
+{
+ local vid=$1 how=$2 fam=${3:-4}
+ local p0=sr0 p1=sr1 r0=rr0 r1=rr1
+ local mac=02:00:00:09:01:99
+ local what="roaming on bridge${vid:+ with vlan ports}${how:+ by fdb $how}"
+ local hn=10.9.1 wn=10.9.2 dot=. len=24 nodad= ip6= srv=10.9.2.99
+ local min=10000000
+ local dev old new slow lpid cpid max=100 win=2
+
+ if [ "$fam" = 6 ]; then
+ hn=dead:9:1 wn=dead:9:2 dot=:: len=64 nodad=nodad ip6=-6
+ srv="[dead:9:2::99]"
+ what="$what over IPv6"
+ fi
+
+ setup_ns rns1 rnsr rns2
+
+ if ! ip -net "$rnsr" link add br0 type bridge 2>/dev/null; then
+ echo "SKIP: could not add bridge for $what"
+ [ "$ret" -eq 0 ] && ret=$ksft_skip
+ return
+ fi
+
+ ip link add sr0 netns "$rns1" type veth peer name rr0 netns "$rnsr"
+ ip link add sr1 netns "$rns1" type veth peer name rr1 netns "$rnsr"
+ ip link add rw0 netns "$rnsr" type veth peer name w0 netns "$rns2"
+ for dev in sr0 sr1; do
+ ip -net "$rns1" link set "$dev" up
+ done
+ for dev in rr0 rr1; do
+ ip -net "$rnsr" link set "$dev" up
+ done
+
+ if [ -n "$vid" ]; then
+ for dev in sr0 sr1; do
+ ip -net "$rns1" link add link "$dev" name "$dev.$vid" type vlan id "$vid"
+ ip -net "$rns1" link set "$dev.$vid" up
+ done
+ for dev in rr0 rr1; do
+ ip -net "$rnsr" link add link "$dev" name "$dev.$vid" type vlan id "$vid"
+ ip -net "$rnsr" link set "$dev.$vid" up
+ done
+ p0=sr0.$vid; p1=sr1.$vid; r0=rr0.$vid; r1=rr1.$vid
+ fi
+
+ ip -net "$rnsr" link set "$r0" master br0
+ ip -net "$rnsr" link set "$r1" master br0
+ ip -net "$rnsr" addr add $hn${dot}1/$len dev br0 $nodad
+ ip -net "$rnsr" link set br0 up
+ ip -net "$rnsr" addr add $wn${dot}1/$len dev rw0 $nodad
+ ip -net "$rnsr" link set rw0 up
+ ip netns exec "$rnsr" sysctl -q net.ipv4.ip_forward=1
+ ip netns exec "$rnsr" sysctl -q net.ipv6.conf.all.forwarding=1
+
+ ip -net "$rns1" link add stbr type bridge
+ ip -net "$rns1" link set stbr address $mac
+ ip -net "$rns1" link set "$p0" master stbr
+ ip -net "$rns1" link set stbr up
+ ip -net "$rns1" addr add $hn${dot}99/$len dev stbr $nodad
+ ip $ip6 -net "$rns1" route add default via $hn${dot}1
+
+ ip -net "$rns2" addr add $wn${dot}99/$len dev w0 $nodad
+ ip -net "$rns2" link set w0 up
+ ip $ip6 -net "$rns2" route add default via $wn${dot}1
+
+ # no neighbour probes, so the host only sends when it receives
+ ip -net "$rnsr" neigh replace $hn${dot}99 lladdr $mac nud permanent dev br0
+ ip -net "$rns1" neigh replace $hn${dot}1 nud permanent dev stbr \
+ lladdr "$(ip -net "$rnsr" -br link show br0 | awk '{print $3}')"
+
+ if ! ip netns exec "$rnsr" nft -f - <<EOF
+table inet filter {
+ counter roam_repl { }
+ flowtable f1 { hook ingress priority 0; devices = { rr0, rr1, rw0 }; }
+ chain forward {
+ type filter hook forward priority 0; policy accept;
+ meta oif "rw0" tcp dport 12345 flow add @f1
+ meta iif "rw0" tcp sport 12345 counter name roam_repl flow add @f1
+ }
+}
+EOF
+ then
+ echo "SKIP: could not load ruleset for $what"
+ [ "$ret" -eq 0 ] && ret=$ksft_skip
+ return
+ fi
+
+ # the host discards what it receives, its socket byte counter is read
+ rx_bytes() { ip netns exec "$rns1" ss -tin "dst $srv" 2>/dev/null |
+ grep -o 'bytes_received:[0-9]*' | head -1 | cut -d: -f2; }
+ rx_started() { local n; n=$(rx_bytes); [ "${n:-0}" -gt 0 ]; }
+ # forward chain packets of the reply direction since the last call
+ slow_pkts() { ip netns exec "$rnsr" nft reset counter inet filter roam_repl |
+ grep -o 'packets [0-9]*' | cut -d' ' -f2; }
+ roam_listener_ready() { ss -N "$rns2" -lnt -o "sport = :12345" | grep -q 12345; }
+
+ roam_stop() { kill "$cpid" "$lpid" 2>/dev/null; wait "$cpid" "$lpid" 2>/dev/null; }
+
+ timeout "$SOCAT_TIMEOUT" ip netns exec "$rns2" socat -u \
+ OPEN:/dev/zero TCP$fam-LISTEN:12345,reuseaddr &
+ lpid=$!
+ if ! busywait "$BUSYWAIT_TIMEOUT" roam_listener_ready; then
+ echo "FAIL: $what: listener did not start" 1>&2
+ ret=1
+ kill "$lpid" 2>/dev/null; wait "$lpid" 2>/dev/null
+ return
+ fi
+ timeout "$SOCAT_TIMEOUT" ip netns exec "$rns1" socat -u \
+ TCP$fam:$srv:12345 OPEN:/dev/null &
+ cpid=$!
+
+ if ! busywait "$BUSYWAIT_TIMEOUT" rx_started; then
+ echo "FAIL: $what: transfer did not start" 1>&2
+ ret=1
+ roam_stop
+ return
+ fi
+ sleep 2
+ slow_pkts > /dev/null
+ old=$(rx_bytes)
+ sleep 2
+ new=$(rx_bytes)
+ slow=$(slow_pkts)
+ if [ -z "$slow" ] || [ $((${new:-0} - ${old:-0})) -lt $min ] ||
+ [ "$slow" -gt 100 ]; then
+ echo "FAIL: $what: not offloaded before the roam:" \
+ "$((${new:-0} - ${old:-0})) bytes, ${slow:-no} forward chain packets" 1>&2
+ ret=1
+ roam_stop
+ return
+ fi
+
+ if [ "$how" = age ]; then
+ # ageing_time is in centiseconds
+ ip -net "$rnsr" link set br0 type bridge ageing_time 200
+ max=0 win=10
+ elif [ "$how" = del ]; then
+ bridge -n "$rnsr" fdb replace $mac dev "$r1" master static sticky
+ sleep 2
+ bridge -n "$rnsr" fdb del $mac dev "$r1" master static
+ ip netns exec "$rns1" ping $ip6 -c 2 -W 1 -q $hn${dot}1 >/dev/null 2>&1
+ else
+ # roam: move the host bridge to the other port and send on it
+ if [ "$how" = replace ]; then
+ bridge -n "$rnsr" fdb replace $mac dev "$r0" master static sticky
+ fi
+ ip -net "$rns1" link set "$p0" nomaster
+ ip -net "$rns1" link set "$p1" master stbr
+ ip netns exec "$rns1" ping $ip6 -c 2 -W 1 -q $hn${dot}1 >/dev/null 2>&1
+ if [ "$how" = replace ]; then
+ bridge -n "$rnsr" fdb replace $mac dev "$r1" master static sticky
+ fi
+ fi
+
+ # the bytes in flight at the roam drain within the first seconds; a
+ # stale flow delivers nothing after that, a torn down one resumes and
+ # is offloaded again.
+ sleep 4
+ old=$(rx_bytes)
+ sleep 4
+ new=$(rx_bytes)
+ slow_pkts > /dev/null
+ sleep $win
+ slow=$(slow_pkts)
+ roam_stop
+
+ if [ "$how" = age ] &&
+ bridge -n "$rnsr" fdb show br br0 | grep -q "$mac.*master"; then
+ echo "FAIL: $what: the host's fdb entry did not age out" 1>&2
+ ret=1
+ elif [ $((${new:-0} - ${old:-0})) -lt $min ]; then
+ echo "FAIL: $what: transfer stalled after the roam" 1>&2
+ ret=1
+ elif [ -z "$slow" ] || [ "$slow" -gt $max ]; then
+ echo "FAIL: $what: $slow forward chain packets after the roam" 1>&2
+ ret=1
+ else
+ echo "PASS: flow offload for $what"
+ fi
+}
+
+test_bridge_roam ""
+test_bridge_roam 100
+test_bridge_roam "" replace
+test_bridge_roam "" del
+test_bridge_roam "" age
+test_bridge_roam "" "" 6
+
KEY_SHA="0x"$(ps -af | sha1sum | cut -d " " -f 1)
KEY_AES="0x"$(ps -af | md5sum | cut -d " " -f 1)
SPI1=$RANDOM
--
2.53.0
^ permalink raw reply related [flat|nested] 3+ messages in thread