[PATCH net-next 5/5] selftests: mptcp: convert iptables to nftables for mptcp_join.sh
From: Matthieu Baerts (NGI0)
Date: Sat Sep 26 2026 - 11:33:03 EST
From: Hangbin Liu <liuhangbin@xxxxxxxxxx>
During conversion, we retain the same table and chain names used by the
original iptables/ip6tables setup, so rule output is identical to the
former iptables/ip6tables output. Add new function init_nftables so we
only do nft table setup when test need set nft rules.
Unlike iptables, nftables cannot match rules based on their full
specification. Since endpoint_tests is the only test that delete rules,
which inserts and deletes rules one by one. Flushing the table
directly is safe and easier than using nft_handle.
The cBPF bytecode matching MPTCP add‑addr and remove‑addr suboptions
is replaced with native nft matching using "tcp option mptcp subtype".
The config file adds CONFIG_NFT_NUMGEN (replaces iptables statistic nth),
CONFIG_NFT_REJECT and CONFIG_NFT_REJECT_INET for reject‑related rules.
Remove CONFIG_NFT_COMPAT since we don't need it now.
Remove the iptables/ip6tables check in mptcp_lib.sh since no script use
it now.
Signed-off-by: Hangbin Liu <liuhangbin@xxxxxxxxxx>
Reviewed-by: Matthieu Baerts (NGI0) <matttbe@xxxxxxxxxx>
Signed-off-by: Matthieu Baerts (NGI0) <matttbe@xxxxxxxxxx>
---
To: Shuah Khan <shuah@xxxxxxxxxx>
Cc: linux-kselftest@xxxxxxxxxxxxxxx
Cc: bpf@xxxxxxxxxxxxxxx ## Note: no eBPF code has been modified
---
tools/testing/selftests/net/mptcp/config | 4 +-
tools/testing/selftests/net/mptcp/mptcp_join.sh | 149 +++++++++---------------
tools/testing/selftests/net/mptcp/mptcp_lib.sh | 2 +-
3 files changed, 62 insertions(+), 93 deletions(-)
diff --git a/tools/testing/selftests/net/mptcp/config b/tools/testing/selftests/net/mptcp/config
index bb9c4c97c620..61057a066719 100644
--- a/tools/testing/selftests/net/mptcp/config
+++ b/tools/testing/selftests/net/mptcp/config
@@ -29,7 +29,9 @@ CONFIG_NET_SCH_INGRESS=m
CONFIG_NET_SCH_NETEM=m
CONFIG_NF_TABLES=m
CONFIG_NF_TABLES_INET=y
-CONFIG_NFT_COMPAT=m
+CONFIG_NFT_NUMGEN=m
+CONFIG_NFT_REJECT=m
+CONFIG_NFT_REJECT_INET=m
CONFIG_NFT_SOCKET=m
CONFIG_NFT_TPROXY=m
CONFIG_SYN_COOKIES=y
diff --git a/tools/testing/selftests/net/mptcp/mptcp_join.sh b/tools/testing/selftests/net/mptcp/mptcp_join.sh
index 18ce7136a2b0..b16e24418e73 100755
--- a/tools/testing/selftests/net/mptcp/mptcp_join.sh
+++ b/tools/testing/selftests/net/mptcp/mptcp_join.sh
@@ -26,8 +26,6 @@ capout=""
cappid=""
ns1=""
ns2=""
-iptables="iptables"
-ip6tables="ip6tables"
timeout_poll=30
timeout_test=$((timeout_poll * 2 + 1))
capture=false
@@ -99,42 +97,6 @@ unset add_addr_tx_nr
unset add_addr_echo_tx_nr
unset add_addr_drop_tx_nr
-# generated using "nfbpf_compile '(ip && (ip[54] & 0xf0) == 0x30) ||
-# (ip6 && (ip6[74] & 0xf0) == 0x30)'"
-CBPF_MPTCP_SUBOPTION_ADD_ADDR="14,
- 48 0 0 0,
- 84 0 0 240,
- 21 0 3 64,
- 48 0 0 54,
- 84 0 0 240,
- 21 6 7 48,
- 48 0 0 0,
- 84 0 0 240,
- 21 0 4 96,
- 48 0 0 74,
- 84 0 0 240,
- 21 0 1 48,
- 6 0 0 65535,
- 6 0 0 0"
-
-# IPv4: TCP hdr of 48B, a first suboption of 12B (DACK8), the RM_ADDR suboption
-# generated using "nfbpf_compile '(ip[32] & 0xf0) == 0xc0 && ip[53] == 0x0c &&
-# (ip[66] & 0xf0) == 0x40'"
-CBPF_MPTCP_SUBOPTION_RM_ADDR="13,
- 48 0 0 0,
- 84 0 0 240,
- 21 0 9 64,
- 48 0 0 32,
- 84 0 0 240,
- 21 0 6 192,
- 48 0 0 53,
- 21 0 4 12,
- 48 0 0 66,
- 84 0 0 240,
- 21 0 1 64,
- 6 0 0 65535,
- 6 0 0 0"
-
init_partial()
{
capout=$(mktemp)
@@ -184,6 +146,26 @@ init_shapers()
done
}
+init_nftables()
+{
+ local netns table
+ for netns in "$ns1" "$ns2"; do
+ for table in ip ip6; do
+ ip netns exec "$netns" nft -f - <<-EOF
+ add table $table filter
+ add chain $table filter INPUT \
+ { type filter hook input priority filter; policy accept; }
+ add chain $table filter OUTPUT \
+ { type filter hook output priority filter; policy accept; }
+
+ add table $table mangle
+ add chain $table mangle OUTPUT \
+ { type route hook output priority mangle; policy accept; }
+ EOF
+ done
+ done
+}
+
cleanup_partial()
{
rm -f "$capout"
@@ -196,7 +178,7 @@ init() {
mptcp_lib_check_mptcp
mptcp_lib_check_kallsyms
- mptcp_lib_check_tools ip tc ss "${iptables}" "${ip6tables}"
+ mptcp_lib_check_tools ip tc ss nft
sin=$(mktemp)
sout=$(mktemp)
@@ -380,24 +362,18 @@ reset_with_cookies()
# $1: test name
reset_with_add_addr_timeout()
{
- local ip="${2:-4}"
- local tables
+ local ip="${2:-}"
reset "${1}" || return 1
-
- tables="${iptables}"
- if [ $ip -eq 6 ]; then
- tables="${ip6tables}"
- fi
+ init_nftables
# set a maximum, to avoid too long timeout with exponential backoff
ip netns exec $ns1 sysctl -q net.mptcp.add_addr_timeout=1
- if ! ip netns exec $ns2 $tables -A OUTPUT -p tcp \
- -m tcp --tcp-option 30 \
- -m bpf --bytecode \
- "$CBPF_MPTCP_SUBOPTION_ADD_ADDR" \
- -j DROP; then
+ if ! ip netns exec "$ns2" nft add rule \
+ ip"$ip" filter OUTPUT meta l4proto tcp \
+ tcp option mptcp subtype add-addr \
+ drop; then
mark_as_skipped "unable to set the 'add addr' rule"
return 1
fi
@@ -449,22 +425,14 @@ setup_fail_rules()
check_invert=1
validate_checksum=true
local i="$1"
- local ip="${2:-4}"
- local tables
+ local ip="${2:-}"
- tables="${iptables}"
- if [ $ip -eq 6 ]; then
- tables="${ip6tables}"
- fi
-
- ip netns exec $ns2 $tables \
- -t mangle \
- -A OUTPUT \
- -o ns2eth$i \
- -p tcp \
- -m length --length 150:9999 \
- -m statistic --mode nth --packet 1 --every 99999 \
- -j MARK --set-mark 42 || return ${KSFT_SKIP}
+ init_nftables
+ ip netns exec "$ns2" nft add rule \
+ ip"$ip" mangle OUTPUT oifname ns2eth$i \
+ meta l4proto tcp \
+ meta length 150-9999 numgen inc mod 99999 1 \
+ meta mark set 42 || return ${KSFT_SKIP}
tc -n $ns2 qdisc add dev ns2eth$i clsact || return ${KSFT_SKIP}
tc -n $ns2 filter add dev ns2eth$i egress \
@@ -510,16 +478,15 @@ reset_with_tcp_filter()
reset "${1}" || return 1
shift
+ init_nftables
+
local ns="${!1}"
local src="${2}"
local target="${3}"
local chain="${4:-INPUT}"
- if ! ip netns exec "${ns}" ${iptables} \
- -A "${chain}" \
- -s "${src}" \
- -p tcp \
- -j "${target}"; then
+ if ! ip netns exec "$ns" nft add rule ip filter "${chain}" \
+ ip saddr "${src}" meta l4proto tcp "${target,,}"; then
mark_as_skipped "unable to set the filter rules"
return 1
fi
@@ -4313,12 +4280,15 @@ userspace_tests()
chk_mptcp_info subflows 1 subflows 1
chk_subflows_total 2 2
+ init_nftables
# force quick loss
ip netns exec $ns2 sysctl -q net.ipv4.tcp_syn_retries=1
- if ip netns exec "${ns1}" ${iptables} -A INPUT -s "10.0.1.2" \
- -p tcp --tcp-option 30 -j REJECT --reject-with tcp-reset &&
- ip netns exec "${ns2}" ${iptables} -A INPUT -d "10.0.1.2" \
- -p tcp --tcp-option 30 -j REJECT --reject-with tcp-reset; then
+ if ip netns exec "${ns1}" nft add rule ip filter INPUT \
+ ip saddr "10.0.1.2" meta l4proto tcp \
+ tcp option mptcp exists reject with tcp reset &&
+ ip netns exec "${ns2}" nft add rule ip filter INPUT \
+ ip daddr "10.0.1.2" meta l4proto tcp \
+ tcp option mptcp exists reject with tcp reset; then
wait_event ns2 MPTCP_LIB_EVENT_SUB_CLOSED 1
wait_event ns1 MPTCP_LIB_EVENT_SUB_CLOSED 1
chk_subflows_total 1 1
@@ -4393,7 +4363,7 @@ endpoint_tests()
chk_subflow_nr "after new reject" 2
chk_mptcp_info subflows 1 subflows 1
- ip netns exec "${ns2}" ${iptables} -D OUTPUT -s "10.0.3.2" -p tcp -j REJECT
+ ip netns exec "${ns2}" nft flush chain ip filter OUTPUT
pm_nl_del_endpoint $ns2 3 10.0.3.2
pm_nl_add_endpoint $ns2 10.0.3.2 id 3 flags subflow
wait_mpj 3
@@ -4402,12 +4372,10 @@ endpoint_tests()
# To make sure RM_ADDR are sent over a different subflow, but
# allow the rest to quickly and cleanly close the subflow
- local ipt=1
- ip netns exec "${ns2}" ${iptables} -I OUTPUT -s "10.0.1.2" \
- -p tcp -m tcp --tcp-option 30 \
- -m bpf --bytecode \
- "$CBPF_MPTCP_SUBOPTION_RM_ADDR" \
- -j DROP || ipt=0
+ local nft=1
+ ip netns exec "${ns2}" nft insert rule ip filter OUTPUT \
+ ip saddr 10.0.1.2 meta l4proto tcp \
+ tcp option mptcp subtype remove-addr drop || nft=0
local i
for i in $(seq 3); do
pm_nl_del_endpoint $ns2 1 10.0.1.2
@@ -4420,7 +4388,7 @@ endpoint_tests()
chk_subflow_nr "after re-add id 0 ($i)" 3
chk_mptcp_info subflows 3 subflows 3
done
- [ ${ipt} = 1 ] && ip netns exec "${ns2}" ${iptables} -D OUTPUT 1
+ [ "${nft}" = 1 ] && ip netns exec "${ns2}" nft flush chain ip filter OUTPUT
mptcp_lib_kill_group_wait $tests_pid
@@ -4480,20 +4448,19 @@ endpoint_tests()
chk_mptcp_info subflows 2 subflows 2
chk_mptcp_info add_addr_signal 2 add_addr_accepted 2
+ init_nftables
# To make sure RM_ADDR are sent over a different subflow, but
# allow the rest to quickly and cleanly close the subflow
- local ipt=1
- ip netns exec "${ns1}" ${iptables} -I OUTPUT -s "10.0.1.1" \
- -p tcp -m tcp --tcp-option 30 \
- -m bpf --bytecode \
- "$CBPF_MPTCP_SUBOPTION_RM_ADDR" \
- -j DROP || ipt=0
+ local nft=1
+ ip netns exec "${ns1}" nft insert rule ip filter OUTPUT \
+ ip saddr 10.0.1.1 meta l4proto tcp \
+ tcp option mptcp subtype remove-addr drop || nft=0
pm_nl_del_endpoint $ns1 42 10.0.1.1
sleep 0.5
chk_subflow_nr "after delete ID 0" 2
chk_mptcp_info subflows 2 subflows 2
chk_mptcp_info add_addr_signal 2 add_addr_accepted 2
- [ ${ipt} = 1 ] && ip netns exec "${ns1}" ${iptables} -D OUTPUT 1
+ [ "${nft}" = 1 ] && ip netns exec "${ns1}" nft flush chain ip filter OUTPUT
pm_nl_add_endpoint $ns1 10.0.1.1 id 42 flags signal
wait_mpj 4
@@ -4555,7 +4522,7 @@ endpoint_tests()
pm_nl_flush_endpoint $ns2
pm_nl_flush_endpoint $ns1
wait_rm_addr $ns2 0
- ip netns exec "${ns2}" ${iptables} -D OUTPUT -s "10.0.3.2" -p tcp -j REJECT
+ ip netns exec "${ns2}" nft flush chain ip filter OUTPUT
pm_nl_add_endpoint $ns2 10.0.3.2 id 3 flags subflow
wait_mpj 1
pm_nl_add_endpoint $ns1 10.0.3.1 id 2 flags signal
diff --git a/tools/testing/selftests/net/mptcp/mptcp_lib.sh b/tools/testing/selftests/net/mptcp/mptcp_lib.sh
index e65b4ebee06a..0559bb168203 100644
--- a/tools/testing/selftests/net/mptcp/mptcp_lib.sh
+++ b/tools/testing/selftests/net/mptcp/mptcp_lib.sh
@@ -528,7 +528,7 @@ mptcp_lib_check_tools() {
exit ${KSFT_SKIP}
fi
;;
- "iptables"* | "ip6tables"* | "nft" | "jq")
+ "nft" | "jq")
if ! "${tool}" -V &> /dev/null; then
mptcp_lib_pr_skip "Could not run all tests without ${tool}"
exit ${KSFT_SKIP}
--
2.55.0