DPDK-dev Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Stephen Hemminger <stephen@networkplumber.org>
To: dev@dpdk.org
Cc: Stephen Hemminger <stephen@networkplumber.org>,
	Thomas Monjalon <thomas@monjalon.net>,
	Reshma Pattan <reshma.pattan@intel.com>
Subject: [PATCH v4 2/4] app/rpcapd: remote pcap daemon
Date: Wed, 30 Sep 2026 19:35:27 -0700	[thread overview]
Message-ID: <20261001025853.319860-3-stephen@networkplumber.org> (raw)
In-Reply-To: <20261001025853.319860-1-stephen@networkplumber.org>

Add RPCAP support over localhost TCP integrated with DPDK.
It runs as a secondary process that allows connections from
tools using tcpdump's defacto protocol rpcap.

See: doc/guides/tools/rpcapd.rst for more info

Signed-off-by: Stephen Hemminger <stephen@networkplumber.org>
---
 MAINTAINERS                            |   2 +
 app/meson.build                        |   1 +
 app/rpcapd/capture.c                   | 526 ++++++++++++++++++++++
 app/rpcapd/filter.c                    | 164 +++++++
 app/rpcapd/main.c                      | 577 +++++++++++++++++++++++++
 app/rpcapd/meson.build                 |  25 ++
 app/rpcapd/rpcap-protocol.h            | 142 ++++++
 app/rpcapd/rpcapd.h                    | 105 +++++
 app/rpcapd/session.c                   | 136 ++++++
 app/rpcapd/sock.c                      | 315 ++++++++++++++
 doc/guides/rel_notes/release_26_11.rst |   5 +
 doc/guides/tools/index.rst             |   1 +
 doc/guides/tools/rpcapd.rst            | 199 +++++++++
 13 files changed, 2198 insertions(+)
 create mode 100644 app/rpcapd/capture.c
 create mode 100644 app/rpcapd/filter.c
 create mode 100644 app/rpcapd/main.c
 create mode 100644 app/rpcapd/meson.build
 create mode 100644 app/rpcapd/rpcap-protocol.h
 create mode 100644 app/rpcapd/rpcapd.h
 create mode 100644 app/rpcapd/session.c
 create mode 100644 app/rpcapd/sock.c
 create mode 100644 doc/guides/tools/rpcapd.rst

diff --git a/MAINTAINERS b/MAINTAINERS
index 482bc7df76..fd56b6b440 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -1722,6 +1722,8 @@ F: app/pdump/
 F: doc/guides/tools/pdump.rst
 F: app/dumpcap/
 F: doc/guides/tools/dumpcap.rst
+F: app/rpcapd/
+F: doc/guides/tools/rpcapd.rst
 
 
 Packet Framework
diff --git a/app/meson.build b/app/meson.build
index 4515688471..b9227f1fe4 100644
--- a/app/meson.build
+++ b/app/meson.build
@@ -17,6 +17,7 @@ apps = [
         'graph',
         'pdump',
         'proc-info',
+        'rpcapd',
         'test-acl',
         'test-bbdev',
         'test-cmdline',
diff --git a/app/rpcapd/capture.c b/app/rpcapd/capture.c
new file mode 100644
index 0000000000..17fce3fbca
--- /dev/null
+++ b/app/rpcapd/capture.c
@@ -0,0 +1,526 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * Starting and stopping a capture, and streaming the captured packets
+ * to the client over the data connection.
+ */
+
+#include <errno.h>
+#include <poll.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/socket.h>
+#include <sys/time.h>
+#include <sys/uio.h>
+#include <time.h>
+#include <unistd.h>
+
+#include <rte_byteorder.h>
+#include <rte_common.h>
+#include <rte_cycles.h>
+#include <rte_errno.h>
+#include <rte_ethdev.h>
+#include <rte_ether.h>
+#include <rte_malloc.h>
+#include <rte_mbuf.h>
+#include <rte_mempool.h>
+#include <rte_pcapng.h>
+#include <rte_pdump.h>
+#include <rte_ring.h>
+#include <rte_stdatomic.h>
+#include <rte_time.h>
+
+#include "rpcap-protocol.h"
+#include "rpcapd.h"
+
+#define BURST_SIZE                    32
+#define MBUF_CACHE_SIZE               32
+#define SLEEP_THRESHOLD		      100
+#define SLEEP_US		      100
+#define DATA_ACCEPT_TIMEOUT_MS        10000
+
+/* Reference point for converting a captured TSC to a time of day.
+ * The TSC is the same counter in the primary that did the capture.
+ */
+static uint64_t tsc_base;
+static uint64_t ns_base;
+
+void
+timestamp_init(void)
+{
+	struct timespec ts;
+	uint64_t cycles;
+
+	cycles = rte_get_tsc_cycles();
+	clock_gettime(CLOCK_REALTIME, &ts);
+	ns_base = rte_timespec_to_ns(&ts);
+	tsc_base = (cycles + rte_get_tsc_cycles()) / 2;
+}
+
+/* Convert a captured TSC to nanoseconds since the Unix epoch.  Whole
+ * seconds come out first so scaling the remainder cannot overflow, and
+ * a packet copied before startup is behind the reference point.
+ */
+static uint64_t
+timestamp_to_ns(uint64_t cycles)
+{
+	const uint64_t hz = rte_get_tsc_hz();
+	uint64_t delta, secs, rem;
+	bool before;
+
+	before = cycles < tsc_base;
+	delta = before ? tsc_base - cycles : cycles - tsc_base;
+
+	secs = delta / hz;
+	rem = delta % hz;
+	delta = secs * NSEC_PER_SEC + (rem * NSEC_PER_SEC) / hz;
+
+	return before ? ns_base - delta : ns_base + delta;
+}
+
+
+/* Open an ephemeral TCP listening socket; return fd, set *port_out. */
+static int
+open_data_listener(uint16_t *port_out)
+{
+	struct sockaddr_storage addr = listen_addr;
+	socklen_t alen;
+	int fd;
+
+	set_sockaddr_port(&addr, 0);
+
+	fd = socket(addr.ss_family, SOCK_STREAM, 0);
+	if (fd < 0) {
+		RPCAPD_LOG(ERR, "data socket: %s", strerror(errno));
+		return -1;
+	}
+
+	alen = listen_addrlen;
+	if (bind(fd, (struct sockaddr *)&addr, alen) < 0 ||
+	    listen(fd, 1) < 0 ||
+	    getsockname(fd, (struct sockaddr *)&addr, &alen) < 0) {
+		RPCAPD_LOG(ERR, "data port bind/listen: %s", strerror(errno));
+		close(fd);
+		return -1;
+	}
+	*port_out = get_sockaddr_port(&addr);
+	return fd;
+}
+
+static struct rte_ring *
+create_capture_ring(uint16_t port)
+{
+	char name[RTE_RING_NAMESIZE];
+
+	snprintf(name, sizeof(name), "rpcapd_r_%u_%d", port, getpid());
+	return rte_ring_create(name, ring_size, rte_socket_id(), 0);
+}
+
+static struct rte_mempool *
+create_capture_mempool(uint16_t port, uint32_t snaplen)
+{
+	char name[RTE_MEMPOOL_NAMESIZE];
+	/* Leaves room for the pcapng block header, the options and the
+	 * trailer, as well as the packet itself.
+	 */
+	uint32_t mbuf_size = rte_pcapng_mbuf_size(snaplen);
+
+	snprintf(name, sizeof(name), "rpcapd_p_%u_%d", port, getpid());
+	return rte_pktmbuf_pool_create(name, ring_size * 2, MBUF_CACHE_SIZE, 0,
+				       mbuf_size, rte_socket_id());
+}
+
+
+/* Tear down anything that handle_startcap brought up.
+ * Safe to call after partial setup as well as after a successful capture.
+ */
+void
+stop_capture(struct session *s)
+{
+	struct rte_mbuf *pkts[BURST_SIZE];
+	unsigned int n;
+
+	if (s->capture_on) {
+		rte_pdump_disable(s->port, RTE_PDUMP_ALL_QUEUES, s->pdump_flags);
+		RPCAPD_LOG(NOTICE, "capture stopped on %s (%u packets)",
+			s->name, s->npkt);
+	}
+	s->capture_on = false;
+
+	if (s->promisc_set) {
+		rte_eth_promiscuous_disable(s->port);
+		s->promisc_set = false;
+	}
+
+	if (s->ring != NULL) {
+		while ((n = rte_ring_sc_dequeue_burst(s->ring, (void **)pkts,
+						      BURST_SIZE, NULL)) > 0)
+			rte_pktmbuf_free_bulk(pkts, n);
+		rte_ring_free(s->ring);
+		s->ring = NULL;
+	}
+	if (s->mp != NULL) {
+		rte_mempool_free(s->mp);
+		s->mp = NULL;
+	}
+
+	/* Only safe once pdump is disabled */
+	rte_free(s->prm);
+	s->prm = NULL;
+	if (s->data.fd >= 0) {
+		close(s->data.fd);
+		s->data.fd = -1;
+	}
+}
+
+/*
+ * STARTCAP_REQ: open the data connection and arm the pdump callback.
+ * We use passive mode with the server-allocated data port:
+ *   - the server picks an ephemeral port and listens on it
+ *   - the server returns that port in startcapreply.portdata
+ *   - the client connects back to that port for the packet stream
+ */
+int
+handle_startcap(const struct conn *c, uint32_t plen, struct session *s)
+{
+	struct rpcap_startcapreq req;
+	uint16_t data_port;
+	uint16_t flags;
+	struct rte_bpf_prm *recorded;
+	int data_listen;
+	int data_fd;
+	int ret;
+
+	/* Keep a filter set before the capture started, drop one from a
+	 * capture being restarted: this request brings its own.
+	 */
+	recorded = s->capture_on ? NULL : s->prm;
+	if (recorded != NULL)
+		s->prm = NULL;
+	stop_capture(s);
+	s->prm = recorded;
+
+	if (!s->opened) {
+		rpcap_discard(c, plen);
+		return rpcap_send_error(c, 0, "no interface open");
+	}
+
+	if (plen < sizeof(req)) {
+		rpcap_discard(c, plen);
+		return rpcap_send_error(c, 0, "short startcap request");
+	}
+	if (recv_full(c, &req, sizeof(req)) < 0)
+		return -1;
+
+	flags = rte_be_to_cpu_16(req.flags);
+	if (flags & RPCAP_STARTCAPREQ_FLAG_DGRAM) {
+		rpcap_discard(c, plen - sizeof(req));
+		return rpcap_send_error(c, 0, "UDP data transfer not supported");
+	}
+
+	ret = read_filter(c, plen - sizeof(req), s);
+	if (ret != 0)
+		return ret < 0 ? -1 : 0;	/* error already reported to client */
+
+	/* Direction flags map onto pdump's RX/TX selection; neither (or both)
+	 * means capture in both directions.
+	 */
+	s->pdump_flags = RTE_PDUMP_FLAG_RXTX;
+	if ((flags & (RPCAP_STARTCAPREQ_FLAG_INBOUND |
+		      RPCAP_STARTCAPREQ_FLAG_OUTBOUND)) ==
+	    RPCAP_STARTCAPREQ_FLAG_INBOUND)
+		s->pdump_flags = RTE_PDUMP_FLAG_RX;
+	else if ((flags & (RPCAP_STARTCAPREQ_FLAG_INBOUND |
+			   RPCAP_STARTCAPREQ_FLAG_OUTBOUND)) ==
+		 RPCAP_STARTCAPREQ_FLAG_OUTBOUND)
+		s->pdump_flags = RTE_PDUMP_FLAG_TX;
+
+	s->snaplen = rte_be_to_cpu_32(req.snaplen);
+	if (s->snaplen == 0 || s->snaplen > DEFAULT_SNAPLEN)
+		s->snaplen = DEFAULT_SNAPLEN;
+
+	s->ring = create_capture_ring(s->port);
+	s->mp = create_capture_mempool(s->port, s->snaplen);
+	if (s->ring == NULL || s->mp == NULL) {
+		RPCAPD_LOG(ERR, "ring/mempool alloc failed: %s",
+			rte_strerror(rte_errno));
+		stop_capture(s);
+		return rpcap_send_error(c, 0, "DPDK alloc failed");
+	}
+
+	data_listen = open_data_listener(&data_port);
+	if (data_listen < 0) {
+		stop_capture(s);
+		return rpcap_send_error(c, 0, "data port setup failed");
+	}
+
+	/* Leave the port alone if it is already promiscuous: it belongs to
+	 * the primary process, and stop_capture() must not turn off
+	 * something this daemon did not turn on.
+	 */
+	if ((flags & RPCAP_STARTCAPREQ_FLAG_PROMISC) &&
+	    rte_eth_promiscuous_get(s->port) != 1) {
+		if (rte_eth_promiscuous_enable(s->port) == 0)
+			s->promisc_set = true;
+		else
+			RPCAPD_LOG(NOTICE, "cannot enable promiscuous mode on %s",
+				s->name);
+	}
+
+	/* Setup packet capture callbacks. */
+	if (rte_pdump_enable_bpf(s->port, RTE_PDUMP_ALL_QUEUES,
+				 s->pdump_flags | RTE_PDUMP_FLAG_PCAPNG,
+				 s->snaplen, s->ring, s->mp, s->prm) < 0) {
+		RPCAPD_LOG(ERR, "rte_pdump_enable_bpf port %u failed: %s",
+			s->port, rte_strerror(rte_errno));
+		close(data_listen);
+		stop_capture(s);
+		return rpcap_send_error(c, 0, "cannot enable capture");
+	}
+	s->capture_on = true;
+	s->npkt = 0;
+
+	struct rpcap_startcapreply reply = {
+		.bufsize = rte_cpu_to_be_32(s->snaplen * BURST_SIZE),
+		.portdata = rte_cpu_to_be_16(data_port),
+	};
+	if (rpcap_send_msg(c, RPCAP_MSG_STARTCAP_REPLY, 0, &reply, sizeof(reply)) < 0) {
+		close(data_listen);
+		stop_capture(s);
+		return -1;
+	}
+
+	RPCAPD_LOG(DEBUG, "awaiting connection");
+
+	data_fd = accept_from(data_listen, &s->peer, DATA_ACCEPT_TIMEOUT_MS);
+	close(data_listen);
+	if (data_fd < 0) {
+		stop_capture(s);
+		return -1;
+	}
+
+	/* Bound how long a send can block. */
+	if (send_timeout > 0) {
+		struct timeval tv = {
+			.tv_sec = send_timeout,
+		};
+
+		if (setsockopt(data_fd, SOL_SOCKET, SO_SNDTIMEO, &tv, sizeof(tv)) < 0)
+			RPCAPD_LOG(NOTICE, "cannot set data send timeout: %s",
+				   strerror(errno));
+	}
+
+	s->data.fd = data_fd;
+
+	RPCAPD_LOG(NOTICE,
+		   "capture started on %s (snaplen %u, data port %u)",
+		   s->name, s->snaplen, data_port);
+	return 0;
+}
+
+/*
+ * Frame each packet from the ring into an RPCAP_MSG_PACKET message and
+ * send it on the data connection.  MSG_MORE corks the socket until the
+ * ring drains, so a backlog coalesces into full segments.  pdump wraps
+ * packets in a pcapng enhanced packet block, which carries the capture
+ * time and the pre-truncation length.
+ */
+static ssize_t
+process_ring(struct session *s, unsigned int *avail)
+{
+	struct rte_mbuf *pkts[BURST_SIZE];
+	unsigned int i, n;
+	ssize_t written = 0;
+
+	n = rte_ring_sc_dequeue_burst(s->ring, (void **)pkts, BURST_SIZE, avail);
+	if (n == 0)
+		return 0;
+
+	for (i = 0; i < n; i++) {
+		struct rte_mbuf *m = pkts[i];
+		uint8_t buf[MAX_CAPTURE_LEN];
+		struct rte_pcapng_pkt pkt;
+		uint32_t caplen, wirelen;
+		const void *data;
+
+		if (unlikely(rte_pcapng_pkt_info(m, &pkt) != 0)) {
+			RPCAPD_LOG(ERR, "malformed capture mbuf on %s", s->name);
+			goto error;
+		}
+
+		caplen = pkt.captured_len;
+		if (unlikely(caplen > sizeof(buf)))
+			caplen = sizeof(buf);
+
+		/* clients reject a packet whose len is below its caplen */
+		wirelen = RTE_MAX(pkt.original_len, caplen);
+		data = rte_pktmbuf_read(m, pkt.data_offset, caplen, buf);
+		if (unlikely(data == NULL)) {
+			RPCAPD_LOG(ERR, "short capture mbuf on %s", s->name);
+			goto error;
+		}
+
+		s->npkt++;
+
+		struct rpcap_header hdr = {
+			.ver = RPCAP_VERSION,
+			.type = RPCAP_MSG_PACKET,
+			.plen = rte_cpu_to_be_32(sizeof(struct rpcap_pkthdr) + caplen),
+		};
+
+		/* rpcap protocol has timestamp in microseconds. */
+		uint64_t us = timestamp_to_ns(pkt.cycles) / 1000;
+		struct rpcap_pkthdr pkthdr = {
+			.timestamp_sec = rte_cpu_to_be_32(us / US_PER_S),
+			.timestamp_usec = rte_cpu_to_be_32(us % US_PER_S),
+			.caplen = rte_cpu_to_be_32(caplen),
+			.len = rte_cpu_to_be_32(wirelen),
+			.npkt = rte_cpu_to_be_32(s->npkt),
+		};
+
+		struct iovec iov[3] = {
+			{
+				.iov_base = &hdr,
+				.iov_len = sizeof(hdr),
+			},
+			{
+				.iov_base = &pkthdr,
+				.iov_len = sizeof(pkthdr),
+			},
+			{
+				.iov_base = (void *)(uintptr_t)data,
+				.iov_len = caplen,
+			},
+		};
+
+		/* more to come in this burst, or still queued in the ring */
+		bool more = (i + 1 < n) || (*avail > 0);
+
+		if (send_iov_full(&s->data, iov, 3, more ? MSG_MORE : 0) < 0) {
+			if (errno == EPIPE || errno == ECONNRESET)
+				RPCAPD_LOG(DEBUG, "data connection closed by client");
+			else if (errno == EAGAIN || errno == EWOULDBLOCK)
+				RPCAPD_LOG(NOTICE,
+					   "client stopped reading data connection, closing");
+			else
+				RPCAPD_LOG(NOTICE, "send on data connection failed: %s",
+					   strerror(errno));
+			goto error;
+		}
+		rte_pktmbuf_free(m);
+		written += sizeof(hdr) + sizeof(pkthdr) + caplen;
+	}
+
+	return written;
+
+error:
+	rte_pktmbuf_free_bulk(pkts + i, n - i);
+	return -1;
+}
+
+/* Poll the control socket while idle.
+ * Returns 0 to keep capturing, 1 if a control message (typically
+ * ENDCAP) is pending, or -1 if the client has gone away.
+ */
+static int
+check_socket_status(const struct conn *ctrl)
+{
+	struct pollfd pfd = { .fd = ctrl->fd, .events = POLLIN };
+
+	if (poll(&pfd, 1, 0) < 0) {
+		if (errno == EINTR)
+			return 0;
+		RPCAPD_LOG(ERR, "poll failed: %s", strerror(errno));
+		return -1;
+	}
+	if (pfd.revents & (POLLERR | POLLHUP | POLLNVAL)) {
+		RPCAPD_LOG(DEBUG, "client closed control connection");
+		return -1;
+	}
+	if (pfd.revents & POLLIN)
+		return 1;
+	return 0;
+}
+
+/*
+ * Drain the ring until a control message arrives, the data connection
+ * breaks, or a quit signal is delivered.  Returns 0 if the session
+ * should continue, -1 if the client is gone.
+ *
+ * The control socket is polled every iteration, not only when the ring
+ * is empty: a client waiting for a reply stops draining the data
+ * socket, and both ends wedge once the buffers fill.
+ */
+int
+capture_loop(const struct conn *ctrl, struct session *s)
+{
+	unsigned int empty_count = 0;
+
+	while (!rte_atomic_load_explicit(&quit_signal, rte_memory_order_relaxed)) {
+		ssize_t written;
+		unsigned int avail = 0;
+
+		switch (check_socket_status(ctrl)) {
+		case 1:
+			/* control message pending, let caller service it */
+			return 0;
+		case 0:
+			break;
+		default:
+			/* client is gone */
+			return -1;
+		}
+
+		written = process_ring(s, &avail);
+		if (written < 0) {
+			/* process_ring has already logged the reason */
+			return -1;
+		}
+
+		if (written > 0) {
+			/* are there more packets? */
+			empty_count = (avail == 0);
+			continue;
+		}
+
+		if (empty_count < SLEEP_THRESHOLD) {
+			/* spin a few times before checking */
+			++empty_count;
+			rte_pause();
+			continue;
+		}
+
+		/* ring has been empty for a while: stop spinning */
+		rte_delay_us_sleep(SLEEP_US);
+	}
+	return 0;
+}
+
+int
+handle_endcap(const struct conn *c, uint32_t plen, struct session *s)
+{
+	if (rpcap_discard(c, plen) < 0)
+		return -1;
+	stop_capture(s);
+	return rpcap_send_msg(c, RPCAP_MSG_ENDCAP_REPLY, 0, NULL, 0);
+}
+
+int
+handle_stats(const struct conn *c, uint32_t plen, const struct session *s)
+{
+	struct rte_eth_stats es = { 0 };
+
+	if (rpcap_discard(c, plen) < 0)
+		return -1;
+
+	if (s->capture_on)
+		rte_eth_stats_get(s->port, &es);
+
+	struct rpcap_stats reply = {
+		.ifrecv   = rte_cpu_to_be_32((uint32_t)es.ipackets),
+		.ifdrop   = rte_cpu_to_be_32((uint32_t)es.ierrors),
+		.krnldrop = 0,
+		.svrcapt  = rte_cpu_to_be_32(s->npkt),
+	};
+	return rpcap_send_msg(c, RPCAP_MSG_STATS_REPLY, 0, &reply, sizeof(reply));
+}
diff --git a/app/rpcapd/filter.c b/app/rpcapd/filter.c
new file mode 100644
index 0000000000..bdf7b8eaf1
--- /dev/null
+++ b/app/rpcapd/filter.c
@@ -0,0 +1,164 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * Capture filters.  The client compiles the filter, so it arrives as
+ * cBPF and has to be converted to the DPDK form that pdump takes.
+ */
+
+#include <stdlib.h>
+
+#include <pcap/pcap.h>
+
+#include <rte_bpf.h>
+#include <rte_byteorder.h>
+#include <rte_errno.h>
+#include <rte_malloc.h>
+#include <rte_pdump.h>
+
+#include "rpcap-protocol.h"
+#include "rpcapd.h"
+
+#define MAX_FILTER_INSNS              4096
+
+/*
+ * Read the optional capture filter that follows a start-capture request,
+ * and convert it for pdump. Client passes cBPF.
+ */
+int
+read_filter(const struct conn *c, uint32_t plen, struct session *s)
+{
+	struct rpcap_filterbpf_insn winsn;
+	struct rpcap_filter filter;
+	struct bpf_program bf;
+	struct bpf_insn *insns;
+	uint32_t i, nitems;
+
+	if (plen == 0)
+		return 0;		/* no filter: capture everything */
+
+	if (plen < sizeof(filter)) {
+		if (rpcap_discard(c, plen) < 0)
+			return -1;
+		return rpcap_send_error(c, 0, "short filter header") < 0 ? -1 : 1;
+	}
+
+	if (recv_full(c, &filter, sizeof(filter)) < 0)
+		return -1;
+	plen -= sizeof(filter);
+
+	if (rte_be_to_cpu_16(filter.filtertype) != RPCAP_UPDATEFILTER_BPF) {
+		if (rpcap_discard(c, plen) < 0)
+			return -1;
+		return rpcap_send_error(c, 0, "unsupported filter type") < 0 ? -1 : 1;
+	}
+
+	/* nitems is client-supplied; bound it before trusting the length. */
+	nitems = rte_be_to_cpu_32(filter.nitems);
+	if (nitems == 0)
+		return rpcap_discard(c, plen) < 0 ? -1 : 0;
+
+	if (nitems > MAX_FILTER_INSNS || plen < nitems * sizeof(winsn)) {
+		if (rpcap_discard(c, plen) < 0)
+			return -1;
+		return rpcap_send_error(c, 0, "bad filter length") < 0 ? -1 : 1;
+	}
+
+	insns = calloc(nitems, sizeof(*insns));
+	if (insns == NULL) {
+		if (rpcap_discard(c, plen) < 0)
+			return -1;
+		return rpcap_send_error(c, 0, "out of memory") < 0 ? -1 : 1;
+	}
+
+	for (i = 0; i < nitems; i++) {
+		if (recv_full(c, &winsn, sizeof(winsn)) < 0) {
+			free(insns);
+			return -1;
+		}
+		insns[i].code = rte_be_to_cpu_16(winsn.code);
+		insns[i].jt   = winsn.jt;
+		insns[i].jf   = winsn.jf;
+		insns[i].k    = rte_be_to_cpu_32(winsn.k);
+	}
+	plen -= nitems * sizeof(winsn);
+
+	/* Anything after the instructions is padding we do not need. */
+	if (rpcap_discard(c, plen) < 0) {
+		free(insns);
+		return -1;
+	}
+
+	bf.bf_len = nitems;
+	bf.bf_insns = insns;
+
+	/* Reject a malformed program here */
+	if (!bpf_validate(bf.bf_insns, bf.bf_len)) {
+		free(insns);
+		return rpcap_send_error(c, 0, "invalid filter program") < 0 ? -1 : 1;
+	}
+
+	/* A filter recorded by an earlier UPDATEFILTER may still be here */
+	rte_free(s->prm);
+	s->prm = rte_bpf_convert(&bf);
+	free(insns);
+	if (s->prm == NULL) {
+		RPCAPD_LOG(ERR, "rte_bpf_convert failed: %s",
+			rte_strerror(rte_errno));
+		return rpcap_send_error(c, 0, "cannot convert filter") < 0 ? -1 : 1;
+	}
+
+	RPCAPD_LOG(DEBUG, "capture filter: %u instructions", nitems);
+	return 0;
+}
+
+/*
+ * UPDATEFILTER_REQ: replace the capture filter.
+ *
+ * pdump takes its filter when the callback is setup.
+ * To replace need to drop old callback and put in new one.
+ * Packets already in the ring are kept.
+ *
+ * Before the capture starts this just records the filter for the
+ * eventual STARTCAP.
+ */
+int
+handle_updatefilter(const struct conn *c, uint32_t plen, struct session *s)
+{
+	struct rte_bpf_prm *old = s->prm;
+	int ret;
+
+	s->prm = NULL;
+	ret = read_filter(c, plen, s);
+	if (ret != 0) {
+		/* Malformed request: keep running with the old filter. */
+		rte_free(s->prm);
+		s->prm = old;
+		return ret < 0 ? -1 : 0;	/* error already reported */
+	}
+
+	if (!s->capture_on) {
+		rte_free(old);
+		return rpcap_send_msg(c, RPCAP_MSG_UPDATEFILTER_REPLY, 0, NULL, 0);
+	}
+
+	rte_pdump_disable(s->port, RTE_PDUMP_ALL_QUEUES, s->pdump_flags);
+	s->capture_on = false;
+
+	if (rte_pdump_enable_bpf(s->port, RTE_PDUMP_ALL_QUEUES,
+				 s->pdump_flags | RTE_PDUMP_FLAG_PCAPNG,
+				 s->snaplen, s->ring, s->mp, s->prm) < 0) {
+		RPCAPD_LOG(ERR, "rte_pdump_enable_bpf port %u failed: %s",
+			s->port, rte_strerror(rte_errno));
+		rte_free(old);
+		/* The capture cannot be resumed */
+		stop_capture(s);
+		return rpcap_send_error(c, 0, "cannot apply filter");
+	}
+	s->capture_on = true;
+
+	/* Safe now that the old program is no longer referenced. */
+	rte_free(old);
+
+	RPCAPD_LOG(DEBUG, "capture filter updated on %s", s->name);
+	return rpcap_send_msg(c, RPCAP_MSG_UPDATEFILTER_REPLY, 0, NULL, 0);
+}
diff --git a/app/rpcapd/main.c b/app/rpcapd/main.c
new file mode 100644
index 0000000000..7b7288785d
--- /dev/null
+++ b/app/rpcapd/main.c
@@ -0,0 +1,577 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * Demonstration server for the rpcap protocol for DPDK.
+ * This allows a libpcap client (e.g. Wireshark or tcpdump)
+ * to use "rpcap://host[:port]/portname" as capture device.
+ *
+ * Based on the DPDK dumpcap application and on rpcapd from libpcap:
+ *   https://github.com/the-tcpdump-group/libpcap/tree/master/rpcapd
+ *
+ * Only the bits of the RPCAP protocol that are needed for an
+ * unauthenticated, passive-mode capture session are implemented.
+ * Configuration files, active mode, sampling and concurrent clients
+ * are intentionally omitted.
+ *
+ * Options, startup and the control connection dispatcher live here; the
+ * request handlers are in session.c, capture.c and filter.c.
+ */
+
+#include <arpa/inet.h>
+#include <errno.h>
+#include <getopt.h>
+#include <netinet/in.h>
+#include <netdb.h>
+#include <signal.h>
+#include <stdbool.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/socket.h>
+#include <sys/types.h>
+#include <unistd.h>
+
+#include <rte_alarm.h>
+#include <rte_byteorder.h>
+#include <rte_common.h>
+#include <rte_debug.h>
+#include <rte_eal.h>
+#include <rte_ethdev.h>
+#include <rte_lcore.h>
+#include <rte_log.h>
+#include <rte_pdump.h>
+#include <rte_stdatomic.h>
+#include <rte_version.h>
+
+#include "rpcap-protocol.h"
+#include "rpcapd.h"
+
+#define DEFAULT_RING_SIZE             2048
+#define MAX_RING_SIZE                 (1U << 20)
+#define PRIMARY_MONITOR_INTERVAL_US   (500 * 1000)
+#define DATA_SEND_TIMEOUT_SEC         10
+
+/* Command-line options */
+static uint16_t listen_port = RPCAP_DEFAULT_NETPORT;
+uint32_t ring_size = DEFAULT_RING_SIZE;
+static const char *lcore_arg;
+static const char *file_prefix;
+static const char *bind_addr;		/* -b argument; NULL means loopback */
+static int bind_family = AF_UNSPEC;
+static const char *debug_file;		/* --debug-file argument */
+static unsigned int debug_log;		/* -D count: raise RPCAPD log verbosity */
+uint32_t send_timeout = DATA_SEND_TIMEOUT_SEC;	/* 0 means no limit */
+
+struct sockaddr_storage listen_addr;
+socklen_t               listen_addrlen;
+
+RTE_ATOMIC(bool) quit_signal;
+
+static bool
+is_loopback(const struct sockaddr_storage *ss)
+{
+	if (ss->ss_family == AF_INET) {
+		const struct sockaddr_in *sin = (const void *)ss;
+
+		return (ntohl(sin->sin_addr.s_addr) >> 24) == 127;
+	}
+	if (ss->ss_family == AF_INET6) {
+		const struct sockaddr_in6 *sin6 = (const void *)ss;
+
+		/* A v4 client on a dual-stack socket arrives as
+		 * ::ffff:127.0.0.1, which is loopback too.
+		 */
+		if (IN6_IS_ADDR_V4MAPPED(&sin6->sin6_addr))
+			return sin6->sin6_addr.s6_addr[12] == 127;
+
+		return IN6_IS_ADDR_LOOPBACK(&sin6->sin6_addr);
+	}
+	return false;
+}
+
+static void
+parse_bind_addr(void)
+{
+	struct addrinfo hints = {
+		.ai_family   = bind_family,
+		.ai_socktype = SOCK_STREAM,
+		.ai_flags    = AI_NUMERICHOST | AI_PASSIVE,
+	};
+	struct addrinfo *res;
+	int rc;
+
+	/* Loopback by default; the wildcard address is not a safe default. */
+	if (bind_addr == NULL)
+		bind_addr = (bind_family == AF_INET6) ? "::1" : "127.0.0.1";
+
+	rc = getaddrinfo(bind_addr, NULL, &hints, &res);
+	if (rc != 0)
+		rte_exit(EXIT_FAILURE, "Invalid bind address '%s': %s\n",
+			 bind_addr, gai_strerror(rc));
+	memcpy(&listen_addr, res->ai_addr, res->ai_addrlen);
+	listen_addrlen = res->ai_addrlen;
+	freeaddrinfo(res);
+}
+
+
+static void
+signal_handler(int sig __rte_unused)
+{
+	rte_atomic_store_explicit(&quit_signal, true, rte_memory_order_relaxed);
+}
+
+/* Service a single client until it disconnects. */
+static void
+handle_client(int ctrl_fd)
+{
+	struct sockaddr_storage peer;
+	socklen_t peerlen = sizeof(peer);
+	char host[NI_MAXHOST] = "?";
+	struct conn ctrl = { .fd = ctrl_fd };
+	struct session s = { .data.fd = -1 };
+
+	/* Remembered so the data connection can be restricted to this peer. */
+	if (getpeername(ctrl_fd, (struct sockaddr *)&peer, &peerlen) != 0) {
+		RPCAPD_LOG(ERR, "getpeername: %s", strerror(errno));
+		return;
+	}
+	s.peer = peer;
+	getnameinfo((struct sockaddr *)&peer, peerlen,
+		    host, sizeof(host), NULL, 0, NI_NUMERICHOST);
+	RPCAPD_LOG(NOTICE, "client %s connected", host);
+
+	while (!rte_atomic_load_explicit(&quit_signal, rte_memory_order_relaxed)) {
+		struct rpcap_header hdr;
+		uint32_t plen;
+
+		/* Drain the ring whenever a capture is running */
+		if (s.capture_on && capture_loop(&ctrl, &s) < 0)
+			goto done;
+
+		if (recv_full(&ctrl, &hdr, sizeof(hdr)) < 0)
+			break;
+
+		plen = rte_be_to_cpu_32(hdr.plen);
+
+		/* Only version 0 is spoken here */
+		if (hdr.ver != RPCAP_VERSION) {
+			RPCAPD_LOG(WARNING, "unsupported protocol version %u",
+				hdr.ver);
+			if (rpcap_discard(&ctrl, plen) < 0 ||
+			    rpcap_send_error(&ctrl, PCAP_ERR_WRONGVER,
+					     "unsupported protocol version") < 0)
+				goto done;
+			continue;
+		}
+
+		switch (hdr.type) {
+		case RPCAP_MSG_AUTH_REQ:
+			/* libpcap treats a zero-length AUTH_REPLY as "version
+			 * 0 only, same byte order".
+			 */
+			if (handle_auth(&ctrl, plen) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_FINDALLIF_REQ:
+			if (rpcap_discard(&ctrl, plen) < 0 || handle_findallif(&ctrl) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_OPEN_REQ:
+			if (handle_open(&ctrl, plen, &s) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_STARTCAP_REQ:
+			if (handle_startcap(&ctrl, plen, &s) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_UPDATEFILTER_REQ:
+			if (handle_updatefilter(&ctrl, plen, &s) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_ENDCAP_REQ:
+			if (handle_endcap(&ctrl, plen, &s) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_STATS_REQ:
+			if (handle_stats(&ctrl, plen, &s) < 0)
+				goto done;
+			break;
+		case RPCAP_MSG_CLOSE:
+			rpcap_discard(&ctrl, plen);
+			goto done;
+		default:
+			RPCAPD_LOG(WARNING, "unsupported request type 0x%02x", hdr.type);
+			if (rpcap_discard(&ctrl, plen) < 0 ||
+			    rpcap_send_error(&ctrl, 0, "unsupported request") < 0)
+				goto done;
+			break;
+		}
+	}
+done:
+	stop_capture(&s);
+	close(ctrl_fd);
+	RPCAPD_LOG(NOTICE, "client %s disconnected", host);
+}
+
+static int
+open_listen_socket(uint16_t port)
+{
+	struct sockaddr_storage addr = listen_addr;
+	char host[NI_MAXHOST];
+	int fd, one = 1;
+
+	set_sockaddr_port(&addr, port);
+
+	fd = socket(addr.ss_family, SOCK_STREAM, 0);
+	if (fd < 0)
+		rte_exit(EXIT_FAILURE, "socket: %s\n", strerror(errno));
+	setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &one, sizeof(one));
+
+	if (bind(fd, (struct sockaddr *)&addr, listen_addrlen) < 0)
+		rte_exit(EXIT_FAILURE, "bind(%u): %s\n", port, strerror(errno));
+
+	int err = getnameinfo((struct sockaddr *)&listen_addr, listen_addrlen,
+			      host, sizeof(host), NULL, 0, NI_NUMERICHOST);
+	if (err != 0)
+		rte_exit(EXIT_FAILURE, "Listen address lookup failed: %s\n",
+			 gai_strerror(err));
+
+	RPCAPD_LOG(NOTICE, "listening on %s port %u", host, listen_port);
+
+	if (!is_loopback(&listen_addr))
+		RPCAPD_LOG(WARNING,
+			"non-loopback address %s; "
+			"rpcap is unauthenticated and unencrypted, captured traffic is exposed to the network",
+			host);
+
+	if (listen(fd, 1) < 0)
+		rte_exit(EXIT_FAILURE, "listen: %s\n", strerror(errno));
+
+	return fd;
+}
+
+static void
+usage(FILE *f, const char *progname)
+{
+	fprintf(f, "Usage: %s [options]\n", progname);
+	fprintf(f,
+		"  -p, --port <port>     listen port (default %u)\n"
+		"  -b, --bind <addr>     bind address (default 127.0.0.1, ::1 with -6)\n"
+		"  -4                    use only IPv4\n"
+		"  -6                    use only IPv6\n"
+		"  -N <ring size>        ring size in packets (default %u)\n"
+		"  -D, --debug           increase log verbosity (-D info, -DD debug)\n"
+		"      --debug-file <f>  redirect log output to file <f> (append mode)\n"
+		"      --send-timeout <s> seconds a data send may block before the\n"
+		"                        client is treated as dead (default %u, 0 waits\n"
+		"                        forever)\n"
+		"      --version         print version and exit\n"
+		"  -h, --help            print this help and exit\n"
+		"      --lcore=<core>    CPU core to run on (default: any)\n"
+		"      --file-prefix=<p> prefix to use for multi-process\n"
+		"\n"
+		"WARNING: rpcap is unauthenticated and unencrypted.  Binding to\n"
+		"any non-loopback address exposes captured traffic to the\n"
+		"network.  Not for production use.\n",
+		RPCAP_DEFAULT_NETPORT, DEFAULT_RING_SIZE,
+		DATA_SEND_TIMEOUT_SEC);
+}
+
+static void
+print_version(void)
+{
+	printf("rpcapd, a remote packet capture daemon (DPDK pdump backend)\n"
+	       "Built against %s\n", rte_version());
+}
+
+static void
+parse_opts(int argc, char **argv)
+{
+	enum {
+		OPT_LONG_ONLY = 0x100,
+		OPT_DEBUG_FILE,
+		OPT_VERSION,
+		OPT_SEND_TIMEOUT,
+	};
+	static const struct option long_options[] = {
+		{ "port",         required_argument, NULL, 'p' },
+		{ "bind",         required_argument, NULL, 'b' },
+		{ "debug",        no_argument,       NULL, 'D' },
+		{ "help",         no_argument,       NULL, 'h' },
+		{ "version",      no_argument,       NULL, OPT_VERSION },
+		{ "debug-file",   required_argument, NULL, OPT_DEBUG_FILE },
+		{ "send-timeout", required_argument, NULL, OPT_SEND_TIMEOUT },
+		{ "file-prefix",  required_argument, NULL, 0 },
+		{ "lcore",        required_argument, NULL, 0 },
+		{ NULL, 0, NULL, 0 },
+	};
+	int option_index, c;
+
+	while ((c = getopt_long(argc, argv, "hD46p:b:N:",
+				long_options, &option_index)) != -1) {
+		switch (c) {
+		case 'p': {
+			unsigned long u = strtoul(optarg, NULL, 0);
+
+			if (u == 0 || u > UINT16_MAX)
+				rte_exit(EXIT_FAILURE, "Invalid port: %s\n", optarg);
+			listen_port = (uint16_t)u;
+			break;
+		}
+		case 'b':
+			bind_addr = optarg;
+			break;
+		case '4':
+			bind_family = AF_INET;
+			break;
+		case '6':
+			bind_family = AF_INET6;
+			break;
+		case 'N': {
+			unsigned long u = strtoul(optarg, NULL, 0);
+
+			/* Check the full value before narrowing it: an upper
+			 * bound is needed anyway because rte_align32pow2()
+			 * wraps to zero above 2^31, and that failure would
+			 * otherwise only surface in rte_ring_create() on the
+			 * first capture.
+			 */
+			if (u < 64 || u > MAX_RING_SIZE)
+				rte_exit(EXIT_FAILURE,
+					 "Ring size must be between 64 and %u\n",
+					 MAX_RING_SIZE);
+			ring_size = (uint32_t)u;
+			/* rte_ring_create() requires a power of two. */
+			if (!rte_is_power_of_2(ring_size)) {
+				ring_size = rte_align32pow2(ring_size);
+				RPCAPD_LOG(NOTICE, "ring size rounded up to %u",
+					ring_size);
+			}
+			break;
+		}
+		case 'D':
+			debug_log++;
+			break;
+		case 'h':
+			usage(stdout, argv[0]);
+			exit(0);
+		case OPT_VERSION:
+			print_version();
+			exit(0);
+		case OPT_DEBUG_FILE:
+			debug_file = optarg;
+			break;
+		case OPT_SEND_TIMEOUT: {
+			unsigned long u = strtoul(optarg, NULL, 0);
+
+			/* Zero means wait forever, which is what the socket
+			 * does without SO_SNDTIMEO.
+			 */
+			if (u > INT32_MAX)
+				rte_exit(EXIT_FAILURE,
+					 "Invalid send timeout: %s\n", optarg);
+			send_timeout = (uint32_t)u;
+			break;
+		}
+		case 0: {
+			const char *longopt = long_options[option_index].name;
+
+			if (!strcmp(longopt, "lcore")) {
+				lcore_arg = optarg;
+				break;
+			} else if (!strcmp(longopt, "file-prefix")) {
+				file_prefix = optarg;
+				break;
+			}
+		}
+			/* fallthrough */
+		default:
+			usage(stderr, argv[0]);
+			exit(EXIT_FAILURE);
+		}
+	}
+
+	/* Resolve the bind address now that -4/-6/-b have been seen. */
+	parse_bind_addr();
+}
+
+/*
+ * Periodic check that the DPDK primary process is still alive.
+ * If it dies our shared-memory state (rings, mempools, pdump) becomes
+ * unsafe to touch, so we set quit_signal and let the main loop tear
+ * down cleanly on its next iteration.  The callback runs on the EAL
+ * interrupt thread; quit_signal is atomic so the read in the main
+ * loop is well-defined.
+ */
+static void
+monitor_primary(void *arg __rte_unused)
+{
+	if (rte_atomic_load_explicit(&quit_signal, rte_memory_order_relaxed))
+		return;
+
+	if (rte_eal_primary_proc_alive(NULL)) {
+		rte_eal_alarm_set(PRIMARY_MONITOR_INTERVAL_US, monitor_primary, NULL);
+		return;
+	}
+
+	RPCAPD_LOG(NOTICE, "primary process exited, shutting down");
+	rte_atomic_store_explicit(&quit_signal, true, rte_memory_order_relaxed);
+}
+
+static void
+enable_primary_monitor(void)
+{
+	if (rte_eal_alarm_set(PRIMARY_MONITOR_INTERVAL_US, monitor_primary, NULL) < 0)
+		RPCAPD_LOG(WARNING, "failed to install primary process monitor");
+}
+
+static void
+disable_primary_monitor(void)
+{
+	rte_eal_alarm_cancel(monitor_primary, NULL);
+}
+
+/*
+ * Bring up EAL as a secondary process so that pdump can attach to a
+ * running primary DPDK application. Hide most of the EAL
+ * complexity and only show serious messages from EAL.
+ */
+static int
+dpdk_init(void)
+{
+	static const char * const args[] = {
+		"rpcapd",
+		"--proc-type", "secondary",
+		"--log-level", "lib.eal:warning",
+	};
+	int eal_argc = RTE_DIM(args);
+	rte_cpuset_t cpuset = { };
+	char **eal_argv;
+	unsigned int i;
+
+	if (file_prefix != NULL)
+		eal_argc += 2;
+
+	if (lcore_arg != NULL)
+		eal_argc += 2;
+
+	eal_argv = calloc(eal_argc + 1, sizeof(char *));
+	if (eal_argv == NULL)
+		return -1;
+
+	for (i = 0; i < RTE_DIM(args); i++) {
+		eal_argv[i] = strdup(args[i]);
+		if (eal_argv[i] == NULL)
+			return -1;
+	}
+
+	if (file_prefix != NULL && *file_prefix != '\0') {
+		eal_argv[i++] = strdup("--file-prefix");
+		eal_argv[i++] = strdup(file_prefix);
+		if (eal_argv[i - 1] == NULL || eal_argv[i - 2] == NULL)
+			return -1;
+	}
+
+	if (lcore_arg != NULL) {
+		eal_argv[i++] = strdup("--lcores");
+		eal_argv[i++] = strdup(lcore_arg);
+		if (eal_argv[i - 1] == NULL || eal_argv[i - 2] == NULL)
+			return -1;
+	}
+	eal_argc = i;
+
+	/*
+	 * Need to get the original cpuset, before EAL init changes
+	 * the affinity of this thread (main lcore).
+	 */
+	if (lcore_arg == NULL &&
+	    rte_thread_get_affinity_by_id(rte_thread_self(), &cpuset) != 0)
+		rte_panic("rte_thread_getaffinity failed\n");
+
+	if (rte_eal_init(eal_argc, eal_argv) < 0)
+		rte_exit(EXIT_FAILURE, "EAL init failed: is the primary process running?\n");
+
+	/*
+	 * If no lcore argument was specified,
+	 * then run this program as a normal process
+	 * which can be scheduled on any non-isolated CPU.
+	 */
+	if (lcore_arg == NULL &&
+	    rte_thread_set_affinity_by_id(rte_thread_self(), &cpuset) != 0)
+		RPCAPD_LOG(INFO, "Can not restore original CPU affinity");
+
+	if (rte_pdump_init() < 0)
+		rte_exit(EXIT_FAILURE, "rte_pdump_init failed\n");
+
+	/* Needs the TSC frequency, so must follow rte_eal_init(). */
+	timestamp_init();
+
+	return 0;
+}
+
+int
+main(int argc, char **argv)
+{
+	struct sigaction action = {
+		.sa_handler = signal_handler,
+	};
+	int srv_fd;
+
+	parse_opts(argc, argv);
+
+	/*
+	 * Redirect log output before EAL init so EAL's own messages are
+	 * captured too.  The FILE handle is intentionally never closed:
+	 * the kernel reclaims it at process exit.
+	 */
+	if (debug_file != NULL) {
+		FILE *fp = fopen(debug_file, "a");
+
+		if (fp == NULL)
+			rte_exit(EXIT_FAILURE, "Cannot open debug file '%s': %s\n",
+				 debug_file, strerror(errno));
+		setvbuf(fp, NULL, _IOLBF, 0);
+		rte_openlog_stream(fp);
+	}
+
+	if (dpdk_init() < 0)
+		rte_exit(EXIT_FAILURE, "EAL init failure\n");
+
+	/* Default to NOTICE: only things the operator needs to see.
+	 * Each -D steps down one level, to INFO then DEBUG.
+	 */
+	rte_log_set_level(RTE_LOGTYPE_RPCAPD,
+			  debug_log >= 2 ? RTE_LOG_DEBUG :
+			  debug_log == 1 ? RTE_LOG_INFO : RTE_LOG_NOTICE);
+
+	if (rte_eth_dev_count_avail() == 0)
+		rte_exit(EXIT_FAILURE, "No Ethernet ports found\n");
+
+	sigaction(SIGTERM, &action, NULL);
+	sigaction(SIGINT, &action, NULL);
+
+	/* If peer closes, this detected in next recv() */
+	signal(SIGPIPE, SIG_IGN);
+
+	srv_fd = open_listen_socket(listen_port);
+
+	enable_primary_monitor();
+
+	while (!rte_atomic_load_explicit(&quit_signal, rte_memory_order_relaxed)) {
+		int cfd = accept_timeout(srv_fd, -1);
+
+		if (cfd < 0) {
+			if (errno == EINTR)
+				continue;
+			break;
+		}
+		handle_client(cfd);
+	}
+
+	disable_primary_monitor();
+	RPCAPD_LOG(NOTICE, "shutting down");
+	close(srv_fd);
+	rte_pdump_uninit();
+	return rte_eal_cleanup() ? EXIT_FAILURE : 0;
+}
diff --git a/app/rpcapd/meson.build b/app/rpcapd/meson.build
new file mode 100644
index 0000000000..61f4dc0a95
--- /dev/null
+++ b/app/rpcapd/meson.build
@@ -0,0 +1,25 @@
+# SPDX-License-Identifier: BSD-3-Clause
+# Copyright(c) 2026 Stephen Hemminger
+
+# relies on primary/secondary process, so Linux only
+if not is_linux
+    build = false
+    reason = 'only supported on Linux'
+    subdir_done()
+endif
+
+if not dpdk_conf.has('RTE_HAS_LIBPCAP')
+    build = false
+    reason = 'missing dependency, "libpcap"'
+    subdir_done()
+endif
+
+sources = files(
+        'capture.c',
+        'filter.c',
+        'main.c',
+        'session.c',
+        'sock.c',
+)
+ext_deps += pcap_dep
+deps += ['ethdev', 'pdump', 'bpf', 'pcapng']
diff --git a/app/rpcapd/rpcap-protocol.h b/app/rpcapd/rpcap-protocol.h
new file mode 100644
index 0000000000..438fd8dd84
--- /dev/null
+++ b/app/rpcapd/rpcap-protocol.h
@@ -0,0 +1,142 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * On-the-wire RPCAP protocol definitions, transcribed from libpcap's
+ * rpcap-protocol.h which is an internal file and not exported.
+ * See:
+ *   https://github.com/the-tcpdump-group/libpcap/blob/master/rpcap-protocol.h
+ *
+ * Only the subset needed by dpdk-rpcapd is included here.
+ * All multi-byte fields in the structures below are big-endian on the wire.
+ */
+
+#ifndef _RPCAP_PROTOCOL_H_
+#define _RPCAP_PROTOCOL_H_
+
+#include <stdint.h>
+
+#include <rte_byteorder.h>
+
+#define RPCAP_VERSION              0
+#define RPCAP_DEFAULT_NETPORT      2002
+
+/* Message types */
+#define RPCAP_MSG_ERROR            0x01
+#define RPCAP_MSG_FINDALLIF_REQ    0x02
+#define RPCAP_MSG_OPEN_REQ         0x03
+#define RPCAP_MSG_STARTCAP_REQ     0x04
+#define RPCAP_MSG_UPDATEFILTER_REQ 0x05
+#define RPCAP_MSG_CLOSE            0x06
+#define RPCAP_MSG_PACKET           0x07
+#define RPCAP_MSG_AUTH_REQ         0x08
+#define RPCAP_MSG_STATS_REQ        0x09
+#define RPCAP_MSG_ENDCAP_REQ       0x0a
+#define RPCAP_MSG_IS_REPLY         0x80
+
+#define RPCAP_MSG_FINDALLIF_REPLY    (RPCAP_MSG_FINDALLIF_REQ    | RPCAP_MSG_IS_REPLY)
+#define RPCAP_MSG_OPEN_REPLY         (RPCAP_MSG_OPEN_REQ         | RPCAP_MSG_IS_REPLY)
+#define RPCAP_MSG_STARTCAP_REPLY     (RPCAP_MSG_STARTCAP_REQ     | RPCAP_MSG_IS_REPLY)
+#define RPCAP_MSG_UPDATEFILTER_REPLY (RPCAP_MSG_UPDATEFILTER_REQ | RPCAP_MSG_IS_REPLY)
+#define RPCAP_MSG_AUTH_REPLY         (RPCAP_MSG_AUTH_REQ         | RPCAP_MSG_IS_REPLY)
+#define RPCAP_MSG_ENDCAP_REPLY       (RPCAP_MSG_ENDCAP_REQ       | RPCAP_MSG_IS_REPLY)
+#define RPCAP_MSG_STATS_REPLY	     (RPCAP_MSG_STATS_REQ	 | RPCAP_MSG_IS_REPLY)
+
+/* Error codes carried in the 'value' field of RPCAP_MSG_ERROR */
+#define PCAP_ERR_WRONGVER          17
+#define PCAP_ERR_AUTH_TYPE_NOTSUP  20
+
+/* Authentication types in rpcap_auth.type */
+#define RPCAP_RMTAUTH_NULL         0	/* no credentials supplied */
+#define RPCAP_RMTAUTH_PWD          1	/* username and password follow */
+
+/* Filter encoding: the filter is a BPF/NPF program */
+#define RPCAP_UPDATEFILTER_BPF     1
+
+/* Flags in rpcap_startcapreq.flags */
+#define RPCAP_STARTCAPREQ_FLAG_PROMISC     0x00000001	/* promiscuous mode */
+#define RPCAP_STARTCAPREQ_FLAG_DGRAM       0x00000002	/* use UDP for data */
+#define RPCAP_STARTCAPREQ_FLAG_SERVEROPEN  0x00000004	/* server connects out */
+#define RPCAP_STARTCAPREQ_FLAG_INBOUND     0x00000008	/* capture inbound only */
+#define RPCAP_STARTCAPREQ_FLAG_OUTBOUND    0x00000010	/* capture outbound only */
+
+/* Subset of pcap interface flags (pcap.h) */
+#define PCAP_IF_UP                 0x00000002
+#define PCAP_IF_RUNNING            0x00000004
+
+/* DLT_EN10MB - ethernet, the only link type we report */
+#define DLT_EN10MB                 1
+
+struct rpcap_header {
+	uint8_t     ver;
+	uint8_t     type;
+	rte_be16_t  value;
+	rte_be32_t  plen;
+};
+
+struct rpcap_findalldevs_if {
+	rte_be16_t  namelen;
+	rte_be16_t  desclen;
+	rte_be32_t  flags;
+	rte_be16_t  naddr;
+	uint16_t    dummy;
+};
+
+struct rpcap_openreply {
+	rte_be32_t  linktype;
+	rte_be32_t  tzoff;
+};
+
+struct rpcap_auth {
+	rte_be16_t  type;	/* RPCAP_RMTAUTH_* */
+	uint16_t    dummy;
+	rte_be16_t  slen1;	/* length of username, if any */
+	rte_be16_t  slen2;	/* length of password, if any */
+};
+
+struct rpcap_startcapreq {
+	rte_be32_t  snaplen;
+	rte_be32_t  read_timeout;
+	rte_be16_t  flags;
+	rte_be16_t  portdata;
+};
+
+struct rpcap_startcapreply {
+	rte_be32_t  bufsize;
+	rte_be16_t  portdata;
+	uint16_t    dummy;
+};
+
+/*
+ * A filter, sent either after rpcap_startcapreq or in an
+ * RPCAP_MSG_UPDATEFILTER_REQ, followed by nitems instructions.
+ */
+struct rpcap_filter {
+	rte_be16_t  filtertype;
+	uint16_t    dummy;
+	rte_be32_t  nitems;
+};
+
+/* One cBPF instruction, repeated nitems times after rpcap_filter. */
+struct rpcap_filterbpf_insn {
+	rte_be16_t  code;
+	uint8_t     jt;
+	uint8_t     jf;
+	rte_be32_t  k;
+};
+
+struct rpcap_stats {
+	rte_be32_t  ifrecv;
+	rte_be32_t  ifdrop;
+	rte_be32_t  krnldrop;
+	rte_be32_t  svrcapt;
+};
+
+struct rpcap_pkthdr {
+	rte_be32_t  timestamp_sec;
+	rte_be32_t  timestamp_usec;
+	rte_be32_t  caplen;
+	rte_be32_t  len;
+	rte_be32_t  npkt;
+};
+
+#endif /* _RPCAP_PROTOCOL_H_ */
diff --git a/app/rpcapd/rpcapd.h b/app/rpcapd/rpcapd.h
new file mode 100644
index 0000000000..df38231bfb
--- /dev/null
+++ b/app/rpcapd/rpcapd.h
@@ -0,0 +1,105 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * State and helpers shared between the parts of the rpcap daemon.
+ */
+
+#ifndef _RPCAPD_H_
+#define _RPCAPD_H_
+
+#include <stdbool.h>
+#include <stdint.h>
+#include <sys/socket.h>
+#include <sys/uio.h>
+
+#include <rte_ethdev.h>
+#include <rte_ether.h>
+#include <rte_log.h>
+#include <rte_mbuf.h>
+#include <rte_stdatomic.h>
+
+struct rte_bpf_prm;
+struct rte_mempool;
+struct rte_ring;
+
+#define RTE_LOGTYPE_RPCAPD RTE_LOGTYPE_USER1
+#define RPCAPD_LOG(level, ...) \
+	RTE_LOG_LINE_PREFIX(level, RPCAPD, "%s(): ", __func__, __VA_ARGS__)
+
+/* Largest snaplen a client can be given. */
+#define DEFAULT_SNAPLEN		RTE_MBUF_DEFAULT_DATAROOM
+
+/*
+ * rte_pcapng_copy() truncates to the snaplen and then re-inserts any
+ * VLAN or QinQ tag the NIC stripped, so a capture can exceed the
+ * snaplen by up to two tags.
+ */
+#define MAX_CAPTURE_LEN		(DEFAULT_SNAPLEN + 2 * sizeof(struct rte_vlan_hdr))
+
+/* A connection to the client. */
+struct conn {
+	int fd;
+};
+
+/* Per-client capture session state. */
+struct session {
+	struct conn data;			/* data connection */
+	struct sockaddr_storage peer;		/* control connection peer */
+	uint16_t port;				/* DPDK ethdev port being captured */
+	char     name[RTE_ETH_NAME_MAX_LEN];
+	uint32_t snaplen;
+	uint32_t npkt;				/* packet sequence for rpcap_pkthdr */
+	uint32_t pdump_flags;			/* direction bits handed to pdump */
+	bool     opened;			/* OPEN_REQ has selected a port */
+	bool     capture_on;
+	bool     promisc_set;			/* we enabled promiscuous mode */
+	struct rte_ring    *ring;
+	struct rte_mempool *mp;
+	struct rte_bpf_prm *prm;		/* capture filter, NULL if none */
+};
+
+/* Set once by the signal handler to unwind the main and capture loops. */
+extern RTE_ATOMIC(bool) quit_signal;
+
+/* Command-line settings needed outside of main.c */
+extern uint32_t ring_size;
+extern uint32_t send_timeout;		/* seconds; 0 means no limit */
+
+/* Address the control socket is bound to; the data socket uses the same
+ * address with an ephemeral port.
+ */
+extern struct sockaddr_storage listen_addr;
+extern socklen_t               listen_addrlen;
+
+/* sock.c: transport and message framing */
+int wait_readable(const struct conn *c, int timeout_ms);
+int accept_timeout(int listen_fd, int timeout_ms);
+int accept_from(int listen_fd, const struct sockaddr_storage *want,
+		int timeout_ms);
+int recv_full(const struct conn *c, void *buf, size_t len);
+int send_iov_full(const struct conn *c, struct iovec *iov, int iovcnt, int flags);
+int rpcap_send_msg(const struct conn *c, uint8_t type, uint16_t value,
+		   const void *payload, uint32_t plen);
+int rpcap_send_error(const struct conn *c, uint16_t errcode, const char *msg);
+int rpcap_discard(const struct conn *c, uint32_t plen);
+void set_sockaddr_port(struct sockaddr_storage *ss, uint16_t port);
+uint16_t get_sockaddr_port(const struct sockaddr_storage *ss);
+
+/* session.c: control requests handled before a capture starts */
+int handle_auth(const struct conn *c, uint32_t plen);
+int handle_findallif(const struct conn *c);
+int handle_open(const struct conn *c, uint32_t plen, struct session *s);
+
+/* filter.c */
+int read_filter(const struct conn *c, uint32_t plen, struct session *s);
+int handle_updatefilter(const struct conn *c, uint32_t plen, struct session *s);
+
+/* capture.c */
+void timestamp_init(void);
+int handle_startcap(const struct conn *c, uint32_t plen, struct session *s);
+int handle_endcap(const struct conn *c, uint32_t plen, struct session *s);
+int handle_stats(const struct conn *c, uint32_t plen, const struct session *s);
+void stop_capture(struct session *s);
+int capture_loop(const struct conn *ctrl, struct session *s);
+
+#endif /* _RPCAPD_H_ */
diff --git a/app/rpcapd/session.c b/app/rpcapd/session.c
new file mode 100644
index 0000000000..cc14f26335
--- /dev/null
+++ b/app/rpcapd/session.c
@@ -0,0 +1,136 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * Control requests handled before a capture starts: authentication,
+ * the interface list, and selecting an interface.
+ */
+
+#include <stdlib.h>
+#include <string.h>
+
+#include <rte_byteorder.h>
+#include <rte_ethdev.h>
+
+#include "rpcap-protocol.h"
+#include "rpcapd.h"
+
+/* Build and send the list of available DPDK ports. */
+int
+handle_findallif(const struct conn *c)
+{
+	uint8_t *buf = NULL;
+	size_t buflen = 0;
+	uint16_t nif = 0;
+	uint16_t p;
+	int rc;
+
+	RTE_ETH_FOREACH_DEV(p) {
+		static const char desc[] = "DPDK port";
+		char name[RTE_ETH_NAME_MAX_LEN];
+		size_t namelen, desclen, entry;
+		uint8_t *nb;
+
+		if (rte_eth_dev_get_name_by_port(p, name) < 0) {
+			RPCAPD_LOG(DEBUG, "can not find name for port %u", p);
+			continue;
+		}
+
+		RPCAPD_LOG(DEBUG, "findallif: port %u -> '%s'", p, name);
+		namelen = strlen(name);
+		desclen = strlen(desc);
+		entry = sizeof(struct rpcap_findalldevs_if) + namelen + desclen;
+
+		nb = realloc(buf, buflen + entry);
+		if (nb == NULL) {
+			RPCAPD_LOG(ERR, "out of memory in findallif");
+			free(buf);
+			return rpcap_send_error(c, 0, "out of memory");
+		}
+		buf = nb;
+
+		struct rpcap_findalldevs_if iface = {
+			.namelen = rte_cpu_to_be_16(namelen),
+			.desclen = rte_cpu_to_be_16(desclen),
+			.flags = rte_cpu_to_be_32(PCAP_IF_UP | PCAP_IF_RUNNING),
+		};
+		memcpy(buf + buflen, &iface, sizeof(iface));
+		memcpy(buf + buflen + sizeof(iface), name, namelen);
+		memcpy(buf + buflen + sizeof(iface) + namelen, desc, desclen);
+		buflen += entry;
+		nif++;
+	}
+
+	RPCAPD_LOG(DEBUG, "findallif: %u interface(s)", nif);
+	rc = rpcap_send_msg(c, RPCAP_MSG_FINDALLIF_REPLY, nif, buf, buflen);
+	free(buf);
+	return rc;
+}
+
+/*
+ * AUTH_REQ: check the authentication type only.
+ *
+ * There is no credential store, so a username and password cannot be
+ * verified; refuse them rather than reply that they were accepted.
+ */
+int
+handle_auth(const struct conn *c, uint32_t plen)
+{
+	struct rpcap_auth auth;
+	uint16_t type;
+
+	if (plen < sizeof(auth)) {
+		rpcap_discard(c, plen);
+		return rpcap_send_error(c, 0, "short authentication request");
+	}
+
+	if (recv_full(c, &auth, sizeof(auth)) < 0)
+		return -1;
+
+	/* Discard any username and password that followed. */
+	if (rpcap_discard(c, plen - sizeof(auth)) < 0)
+		return -1;
+
+	type = rte_be_to_cpu_16(auth.type);
+	if (type != RPCAP_RMTAUTH_NULL) {
+		RPCAPD_LOG(NOTICE, "rejecting authentication type %u", type);
+		return rpcap_send_error(c, PCAP_ERR_AUTH_TYPE_NOTSUP,
+					"this server cannot check credentials; "
+					"connect without a username or password");
+	}
+
+	return rpcap_send_msg(c, RPCAP_MSG_AUTH_REPLY, 0, NULL, 0);
+}
+
+/* OPEN_REQ: payload is the interface name (no NUL). */
+int
+handle_open(const struct conn *c, uint32_t plen, struct session *s)
+{
+	struct rpcap_openreply reply = {
+		.linktype = rte_cpu_to_be_32(DLT_EN10MB),
+	};
+	uint16_t port;
+
+	stop_capture(s);
+
+	if (plen >= sizeof(s->name)) {
+		rpcap_discard(c, plen);
+		return rpcap_send_error(c, 0, "interface name too long");
+	}
+	if (recv_full(c, s->name, plen) < 0)
+		return -1;
+	s->name[plen] = '\0';
+
+	if (rte_eth_dev_get_port_by_name(s->name, &port) < 0) {
+		RPCAPD_LOG(WARNING, "open: no such port '%s'", s->name);
+		/* s->name has already been overwritten; make sure a later
+		 * STARTCAP cannot capture the previously opened port.
+		 */
+		s->opened = false;
+		return rpcap_send_error(c, 0, "unknown interface");
+	}
+	s->port = port;
+	s->opened = true;
+
+	RPCAPD_LOG(DEBUG, "open: '%s' -> dpdk port %u", s->name, port);
+	return rpcap_send_msg(c, RPCAP_MSG_OPEN_REPLY, 0, &reply, sizeof(reply));
+}
diff --git a/app/rpcapd/sock.c b/app/rpcapd/sock.c
new file mode 100644
index 0000000000..179c1a31c1
--- /dev/null
+++ b/app/rpcapd/sock.c
@@ -0,0 +1,315 @@
+/* SPDX-License-Identifier: BSD-3-Clause
+ * Copyright(c) 2026 Stephen Hemminger
+ *
+ * Socket helpers and rpcap message framing, used by both the control
+ * connection and the data connection.
+ */
+
+#include <errno.h>
+#include <netdb.h>
+#include <netinet/in.h>
+#include <poll.h>
+#include <stdbool.h>
+#include <string.h>
+#include <sys/socket.h>
+#include <sys/uio.h>
+#include <time.h>
+#include <unistd.h>
+
+#include <rte_byteorder.h>
+#include <rte_common.h>
+#include <rte_stdatomic.h>
+
+#include "rpcap-protocol.h"
+#include "rpcapd.h"
+
+#define POLL_INTERVAL_MS              500
+
+/* Monotonic milliseconds, for timing out across repeated waits. */
+static int64_t
+get_monotonic_ms(void)
+{
+	struct timespec ts;
+
+	clock_gettime(CLOCK_MONOTONIC, &ts);
+	return (int64_t)ts.tv_sec * 1000 + ts.tv_nsec / 1000000;
+}
+
+void
+set_sockaddr_port(struct sockaddr_storage *ss, uint16_t port)
+{
+	if (ss->ss_family == AF_INET6)
+		((struct sockaddr_in6 *)ss)->sin6_port = htons(port);
+	else
+		((struct sockaddr_in *)ss)->sin_port = htons(port);
+}
+
+uint16_t
+get_sockaddr_port(const struct sockaddr_storage *ss)
+{
+	if (ss->ss_family == AF_INET6)
+		return ntohs(((const struct sockaddr_in6 *)ss)->sin6_port);
+	return ntohs(((const struct sockaddr_in *)ss)->sin_port);
+}
+
+
+/* Wait for a connection to become readable with timeout */
+int
+wait_readable(const struct conn *c, int timeout_ms)
+{
+	struct pollfd pfd = { .fd = c->fd, .events = POLLIN };
+
+	while (!rte_atomic_load_explicit(&quit_signal, rte_memory_order_relaxed)) {
+		int wait_ms = POLL_INTERVAL_MS;
+		int rc;
+
+		if (timeout_ms >= 0) {
+			if (timeout_ms == 0)
+				return 0;
+			if (timeout_ms < wait_ms)
+				wait_ms = timeout_ms;
+			timeout_ms -= wait_ms;
+		}
+
+		rc = poll(&pfd, 1, wait_ms);
+		if (rc < 0) {
+			if (errno == EINTR)
+				continue;
+			RPCAPD_LOG(ERR, "poll failed: %s", strerror(errno));
+			return -1;
+		}
+		if (rc > 0)
+			return 1;
+	}
+	return -1;
+}
+
+/* accept() with a timeout, so a stalled client cannot wedge the daemon. */
+int
+accept_timeout(int listen_fd, int timeout_ms)
+{
+	struct conn listener = { .fd = listen_fd };
+	int fd;
+
+	switch (wait_readable(&listener, timeout_ms)) {
+	case 1:
+		break;
+	case 0:
+		RPCAPD_LOG(ERR, "timed out waiting for data connection");
+		return -1;
+	default:
+		return -1;
+	}
+
+	fd = accept(listen_fd, NULL, NULL);
+	if (fd < 0)
+		RPCAPD_LOG(ERR, "accept: %s", strerror(errno));
+	return fd;
+}
+
+/* Compare the host part of two addresses, ignoring the port: the data
+ * connection comes from an ephemeral port, not the control one.
+ */
+static bool
+same_host(const struct sockaddr_storage *a, const struct sockaddr_storage *b)
+{
+	if (a->ss_family != b->ss_family)
+		return false;
+
+	if (a->ss_family == AF_INET) {
+		const struct sockaddr_in *sa = (const void *)a;
+		const struct sockaddr_in *sb = (const void *)b;
+
+		return sa->sin_addr.s_addr == sb->sin_addr.s_addr;
+	}
+	if (a->ss_family == AF_INET6) {
+		const struct sockaddr_in6 *sa = (const void *)a;
+		const struct sockaddr_in6 *sb = (const void *)b;
+
+		return IN6_ARE_ADDR_EQUAL(&sa->sin6_addr, &sb->sin6_addr);
+	}
+	return false;
+}
+
+/*
+ * Accept a data connection only from the control connection's peer;
+ * the port is handed to the client in the clear, so any local user
+ * could otherwise race for the stream.  A mismatch is rejected and the
+ * wait continues.
+ */
+int
+accept_from(int listen_fd, const struct sockaddr_storage *want, int timeout_ms)
+{
+	struct conn listener = { .fd = listen_fd };
+	int remaining = timeout_ms;
+
+	while (!rte_atomic_load_explicit(&quit_signal, rte_memory_order_relaxed)) {
+		struct sockaddr_storage peer;
+		socklen_t peerlen = sizeof(peer);
+		char host[NI_MAXHOST] = "?";
+		int64_t start, waited;
+		int fd;
+
+		start = get_monotonic_ms();
+		switch (wait_readable(&listener, remaining)) {
+		case 1:
+			break;
+		case 0:
+			RPCAPD_LOG(ERR, "timed out waiting for data connection");
+			return -1;
+		default:
+			return -1;
+		}
+
+		fd = accept(listen_fd, (struct sockaddr *)&peer, &peerlen);
+		if (fd < 0) {
+			if (errno == EINTR || errno == ECONNABORTED)
+				goto next;
+			RPCAPD_LOG(ERR, "accept: %s", strerror(errno));
+			return -1;
+		}
+
+		if (same_host(&peer, want))
+			return fd;
+
+		getnameinfo((struct sockaddr *)&peer, peerlen,
+			    host, sizeof(host), NULL, 0, NI_NUMERICHOST);
+		RPCAPD_LOG(WARNING,
+			   "rejected data connection from %s: does not match control peer",
+			   host);
+		close(fd);
+next:
+		if (remaining >= 0) {
+			waited = get_monotonic_ms() - start;
+			remaining -= (waited > 0) ? (int)waited : 0;
+			if (remaining <= 0) {
+				RPCAPD_LOG(ERR,
+					   "timed out waiting for data connection");
+				return -1;
+			}
+		}
+	}
+	return -1;
+}
+
+/* Read exactly len bytes; return 0 on success, -1 on error or EOF. */
+int
+recv_full(const struct conn *c, void *buf, size_t len)
+{
+	uint8_t *p = buf;
+
+	while (len > 0) {
+		ssize_t n;
+
+		/* Timed wait, so a quit signal or a dead primary is acted
+		 * on promptly.
+		 */
+		if (wait_readable(c, -1) != 1)
+			return -1;
+
+		n = recv(c->fd, p, len, 0);
+		if (n < 0 && errno == EINTR)
+			continue;
+
+		if (n <= 0)
+			return -1;
+
+		p += n;
+		len -= n;
+	}
+	return 0;
+}
+
+/*
+ * Send all of iov, resending the remainder if sendmsg() reports a short
+ * count (possible when the connection breaks or a signal arrives after
+ * some bytes were copied).  Consumes iov, so pass a scratch copy.
+ */
+int
+send_iov_full(const struct conn *c, struct iovec *iov, int iovcnt, int flags)
+{
+	struct msghdr msg = {
+		.msg_iov    = iov,
+		.msg_iovlen = iovcnt,
+	};
+
+	while (msg.msg_iovlen > 0) {
+		ssize_t n = sendmsg(c->fd, &msg, flags | MSG_NOSIGNAL);
+
+		if (n < 0) {
+			/*
+			 * Send blocks rather than polling first; the data
+			 * socket has a send timeout so a client that stops
+			 * reading fails with EAGAIN.
+			 */
+			if (errno == EINTR &&
+			    !rte_atomic_load_explicit(&quit_signal,
+						      rte_memory_order_relaxed))
+				continue;
+			return -1;
+		}
+		if (n == 0)
+			return -1;
+
+		/* Drop whole iovecs that were fully sent, then trim the
+		 * partially sent one.
+		 */
+		while (msg.msg_iovlen > 0 && (size_t)n >= msg.msg_iov->iov_len) {
+			n -= msg.msg_iov->iov_len;
+			msg.msg_iov++;
+			msg.msg_iovlen--;
+		}
+		if (n > 0) {
+			msg.msg_iov->iov_base = (char *)msg.msg_iov->iov_base + n;
+			msg.msg_iov->iov_len -= n;
+		}
+	}
+	return 0;
+}
+
+int
+rpcap_send_msg(const struct conn *c, uint8_t type, uint16_t value,
+	       const void *payload, uint32_t plen)
+{
+	struct rpcap_header hdr = {
+		.ver = RPCAP_VERSION,
+		.type = type,
+		.value = rte_cpu_to_be_16(value),
+		.plen = rte_cpu_to_be_32(plen),
+	};
+	struct iovec iov[2] = {
+		{
+			.iov_base = &hdr,
+			.iov_len = sizeof(hdr),
+		},
+		{
+			.iov_base = (void *)(uintptr_t)payload,
+			.iov_len = plen,
+		},
+	};
+
+	return send_iov_full(c, iov, plen > 0 ? 2 : 1, 0);
+}
+
+int
+rpcap_send_error(const struct conn *c, uint16_t errcode, const char *msg)
+{
+	RPCAPD_LOG(WARNING, "sending error to client: %s", msg);
+	return rpcap_send_msg(c, RPCAP_MSG_ERROR, errcode, msg, strlen(msg));
+}
+
+/* Throw away plen bytes of payload we don't care about. */
+int
+rpcap_discard(const struct conn *c, uint32_t plen)
+{
+	uint8_t buf[256];
+
+	while (plen > 0) {
+		size_t chunk = plen > sizeof(buf) ? sizeof(buf) : plen;
+
+		if (recv_full(c, buf, chunk) < 0)
+			return -1;
+		plen -= chunk;
+	}
+	return 0;
+}
diff --git a/doc/guides/rel_notes/release_26_11.rst b/doc/guides/rel_notes/release_26_11.rst
index 5b5a9f006e..8e107b48b6 100644
--- a/doc/guides/rel_notes/release_26_11.rst
+++ b/doc/guides/rel_notes/release_26_11.rst
@@ -143,6 +143,11 @@ New Features
   Added ``rte_bbdev_queue_stats_get()`` function to retrieve statistics
   for a specific queue, complementing the existing device-level statistics API.
 
+* **Added libpcap remote capture daemon.**
+
+  Added the ``dpdk-rpcapd`` application, which implements the rpcap
+  protocol to allow live capture in tcpdump and Wireshark.
+
 
 Removed Items
 -------------
diff --git a/doc/guides/tools/index.rst b/doc/guides/tools/index.rst
index 13f75a5bc6..a23333f763 100644
--- a/doc/guides/tools/index.rst
+++ b/doc/guides/tools/index.rst
@@ -13,6 +13,7 @@ DPDK Tools User Guides
     proc_info
     pmdinfo
     dumpcap
+    rpcapd
     pdump
     telemetrywatcher
     dmaperf
diff --git a/doc/guides/tools/rpcapd.rst b/doc/guides/tools/rpcapd.rst
new file mode 100644
index 0000000000..a8b026a409
--- /dev/null
+++ b/doc/guides/tools/rpcapd.rst
@@ -0,0 +1,199 @@
+..  SPDX-License-Identifier: BSD-3-Clause
+    Copyright(c) 2026 Stephen Hemminger
+
+.. _rpcapd_tool:
+
+dpdk-rpcapd Application
+=======================
+
+The ``dpdk-rpcapd`` application is a Data Plane Development Kit
+(DPDK) implementation of the remote packet capture daemon protocol
+(``rpcap``) used by libpcap.  It runs as a DPDK secondary process and
+allows libpcap-aware tools such as ``tcpdump`` and Wireshark to capture
+packets from a DPDK primary process live, without writing to an
+intermediate file.
+
+The ``dpdk-rpcapd`` tool implements a subset of the protocol spoken by
+the libpcap project's ``rpcapd``.
+See
+https://github.com/the-tcpdump-group/libpcap/tree/master/rpcapd
+for the reference implementation.
+Clients connect to ``dpdk-rpcapd`` using a ``rpcap://`` URL,
+request the list of available interfaces(which are the ports of the DPDK primary),
+open one, and stream packets from it.
+
+.. warning::
+
+   ``dpdk-rpcapd`` listens on an unauthenticated, unencrypted TCP port
+   (default 2002).  Anyone able to reach the port can list DPDK ports
+   and capture all traffic flowing through them.  The default bind
+   address is ``127.0.0.1``, so the listener is not reachable from
+   other hosts; overriding this with ``--bind`` exposes captured
+   traffic to anyone who can reach that address.  **Do not run
+   ``dpdk-rpcapd`` on a production system.**
+
+
+Running the Application
+-----------------------
+
+The application has a small set of command-line options:
+
+*   ``-p <port>``, ``--port <port>``
+
+    TCP port to listen on.  Default is 2002, the IANA-assigned rpcap
+    port.
+
+*   ``-b <addr>``, ``--bind <addr>``
+
+    Numeric IPv4 or IPv6 address to bind the listener to.  Default is
+    ``127.0.0.1``, or ``::1`` when ``-6`` is given (loopback only).
+    See the warning above before using any other address.
+
+*   ``-4``
+
+    Use only IPv4; an IPv6 argument to ``-b`` is rejected.
+
+*   ``-6``
+
+    Use only IPv6; an IPv4 argument to ``-b`` is rejected.  The default
+    bind address becomes ``::1``.
+
+*   ``-N <ring_size>``
+
+    Size of the per-session capture ring in packets.  Default is 2048.
+    Rounded up to a power of two if necessary.
+
+*   ``-D``, ``--debug``
+
+    Increase log verbosity.  A single ``-D`` adds informational
+    messages; ``-DD`` adds per-request protocol detail.
+
+*   ``--debug-file <file>``
+
+    Append log output to ``<file>`` instead of writing it to standard
+    error.
+
+*   ``--send-timeout <seconds>``
+
+    How long a send on the data connection may block before the client
+    is treated as dead and the capture stopped.  Default is 10 seconds;
+    zero waits forever.
+
+*   ``--lcore <core>``
+
+    CPU core to run on.  By default the daemon runs as an ordinary
+    process on any non-isolated CPU.
+
+*   ``--file-prefix <prefix>``
+
+    EAL file prefix of the primary process to attach to.  Needed when
+    the primary was started with a non-default prefix.
+
+*   ``--version``
+
+    Print the version and exit.
+
+*   ``-h``, ``--help``
+
+    Print usage and exit.
+
+EAL options are supplied automatically; the application runs as a
+secondary process and does not need EAL options on its command line for
+typical use.
+
+
+Client Setup
+------------
+
+Most Linux distributions ship libpcap built without ``rpcap`` support,
+since ``--enable-remote`` is off by default.  To use ``dpdk-rpcapd``
+from ``tcpdump`` or Wireshark on Linux, rebuild libpcap with it:
+
+.. code-block:: console
+
+    wget https://www.tcpdump.org/release/libpcap-1.10.7.tar.xz
+    tar xf libpcap-1.10.7.tar.xz
+    cd libpcap-1.10.7
+    ./configure --enable-remote
+    make
+    sudo make install
+
+Only the client side of ``rpcap`` is used for ``dpdk-rpcapd``.
+Do not run libpcap's version of ``rpcapd``.
+
+``tcpdump`` rebuilt against this libpcap can be used as a client without
+further changes.  Wireshark on Windows and macOS ships with rpcap support
+enabled by default.
+
+
+Example
+-------
+
+Start a primary application with the packet capture framework
+initialized.  ``dpdk-testpmd`` is the simplest:
+
+.. code-block:: console
+
+    sudo ./<build_dir>/app/dpdk-testpmd --vdev=net_tap0 -- -i
+
+In another window, start ``dpdk-rpcapd``:
+
+.. code-block:: console
+
+    sudo ./<build_dir>/app/dpdk-rpcapd
+    RPCAPD: open_listen_socket(): listening on 127.0.0.1 port 2002
+
+In a third window, list available interfaces using a libpcap-based
+``tcpdump`` rebuilt with remote support:
+
+.. code-block:: console
+
+    sudo /usr/local/sbin/tcpdump --list-remote-interfaces=rpcap://localhost:2002/
+    rpcap://localhost:2002/net_tap0  Network adapter 'DPDK port' on remote node localhost
+
+Capture live from a port:
+
+.. code-block:: console
+
+    sudo /usr/local/sbin/tcpdump -i rpcap://localhost:2002/net_tap0 -nn -c 20
+
+Or save to a file readable by any pcap consumer:
+
+.. code-block:: console
+
+    sudo /usr/local/sbin/tcpdump -i rpcap://localhost:2002/net_tap0 -w /tmp/capture.pcap
+
+
+Limitations
+-----------
+
+The following features of the reference ``rpcapd`` are not implemented
+in this initial version:
+
+*   **Single client.** Only one client may be connected at a time.
+    Subsequent clients are queued by the listening socket but not
+    serviced until the first disconnects.
+
+*   **No authentication.** Password authentication is refused with
+    ``PCAP_ERR_AUTH_TYPE_NOTSUP``; clients must connect without
+    credentials, which is what a ``rpcap://`` URL with no userinfo does.
+    With the default loopback bind, reaching the port already requires
+    an account on the host.
+
+*   **No TLS.** The ``-S`` option of the reference ``rpcapd`` is not
+    implemented, so the connection is always in the clear.  This is
+    reasonable for the default loopback bind, where the traffic never
+    leaves the host, but means ``--bind`` to any other address sends
+    captured packets over the network unencrypted.
+
+*   **TCP data transport only.** A client requesting UDP is refused.
+
+
+See Also
+--------
+
+*   :doc:`dumpcap` -- file-based capture writing pcapng
+    output.
+
+*   The libpcap project's ``rpcapd`` reference implementation:
+    https://github.com/the-tcpdump-group/libpcap/tree/master/rpcapd
-- 
2.53.0


  parent reply	other threads:[~2026-10-01  2:59 UTC|newest]

Thread overview: 21+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-08 21:07 [PATCH] examples/rpcapd: demo version of packet capture daemon Stephen Hemminger
2026-09-20 18:59 ` [PATCH v2] " Stephen Hemminger
2026-09-21  9:51   ` Marat Khalili
2026-09-21 15:57     ` Stephen Hemminger
2026-09-21 15:58     ` Stephen Hemminger
2026-09-21 16:43       ` Marat Khalili
2026-09-21 17:31         ` Stephen Hemminger
2026-09-21 17:53           ` Marat Khalili
2026-09-21 16:17     ` Stephen Hemminger
2026-09-22 18:45   ` Stephen Hemminger
2026-09-22 21:31 ` [PATCH v3] " Stephen Hemminger
2026-09-28 16:18   ` Marat Khalili
2026-09-28 17:24     ` Stephen Hemminger
2026-10-01  2:35 ` [PATCH v4 0/4] add rpcap remote " Stephen Hemminger
2026-10-01  2:35   ` [PATCH v4 1/4] pcapng: add API to read back capture mbuf header Stephen Hemminger
2026-10-01  2:35   ` Stephen Hemminger [this message]
2026-10-01 18:50     ` [PATCH v4 2/4] app/rpcapd: remote pcap daemon Marat Khalili
2026-10-01  2:35   ` [PATCH v4 3/4] app/rpcapd: add TLS support Stephen Hemminger
2026-10-01  2:35   ` [PATCH v4 4/4] app/rpcapd: add host list option Stephen Hemminger
2026-10-01 18:50   ` [PATCH v4 0/4] add rpcap remote capture daemon Marat Khalili
2026-10-01 23:00     ` Stephen Hemminger

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261001025853.319860-3-stephen@networkplumber.org \
    --to=stephen@networkplumber.org \
    --cc=dev@dpdk.org \
    --cc=reshma.pattan@intel.com \
    --cc=thomas@monjalon.net \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox