Netdev List
 help / color / mirror / Atom feed
From: Koichiro Den <den@valinux.co.jp>
To: Jon Mason <jdmason@kudzu.us>, Dave Jiang <dave.jiang@intel.com>,
	Allen Hubbe <allenbh@gmail.com>,
	Andrew Lunn <andrew+netdev@lunn.ch>,
	"David S. Miller" <davem@davemloft.net>,
	Eric Dumazet <edumazet@google.com>,
	Jakub Kicinski <kuba@kernel.org>, Paolo Abeni <pabeni@redhat.com>
Cc: ntb@lists.linux.dev, netdev@vger.kernel.org,
	linux-kernel@vger.kernel.org
Subject: [PATCH net-next v4 10/10] net: ntb_netdev: Preserve CHECKSUM_PARTIAL across NTB
Date: Mon, 14 Sep 2026 17:48:38 +0900	[thread overview]
Message-ID: <20260914084838.2158249-11-den@valinux.co.jp> (raw)
In-Reply-To: <20260914084838.2158249-1-den@valinux.co.jp>

Calculating L4 checksums can limit ntb_netdev throughput especially on
embedded systems, where CPU resources are often limited. On a trusted PCIe
fabric, we can skip that work.

Use the optional ntb_netdev header to carry CHECKSUM_PARTIAL with its
csum_start and csum_offset.

Exchange receive support at QP link-up and fall back to software if the
peer does not advertise it. This preserves netdev checksum semantics and
interoperability with existing transport version 4 peers.

Keep NETIF_F_RXCSUM unchanged while the interface is up so peers can
rely on the negotiated receive support. Drop partial-checksum frames
when it is disabled.

Leave the TX and RX checksum features disabled by default. Users can
just enable them explicitly for links they trust for lower CPU usage
and/or higher throughput.

Signed-off-by: Koichiro Den <den@valinux.co.jp>
---
Changes in v4:
  - Move offload metadata into an ntb_netdev header (Jakub)
  - Exchange receive support at QP link-up, not on RX (Sashiko)
  - Keep peer checksum state per QP (Sashiko)
  - Advertise receive support only with RXCSUM enabled. Drop
    partial-checksum frames when it is disabled (Sashiko)
  - Keep RXCSUM unchanged while the interface is up

 drivers/net/ntb_netdev.c | 100 ++++++++++++++++++++++++++++++++++++---
 1 file changed, 93 insertions(+), 7 deletions(-)

diff --git a/drivers/net/ntb_netdev.c b/drivers/net/ntb_netdev.c
index 67cfe0f1a49a..d889c8538be0 100644
--- a/drivers/net/ntb_netdev.c
+++ b/drivers/net/ntb_netdev.c
@@ -4,6 +4,7 @@
  */
 #include <linux/etherdevice.h>
 #include <linux/ethtool.h>
+#include <linux/if_vlan.h>
 #include <linux/module.h>
 #include <linux/pci.h>
 #include <linux/ntb.h>
@@ -29,12 +30,19 @@ static unsigned int tx_stop = 5;
 #define NTB_NETDEV_MAX_QUEUES		64
 #define NTB_NETDEV_DEFAULT_QUEUES	1
 
+#define NTB_NETDEV_CAP_CSUM		BIT(0)
+
 /* An ntb_netdev_hdr precedes the packet. */
 #define NTB_NETDEV_META_HDR		BIT(0)
 
+#define NTB_NETDEV_HDR_F_CSUM		BIT(0)
+
 struct ntb_netdev_hdr {
 	__le16 len;		/* Header length in bytes, a multiple of 2. */
 	__le16 flags;
+	/* From the packet start, excluding this header. */
+	__le16 csum_start;
+	__le16 csum_offset;
 };
 
 struct ntb_netdev;
@@ -44,6 +52,7 @@ struct ntb_netdev_queue {
 	struct ntb_transport_qp *qp;
 	struct timer_list tx_timer;
 	u16 qid;
+	bool peer_csum;
 };
 
 struct ntb_netdev {
@@ -117,6 +126,7 @@ static void ntb_netdev_event_handler(void *data, int link_is_up, u32 peer_caps)
 	struct net_device *ndev;
 
 	ndev = dev->ndev;
+	WRITE_ONCE(q->peer_csum, link_is_up && (peer_caps & NTB_NETDEV_CAP_CSUM));
 
 	netdev_dbg(ndev, "Event %x, Link %x, qp %u\n", link_is_up,
 		   ntb_transport_link_query(q->qp), q->qid);
@@ -189,16 +199,29 @@ static void ntb_netdev_rx_handler(struct ntb_transport_qp *qp, void *qp_data,
 
 	skb_put(skb, len + hdr_len);
 	if (hdr) {
+		u16 offset = le16_to_cpu(hdr->csum_offset);
+		u16 start = le16_to_cpu(hdr->csum_start);
 		u16 flags = le16_to_cpu(hdr->flags);
 
 		skb_pull(skb, hdr_len);
-		if (flags)
+		if (flags & ~NTB_NETDEV_HDR_F_CSUM)
 			goto rx_drop;
+
+		if (flags & NTB_NETDEV_HDR_F_CSUM) {
+			if (!(ndev->features & NETIF_F_RXCSUM)) {
+				ntb_netdev_rx_stats_add(ndev, len);
+				DEV_STATS_INC(ndev, rx_dropped);
+				goto rx_free;
+			}
+
+			if (start < ETH_HLEN ||
+			    !skb_partial_csum_set(skb, start, offset))
+				goto rx_drop;
+		}
 	}
 
 	ntb_netdev_rx_stats_add(ndev, len);
 	skb->protocol = eth_type_trans(skb, ndev);
-	skb->ip_summed = CHECKSUM_NONE;
 	skb_record_rx_queue(skb, q->qid);
 
 	netif_rx(skb);
@@ -216,6 +239,7 @@ static void ntb_netdev_rx_handler(struct ntb_transport_qp *qp, void *qp_data,
 
 rx_drop:
 	DEV_STATS_INC(ndev, rx_errors);
+rx_free:
 	dev_kfree_skb_any(skb);
 	skb = new_skb;
 	goto enqueue_again;
@@ -316,6 +340,8 @@ static netdev_tx_t ntb_netdev_start_xmit(struct sk_buff *skb,
 	struct ntb_netdev *dev = netdev_priv(ndev);
 	u16 qid = skb_get_queue_mapping(skb);
 	struct ntb_netdev_queue *q;
+	unsigned int hdr_len = 0;
+	unsigned int meta = 0;
 	int rc;
 
 	q = &dev->queues[qid];
@@ -323,7 +349,29 @@ static netdev_tx_t ntb_netdev_start_xmit(struct sk_buff *skb,
 	if (unlikely(ntb_netdev_maybe_stop_tx(ndev, q, tx_stop)))
 		return NETDEV_TX_BUSY;
 
-	rc = ntb_transport_tx_enqueue(q->qp, skb, skb->data, skb->len, 0);
+	if (skb->ip_summed == CHECKSUM_PARTIAL) {
+		if (READ_ONCE(q->peer_csum)) {
+			struct ntb_netdev_hdr hdr = {
+				.len = cpu_to_le16(sizeof(hdr)),
+				.flags = cpu_to_le16(NTB_NETDEV_HDR_F_CSUM),
+			};
+
+			if (skb_cow_head(skb, sizeof(hdr)))
+				goto drop;
+
+			hdr.csum_start = cpu_to_le16(skb_checksum_start_offset(skb));
+			hdr.csum_offset = cpu_to_le16(skb->csum_offset);
+			hdr_len = sizeof(hdr);
+			/* Keep skb->len unchanged for retries and byte accounting. */
+			memcpy(skb->data - hdr_len, &hdr, hdr_len);
+			meta = NTB_NETDEV_META_HDR;
+		} else if (skb_checksum_help(skb)) {
+			goto drop;
+		}
+	}
+
+	rc = ntb_transport_tx_enqueue(q->qp, skb, skb->data - hdr_len,
+				      skb->len + hdr_len, meta);
 	if (rc) {
 		if (rc == -EAGAIN || rc == -EBUSY) {
 			netif_stop_subqueue(ndev, q->qid);
@@ -346,6 +394,28 @@ static netdev_tx_t ntb_netdev_start_xmit(struct sk_buff *skb,
 	return NETDEV_TX_OK;
 }
 
+static netdev_features_t ntb_netdev_features_check(struct sk_buff *skb,
+						   struct net_device *ndev,
+						   netdev_features_t features)
+{
+	if (skb->ip_summed == CHECKSUM_PARTIAL &&
+	    skb_checksum_start_offset(skb) < ETH_HLEN)
+		features &= ~NETIF_F_CSUM_MASK;
+
+	return vlan_features_check(skb, features);
+}
+
+static netdev_features_t ntb_netdev_fix_features(struct net_device *ndev,
+						 netdev_features_t features)
+{
+	/* RX checksum support is exchanged at link-up. */
+	if (netif_running(ndev))
+		features = (features & ~NETIF_F_RXCSUM) |
+			   (ndev->features & NETIF_F_RXCSUM);
+
+	return features;
+}
+
 static void ntb_netdev_tx_timer(struct timer_list *t)
 {
 	struct ntb_netdev_queue *q = timer_container_of(q, t, tx_timer);
@@ -369,6 +439,16 @@ static void ntb_netdev_tx_timer(struct timer_list *t)
 	}
 }
 
+static void ntb_netdev_link_up(struct ntb_netdev_queue *q)
+{
+	u32 caps = 0;
+
+	WRITE_ONCE(q->peer_csum, false);
+	if (q->ntdev->ndev->features & NETIF_F_RXCSUM)
+		caps = NTB_NETDEV_CAP_CSUM;
+	ntb_transport_link_up(q->qp, caps);
+}
+
 static int ntb_netdev_open(struct net_device *ndev)
 {
 	struct ntb_netdev *dev = netdev_priv(ndev);
@@ -391,7 +471,7 @@ static int ntb_netdev_open(struct net_device *ndev)
 	netif_tx_stop_all_queues(ndev);
 
 	for (q = 0; q < dev->num_queues; q++)
-		ntb_transport_link_up(dev->queues[q].qp, 0);
+		ntb_netdev_link_up(&dev->queues[q]);
 
 	return 0;
 
@@ -420,6 +500,9 @@ static int ntb_netdev_close(struct net_device *ndev)
 		timer_delete_sync(&queue->tx_timer);
 	}
 
+	/* Apply RX checksum changes deferred while the interface was up. */
+	netdev_update_features(ndev);
+
 	return 0;
 }
 
@@ -475,7 +558,7 @@ static int ntb_netdev_change_mtu(struct net_device *ndev, int new_mtu)
 	WRITE_ONCE(ndev->mtu, new_mtu);
 
 	for (q = 0; q < dev->num_queues; q++)
-		ntb_transport_link_up(dev->queues[q].qp, 0);
+		ntb_netdev_link_up(&dev->queues[q]);
 
 	return 0;
 
@@ -496,6 +579,8 @@ static const struct net_device_ops ntb_netdev_ops = {
 	.ndo_open = ntb_netdev_open,
 	.ndo_stop = ntb_netdev_close,
 	.ndo_start_xmit = ntb_netdev_start_xmit,
+	.ndo_features_check = ntb_netdev_features_check,
+	.ndo_fix_features = ntb_netdev_fix_features,
 	.ndo_change_mtu = ntb_netdev_change_mtu,
 	.ndo_set_mac_address = eth_mac_addr,
 };
@@ -583,7 +668,7 @@ static int ntb_inc_channels(struct net_device *ndev,
 
 	if (running)
 		for (q = old; q < new; q++)
-			ntb_transport_link_up(dev->queues[q].qp, 0);
+			ntb_netdev_link_up(&dev->queues[q]);
 
 	return 0;
 
@@ -716,7 +801,8 @@ static int ntb_netdev_probe(struct device *client_dev)
 
 	ndev->priv_flags |= IFF_LIVE_ADDR_CHANGE;
 
-	ndev->hw_features = ndev->features;
+	/* Checksum bypass assumes a trusted NTB link, so keep it opt-in. */
+	ndev->hw_features = ndev->features | NETIF_F_HW_CSUM | NETIF_F_RXCSUM;
 	ndev->watchdog_timeo = msecs_to_jiffies(NTB_TX_TIMEOUT_MS);
 
 	eth_random_addr(ndev->perm_addr);
-- 
2.51.0


  parent reply	other threads:[~2026-09-14  8:48 UTC|newest]

Thread overview: 23+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-14  8:48 [PATCH net-next v4 00/10] net: ntb_netdev: Preserve checksum offload across NTB Koichiro Den
2026-09-14  8:48 ` [PATCH net-next v4 01/10] NTB: ntb_transport: Order RX descriptor reads after completion Koichiro Den
2026-09-19  0:36   ` Joe Damato
2026-09-19 12:37     ` Koichiro Den
2026-09-14  8:48 ` [PATCH net-next v4 02/10] NTB: ntb_transport: Use little-endian shared fields Koichiro Den
2026-09-19  0:54   ` Joe Damato
2026-09-19 13:06     ` Koichiro Den
2026-09-14  8:48 ` [PATCH net-next v4 03/10] NTB: ntb_transport: Order RX entry completion Koichiro Den
2026-09-19  1:10   ` Joe Damato
2026-09-19 12:53     ` Koichiro Den
2026-09-14  8:48 ` [PATCH net-next v4 04/10] NTB: ntb_transport: Keep local QP link requests separate Koichiro Den
2026-09-17 20:49   ` netdev-bot+sashiko
2026-09-14  8:48 ` [PATCH net-next v4 05/10] NTB: ntb_transport: Exchange client capabilities at link-up Koichiro Den
2026-09-17 20:49   ` netdev-bot+sashiko
2026-09-14  8:48 ` [PATCH net-next v4 06/10] NTB: ntb_transport: Add per-payload client metadata Koichiro Den
2026-09-14  8:48 ` [PATCH net-next v4 07/10] net: ntb_netdev: Reject short RX frames Koichiro Den
2026-09-19  1:14   ` Joe Damato
2026-09-14  8:48 ` [PATCH net-next v4 08/10] net: ntb_netdev: Factor out RX statistics update Koichiro Den
2026-09-14  8:48 ` [PATCH net-next v4 09/10] net: ntb_netdev: Introduce an optional packet header, ntb_netdev_hdr Koichiro Den
2026-09-17 20:49   ` netdev-bot+sashiko
2026-09-14  8:48 ` Koichiro Den [this message]
2026-09-17 20:49   ` [PATCH net-next v4 10/10] net: ntb_netdev: Preserve CHECKSUM_PARTIAL across NTB netdev-bot+sashiko
2026-10-09  5:10 ` [PATCH net-next v4 00/10] net: ntb_netdev: Preserve checksum offload " Koichiro Den

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260914084838.2158249-11-den@valinux.co.jp \
    --to=den@valinux.co.jp \
    --cc=allenbh@gmail.com \
    --cc=andrew+netdev@lunn.ch \
    --cc=dave.jiang@intel.com \
    --cc=davem@davemloft.net \
    --cc=edumazet@google.com \
    --cc=jdmason@kudzu.us \
    --cc=kuba@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=netdev@vger.kernel.org \
    --cc=ntb@lists.linux.dev \
    --cc=pabeni@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox