Netdev List
 help / color / mirror / Atom feed
* [PATCH net-next 2/2] gue: Call remcsum_adjust
From: Tom Herbert @ 2014-11-25 19:21 UTC (permalink / raw)
  To: davem, netdev
In-Reply-To: <1416943280-28473-1-git-send-email-therbert@google.com>

Change remote checksum offload to call remcsum_adjust. This also
eliminates the optimization to skip an IP header as part of the
adjustment (really does not seem to be much of a win).

Signed-off-by: Tom Herbert <therbert@google.com>
---
 net/ipv4/fou.c | 84 ++++++++++++----------------------------------------------
 1 file changed, 17 insertions(+), 67 deletions(-)

diff --git a/net/ipv4/fou.c b/net/ipv4/fou.c
index 3dfe982..b986298 100644
--- a/net/ipv4/fou.c
+++ b/net/ipv4/fou.c
@@ -64,15 +64,13 @@ static int fou_udp_recv(struct sock *sk, struct sk_buff *skb)
 }
 
 static struct guehdr *gue_remcsum(struct sk_buff *skb, struct guehdr *guehdr,
-				  void *data, int hdrlen, u8 ipproto)
+				  void *data, size_t hdrlen, u8 ipproto)
 {
 	__be16 *pd = data;
-	u16 start = ntohs(pd[0]);
-	u16 offset = ntohs(pd[1]);
-	u16 poffset = 0;
-	u16 plen;
-	__wsum csum, delta;
-	__sum16 *psum;
+	size_t start = ntohs(pd[0]);
+	size_t offset = ntohs(pd[1]);
+	size_t plen = hdrlen + max_t(size_t, offset + sizeof(u16), start);
+	__wsum delta;
 
 	if (skb->remcsum_offload) {
 		/* Already processed in GRO path */
@@ -80,35 +78,15 @@ static struct guehdr *gue_remcsum(struct sk_buff *skb, struct guehdr *guehdr,
 		return guehdr;
 	}
 
-	if (start > skb->len - hdrlen ||
-	    offset > skb->len - hdrlen - sizeof(u16))
-		return NULL;
-
-	if (unlikely(skb->ip_summed != CHECKSUM_COMPLETE))
-		__skb_checksum_complete(skb);
-
-	plen = hdrlen + offset + sizeof(u16);
 	if (!pskb_may_pull(skb, plen))
 		return NULL;
 	guehdr = (struct guehdr *)&udp_hdr(skb)[1];
 
-	if (ipproto == IPPROTO_IP && sizeof(struct iphdr) < plen) {
-		struct iphdr *ip = (struct iphdr *)(skb->data + hdrlen);
-
-		/* If next header happens to be IP we can skip that for the
-		 * checksum calculation since the IP header checksum is zero
-		 * if correct.
-		 */
-		poffset = ip->ihl * 4;
-	}
-
-	csum = csum_sub(skb->csum, skb_checksum(skb, poffset + hdrlen,
-						start - poffset - hdrlen, 0));
+	if (unlikely(skb->ip_summed != CHECKSUM_COMPLETE))
+		__skb_checksum_complete(skb);
 
-	/* Set derived checksum in packet */
-	psum = (__sum16 *)(skb->data + hdrlen + offset);
-	delta = csum_sub(csum_fold(csum), *psum);
-	*psum = csum_fold(csum);
+	delta = remcsum_adjust((void *)guehdr + hdrlen,
+			       skb->csum, start, offset);
 
 	/* Adjust skb->csum since we changed the packet */
 	skb->csum = csum_add(skb->csum, delta);
@@ -158,9 +136,6 @@ static int gue_udp_recv(struct sock *sk, struct sk_buff *skb)
 
 	ip_hdr(skb)->tot_len = htons(ntohs(ip_hdr(skb)->tot_len) - len);
 
-	/* Pull UDP header now, skb->data points to guehdr */
-	__skb_pull(skb, sizeof(struct udphdr));
-
 	/* Pull csum through the guehdr now . This can be used if
 	 * there is a remote checksum offload.
 	 */
@@ -188,7 +163,7 @@ static int gue_udp_recv(struct sock *sk, struct sk_buff *skb)
 	if (unlikely(guehdr->control))
 		return gue_control_message(skb, guehdr);
 
-	__skb_pull(skb, hdrlen);
+	__skb_pull(skb, sizeof(struct udphdr) + hdrlen);
 	skb_reset_transport_header(skb);
 
 	return -guehdr->proto_ctype;
@@ -248,24 +223,17 @@ static struct guehdr *gue_gro_remcsum(struct sk_buff *skb, unsigned int off,
 				      size_t hdrlen, u8 ipproto)
 {
 	__be16 *pd = data;
-	u16 start = ntohs(pd[0]);
-	u16 offset = ntohs(pd[1]);
-	u16 poffset = 0;
-	u16 plen;
-	void *ptr;
-	__wsum csum, delta;
-	__sum16 *psum;
+	size_t start = ntohs(pd[0]);
+	size_t offset = ntohs(pd[1]);
+	size_t plen = hdrlen + max_t(size_t, offset + sizeof(u16), start);
+	__wsum delta;
 
 	if (skb->remcsum_offload)
 		return guehdr;
 
-	if (start > skb_gro_len(skb) - hdrlen ||
-	    offset > skb_gro_len(skb) - hdrlen - sizeof(u16) ||
-	    !NAPI_GRO_CB(skb)->csum_valid || skb->remcsum_offload)
+	if (!NAPI_GRO_CB(skb)->csum_valid)
 		return NULL;
 
-	plen = hdrlen + offset + sizeof(u16);
-
 	/* Pull checksum that will be written */
 	if (skb_gro_header_hard(skb, off + plen)) {
 		guehdr = skb_gro_header_slow(skb, off + plen, off);
@@ -273,26 +241,8 @@ static struct guehdr *gue_gro_remcsum(struct sk_buff *skb, unsigned int off,
 			return NULL;
 	}
 
-	ptr = (void *)guehdr + hdrlen;
-
-	if (ipproto == IPPROTO_IP &&
-	    (hdrlen + sizeof(struct iphdr) < plen)) {
-		struct iphdr *ip = (struct iphdr *)(ptr + hdrlen);
-
-		/* If next header happens to be IP we can skip
-		 * that for the checksum calculation since the
-		 * IP header checksum is zero if correct.
-		 */
-		poffset = ip->ihl * 4;
-	}
-
-	csum = csum_sub(NAPI_GRO_CB(skb)->csum,
-			csum_partial(ptr + poffset, start - poffset, 0));
-
-	/* Set derived checksum in packet */
-	psum = (__sum16 *)(ptr + offset);
-	delta = csum_sub(csum_fold(csum), *psum);
-	*psum = csum_fold(csum);
+	delta = remcsum_adjust((void *)guehdr + hdrlen,
+			       NAPI_GRO_CB(skb)->csum, start, offset);
 
 	/* Adjust skb->csum since we changed the packet */
 	skb->csum = csum_add(skb->csum, delta);
-- 
2.1.0.rc2.206.gedb03e5

^ permalink raw reply related

* [PATCH net-next 1/2] net: Add remcsum_adjust as common function for remote checksum offload
From: Tom Herbert @ 2014-11-25 19:21 UTC (permalink / raw)
  To: davem, netdev
In-Reply-To: <1416943280-28473-1-git-send-email-therbert@google.com>

This function does the work to update a checksum field as part of
remote checksum offload.

remcsum_adjust does the following:

1) Subtract out the calculated checksum from the beginning of the
   packet (ptr arg) to the start offset.
2) Adjust the checksum field indicated by offset based on the modified
   checksum value from above step.
3) Return the difference in the old checksum field value and the
   new one. The caller will use this to update skb->csum and NAPI csum.

Signed-off-by: Tom Herbert <therbert@google.com>
---
 include/net/checksum.h | 16 ++++++++++++++++
 1 file changed, 16 insertions(+)

diff --git a/include/net/checksum.h b/include/net/checksum.h
index 6465bae..e339a95 100644
--- a/include/net/checksum.h
+++ b/include/net/checksum.h
@@ -151,4 +151,20 @@ static inline void inet_proto_csum_replace2(__sum16 *sum, struct sk_buff *skb,
 				 (__force __be32)to, pseudohdr);
 }
 
+static inline __wsum remcsum_adjust(void *ptr, __wsum csum,
+				    int start, int offset)
+{
+	__sum16 *psum = (__sum16 *)(ptr + offset);
+	__wsum delta;
+
+	/* Subtract out checksum up to start */
+	csum = csum_sub(csum, csum_partial(ptr, start, 0));
+
+	/* Set derived checksum in packet */
+	delta = csum_sub(csum_fold(csum), *psum);
+	*psum = csum_fold(csum);
+
+	return delta;
+}
+
 #endif
-- 
2.1.0.rc2.206.gedb03e5

^ permalink raw reply related

* [PATCH net-next 0/2] gue: Generalize remote checksum offload
From: Tom Herbert @ 2014-11-25 19:21 UTC (permalink / raw)
  To: davem, netdev

The remote checksum offload is generalized by creating a common
function (remcsum_adjust) that does the work of modifying the
checksum in remote checksum offload. This function can be called
from normal or GRO path. GUE was modified to use this function.

Remote checksum offload is described in
https://tools.ietf.org/html/draft-herbert-remotecsumoffload-01

Tested by running 200 TCP_STREAM connections over GUE, did not see
any problems with remote checksum offload enabled.

Tom Herbert (2):
  net: Add remcsum_adjust as common function for remote checksum offload
  gue: Call remcsum_adjust

 include/net/checksum.h | 16 ++++++++++
 net/ipv4/fou.c         | 84 ++++++++++----------------------------------------
 2 files changed, 33 insertions(+), 67 deletions(-)

-- 
2.1.0.rc2.206.gedb03e5

^ permalink raw reply

* Re: [patch net-next v3 07/17] rocker: introduce rocker switch driver
From: Scott Feldman @ 2014-11-25 19:19 UTC (permalink / raw)
  To: David Laight
  Cc: Jiri Pirko, netdev@vger.kernel.org, davem@davemloft.net,
	nhorman@tuxdriver.com, andy@greyhouse.net, tgraf@suug.ch,
	dborkman@redhat.com, ogerlitz@mellanox.com, jesse@nicira.com,
	pshelar@nicira.com, azhou@nicira.com, ben@decadent.org.uk,
	stephen@networkplumber.org, jeffrey.t.kirsher@intel.com,
	vyasevic@redhat.com, xiyou.wangcong@gmail.com,
	john.r.fastabend@intel.com, edumazet@google.com, jhs@mojatatu.com,
	"f
In-Reply-To: <063D6719AE5E284EB5DD2968C1650D6D1C9FB50A@AcuExch.aculab.com>

On Tue, Nov 25, 2014 at 6:13 AM, David Laight <David.Laight@aculab.com> wrote:
> From: Jiri Pirko
>>
>> This patch introduces the first driver to benefit from the switchdev
>> infrastructure and to implement newly introduced switch ndos. This is a
>> driver for emulated switch chip implemented in qemu:
>> https://github.com/sfeldma/qemu-rocker/
>
> If this driver caller 'rocker' just to get the (bad) pun 'rocker switch'?
> IMHO A more descriptive name would be a lot better.

Sorry, it's the best we could do since qla3xxx and mlx4 and fm10k were
already taken.

^ permalink raw reply

* Re: [PATCH v4 0/7] kernel tinification: optionally compile out splice family of syscalls (splice, vmsplice, tee and sendfile)
From: Eric W. Biederman @ 2014-11-25 19:16 UTC (permalink / raw)
  To: David Miller
  Cc: josh-iaAMLnmF4UmaiuxdJuQwMA, rdunlap-wEGCiKHe2LqWVfeAwA7xHQ,
	pieter-qeJ+1H9vRZbz+pZb47iToQ,
	alexander.h.duyck-ral2JQCrhuEAvxtiuMwx3w,
	viro-RmSDqhL/yNMiFSDQTTA3OLVCufUGDwFn, ast-uqk4Ao+rVK5Wk0Htik3J/w,
	akpm-de/tnXTf+JLsfHDXvbKv3WD2FQJk+8+b,
	beber-2YnHqweIUXrk1uMJSBkQmQ,
	catalina.mocanu-Re5JQEeQqe8AvxtiuMwx3w,
	dborkman-H+wXaHxf7aLQT0dZR+AlfA, edumazet-hpIqsD4AKlfQT0dZR+AlfA,
	fabf-AgBVmzD5pcezQB+pC5nmwQ,
	fuse-devel-5NWGOfrQmneRv+LV9MX5uipxlwaOVQ5f,
	geert-Td1EMuHUCqxL1ZNQvxDV9g, hughd-hpIqsD4AKlfQT0dZR+AlfA,
	iulia.manda21-Re5JQEeQqe8AvxtiuMwx3w, JBeulich-IBi9RG/b67k,
	bfields-uC3wQj2KruNg9hUCZPvPmw, jlayton-vpEMnDpepFuMZCB2o+C8xQ,
	linux-api-u79uwXL29TY76Z2rM5mHXA,
	linux-fsdevel-u79uwXL29TY76Z2rM5mHXA,
	linux-kernel-u79uwXL29TY76Z2rM5mHXA,
	linux-nfs-u79uwXL29TY76Z2rM5mHXA, mcgrof-IBi9RG/b67k,
	mattst88-Re5JQEeQqe8AvxtiuMwx3w, mgorman-l3A5Bk7waGM,
	mst-H+wXaHxf7aLQT0dZR+AlfA, miklos-sUDqSbJrdHQHWmgEVkV9KA,
	netdev-u79uwXL29TY76Z2rM5mHXA, oleg-H+wXaHxf7aLQT0dZR+AlfA,
	Paul.Durrant-Sxgqhf6Nn4DQT0dZR+AlfA,
	paulmck-23VcF4HTsmIX0ybBhKVfKdBPR1lH4CV8,
	pefoley2-lY0TAiDIAFlBDgjK7y7TUQ, tgraf-G/eBtMaohhA,
	therbert-hpIqsD4AKlfQT0dZR+AlfA,
	trond.myklebust-7I+n7zu2hftEKMMhf/gKZA,
	willemb-hpIqsD4AKlfQT0dZR+AlfA,
	xiaoguangrong-23VcF4HTsmIX0ybBhKVfKdBPR1lH4CV8, zhe
In-Reply-To: <20141125.140441.401150380839514113.davem-fT/PcQaiUtIeIZ0/mPfg9Q@public.gmane.org>

David Miller <davem-fT/PcQaiUtIeIZ0/mPfg9Q@public.gmane.org> writes:

> From: josh-iaAMLnmF4UmaiuxdJuQwMA@public.gmane.org
> Date: Tue, 25 Nov 2014 10:53:10 -0800
>
>> It's not a "slippery slope"; it's been our standard practice for ages.
>
> We've never put an entire class of generic system calls behind
> a config option.

CONFIG_SYSVIPC has been in the kernel as long as I can remember.

I seem to remember a plan to remove that code once userspace had
finished migrating to more unixy interfaces to ipc.  But in 20 years
that migration does does not seem to have finished, or even look
like it ever will.

But if we started a slippery slope it was long long ago.

Eric
--
To unsubscribe from this list: send the line "unsubscribe linux-nfs" in
the body of a message to majordomo-u79uwXL29TY76Z2rM5mHXA@public.gmane.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html

^ permalink raw reply

* Re: [PATCH net] net/mlx4_core: Limit count field to 24 bits in qp_alloc_res
From: David Miller @ 2014-11-25 19:16 UTC (permalink / raw)
  To: ogerlitz; +Cc: netdev, matanb, amirv, jackm
In-Reply-To: <1416909271-28840-1-git-send-email-ogerlitz@mellanox.com>

From: Or Gerlitz <ogerlitz@mellanox.com>
Date: Tue, 25 Nov 2014 11:54:31 +0200

>  	case RES_OP_RESERVE:
> -		count = get_param_l(&in_param);
> +		count = get_param_l(&in_param) & 0xffffff;

I think if these high bits are set, you should be using the maximum
value rather then truncating.

^ permalink raw reply

* Re: [net PATCH 0/2] Fix outer UDP checksums for IPv6 VXLAN tunnels
From: David Miller @ 2014-11-25 19:13 UTC (permalink / raw)
  To: alexander.duyck; +Cc: netdev
In-Reply-To: <20141125035808.13612.52556.stgit@ahduyck-workstation.home>

From: alexander.duyck@gmail.com
Date: Mon, 24 Nov 2014 20:08:25 -0800

> In testing against an older kernel I found a couple issues in the IPv6
> VXLAN tunnel checksum logic for the outer UDP checksum.
> 
> First the default transitioned from using an outer checksum to not using
> one.  Second, sometime after that the checksum inputs were changed
> resulting the checksum not being correct if it were computed.
> 
> These two issues prevented a ping from the newer kernel to the older one.
> With these two changes applied I verified I was able to send traffic over
> the VXLAN tunnel to a link partner on an older kernel.
> 
> The boolean flip fix can be submitted for 3.17 stable as well since the
> patch that introduced the issue was included in that kernel.

Series applied and queued up for -stable, thanks.

^ permalink raw reply

* Re: [PATCH net-next] cxgb4/cxgb4vf/csiostor: Add T4/T5 PCI ID Table
From: David Miller @ 2014-11-25 19:07 UTC (permalink / raw)
  To: hariprasad
  Cc: netdev, linux-scsi, JBottomley, hch, leedom, anish, nirranjan,
	kumaras, praveenm, varun
In-Reply-To: <1416884638-7582-1-git-send-email-hariprasad@chelsio.com>

From: Hariprasad Shenai <hariprasad@chelsio.com>
Date: Tue, 25 Nov 2014 08:33:58 +0530

> Add a new file t4_pci_id_tbl.h that contains T4/T5 PCI ID Table so that for all
> drivers that uses T4/T5 PCI functions changes can be done in one place.
> 
> checkpatch.pl script reports following error, which if tried to fix ends up in
> compilation error.
...
> Signed-off-by: Hariprasad Shenai <hariprasad@chelsio.com>

Applied, thanks.

^ permalink raw reply

* Re: [PATCH v4 0/7] kernel tinification: optionally compile out splice family of syscalls (splice, vmsplice, tee and sendfile)
From: David Miller @ 2014-11-25 19:05 UTC (permalink / raw)
  To: tytso-3s7WtUTddSA
  Cc: paulmck-23VcF4HTsmIX0ybBhKVfKdBPR1lH4CV8,
	rdunlap-wEGCiKHe2LqWVfeAwA7xHQ, pieter-qeJ+1H9vRZbz+pZb47iToQ,
	josh-iaAMLnmF4UmaiuxdJuQwMA,
	alexander.h.duyck-ral2JQCrhuEAvxtiuMwx3w,
	viro-RmSDqhL/yNMiFSDQTTA3OLVCufUGDwFn, ast-uqk4Ao+rVK5Wk0Htik3J/w,
	akpm-de/tnXTf+JLsfHDXvbKv3WD2FQJk+8+b,
	beber-2YnHqweIUXrk1uMJSBkQmQ,
	catalina.mocanu-Re5JQEeQqe8AvxtiuMwx3w,
	dborkman-H+wXaHxf7aLQT0dZR+AlfA, edumazet-hpIqsD4AKlfQT0dZR+AlfA,
	ebiederm-aS9lmoZGLiVWk0Htik3J/w, fabf-AgBVmzD5pcezQB+pC5nmwQ,
	fuse-devel-5NWGOfrQmneRv+LV9MX5uipxlwaOVQ5f,
	geert-Td1EMuHUCqxL1ZNQvxDV9g, hughd-hpIqsD4AKlfQT0dZR+AlfA,
	iulia.manda21-Re5JQEeQqe8AvxtiuMwx3w, JBeulich-IBi9RG/b67k,
	bfields-uC3wQj2KruNg9hUCZPvPmw, jlayton-vpEMnDpepFuMZCB2o+C8xQ,
	linux-api-u79uwXL29TY76Z2rM5mHXA,
	linux-fsdevel-u79uwXL29TY76Z2rM5mHXA,
	linux-kernel-u79uwXL29TY76Z2rM5mHXA,
	linux-nfs-u79uwXL29TY76Z2rM5mHXA, mcgrof-IBi9RG/b67k,
	mattst88-Re5JQEeQqe8AvxtiuMwx3w, mgorman-l3A5Bk7waGM,
	mst-H+wXaHxf7aLQT0dZR+AlfA, miklos-sUDqSbJrdHQHWmgEVkV9KA,
	netdev-u79uwXL29TY76Z2rM5mHXA, oleg-H+wXaHxf7aLQT0dZR+AlfA,
	Paul.Durrant-Sxgqhf6Nn4DQT0dZR+AlfA,
	pefoley2-lY0TAiDIAFlBDgjK7y7TUQ, tgraf-G/eBtMaohhA,
	therbert-hpIqsD4AKlfQT0dZR+AlfA,
	trond.myklebust-7I+n7zu2hftEKMMhf/gKZA,
	willemb-hpIqsD4AKlfQT0dZR+AlfA, xiaoguangrong@
In-Reply-To: <20141125185806.GA28116-AKGzg7BKzIDYtjvyW6yDsg@public.gmane.org>

From: Theodore Ts'o <tytso-3s7WtUTddSA@public.gmane.org>
Date: Tue, 25 Nov 2014 13:58:06 -0500

> On Tue, Nov 25, 2014 at 01:24:45PM -0500, David Miller wrote:
>> 
>> And then if some fundamental part of userland (glibc, klibc, etc.) finds
>> a useful way to use splice for a fundamental operation, we're back to
>> square one.
> 
> I'll note that the applications for these super-tiny kernels are
> places where it's not likely they would be using glibc at all; think
> very tiny embedded systems.  The userspace tends to be highly
> restricted for the same space reasons why there is an effort to make
> the kernel as small as possible.

This is why I mentioned klibc, in order to avoid replies like your's,
it seems I have failed.

^ permalink raw reply

* Re: [PATCH v4 0/7] kernel tinification: optionally compile out splice family of syscalls (splice, vmsplice, tee and sendfile)
From: David Miller @ 2014-11-25 19:04 UTC (permalink / raw)
  To: josh
  Cc: rdunlap, pieter, alexander.h.duyck, viro, ast, akpm, beber,
	catalina.mocanu, dborkman, edumazet, ebiederm, fabf, fuse-devel,
	geert, hughd, iulia.manda21, JBeulich, bfields, jlayton,
	linux-api, linux-fsdevel, linux-kernel, linux-nfs, mcgrof,
	mattst88, mgorman, mst, miklos, netdev, oleg, Paul.Durrant,
	paulmck, pefoley2, tgraf, therbert, trond.myklebust, willemb,
	xiaoguangrong, zhe
In-Reply-To: <20141125185310.GA24891@cloud>

From: josh@joshtriplett.org
Date: Tue, 25 Nov 2014 10:53:10 -0800

> It's not a "slippery slope"; it's been our standard practice for ages.

We've never put an entire class of generic system calls behind
a config option.

^ permalink raw reply

* Re: [patch net-next v3 17/17] rocker: add ndo_bridge_setlnk/getlink support for learning policy
From: Jamal Hadi Salim @ 2014-11-25 19:00 UTC (permalink / raw)
  To: Scott Feldman
  Cc: Jiri Pirko, Netdev, David S. Miller, nhorman, Andy Gospodarek,
	Thomas Graf, dborkman, ogerlitz, jesse, pshelar, azhou, ben,
	stephen, Kirsher, Jeffrey T, vyasevic, Cong Wang,
	Fastabend, John R, Eric Dumazet, Florian Fainelli, Roopa Prabhu,
	John Linville, jasowang, ebiederm, Nicolas Dichtel, ryazanov.s.a,
	buytenh, Aviad Raveh, nbd, Alexei Starovoitov <alexei.
In-Reply-To: <CAE4R7bBNvXFYZiOH+cNm_o6PyyQfCXAWrxU7tJzo+pJ+QQSXGg@mail.gmail.com>

On 11/25/14 13:55, Scott Feldman wrote:

> I disagree.  API changes need a reference implementation to show usage
> and for testing.  If you have have an alternate switch implementation
> that achieves the same goal, bring it forward.
>

Yes, point conceded ;->

/me waits for the next guy who is going to smirk at me for saying the
above and tell Jiri to fix his typo ;->


cheers,
jamal

^ permalink raw reply

* Re: [PATCH v4 0/7] kernel tinification: optionally compile out splice family of syscalls (splice, vmsplice, tee and sendfile)
From: josh-iaAMLnmF4UmaiuxdJuQwMA @ 2014-11-25 19:00 UTC (permalink / raw)
  To: Randy Dunlap
  Cc: Michael S. Tsirkin, Trond Myklebust, Bertrand Jacquin,
	Oleg Nesterov, J. Bruce Fields, Eric Dumazet,
	蔡正龙, Jeff Layton, Tom Herbert,
	Alexei Starovoitov, Miklos Szeredi, Peter Foley, Hugh Dickins,
	Xiao Guangrong, Geert Uytterhoeven, Mel Gorman, Matt Turner,
	Paul E. McKenney, Alexander Duyck, Pieter Smith,
	open list:FUSE: FILESYSTEM...
In-Reply-To: <5474ABB6.3030400-wEGCiKHe2LqWVfeAwA7xHQ@public.gmane.org>

On Tue, Nov 25, 2014 at 08:17:58AM -0800, Randy Dunlap wrote:
> On 11/24/2014 03:00 PM, Pieter Smith wrote:
> >REPO: https://github.com/smipi1/linux-tinification.git
> >
> >BRANCH: tiny/config-syscall-splice
> >
> >BACKGROUND: This patch-set forms part of the Linux Kernel Tinification effort (
> >   https://tiny.wiki.kernel.org/).
> >
> >GOAL: Support compiling out the splice family of syscalls (splice, vmsplice,
> >   tee and sendfile) along with all supporting infrastructure if not needed.
> >   Many embedded systems will not need the splice-family syscalls. Omitting them
> >   saves space.
> 
> Hi,
> 
> Is the splice family of syscalls the only one that tiny has identified
> for optional building or can we expect similar treatment for other
> syscalls?

Pretty much any system call that you could conceive of writing a
userspace without.

There's a partial project list at https://tiny.wiki.kernel.org/projects.

> Why will many embedded systems not need these syscalls?  You know
> exactly what apps they run and you are positive that those apps do
> not use splice?

Yes, precisely.  We're talking about embedded systems small enough that
you're booting with init=/your/app and don't even call fork(), where you
know exactly what code you're putting in and what libraries you use.
And they're almost certainly not running glibc.

> >RESULTS: A tinyconfig bloat-o-meter score for the entire patch-set:
> >
> >add/remove: 0/41 grow/shrink: 5/7 up/down: 23/-8422 (-8399)
> 
> The summary is that this patch saves around 8 KB of code space --
> is that correct?

Right.  For reference, we're talking about kernels where the *total*
size is a few hundred kB.

> How much storage space do embedded systems have nowadays?

For the embedded systems we're targeting for the tinification effort, in
a first pass: 512k-2M of storage (often for an *uncompressed* kernel, to
support execute-in-place), and 128k-512k of memory.  We've successfully
built useful kernels and userspaces for such environments, and we'd like
to go even smaller.

- Josh Triplett

------------------------------------------------------------------------------
Download BIRT iHub F-Type - The Free Enterprise-Grade BIRT Server
from Actuate! Instantly Supercharge Your Business Reports and Dashboards
with Interactivity, Sharing, Native Excel Exports, App Integration & more
Get technology previously reserved for billion-dollar corporations, FREE
http://pubads.g.doubleclick.net/gampad/clk?id=157005751&iu=/4140/ostg.clktrk

^ permalink raw reply

* Re: [PATCH 1/2 3.18] rtlwifi: rtl8821ae: Fix 5G detection problem
From: John W. Linville @ 2014-11-25 18:46 UTC (permalink / raw)
  To: Larry Finger; +Cc: linux-wireless, netdev, Valerio Passini
In-Reply-To: <1416933127-25912-2-git-send-email-Larry.Finger@lwfinger.net>

On Tue, Nov 25, 2014 at 10:32:06AM -0600, Larry Finger wrote:
> The changes associated with moving this driver from staging to the regular
> tree missed one section setting the allowable rates for the 5GHz band.
> 
> This patch is needed to fix the regression reported in Bug #88811
> (https://bugzilla.kernel.org/show_bug.cgi?id=88811).
> 
> Reported-by: Valerio Passini <valerio.passini@unicam.it>
> Tested-by: Valerio Passini <valerio.passini@unicam.it>
> Signed-off-by: Larry Finger <Larry.Finger@lwfinger.net>
> Cc: Valerio Passini <valerio.passini@unicam.it>
> ---
>  drivers/net/wireless/rtlwifi/rtl8821ae/hw.c | 5 +++--
>  1 file changed, 3 insertions(+), 2 deletions(-)
> 
> diff --git a/drivers/net/wireless/rtlwifi/rtl8821ae/hw.c b/drivers/net/wireless/rtlwifi/rtl8821ae/hw.c
> index 310d316..18f34f7 100644
> --- a/drivers/net/wireless/rtlwifi/rtl8821ae/hw.c
> +++ b/drivers/net/wireless/rtlwifi/rtl8821ae/hw.c
> @@ -3672,8 +3672,9 @@ static void rtl8821ae_update_hal_rate_mask(struct ieee80211_hw *hw,
>  		mac->opmode == NL80211_IFTYPE_ADHOC)
>  		macid = sta->aid + 1;
>  	if (wirelessmode == WIRELESS_MODE_N_5G ||
> -	    wirelessmode == WIRELESS_MODE_AC_5G)
> -		ratr_bitmap = sta->supp_rates[NL80211_BAND_5GHZ];
> +	    wirelessmode == WIRELESS_MODE_AC_5G ||
> +	    wirelessmode == WIRELESS_MODE_A)
> +		ratr_bitmap = (sta->supp_rates[NL80211_BAND_5GHZ])<<4;

The parenthesis seem superfluous.  How about this line instead?

+		ratr_bitmap = sta->supp_rates[NL80211_BAND_5GHZ] << 4;

>  	else
>  		ratr_bitmap = sta->supp_rates[NL80211_BAND_2GHZ];
>  
> -- 
> 2.1.2
> 
> 

-- 
John W. Linville		Someday the world will need a hero, and you
linville@tuxdriver.com			might be all we have.  Be ready.

^ permalink raw reply

* Re: [PATCH v4 0/7] kernel tinification: optionally compile out splice family of syscalls (splice, vmsplice, tee and sendfile)
From: Theodore Ts'o @ 2014-11-25 18:58 UTC (permalink / raw)
  To: David Miller
  Cc: mst-H+wXaHxf7aLQT0dZR+AlfA,
	trond.myklebust-7I+n7zu2hftEKMMhf/gKZA,
	beber-2YnHqweIUXrk1uMJSBkQmQ, oleg-H+wXaHxf7aLQT0dZR+AlfA,
	bfields-uC3wQj2KruNg9hUCZPvPmw, edumazet-hpIqsD4AKlfQT0dZR+AlfA,
	willemb-hpIqsD4AKlfQT0dZR+AlfA,
	zhenglong.cai-TJRtMXcVgQTM1kAEIRd3EQ,
	jlayton-vpEMnDpepFuMZCB2o+C8xQ, therbert-hpIqsD4AKlfQT0dZR+AlfA,
	ast-uqk4Ao+rVK5Wk0Htik3J/w, miklos-sUDqSbJrdHQHWmgEVkV9KA,
	pefoley2-lY0TAiDIAFlBDgjK7y7TUQ, hughd-hpIqsD4AKlfQT0dZR+AlfA,
	xiaoguangrong-23VcF4HTsmIX0ybBhKVfKdBPR1lH4CV8,
	geert-Td1EMuHUCqxL1ZNQvxDV9g, mgorman-l3A5Bk7waGM,
	mattst88-Re5JQEeQqe8AvxtiuMwx3w,
	paulmck-23VcF4HTsmIX0ybBhKVfKdBPR1lH4CV8,
	alexander.h.duyck-ral2JQCrhuEAvxtiuMwx3w,
	pieter-qeJ+1H9vRZbz+pZb47iToQ,
	fuse-devel-5NWGOfrQmneRv+LV9MX5uipxlwaOVQ5f, mcgrof-IBi9RG/b67k,
	josh-iaAMLnmF4UmaiuxdJuQwMA,
	catalina.mocanu-Re5JQEeQqe8AvxtiuMwx3w,
	fabf-AgBVmzD5pcezQB+pC5nmwQ, tgraf-G/eBtMaohhA,
	Paul.Durrant-Sxgqhf6Nn4DQT0dZR+AlfA,
	viro-RmSDqhL/yNMiFSDQTTA3OLVCufUGDwFn, JBeulich-IBi9RG/b67k,
	linux-nfs-u79uwXL29TY76Z2rM5mHXA,
	iulia.manda21-Re5JQEeQqe8AvxtiuMwx3w,
	linux-api-u79uwXL29TY76Z2rM5mHXA, rdunlap-wEGCiKHe2LqWVfeAwA7xHQ,
	linux-kernel-u79uwXL29TY76Z2rM5mHXA,
	dborkman-H+wXaHxf7aLQT0dZR+AlfA, ebiederm-aS9lmoZGLiVWk0Htik3J/w,
	netdev-u79uwXL29TY76Z2rM5mHXA, linux-fsdeve
In-Reply-To: <20141125.132445.152609149279137368.davem-fT/PcQaiUtIeIZ0/mPfg9Q@public.gmane.org>

On Tue, Nov 25, 2014 at 01:24:45PM -0500, David Miller wrote:
> 
> And then if some fundamental part of userland (glibc, klibc, etc.) finds
> a useful way to use splice for a fundamental operation, we're back to
> square one.

I'll note that the applications for these super-tiny kernels are
places where it's not likely they would be using glibc at all; think
very tiny embedded systems.  The userspace tends to be highly
restricted for the same space reasons why there is an effort to make
the kernel as small as possible.

In these places, they are using Linux already, but they're using a 2.2
or 2.4 kernel because 3.0 is just too damned big.  So the goal is to
try to provide them an alternative which allows them to use a modern,
but stripped down kernel.  If glibc or klibc isn't going to work
without splice, then it's not going to work on a pre 2.6 kernel
anyway, so things are no worse with these systems anyway.

After all, if we can get these systems to using a 3.x kernel w/o
splice, that's surely better than their using a 2.2 or 2.4 kernel w/o
the splice system, isn't it?

Cheers,

					- Ted

------------------------------------------------------------------------------
Download BIRT iHub F-Type - The Free Enterprise-Grade BIRT Server
from Actuate! Instantly Supercharge Your Business Reports and Dashboards
with Interactivity, Sharing, Native Excel Exports, App Integration & more
Get technology previously reserved for billion-dollar corporations, FREE
http://pubads.g.doubleclick.net/gampad/clk?id=157005751&iu=/4140/ostg.clktrk

^ permalink raw reply

* Re: [patch net-next v3 17/17] rocker: add ndo_bridge_setlnk/getlink support for learning policy
From: Scott Feldman @ 2014-11-25 18:55 UTC (permalink / raw)
  To: Jamal Hadi Salim
  Cc: Jiri Pirko, Netdev, David S. Miller, nhorman, Andy Gospodarek,
	Thomas Graf, dborkman, ogerlitz, jesse, pshelar, azhou, ben,
	stephen, Kirsher, Jeffrey T, vyasevic, Cong Wang,
	Fastabend, John R, Eric Dumazet, Florian Fainelli, Roopa Prabhu,
	John Linville, jasowang, ebiederm, Nicolas Dichtel, ryazanov.s.a,
	buytenh, Aviad Raveh, nbd, Alexei Starovoitov <alexei.
In-Reply-To: <5474A9A4.9070905@mojatatu.com>

On Tue, Nov 25, 2014 at 6:09 AM, Jamal Hadi Salim <jhs@mojatatu.com> wrote:
> On 11/25/14 05:28, Jiri Pirko wrote:
>>
>> From: Scott Feldman <sfeldma@gmail.com>
>>
>> Rocker ports will use new "swdev" hwmode for bridge port offload policy.
>> Current supported policy settings are BR_LEARNING and BR_LEARNING_SYNC.
>> User can turn on/off device port FDB learning and syncing to bridge.
>>
>> Signed-off-by: Scott Feldman <sfeldma@gmail.com>
>> Signed-off-by: Jiri Pirko <jiri@resnulli.us>
>
>
> as previous comments - please submit rocker separately

I disagree.  API changes need a reference implementation to show usage
and for testing.  If you have have an alternate switch implementation
that achieves the same goal, bring it forward.

> cheers,
> jamal

^ permalink raw reply

* Re: [PATCH net-next 3/3] vxlan: Remote checksum offload
From: Jesse Gross @ 2014-11-25 18:54 UTC (permalink / raw)
  To: Tom Herbert; +Cc: David Miller, netdev
In-Reply-To: <CA+mtBx-2gHrFfgWhMjtH=XhyYyYh+57W1YL9x3bUiO8nJjA4YA@mail.gmail.com>

On Mon, Nov 24, 2014 at 6:50 PM, Tom Herbert <therbert@google.com> wrote:
> On Mon, Nov 24, 2014 at 5:06 PM, Jesse Gross <jesse@nicira.com> wrote:
>> On Mon, Nov 24, 2014 at 3:52 PM, Tom Herbert <therbert@google.com> wrote:
>>> Add support for remote checksum offload in VXLAN. This commandeers a
>>> reserved bit to indicate that RCO is being done, and uses the low order
>>> reserved eight bits of the VNI to hold the start and offset values in a
>>> compressed manner.
>>
>> Why do you think that this is OK for you to do? It's clear that there
>> is no consensus for this (and in fact there are other proposals that
>> use that bit in a different way).
>
> I asked on nvo3 list (which I believe is the appropriate forum) what
> the best way to do this is but haven't gotten any response. I will ask
> again-- I would assume that with an implementation and data in hand
> that might be better basis for discussion.
>
> The flag bit is currently unused in the Linux implementation, so I
> don't think it can break anything as of now. I suppose we could make
> RCO for VXLAN a config option and possibly change to use a different
> if consensus is reached on the right approach in the future.

This will definitely break things if this is applied now and the bit
is later used for a different purpose in the future as there will be
no way to update existing deployments.

There are a ton of conflicting proposals in this space so I think
there are only two possible solutions at this point:
 * Potentially support all of them and chose a variant at runtime
though a series of configuration options. This seems ugly,
particularly for GRO.
 * Stick to the version described in the RFC.

I don't think the third alternative of protocol design by order of
patch submission is viable.

^ permalink raw reply

* Re: [patch net-next v3 02/17] net: make vid as a parameter for ndo_fdb_add/ndo_fdb_del
From: Samudrala, Sridhar @ 2014-11-25 18:53 UTC (permalink / raw)
  To: Jiri Pirko, netdev
  Cc: davem, nhorman, andy, tgraf, dborkman, ogerlitz, jesse, pshelar,
	azhou, ben, stephen, jeffrey.t.kirsher, vyasevic, xiyou.wangcong,
	john.r.fastabend, edumazet, jhs, sfeldma, f.fainelli, roopa,
	linville, jasowang, ebiederm, nicolas.dichtel, ryazanov.s.a,
	buytenh, aviadr, nbd, alexei.starovoitov, Neil.Jerram, ronye,
	simon.horman, alexander.h.duyck, john.ronciak, mleitner, shrijeet,
	gospo, bcrl
In-Reply-To: <1416911328-10979-3-git-send-email-jiri@resnulli.us>


On 11/25/2014 2:28 AM, Jiri Pirko wrote:
> Do the work of parsing NDA_VLAN directly in rtnetlink code, pass simple
> u16 vid to drivers from there.
>
> Signed-off-by: Jiri Pirko <jiri@resnulli.us>
> ---
> new in v3
> ---
>   drivers/net/ethernet/intel/i40e/i40e_main.c      |  2 +-
>   drivers/net/ethernet/intel/ixgbe/ixgbe_main.c    |  4 +-
>   drivers/net/ethernet/qlogic/qlcnic/qlcnic_main.c |  9 +++--
>   drivers/net/macvlan.c                            |  4 +-
>   drivers/net/vxlan.c                              |  4 +-
>   include/linux/netdevice.h                        |  8 ++--
>   include/linux/rtnetlink.h                        |  6 ++-
>   net/bridge/br_fdb.c                              | 39 ++----------------
>   net/bridge/br_private.h                          |  4 +-
>   net/core/rtnetlink.c                             | 50 ++++++++++++++++++++----
>   10 files changed, 70 insertions(+), 60 deletions(-)
>
<deleted>
>   
> +static int fbd_vid_parse(struct nlattr *vlan_attr, u16 *p_vid)

looks like a typo? fdb_vid_parse()

^ permalink raw reply

* Re: [PATCH v4 0/7] kernel tinification: optionally compile out splice family of syscalls (splice, vmsplice, tee and sendfile)
From: josh @ 2014-11-25 18:53 UTC (permalink / raw)
  To: David Miller
  Cc: rdunlap, pieter, alexander.h.duyck, viro, ast, akpm, beber,
	catalina.mocanu, dborkman, edumazet, ebiederm, fabf, fuse-devel,
	geert, hughd, iulia.manda21, JBeulich, bfields, jlayton,
	linux-api, linux-fsdevel, linux-kernel, linux-nfs, mcgrof,
	mattst88, mgorman, mst, miklos, netdev, oleg, Paul.Durrant,
	paulmck, pefoley2, tgraf, therbert, trond.myklebust, willemb,
	xiaoguangrong, zhe
In-Reply-To: <20141125.121305.2094097848188324942.davem@davemloft.net>

On Tue, Nov 25, 2014 at 12:13:05PM -0500, David Miller wrote:
> From: Randy Dunlap <rdunlap@infradead.org>
> Date: Tue, 25 Nov 2014 08:17:58 -0800
> 
> > Is the splice family of syscalls the only one that tiny has identified
> > for optional building or can we expect similar treatment for other
> > syscalls?
> > 
> > Why will many embedded systems not need these syscalls?  You know
> > exactly what apps they run and you are positive that those apps do
> > not use splice?
> 
> I think starting to compile out system calls is a very slippery
> slope we should not begin the journey down.
> 
> This changes the forward facing interface to userspace.

It's not a "slippery slope"; it's been our standard practice for ages.
We started down that road long, long ago, when we first introduced
Kconfig and optional/modular features.  /dev/* are user-facing
interfaces, yet you can compile them out or make them modular.  /sys/*
and/proc/* are user-facing interfaces, yet you can compile part or all
of them out.  Filesystem names passed to mount are user-facing
interfaces, yet you can compile them out.  (Not just things like ext4;
think FUSE or overlayfs, which some applications will build upon and
require.)  Some prctls are optional, new syscalls like BPF or inotify or
process_vm_{read,write}v are optional, hardware interfaces are optional,
control groups are optional, containers and namespaces are optional,
checkpoint/restart is optional, KVM is optional, kprobes are optional,
kmsg is optional, /dev/port is optional, ACL support is optional, USB
support (as used by libusb) is optional, sound interfaces are optional,
GPU interfaces are optional, even futexes are optional.

For every single one of those, userspace programs or libraries may
depend on that functionality, and summarily exit if it doesn't exist,
perhaps with a warning that you need to enable options in your kernel,
or perhaps with a simple "Function not implemented" or "No such file or
directory".

Out of the entire list above and the many more where that came from,
what makes syscalls unique?  What's wildly different between
open("/dev/foo", ...) returning an error and sys_foo returning an error?
What makes syscalls so special out of the entire list above?  We're not
breaking the ability to run old userspace on a new kernel, which *must*
be supported, and that includes not just syscalls but all user-facing
interfaces; we don't break userspace.  But we've *never* guaranteed that
you can run old userspace on a new *allnoconfig* kernel.

All of these features will remain behind CONFIG_EXPERT, and all of them
warn that you can only use them if your userspace can cope.

I've actually been thinking of introducing a new CONFIG_ALL_SYSCALLS,
under which all the "enable support for foo syscall" can live, rather
than just piling all of them directly under CONFIG_EXPERT; that option
would then repeat in very clear terms the warning that if you disable
that option and then disable specific syscalls, you need to know exactly
what your target userspace uses.  That would group together this whole
family of options, and make it clearer what the implications are.

- Josh Triplett

^ permalink raw reply

* [PATCH net] Revert "netfilter: conntrack: fix race in __nf_conntrack_confirm against get_next_corpse"
From: Pablo Neira Ayuso @ 2014-11-25 18:54 UTC (permalink / raw)
  To: netfilter-devel; +Cc: davem, netdev, brouer

This reverts commit 5195c14c8b27cc0b18220ddbf0e5ad3328a04187.

If the conntrack clashes with an existing one, it is left out of
the unconfirmed list, thus, crashing when dropping the packet and
releasing the conntrack since golden rule is that conntracks are
always placed in any of the existing lists for traceability reasons.

Reported-by: Daniel Borkmann <dborkman@redhat.com>
Fixes: https://bugzilla.kernel.org/show_bug.cgi?id=88841
Signed-off-by: Pablo Neira Ayuso <pablo@netfilter.org>
---
Hi David,

Could you manually apply this to your net tree? We have a better
candidate fix to replace this broken patch that I will pass to you
once it gets sufficient testing.

Thanks!

 net/netfilter/nf_conntrack_core.c |   14 ++++++--------
 1 file changed, 6 insertions(+), 8 deletions(-)

diff --git a/net/netfilter/nf_conntrack_core.c b/net/netfilter/nf_conntrack_core.c
index 2c69975..5016a69 100644
--- a/net/netfilter/nf_conntrack_core.c
+++ b/net/netfilter/nf_conntrack_core.c
@@ -611,16 +611,12 @@ __nf_conntrack_confirm(struct sk_buff *skb)
 	 */
 	NF_CT_ASSERT(!nf_ct_is_confirmed(ct));
 	pr_debug("Confirming conntrack %p\n", ct);
-
-	/* We have to check the DYING flag after unlink to prevent
-	 * a race against nf_ct_get_next_corpse() possibly called from
-	 * user context, else we insert an already 'dead' hash, blocking
-	 * further use of that particular connection -JM.
-	 */
-	nf_ct_del_from_dying_or_unconfirmed_list(ct);
+	/* We have to check the DYING flag inside the lock to prevent
+	   a race against nf_ct_get_next_corpse() possibly called from
+	   user context, else we insert an already 'dead' hash, blocking
+	   further use of that particular connection -JM */
 
 	if (unlikely(nf_ct_is_dying(ct))) {
-		nf_ct_add_to_dying_list(ct);
 		nf_conntrack_double_unlock(hash, reply_hash);
 		local_bh_enable();
 		return NF_ACCEPT;
@@ -640,6 +636,8 @@ __nf_conntrack_confirm(struct sk_buff *skb)
 		    zone == nf_ct_zone(nf_ct_tuplehash_to_ctrack(h)))
 			goto out;
 
+	nf_ct_del_from_dying_or_unconfirmed_list(ct);
+
 	/* Timer relative to confirmation time, not original
 	   setting time, otherwise we'd get timer wrap in
 	   weird delay cases. */
-- 
1.7.10.4

^ permalink raw reply related

* Re: [PATCH net-next 0/3] net: Remote checksum offload for VXLAN
From: David Miller @ 2014-11-25 18:50 UTC (permalink / raw)
  To: therbert; +Cc: netdev
In-Reply-To: <1416873150-12260-1-git-send-email-therbert@google.com>

From: Tom Herbert <therbert@google.com>
Date: Mon, 24 Nov 2014 15:52:27 -0800

> This patch set adds support for remote checksum offload in VXLAN.
> 
> The remote checksum offload is generalized by creating a common
> function (remcsum_adjust) that does the work of modifying the
> checksum in remote checksum offload. This function can be called
> from normal or GRO path. GUE was modified to use this function.
> 
> To support RCO is VXLAN we use the 9th bit in the reserved
> flags to indicated remote checksum offload. The start and offset
> values are encoded n a compressed form in the low order (reserved)
> byte of the vni field.
> 
> Remote checksum offload is described in
> https://tools.ietf.org/html/draft-herbert-remotecsumoffload-01
> 
> Tested by running 200 TCP_STREAM connections with VXLAN (over IPv4).

What to do with the reserved bit seems to still be up in the air,
so I've marked this series as 'deferred'.

^ permalink raw reply

* Re: [PATCH] bonding: move ipoib_header_ops to vmlinux
From: David Miller @ 2014-11-25 18:44 UTC (permalink / raw)
  To: jay.vosburgh-Z7WLFzj8eWMS+FvcfC7Uqw
  Cc: ogerlitz-VPRAkNaXOzVWk0Htik3J/w,
	wen.gang.wang-QHcLZuEGTsvQT0dZR+AlfA,
	netdev-u79uwXL29TY76Z2rM5mHXA, linux-rdma-u79uwXL29TY76Z2rM5mHXA
In-Reply-To: <19740.1416940877@famine>

From: Jay Vosburgh <jay.vosburgh-Z7WLFzj8eWMS+FvcfC7Uqw@public.gmane.org>
Date: Tue, 25 Nov 2014 10:41:17 -0800

> Or Gerlitz <ogerlitz-VPRAkNaXOzVWk0Htik3J/w@public.gmane.org> wrote:
> 
>>On 11/25/2014 8:07 AM, David Miller wrote:
>>> IPOIB should not work over bonding as it requires that the device
>>> use ARPHRD_ETHER.
>>
>>Hi Dave,
>>
>>IPoIB devices can be enslaved to both bonding and teaming in their HA mode,
>>the bond device type becomes ARPHRD_INFINIBAND when this happens.
> 
> 	The point was that pktgen disallows ARPHRD_INFINIBAND, not that
> bonding does.
> 
> 	Pktgen specifically checks for type != ARPHRD_ETHER, so the
> IPoIB bond should not be able to be used with pkgten.  My suspicion is
> that pktgen is being configured on the bond first, then an IPoIB slave
> is added to the bond; this would change its type in a way that pktgen
> wouldn't notice.

+1
--
To unsubscribe from this list: send the line "unsubscribe linux-rdma" in
the body of a message to majordomo-u79uwXL29TY76Z2rM5mHXA@public.gmane.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html

^ permalink raw reply

* Re: [PATCH rfc 1/4] net-timestamp: pull headers for SOCK_STREAM
From: David Miller @ 2014-11-25 18:42 UTC (permalink / raw)
  To: willemb; +Cc: netdev, luto, richardcochran
In-Reply-To: <1416938286-14147-2-git-send-email-willemb@google.com>

From: Willem de Bruijn <willemb@google.com>
Date: Tue, 25 Nov 2014 12:58:03 -0500

> From: Willem de Bruijn <willemb@google.com>
> 
> When returning timestamped packets on the error queue, only return
> the data that the application initially sent: not the protocol
> headers.
> 
> This changes the ABI. The TCP interface is new enough that it should
> be safe to do so. The UDP interface could be changed analogously with
> 
> +  else if (sk->sk_protocol == IPPROTO_UDP)
> +    skb_pull(skb, skb_transport_offset(skb) + sizeof(struct udphdr));
> 
> Tested with Documentation/networking/timestamping/txtimestamp -l 60 -x
> 
> Signed-off-by: Willem de Bruijn <willemb@google.com>

What's the harm in exposing the headers?  Either it's harmful, and
therefore doing so for UDP is bad too, or it's harmless and we should
probably leave it alone to not risk breaking anyone.

^ permalink raw reply

* Re: [PATCH rfc 3/4] net-timestamp: tcp sockets return v6 errors on v6 sockets
From: David Miller @ 2014-11-25 18:41 UTC (permalink / raw)
  To: willemb; +Cc: netdev, luto, richardcochran
In-Reply-To: <1416938286-14147-4-git-send-email-willemb@google.com>

From: Willem de Bruijn <willemb@google.com>
Date: Tue, 25 Nov 2014 12:58:05 -0500

> From: Willem de Bruijn <willemb@google.com>
> 
> TCP timestamping introduced MSG_ERRQUEUE handling for TCP sockets.
> It always passed errorqueue requests onto ip_recv_error, but the
> same tcp_recvmsg code may also be called for IPv6 sockets. In that
> case, pass to ipv6_recv_error.
> 
> Tested by asking for PKTINFO with
> 
>   Documentation/networking/timestamping/txtimestamp -I
> 
> Before this change, IPv6 sockets would return AF_INET/IP_PKTINFO
> after the change, these sockets return AF_INET6/IPV6_PKTINFO
> 
> Signed-off-by: Willem de Bruijn <willemb@google.com>

This looks like a bug fix to me, and is therefore probably 'net'
material.

^ permalink raw reply

* Re: [PATCH] bonding: move ipoib_header_ops to vmlinux
From: Jay Vosburgh @ 2014-11-25 18:41 UTC (permalink / raw)
  To: Or Gerlitz
  Cc: David Miller, wen.gang.wang-QHcLZuEGTsvQT0dZR+AlfA,
	netdev-u79uwXL29TY76Z2rM5mHXA, linux-rdma-u79uwXL29TY76Z2rM5mHXA
In-Reply-To: <54742D6E.9030605-VPRAkNaXOzVWk0Htik3J/w@public.gmane.org>

Or Gerlitz <ogerlitz-VPRAkNaXOzVWk0Htik3J/w@public.gmane.org> wrote:

>On 11/25/2014 8:07 AM, David Miller wrote:
>> IPOIB should not work over bonding as it requires that the device
>> use ARPHRD_ETHER.
>
>Hi Dave,
>
>IPoIB devices can be enslaved to both bonding and teaming in their HA mode,
>the bond device type becomes ARPHRD_INFINIBAND when this happens.

	The point was that pktgen disallows ARPHRD_INFINIBAND, not that
bonding does.

	Pktgen specifically checks for type != ARPHRD_ETHER, so the
IPoIB bond should not be able to be used with pkgten.  My suspicion is
that pktgen is being configured on the bond first, then an IPoIB slave
is added to the bond; this would change its type in a way that pktgen
wouldn't notice.

	-J

---
	-Jay Vosburgh, jay.vosburgh-Z7WLFzj8eWMS+FvcfC7Uqw@public.gmane.org
--
To unsubscribe from this list: send the line "unsubscribe linux-rdma" in
the body of a message to majordomo-u79uwXL29TY76Z2rM5mHXA@public.gmane.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html

^ permalink raw reply

* Re: [PATCH rfc 2/4] net-errqueue: add IP(V6)_PKTINFO support
From: David Miller @ 2014-11-25 18:41 UTC (permalink / raw)
  To: willemb; +Cc: netdev, luto, richardcochran
In-Reply-To: <1416938286-14147-3-git-send-email-willemb@google.com>

From: Willem de Bruijn <willemb@google.com>
Date: Tue, 25 Nov 2014 12:58:04 -0500

> +	if (inet_sk(sk)->cmsg_flags & IP_CMSG_PKTINFO && skb->dev) {
> +		struct in_pktinfo info = {0};

I think memset(&info... is cleaner, and:

> +		struct in6_pktinfo info;
> +
> +		memset(&info, 0, sizeof(info));

Would make the code consistent with the ipv6 side.

^ permalink raw reply


This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox