From: "J. Bruce Fields" <bfields@fieldses.org>
To: Jeff Layton <jlayton@redhat.com>
Cc: viro@zeniv.linux.org.uk, matthew@wil.cx, dhowells@redhat.com,
sage@inktank.com, smfrench@gmail.com, swhiteho@redhat.com,
Trond.Myklebust@netapp.com, akpm@linux-foundation.org,
linux-kernel@vger.kernel.org, linux-afs@lists.infradead.org,
ceph-devel@vger.kernel.org, linux-cifs@vger.kernel.org,
samba-technical@lists.samba.org, cluster-devel@redhat.com,
linux-nfs@vger.kernel.org, linux-fsdevel@vger.kernel.org,
piastryyy@gmail.com
Subject: Re: [PATCH v2 14/14] locks: move file_lock_list to a set of percpu hlist_heads and convert file_lock_lock to an lglock
Date: Thu, 13 Jun 2013 11:37:42 -0400 [thread overview]
Message-ID: <20130613153742.GE20666@fieldses.org> (raw)
In-Reply-To: <1370948948-31784-15-git-send-email-jlayton@redhat.com>
On Tue, Jun 11, 2013 at 07:09:08AM -0400, Jeff Layton wrote:
> The file_lock_list is only used for /proc/locks. The vastly common case
> is for locks to be put onto the list and come off again, without ever
> being traversed.
>
> Help optimize for this use-case by moving to percpu hlist_head-s. At the
> same time, we can make the locking less contentious by moving to an
> lglock. When iterating over the lists for /proc/locks, we must take the
> global lock and then iterate over each CPU's list in turn.
>
> This change necessitates a new fl_link_cpu field to keep track of which
> CPU the entry is on. On x86_64 at least, this field is placed within an
> existing hole in the struct to avoid growing its real size.
ACK.--b.
>
> Signed-off-by: Jeff Layton <jlayton@redhat.com>
> ---
> fs/locks.c | 57 +++++++++++++++++++++++++++++++++++----------------
> include/linux/fs.h | 1 +
> 2 files changed, 40 insertions(+), 18 deletions(-)
>
> diff --git a/fs/locks.c b/fs/locks.c
> index 8124fc1..094eb4d 100644
> --- a/fs/locks.c
> +++ b/fs/locks.c
> @@ -127,6 +127,8 @@
> #include <linux/rcupdate.h>
> #include <linux/pid_namespace.h>
> #include <linux/hashtable.h>
> +#include <linux/percpu.h>
> +#include <linux/lglock.h>
>
> #include <asm/uaccess.h>
>
> @@ -165,8 +167,8 @@ int lease_break_time = 45;
> static DEFINE_SPINLOCK(blocked_hash_lock);
> static DEFINE_HASHTABLE(blocked_hash, BLOCKED_HASH_BITS);
>
> -static DEFINE_SPINLOCK(file_lock_lock);
> -static HLIST_HEAD(file_lock_list);
> +DEFINE_STATIC_LGLOCK(file_lock_lglock);
> +static DEFINE_PER_CPU(struct hlist_head, file_lock_list);
>
> static struct kmem_cache *filelock_cache __read_mostly;
>
> @@ -512,17 +514,21 @@ locks_delete_global_blocked(struct file_lock *waiter)
> static inline void
> locks_insert_global_locks(struct file_lock *waiter)
> {
> - spin_lock(&file_lock_lock);
> - hlist_add_head(&waiter->fl_link, &file_lock_list);
> - spin_unlock(&file_lock_lock);
> + lg_local_lock(&file_lock_lglock);
> + waiter->fl_link_cpu = smp_processor_id();
> + hlist_add_head(&waiter->fl_link, this_cpu_ptr(&file_lock_list));
> + lg_local_unlock(&file_lock_lglock);
> }
>
> static inline void
> locks_delete_global_locks(struct file_lock *waiter)
> {
> - spin_lock(&file_lock_lock);
> + /* avoid taking lock if already unhashed */
> + if (hlist_unhashed(&waiter->fl_link))
> + return;
> + lg_local_lock_cpu(&file_lock_lglock, waiter->fl_link_cpu);
> hlist_del_init(&waiter->fl_link);
> - spin_unlock(&file_lock_lock);
> + lg_local_unlock_cpu(&file_lock_lglock, waiter->fl_link_cpu);
> }
>
> /* Remove waiter from blocker's block list.
> @@ -2228,6 +2234,11 @@ EXPORT_SYMBOL_GPL(vfs_cancel_lock);
> #include <linux/proc_fs.h>
> #include <linux/seq_file.h>
>
> +struct locks_iterator {
> + int li_cpu;
> + loff_t li_pos;
> +};
> +
> static void lock_get_status(struct seq_file *f, struct file_lock *fl,
> loff_t id, char *pfx)
> {
> @@ -2302,16 +2313,17 @@ static void lock_get_status(struct seq_file *f, struct file_lock *fl,
> static int locks_show(struct seq_file *f, void *v)
> {
> int bkt;
> + struct locks_iterator *iter = f->private;
> struct file_lock *fl, *bfl;
>
> fl = hlist_entry(v, struct file_lock, fl_link);
>
> - lock_get_status(f, fl, *((loff_t *)f->private), "");
> + lock_get_status(f, fl, iter->li_pos, "");
>
> spin_lock(&blocked_hash_lock);
> hash_for_each(blocked_hash, bkt, bfl, fl_link) {
> if (bfl->fl_next == fl)
> - lock_get_status(f, bfl, *((loff_t *)f->private), " ->");
> + lock_get_status(f, bfl, iter->li_pos, " ->");
> }
> spin_unlock(&blocked_hash_lock);
>
> @@ -2320,23 +2332,24 @@ static int locks_show(struct seq_file *f, void *v)
>
> static void *locks_start(struct seq_file *f, loff_t *pos)
> {
> - loff_t *p = f->private;
> + struct locks_iterator *iter = f->private;
>
> - spin_lock(&file_lock_lock);
> - *p = (*pos + 1);
> - return seq_hlist_start(&file_lock_list, *pos);
> + iter->li_pos = *pos + 1;
> + lg_global_lock(&file_lock_lglock);
> + return seq_hlist_start_percpu(&file_lock_list, &iter->li_cpu, *pos);
> }
>
> static void *locks_next(struct seq_file *f, void *v, loff_t *pos)
> {
> - loff_t *p = f->private;
> - ++*p;
> - return seq_hlist_next(v, &file_lock_list, pos);
> + struct locks_iterator *iter = f->private;
> +
> + ++iter->li_pos;
> + return seq_hlist_next_percpu(v, &file_lock_list, &iter->li_cpu, pos);
> }
>
> static void locks_stop(struct seq_file *f, void *v)
> {
> - spin_unlock(&file_lock_lock);
> + lg_global_unlock(&file_lock_lglock);
> }
>
> static const struct seq_operations locks_seq_operations = {
> @@ -2348,7 +2361,8 @@ static const struct seq_operations locks_seq_operations = {
>
> static int locks_open(struct inode *inode, struct file *filp)
> {
> - return seq_open_private(filp, &locks_seq_operations, sizeof(loff_t));
> + return seq_open_private(filp, &locks_seq_operations,
> + sizeof(struct locks_iterator));
> }
>
> static const struct file_operations proc_locks_operations = {
> @@ -2448,9 +2462,16 @@ EXPORT_SYMBOL(lock_may_write);
>
> static int __init filelock_init(void)
> {
> + int i;
> +
> filelock_cache = kmem_cache_create("file_lock_cache",
> sizeof(struct file_lock), 0, SLAB_PANIC, NULL);
>
> + lg_lock_init(&file_lock_lglock, "file_lock_lglock");
> +
> + for_each_possible_cpu(i)
> + INIT_HLIST_HEAD(per_cpu_ptr(&file_lock_list, i));
> +
> return 0;
> }
>
> diff --git a/include/linux/fs.h b/include/linux/fs.h
> index 232a345..18e59b8 100644
> --- a/include/linux/fs.h
> +++ b/include/linux/fs.h
> @@ -953,6 +953,7 @@ struct file_lock {
> unsigned int fl_flags;
> unsigned char fl_type;
> unsigned int fl_pid;
> + int fl_link_cpu; /* what cpu's list is this on? */
> struct pid *fl_nspid;
> wait_queue_head_t fl_wait;
> struct file *fl_file;
> --
> 1.7.1
>
next prev parent reply other threads:[~2013-06-13 15:38 UTC|newest]
Thread overview: 30+ messages / expand[flat|nested] mbox.gz Atom feed top
2013-06-11 11:08 [PATCH v2 00/14] locks: scalability improvements for file locking Jeff Layton
2013-06-11 11:08 ` [PATCH v2 01/14] cifs: use posix_unblock_lock instead of locks_delete_block Jeff Layton
2013-06-11 11:08 ` [PATCH v2 02/14] locks: make generic_add_lease and generic_delete_lease static Jeff Layton
2013-06-11 11:08 ` [PATCH v2 03/14] locks: comment cleanups and clarifications Jeff Layton
2013-06-11 11:08 ` [PATCH v2 04/14] locks: make "added" in __posix_lock_file a bool Jeff Layton
2013-06-11 11:08 ` [PATCH v2 05/14] locks: encapsulate the fl_link list handling Jeff Layton
2013-06-11 11:09 ` [PATCH v2 06/14] locks: don't walk inode->i_flock list in locks_show Jeff Layton
2013-06-13 19:45 ` J. Bruce Fields
2013-06-13 20:26 ` Jeff Layton
[not found] ` <51BB040C.3050101@samba.org>
2013-06-15 11:05 ` Jeff Layton
2013-06-15 15:04 ` Simo
2013-06-11 11:09 ` [PATCH v2 07/14] locks: convert to i_lock to protect i_flock list Jeff Layton
2013-06-13 14:41 ` J. Bruce Fields
2013-06-13 15:09 ` Jeff Layton
2013-06-11 11:09 ` [PATCH v2 08/14] locks: ensure that deadlock detection is atomic with respect to blocked_list modification Jeff Layton
2013-06-11 11:09 ` [PATCH v2 09/14] locks: convert fl_link to a hlist_node Jeff Layton
2013-06-11 11:09 ` [PATCH v2 10/14] locks: turn the blocked_list into a hashtable Jeff Layton
2013-06-13 14:50 ` J. Bruce Fields
2013-06-11 11:09 ` [PATCH v2 11/14] locks: add a new "lm_owner_key" lock operation Jeff Layton
2013-06-13 15:00 ` J. Bruce Fields
2013-06-11 11:09 ` [PATCH v2 12/14] locks: give the blocked_hash its own spinlock Jeff Layton
2013-06-13 15:02 ` J. Bruce Fields
2013-06-13 15:18 ` Jeff Layton
2013-06-13 15:20 ` J. Bruce Fields
2013-06-11 11:09 ` [PATCH v2 13/14] seq_file: add seq_list_*_percpu helpers Jeff Layton
2013-06-13 15:27 ` J. Bruce Fields
2013-06-11 11:09 ` [PATCH v2 14/14] locks: move file_lock_list to a set of percpu hlist_heads and convert file_lock_lock to an lglock Jeff Layton
2013-06-13 15:37 ` J. Bruce Fields [this message]
2013-06-11 16:04 ` [PATCH v2 00/14] locks: scalability improvements for file locking J. Bruce Fields
2013-06-11 16:35 ` Jeff Layton
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20130613153742.GE20666@fieldses.org \
--to=bfields@fieldses.org \
--cc=Trond.Myklebust@netapp.com \
--cc=akpm@linux-foundation.org \
--cc=ceph-devel@vger.kernel.org \
--cc=cluster-devel@redhat.com \
--cc=dhowells@redhat.com \
--cc=jlayton@redhat.com \
--cc=linux-afs@lists.infradead.org \
--cc=linux-cifs@vger.kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-nfs@vger.kernel.org \
--cc=matthew@wil.cx \
--cc=piastryyy@gmail.com \
--cc=sage@inktank.com \
--cc=samba-technical@lists.samba.org \
--cc=smfrench@gmail.com \
--cc=swhiteho@redhat.com \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox;
as well as URLs for NNTP newsgroup(s).