* [PATCH v4 1/9] dax: add fsdev_dax_from_file()
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:34 ` sashiko-bot
2026-10-08 11:19 ` [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder() Miklos Szeredi
` (7 subsequent siblings)
8 siblings, 1 reply; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, John Groves, Amir Goldstein, Darrick J . Wong,
Vishal Verma, Dave Jiang, Alison Schofield, nvdimm, linux-cxl
From: John Groves <John@Groves.net>
If the file is a fsdev_dax open, take the dax_dev from the device inode's
i_cdev. Return with a reference grabbed on the private dax inode.
Make dax_dev_get() static again (internal to super.c for alloc_dax), export
fsdev_dax_from_file() instead.
This fix is in response to a Sashiko review, and some subsequent analysis.
About the 'fixes' tag: this removes the export of dax_dev_get(), which was
flawed, and replaces it with fsdev_dax_from_file(). It feels like the fixes
tag makes sense for correcting an ABI error.
Fixes: 2ae624d5a555d ("dax: export dax_dev_get()")
Originally-by: John Groves <john@groves.net>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
drivers/dax/fsdev.c | 24 ++++++++++++++++++++++++
drivers/dax/super.c | 3 +--
include/linux/dax.h | 3 ++-
3 files changed, 27 insertions(+), 3 deletions(-)
diff --git a/drivers/dax/fsdev.c b/drivers/dax/fsdev.c
index 598604bf5ac5..04b04d0d3da5 100644
--- a/drivers/dax/fsdev.c
+++ b/drivers/dax/fsdev.c
@@ -234,6 +234,30 @@ static const struct file_operations fsdev_fops = {
.release = fsdev_release,
};
+/**
+ * fsdev_dax_from_file - get dax_device from an open file
+ * @file: open device file
+ *
+ * Returns a dax_device pointer if @file refers to a fsdev_dax device.
+ * Otherwise return NULL.
+ *
+ * Caller must put_dax() the returned device when done.
+ */
+struct dax_device *fsdev_dax_from_file(struct file *file)
+{
+ struct dax_device *dax_dev;
+
+ if (file->f_op != &fsdev_fops)
+ return NULL;
+
+ dax_dev = inode_dax(file_inode(file));
+ WARN_ON(!dax_alive(dax_dev));
+ ihold(dax_inode(dax_dev));
+
+ return dax_dev;
+}
+EXPORT_SYMBOL_GPL(fsdev_dax_from_file);
+
/*
* Acquire the dev_pagemap for probe: the static (pre-populated) one if
* present, or a devm-allocated one for the dynamic case. Note that
diff --git a/drivers/dax/super.c b/drivers/dax/super.c
index 45f84b0eb909..25836f389bde 100644
--- a/drivers/dax/super.c
+++ b/drivers/dax/super.c
@@ -565,7 +565,7 @@ static int dax_set(struct inode *inode, void *data)
return 0;
}
-struct dax_device *dax_dev_get(dev_t devt)
+static struct dax_device *dax_dev_get(dev_t devt)
{
struct dax_device *dax_dev;
struct inode *inode;
@@ -588,7 +588,6 @@ struct dax_device *dax_dev_get(dev_t devt)
return dax_dev;
}
-EXPORT_SYMBOL_GPL(dax_dev_get);
struct dax_device *alloc_dax(void *private, const struct dax_operations *ops)
{
diff --git a/include/linux/dax.h b/include/linux/dax.h
index fe6c3ded1b50..2c9fc81f031b 100644
--- a/include/linux/dax.h
+++ b/include/linux/dax.h
@@ -54,7 +54,6 @@ struct dax_device *alloc_dax(void *private, const struct dax_operations *ops);
void *dax_holder(struct dax_device *dax_dev);
void put_dax(struct dax_device *dax_dev);
void kill_dax(struct dax_device *dax_dev);
-struct dax_device *dax_dev_get(dev_t devt);
void dax_write_cache(struct dax_device *dax_dev, bool wc);
bool dax_write_cache_enabled(struct dax_device *dax_dev);
bool dax_synchronous(struct dax_device *dax_dev);
@@ -323,3 +322,5 @@ typedef int (*walk_hmem_fn)(struct device *dev, int target_nid,
const struct resource *res);
int walk_hmem_resources(struct device *dev, walk_hmem_fn fn);
#endif
+
+struct dax_device *fsdev_dax_from_file(struct file *file);
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* Re: [PATCH v4 1/9] dax: add fsdev_dax_from_file()
2026-10-08 11:19 ` [PATCH v4 1/9] dax: add fsdev_dax_from_file() Miklos Szeredi
@ 2026-10-08 11:34 ` sashiko-bot
0 siblings, 0 replies; 16+ messages in thread
From: sashiko-bot @ 2026-10-08 11:34 UTC (permalink / raw)
To: Miklos Szeredi; +Cc: linux-cxl, nvdimm
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] dax: lockdep splat and user-triggerable WARN_ON in fsdev_dax_from_file()
--
commit f952deffb3bddf19404bc61323d8ebb31e8333ad
Author: John Groves <John@Groves.net>
Date: Thu Oct 8 11:19:52 2026 +0000
dax: add fsdev_dax_from_file()
This commit adds the fsdev_dax_from_file() helper to get the dax_dev from a
device inode's i_cdev and grab a reference on the private dax inode. It also
makes dax_dev_get() static again, replacing its export with the new helper.
> diff --git a/drivers/dax/fsdev.c b/drivers/dax/fsdev.c
> --- a/drivers/dax/fsdev.c
> +++ b/drivers/dax/fsdev.c
[ ... ]
> @@ -234,6 +234,30 @@ static const struct file_operations fsdev_fops = {
> .release = fsdev_release,
> };
>
> +/**
> + * fsdev_dax_from_file - get dax_device from an open file
> + * @file: open device file
> + *
> + * Returns a dax_device pointer if @file refers to a fsdev_dax device.
> + * Otherwise return NULL.
> + *
> + * Caller must put_dax() the returned device when done.
> + */
> +struct dax_device *fsdev_dax_from_file(struct file *file)
> +{
> + struct dax_device *dax_dev;
> +
> + if (file->f_op != &fsdev_fops)
> + return NULL;
> +
> + dax_dev = inode_dax(file_inode(file));
> + WARN_ON(!dax_alive(dax_dev));
> + ihold(dax_inode(dax_dev));
> +
> + return dax_dev;
[Severity: High]
Does this code trigger a lockdep splat when calling dax_alive()?
Looking at dax_alive() in drivers/dax/super.c, it explicitly asserts that
the dax_srcu lock is held:
bool dax_alive(struct dax_device *dax_dev)
{
lockdep_assert_held(&dax_srcu);
return test_bit(DAXDEV_ALIVE, &dax_dev->flags);
}
Since fsdev_dax_from_file() does not appear to acquire the dax_srcu lock
before this call, it looks like this will cause a lockdep splat on every
invocation when lockdep is enabled.
Additionally, if an administrator unbinds the driver via sysfs (which
clears DAXDEV_ALIVE), will this WARN_ON become user-triggerable during a
later file lookup? A user-triggerable kernel splat can cause a panic if
panic_on_warn is enabled.
If the device is actually dead, is it safe to proceed to call ihold() and
return the dead device pointer to the caller? Perhaps this path should
return NULL or an error instead of providing a dead device.
> +}
> +EXPORT_SYMBOL_GPL(fsdev_dax_from_file);
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261008112004.1899560-1-mszeredi@redhat.com?part=1
^ permalink raw reply [flat|nested] 16+ messages in thread
* [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder()
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
2026-10-08 11:19 ` [PATCH v4 1/9] dax: add fsdev_dax_from_file() Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:32 ` sashiko-bot
2026-10-08 13:20 ` Gregory Price
2026-10-08 11:19 ` [PATCH v4 3/9] fuse: add helpers for EIO return value with kernel message Miklos Szeredi
` (6 subsequent siblings)
8 siblings, 2 replies; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
A future fuse patch will utilize dax_holder() in RCU context to make sure
that the holder can't be freed by a racing fs_put_dax.
This patch just makes sure that dax_holder() works properly in such a
context by adding READ_ONCE() (equivalent to rcu_dereference minus sanity
checks).
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
drivers/dax/super.c | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/drivers/dax/super.c b/drivers/dax/super.c
index 25836f389bde..76e8d3654629 100644
--- a/drivers/dax/super.c
+++ b/drivers/dax/super.c
@@ -642,7 +642,7 @@ EXPORT_SYMBOL_GPL(put_dax);
*/
void *dax_holder(struct dax_device *dax_dev)
{
- return dax_dev->holder_data;
+ return READ_ONCE(dax_dev->holder_data);
}
EXPORT_SYMBOL_GPL(dax_holder);
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* Re: [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder()
2026-10-08 11:19 ` [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder() Miklos Szeredi
@ 2026-10-08 11:32 ` sashiko-bot
2026-10-08 13:20 ` Gregory Price
1 sibling, 0 replies; 16+ messages in thread
From: sashiko-bot @ 2026-10-08 11:32 UTC (permalink / raw)
To: Miklos Szeredi; +Cc: nvdimm, linux-cxl
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] dax: UAF and NULL deref in xfs_dax_notify_failure due to synchronous holder free
--
commit 27365c0f46f3cc2abb67ac0ebca3b78ca639e39f
Author: Miklos Szeredi <mszeredi@redhat.com>
dax: use READ_ONCE() in dax_holder()
This commit updates dax_holder() to use READ_ONCE() when returning the
holder_data. This is intended to allow dax_holder() to be utilized in RCU
context and ensure the holder cannot be freed by a racing fs_put_dax.
> diff --git a/drivers/dax/super.c b/drivers/dax/super.c
> index 25836f389bded..76e8d36546290 100644
> --- a/drivers/dax/super.c
> +++ b/drivers/dax/super.c
> @@ -642,7 +642,7 @@ EXPORT_SYMBOL_GPL(put_dax);
> */
> void *dax_holder(struct dax_device *dax_dev)
> {
> - return dax_dev->holder_data;
> + return READ_ONCE(dax_dev->holder_data);
[Severity: High]
Does this change and the reliance on RCU context expose a NULL pointer
dereference and Use-After-Free risk for existing users like XFS?
When a hardware memory failure triggers dax_holder_notify_failure(), it
takes the SRCU read lock and calls xfs_dax_notify_failure(). This reads the
holder pointer and unconditionally dereferences it without checking for NULL:
fs/xfs/xfs_notify_failure.c:xfs_dax_notify_failure() {
struct xfs_mount *mp = dax_holder(dax_dev);
if (!(mp->m_super->s_flags & SB_BORN)) {
...
}
Concurrently, if the filesystem is unmounted, xfs_free_buftarg() calls
fs_put_dax(), which clears the holder_data without waiting for active SRCU
or RCU readers to finish:
drivers/dax/super.c:fs_put_dax() {
...
WRITE_ONCE(dax_dev->holder_ops, NULL);
prev = cmpxchg(&dax_dev->holder_data, holder, NULL);
...
}
Immediately after, xfs_mount_free() synchronously frees the xfs_mount object
without waiting for an RCU or SRCU grace period:
fs/xfs/xfs_super.c:xfs_mount_free() {
...
debugfs_remove(mp->m_debugfs);
kfree(mp->m_rtname);
kfree(mp->m_logname);
#ifdef DEBUG
kfree(mp->m_errortag);
#endif
kfree(mp);
}
If xfs_dax_notify_failure() calls dax_holder() after it is cleared, mp is
NULL, causing an immediate crash when accessing mp->m_super. If it reads the
pointer before clearing, but kfree(mp) executes concurrently, it causes a
Use-After-Free.
An asynchronous memory failure during unmount could reliably trigger this.
The commit message mentions utilizing dax_holder() in RCU context to make
sure that the holder cannot be freed by a racing fs_put_dax. Does this
assume that filesystems defer freeing the holder? For XFS, the holder
appears to be synchronously freed without an RCU grace period.
> }
> EXPORT_SYMBOL_GPL(dax_holder);
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261008112004.1899560-1-mszeredi@redhat.com?part=2
^ permalink raw reply [flat|nested] 16+ messages in thread* Re: [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder()
2026-10-08 11:19 ` [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder() Miklos Szeredi
2026-10-08 11:32 ` sashiko-bot
@ 2026-10-08 13:20 ` Gregory Price
1 sibling, 0 replies; 16+ messages in thread
From: Gregory Price @ 2026-10-08 13:20 UTC (permalink / raw)
To: Miklos Szeredi
Cc: fuse-devel, John Groves, Amir Goldstein, Darrick J . Wong,
Vishal Verma, Dave Jiang, Alison Schofield, nvdimm, linux-cxl
On Thu, Oct 08, 2026 at 01:19:53PM +0200, Miklos Szeredi wrote:
> A future fuse patch will utilize dax_holder() in RCU context to make sure
> that the holder can't be freed by a racing fs_put_dax.
>
> This patch just makes sure that dax_holder() works properly in such a
> context by adding READ_ONCE() (equivalent to rcu_dereference minus sanity
> checks).
>
> Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
Reviewed-by: Gregory Price (Meta) <gourry@gourry.net>
> ---
> drivers/dax/super.c | 2 +-
> 1 file changed, 1 insertion(+), 1 deletion(-)
>
> diff --git a/drivers/dax/super.c b/drivers/dax/super.c
> index 25836f389bde..76e8d3654629 100644
> --- a/drivers/dax/super.c
> +++ b/drivers/dax/super.c
> @@ -642,7 +642,7 @@ EXPORT_SYMBOL_GPL(put_dax);
> */
> void *dax_holder(struct dax_device *dax_dev)
> {
> - return dax_dev->holder_data;
> + return READ_ONCE(dax_dev->holder_data);
> }
> EXPORT_SYMBOL_GPL(dax_holder);
>
> --
> 2.54.0
>
^ permalink raw reply [flat|nested] 16+ messages in thread
* [PATCH v4 3/9] fuse: add helpers for EIO return value with kernel message
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
2026-10-08 11:19 ` [PATCH v4 1/9] dax: add fsdev_dax_from_file() Miklos Szeredi
2026-10-08 11:19 ` [PATCH v4 2/9] dax: use READ_ONCE() in dax_holder() Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:19 ` [PATCH v4 4/9] fuse: support 64 bit, server allocated backing ID Miklos Szeredi
` (5 subsequent siblings)
8 siblings, 0 replies; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
Conditions are triggered by a buggy fuse server will result in -EIO which
is hard to interpret without any context.
Add helpers that print a short message to the kernel log as well as
returning the error value. This serves a dual purpose:
- documents the error condition inside the code
- allows the implementor of the fuse server to get the context for the
error
Some debug messages are removed in favor of this.
This also changes the mmap return value from -ETXTBSY to -ENODEV if a
passthrough open raced with the prior check. This makes both cases return
-ENODEV if the file is in passthrough mode.
Additionally change the return value of fuse_file_cached_io_open() from an
error value to a bool (true on success) as the error value is translated
anyway.
Currently these use the pr_notice_once() variant. Possibly should be
changed to a ratelimit of e.g. once per minute.
Reviewed-by: Amir Goldstein <amir73il@gmail.com>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/fuse/file.c | 6 ++--
fs/fuse/fuse_i.h | 4 ++-
fs/fuse/iomode.c | 75 ++++++++++++++++++-------------------------
fs/fuse/passthrough.c | 19 ++++-------
4 files changed, 43 insertions(+), 61 deletions(-)
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
index 6d707f2b3bff..bde07948d1f0 100644
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -2415,7 +2415,6 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma)
struct fuse_file *ff = file->private_data;
struct fuse_conn *fc = ff->fm->fc;
struct inode *inode = file_inode(file);
- int rc;
/* DAX mmap is superior to direct_io mmap */
if (FUSE_IS_VDAX(inode))
@@ -2457,9 +2456,8 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma)
* After first mmap, the inode stays in caching io mode until
* the direct_io file release.
*/
- rc = fuse_file_cached_io_open(inode, ff);
- if (rc)
- return rc;
+ if (!fuse_file_cached_io_open(inode, ff))
+ return -ENODEV;
}
if ((vma->vm_flags & VM_SHARED) && (vma->vm_flags & VM_MAYWRITE))
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 8546855386b5..d9ec88028d7b 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -11,6 +11,8 @@
# define pr_fmt(fmt) "fuse: " fmt
#endif
+#define fuse_EIO(fmt, ...) (pr_notice_once("%s: " fmt "\n", __func__, ##__VA_ARGS__), -EIO)
+
#include "args.h"
#include <linux/fuse.h>
#include <linux/fs.h>
@@ -1251,7 +1253,7 @@ int fuse_fileattr_set(struct mnt_idmap *idmap,
struct dentry *dentry, struct file_kattr *fa);
/* iomode.c */
-int fuse_file_cached_io_open(struct inode *inode, struct fuse_file *ff);
+bool fuse_file_cached_io_open(struct inode *inode, struct fuse_file *ff);
int fuse_inode_uncached_io_start(struct fuse_inode *fi,
struct fuse_backing *fb);
void fuse_inode_uncached_io_end(struct fuse_inode *fi);
diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c
index 79637c09e883..1b8b267d0c0d 100644
--- a/fs/fuse/iomode.c
+++ b/fs/fuse/iomode.c
@@ -27,15 +27,15 @@ static inline bool fuse_is_io_cache_wait(struct fuse_inode *fi)
* Blocks new parallel dio writes and waits for the in-progress parallel dio
* writes to complete.
*/
-int fuse_file_cached_io_open(struct inode *inode, struct fuse_file *ff)
+bool fuse_file_cached_io_open(struct inode *inode, struct fuse_file *ff)
{
struct fuse_inode *fi = get_fuse_inode(inode);
/* There are no io modes if server does not implement open */
if (!ff->args)
- return 0;
+ return true;
- spin_lock(&fi->lock);
+ guard(spinlock)(&fi->lock);
/*
* Setting the bit advises new direct-io writes to use an exclusive
* lock - without it the wait below might be forever.
@@ -53,8 +53,7 @@ int fuse_file_cached_io_open(struct inode *inode, struct fuse_file *ff)
*/
if (fuse_inode_backing(fi)) {
clear_bit(FUSE_I_CACHE_IO_MODE, &fi->state);
- spin_unlock(&fi->lock);
- return -ETXTBSY;
+ return false;
}
WARN_ON(ff->iomode == IOM_UNCACHED);
@@ -64,8 +63,7 @@ int fuse_file_cached_io_open(struct inode *inode, struct fuse_file *ff)
set_bit(FUSE_I_CACHE_IO_MODE, &fi->state);
fi->iocachectr++;
}
- spin_unlock(&fi->lock);
- return 0;
+ return true;
}
static void fuse_file_cached_io_release(struct fuse_file *ff,
@@ -85,19 +83,16 @@ static void fuse_file_cached_io_release(struct fuse_file *ff,
int fuse_inode_uncached_io_start(struct fuse_inode *fi, struct fuse_backing *fb)
{
struct fuse_backing *oldfb;
- int err = 0;
- spin_lock(&fi->lock);
+ guard(spinlock)(&fi->lock);
/* deny conflicting backing files on same fuse inode */
oldfb = fuse_inode_backing(fi);
- if (fb && oldfb && oldfb != fb) {
- err = -EBUSY;
- goto unlock;
- }
- if (fi->iocachectr > 0) {
- err = -ETXTBSY;
- goto unlock;
- }
+ if (fb && oldfb && oldfb != fb)
+ return -EBUSY;
+
+ if (fi->iocachectr > 0)
+ return -ETXTBSY;
+
fi->iocachectr--;
/* fuse inode holds a single refcount of backing file */
@@ -107,9 +102,7 @@ int fuse_inode_uncached_io_start(struct fuse_inode *fi, struct fuse_backing *fb)
} else {
fuse_backing_put(fb);
}
-unlock:
- spin_unlock(&fi->lock);
- return err;
+ return 0;
}
/* Takes uncached_io inode mode reference to be dropped on file release */
@@ -121,8 +114,11 @@ static int fuse_file_uncached_io_open(struct inode *inode,
int err;
err = fuse_inode_uncached_io_start(fi, fb);
- if (err)
- return err;
+ if (err) {
+ if (err == -EBUSY)
+ return fuse_EIO("mismatched backing");
+ return fuse_EIO("conflicting caching mode");
+ }
WARN_ON(ff->iomode != IOM_NONE);
ff->iomode = IOM_UNCACHED;
@@ -173,9 +169,11 @@ static int fuse_file_passthrough_open(struct inode *inode, struct file *file)
int err;
/* Check allowed conditions for file open in passthrough mode */
- if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH) || !fc->passthrough ||
- (ff->open_flags & ~FOPEN_PASSTHROUGH_MASK))
- return -EINVAL;
+ if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH) || !fc->passthrough)
+ return fuse_EIO("passthrough not enabled");
+
+ if (ff->open_flags & ~FOPEN_PASSTHROUGH_MASK)
+ return fuse_EIO("conflicting open flags");
fb = fuse_passthrough_open(file, ff->args->open_outarg.backing_id);
if (IS_ERR(fb))
@@ -197,7 +195,6 @@ int fuse_file_io_open(struct file *file, struct inode *inode)
{
struct fuse_file *ff = file->private_data;
struct fuse_inode *fi = get_fuse_inode(inode);
- int err;
/*
* io modes are not relevant with virtiofs DAX and with server that does not
@@ -208,11 +205,12 @@ int fuse_file_io_open(struct file *file, struct inode *inode)
/*
* Server is expected to use FOPEN_PASSTHROUGH for all opens of an inode
- * which is already open for passthrough.
+ * which is already open for passthrough. Using incorrect open mode is
+ * a server mistake, which results in user visible failure of open()
+ * with EIO error.
*/
- err = -EINVAL;
if (fuse_inode_backing(fi) && !(ff->open_flags & FOPEN_PASSTHROUGH))
- goto fail;
+ return fuse_EIO("FOPEN_PASSTHROUGH expected");
/*
* FOPEN_PARALLEL_DIRECT_WRITES requires FOPEN_DIRECT_IO.
@@ -233,23 +231,12 @@ int fuse_file_io_open(struct file *file, struct inode *inode)
return 0;
if (ff->open_flags & FOPEN_PASSTHROUGH)
- err = fuse_file_passthrough_open(inode, file);
- else
- err = fuse_file_cached_io_open(inode, ff);
- if (err)
- goto fail;
+ return fuse_file_passthrough_open(inode, file);
- return 0;
+ if (!fuse_file_cached_io_open(inode, ff))
+ return fuse_EIO("conflicting passthrough open");
-fail:
- pr_debug("failed to open file in requested io mode (open_flags=0x%x, err=%i).\n",
- ff->open_flags, err);
- /*
- * The file open mode determines the inode io mode.
- * Using incorrect open mode is a server mistake, which results in
- * user visible failure of open() with EIO error.
- */
- return -EIO;
+ return 0;
}
/* No more pending io and no new io possible to inode via open/mmapped file */
diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c
index b43d3e0f7081..489838462d1e 100644
--- a/fs/fuse/passthrough.c
+++ b/fs/fuse/passthrough.c
@@ -159,34 +159,29 @@ struct fuse_backing *fuse_passthrough_open(struct file *file, int backing_id)
struct fuse_conn *fc = ff->fm->fc;
struct fuse_backing *fb = NULL;
struct file *backing_file;
- int err;
- err = -EINVAL;
if (backing_id <= 0)
- goto out;
+ return ERR_PTR(fuse_EIO("invalid backing_id"));
- err = -ENOENT;
fb = fuse_backing_lookup(fc, backing_id);
if (!fb)
- goto out;
+ return ERR_PTR(fuse_EIO("backing not found"));
/* Allocate backing file per fuse file to store fuse path */
backing_file = backing_file_open(file, file->f_flags,
&fb->file->f_path, fb->cred);
- err = PTR_ERR(backing_file);
if (IS_ERR(backing_file)) {
fuse_backing_put(fb);
- goto out;
+ return ERR_PTR(fuse_EIO("failed to open backing file (%ld)", PTR_ERR(backing_file)));
}
- err = 0;
ff->passthrough = backing_file;
ff->cred = get_cred(fb->cred);
-out:
- pr_debug("%s: backing_id=%d, fb=0x%p, backing_file=0x%p, err=%i\n", __func__,
- backing_id, fb, ff->passthrough, err);
- return err ? ERR_PTR(err) : fb;
+ pr_debug("%s: backing_id=%d, fb=0x%p, backing_file=0x%p\n", __func__,
+ backing_id, fb, ff->passthrough);
+
+ return fb;
}
void fuse_passthrough_release(struct fuse_file *ff, struct fuse_backing *fb)
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* [PATCH v4 4/9] fuse: support 64 bit, server allocated backing ID
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
` (2 preceding siblings ...)
2026-10-08 11:19 ` [PATCH v4 3/9] fuse: add helpers for EIO return value with kernel message Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-09 7:36 ` Amir Goldstein
2026-10-08 11:19 ` [PATCH v4 5/9] fuse: support opening 64 bit " Miklos Szeredi
` (4 subsequent siblings)
8 siblings, 1 reply; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
Add support for server allocated 64-bit backing IDs alongside the
existing kernel allocated 32-bit IDs.
Opened with FUSE_DEV_IOC_BACKING_CREATE, the backing ID sent via
fuse_backing_create_in.backing_id.
Closed with FUSE_NOTIFY_BACKING_REMOVE. Since close only provides the
backing ID, not the file descriptor, it doesn't have to be done with an
ioctl.
Backing ID can take any 64 bit value other than zero.
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/file.c | 1 +
fs/fuse/backing.c | 178 ++++++++++++++++++++++++++++----------
fs/fuse/dev.c | 54 +++++++++---
fs/fuse/dev.h | 3 +
fs/fuse/fuse_i.h | 21 ++++-
fs/fuse/inode.c | 8 +-
fs/fuse/notify.c | 28 ++++++
fs/fuse/passthrough.c | 3 +
include/uapi/linux/fuse.h | 22 ++++-
9 files changed, 252 insertions(+), 66 deletions(-)
diff --git a/fs/file.c b/fs/file.c
index 628ca07dc4b1..7a9593297463 100644
--- a/fs/file.c
+++ b/fs/file.c
@@ -1213,6 +1213,7 @@ struct fd fdget_raw(unsigned int fd)
{
return __fget_light(fd, 0);
}
+EXPORT_SYMBOL(fdget_raw);
/*
* Try to avoid f_pos locking. We only need it if the
diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c
index 433fa3098d71..15132150535a 100644
--- a/fs/fuse/backing.c
+++ b/fs/fuse/backing.c
@@ -9,6 +9,7 @@
#include "fuse_i.h"
#include <linux/file.h>
+#include <linux/rhashtable.h>
static struct fuse_backing *fuse_backing_get(struct fuse_backing *fb)
{
@@ -48,6 +49,8 @@ static int fuse_backing_id_alloc(struct fuse_conn *fc, struct fuse_backing *fb)
id = idr_alloc_cyclic(&fc->backing_files_map, fb, 1, 0, GFP_ATOMIC);
spin_unlock(&fc->lock);
idr_preload_end();
+ if (id > 0)
+ fb->backing_id = id;
WARN_ON_ONCE(id == 0);
return id;
@@ -61,81 +64,130 @@ static struct fuse_backing *fuse_backing_id_remove(struct fuse_conn *fc,
spin_lock(&fc->lock);
fb = idr_remove(&fc->backing_files_map, id);
spin_unlock(&fc->lock);
+ if (fb)
+ fb->backing_id = 0;
return fb;
}
-static int fuse_backing_id_free(int id, void *p, void *data)
-{
- struct fuse_backing *fb = p;
+static const struct rhashtable_params fuse_backing_prm = {
+ .head_offset = offsetof(struct fuse_backing, hash_node),
+ .key_offset = offsetof(struct fuse_backing, backing_id),
+ .key_len = sizeof_field(struct fuse_backing, backing_id),
+};
- WARN_ON_ONCE(refcount_read(&fb->count) != 1);
- fuse_backing_free(fb);
- return 0;
+static int fuse_backing_add_64(struct fuse_conn *fc, struct fuse_backing *fb)
+{
+ return rhashtable_lookup_insert_fast(&fc->backing_64_ht, &fb->hash_node, fuse_backing_prm);
}
-void fuse_backing_files_free(struct fuse_conn *fc)
+int fuse_backing_close_64(struct fuse_conn *fc, u64 backing_id)
{
- idr_for_each(&fc->backing_files_map, fuse_backing_id_free, NULL);
- idr_destroy(&fc->backing_files_map);
+ struct fuse_backing *fb;
+ int err;
+
+ if (!fc->backing_id_64)
+ return -EINVAL;
+
+ scoped_guard(spinlock, &fc->lock) {
+ fb = rhashtable_lookup_fast(&fc->backing_64_ht, &backing_id, fuse_backing_prm);
+ if (!fb)
+ return -ENOENT;
+
+ err = rhashtable_remove_fast(&fc->backing_64_ht, &fb->hash_node, fuse_backing_prm);
+ WARN_ON(err);
+ }
+ fb->backing_id = 0;
+ fuse_backing_put(fb);
+
+ return 0;
}
-int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map)
+static struct fuse_backing *fuse_backing_new(struct fuse_conn *fc, int fd)
{
- struct file *file;
+ struct fuse_backing *fb;
struct super_block *backing_sb;
- struct fuse_backing *fb = NULL;
- int res;
-
- pr_debug("%s: fd=%d flags=0x%x\n", __func__, map->fd, map->flags);
+ struct file *file;
/* TODO: relax CAP_SYS_ADMIN once backing files are visible to lsof */
- res = -EPERM;
if (!fc->passthrough || !capable(CAP_SYS_ADMIN))
- goto out;
+ return ERR_PTR(-EPERM);
- res = -EINVAL;
- if (map->flags || map->padding)
- goto out;
+ CLASS(fd_raw, f)(fd);
+ if (fd_empty(f))
+ return ERR_PTR(-EBADF);
- file = fget_raw(map->fd);
- res = -EBADF;
- if (!file)
- goto out;
+ file = fd_file(f);
/* read/write/splice/mmap passthrough only relevant for regular files */
- res = d_is_dir(file->f_path.dentry) ? -EISDIR : -EINVAL;
if (!d_is_reg(file->f_path.dentry))
- goto out_fput;
+ return d_is_dir(file->f_path.dentry) ? ERR_PTR(-EISDIR) : ERR_PTR(-EINVAL);
backing_sb = file_inode(file)->i_sb;
- res = -ELOOP;
if (backing_sb->s_stack_depth >= fc->max_stack_depth)
- goto out_fput;
+ return ERR_PTR(-ELOOP);
fb = kmalloc_obj(struct fuse_backing);
- res = -ENOMEM;
if (!fb)
- goto out_fput;
+ return ERR_PTR(-ENOMEM);
- fb->file = file;
+ fb->file = get_file(file);
fb->cred = get_current_cred();
refcount_set(&fb->count, 1);
- res = fuse_backing_id_alloc(fc, fb);
- if (res < 0) {
+ return fb;
+}
+
+int fuse_backing_open_64(struct fuse_conn *fc, struct fuse_backing_create_in *map)
+{
+ struct fuse_backing *fb;
+ int res;
+
+ if (map->padding || map->spare[0] || map->spare[1] || !map->backing_id)
+ return -EINVAL;
+
+ if (!fc->backing_id_64)
+ return -EINVAL;
+
+ fb = fuse_backing_new(fc, map->fd);
+ if (IS_ERR(fb))
+ return PTR_ERR(fb);
+
+ fb->backing_id = map->backing_id;
+ res = fuse_backing_add_64(fc, fb);
+ if (res < 0)
fuse_backing_free(fb);
- fb = NULL;
- }
+ return res;
+}
+
+int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map)
+{
+ struct fuse_backing *fb = NULL;
+ int res;
+
+ pr_debug("%s: fd=%d flags=0x%x\n", __func__, map->fd, map->flags);
+
+ res = -EINVAL;
+ if (map->flags || map->padding)
+ goto out;
+
+ if (fc->backing_id_64)
+ goto out;
+
+ fb = fuse_backing_new(fc, map->fd);
+ res = PTR_ERR(fb);
+ if (!IS_ERR(fb)) {
+ res = fuse_backing_id_alloc(fc, fb);
+ if (res < 0) {
+ fuse_backing_free(fb);
+ fb = NULL;
+ }
+ }
out:
pr_debug("%s: fb=0x%p, ret=%i\n", __func__, fb, res);
return res;
-
-out_fput:
- fput(file);
- goto out;
}
int fuse_backing_close(struct fuse_conn *fc, int backing_id)
@@ -145,6 +197,9 @@ int fuse_backing_close(struct fuse_conn *fc, int backing_id)
pr_debug("%s: backing_id=%d\n", __func__, backing_id);
+ if (fc->backing_id_64)
+ return -EINVAL;
+
/* TODO: relax CAP_SYS_ADMIN once backing files are visible to lsof */
err = -EPERM;
if (!fc->passthrough || !capable(CAP_SYS_ADMIN))
@@ -167,14 +222,47 @@ int fuse_backing_close(struct fuse_conn *fc, int backing_id)
return err;
}
-struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, int backing_id)
+struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, u64 backing_id)
{
struct fuse_backing *fb;
- rcu_read_lock();
- fb = idr_find(&fc->backing_files_map, backing_id);
- fb = fuse_backing_get(fb);
- rcu_read_unlock();
+ guard(rcu)();
+ if (!fc->backing_id_64)
+ fb = idr_find(&fc->backing_files_map, backing_id);
+ else
+ fb = rhashtable_lookup(&fc->backing_64_ht, &backing_id, fuse_backing_prm);
- return fb;
+ return fuse_backing_get(fb);
+}
+
+static void fuse_backing_check_free(struct fuse_backing *fb)
+{
+ WARN_ON_ONCE(refcount_read(&fb->count) != 1);
+ fuse_backing_free(fb);
+}
+
+static int fuse_backing_idr_free(int id, void *p, void *data)
+{
+ fuse_backing_check_free(p);
+ return 0;
+}
+
+static void fuse_backing_rht_free(void *p, void *data)
+{
+ fuse_backing_check_free(p);
+}
+
+void fuse_backing_files_free(struct fuse_conn *fc)
+{
+ if (fc->backing_id_64) {
+ rhashtable_free_and_destroy(&fc->backing_64_ht, fuse_backing_rht_free, NULL);
+ } else {
+ idr_for_each(&fc->backing_files_map, fuse_backing_idr_free, NULL);
+ idr_destroy(&fc->backing_files_map);
+ }
+}
+
+void fuse_backing_files_init_64(struct fuse_conn *fc)
+{
+ rhashtable_init(&fc->backing_64_ht, &fuse_backing_prm);
}
diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
index 2d7ee5498f1c..eaa55fd9c106 100644
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -2321,39 +2321,64 @@ static long fuse_dev_ioctl_clone(struct file *file, __u32 __user *argp)
return 0;
}
-static long fuse_dev_ioctl_backing_open(struct file *file,
- struct fuse_backing_map __user *argp)
+static struct fuse_conn *fuse_get_conn_for_passthrough(struct file *file)
{
struct fuse_dev *fud = fuse_get_dev(file);
- struct fuse_backing_map map;
if (IS_ERR(fud))
- return PTR_ERR(fud);
+ return ERR_CAST(fud);
+
+ if (!smp_load_acquire(&fud->chan->initialized))
+ return ERR_PTR(-ENOTCONN);
if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
- return -EOPNOTSUPP;
+ return ERR_PTR(-EOPNOTSUPP);
+
+ return fud->chan->conn;
+}
+
+static long fuse_dev_ioctl_backing_open(struct file *file,
+ struct fuse_backing_map __user *argp)
+{
+ struct fuse_conn *fc = fuse_get_conn_for_passthrough(file);
+ struct fuse_backing_map map;
+
+ if (IS_ERR(fc))
+ return PTR_ERR(fc);
if (copy_from_user(&map, argp, sizeof(map)))
return -EFAULT;
- return fuse_backing_open(fud->chan->conn, &map);
+ return fuse_backing_open(fc, &map);
+}
+
+static long fuse_dev_ioctl_backing_create(struct file *file,
+ struct fuse_backing_create_in __user *argp)
+{
+ struct fuse_conn *fc = fuse_get_conn_for_passthrough(file);
+ struct fuse_backing_create_in map;
+
+ if (IS_ERR(fc))
+ return PTR_ERR(fc);
+
+ if (copy_from_user(&map, argp, sizeof(map)))
+ return -EFAULT;
+
+ return fuse_backing_open_64(fc, &map);
}
static long fuse_dev_ioctl_backing_close(struct file *file, __u32 __user *argp)
{
- struct fuse_dev *fud = fuse_get_dev(file);
+ struct fuse_conn *fc = fuse_get_conn_for_passthrough(file);
int backing_id;
- if (IS_ERR(fud))
- return PTR_ERR(fud);
-
- if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
- return -EOPNOTSUPP;
+ if (IS_ERR(fc))
+ return PTR_ERR(fc);
if (get_user(backing_id, argp))
return -EFAULT;
- return fuse_backing_close(fud->chan->conn, backing_id);
+ return fuse_backing_close(fc, backing_id);
}
static long fuse_dev_ioctl_sync_init(struct file *file)
@@ -2379,6 +2404,9 @@ static long fuse_dev_ioctl(struct file *file, unsigned int cmd,
case FUSE_DEV_IOC_BACKING_OPEN:
return fuse_dev_ioctl_backing_open(file, argp);
+ case FUSE_DEV_IOC_BACKING_CREATE:
+ return fuse_dev_ioctl_backing_create(file, argp);
+
case FUSE_DEV_IOC_BACKING_CLOSE:
return fuse_dev_ioctl_backing_close(file, argp);
diff --git a/fs/fuse/dev.h b/fs/fuse/dev.h
index f6c47ae0395b..dbffd5bed2f5 100644
--- a/fs/fuse/dev.h
+++ b/fs/fuse/dev.h
@@ -14,6 +14,7 @@ struct fuse_dev;
struct fuse_args;
struct fuse_copy_state;
struct fuse_backing_map;
+struct fuse_backing_create_in;
struct file;
struct folio;
enum fuse_notify_code;
@@ -87,6 +88,8 @@ int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map);
int fuse_backing_close(struct fuse_conn *fc, int backing_id);
+int fuse_backing_open_64(struct fuse_conn *fc, struct fuse_backing_create_in *map);
+int fuse_backing_close_64(struct fuse_conn *fc, u64 backing_id);
int fuse_copy_one(struct fuse_copy_state *cs, void *val, unsigned size);
int fuse_copy_folio(struct fuse_copy_state *cs, struct folio **foliop,
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index d9ec88028d7b..b1e450f096f6 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -32,6 +32,7 @@
#include <linux/pid_namespace.h>
#include <linux/refcount.h>
#include <linux/user_namespace.h>
+#include <linux/rhashtable-types.h>
/** Default max number of pages that can be used in a single read request */
#define FUSE_DEFAULT_MAX_PAGES_PER_REQ 32
@@ -92,7 +93,8 @@ struct fuse_submount_lookup {
struct fuse_backing {
struct file *file;
const struct cred *cred;
-
+ u64 backing_id;
+ struct rhash_head hash_node;
/* refcount */
refcount_t count;
struct rcu_head rcu;
@@ -713,6 +715,9 @@ struct fuse_conn {
/** @passthrough: Passthrough support for read/write IO */
unsigned int passthrough:1;
+ /** @backing_id_64: Backing ID is 64 bit and allocated by the server */
+ bool backing_id_64:1;
+
/** @use_pages_for_kvec_io: Use pages instead of pointer for kernel I/O */
unsigned int use_pages_for_kvec_io:1;
@@ -770,8 +775,14 @@ struct fuse_conn {
struct fuse_sync_bucket __rcu *curr_bucket;
#ifdef CONFIG_FUSE_PASSTHROUGH
- /** @backing_files_map: IDR for backing files ids */
- struct idr backing_files_map;
+ /* Selected by backing_id_64 */
+ union {
+ /** @backing_files_map: IDR for backing files ids */
+ struct idr backing_files_map;
+
+ /** @backing_64_ht: 64 bit ID lookup hash table */
+ struct rhashtable backing_64_ht;
+ };
#endif
};
@@ -1270,7 +1281,7 @@ void fuse_file_release(struct inode *inode, struct fuse_file *ff,
/* backing.c */
#ifdef CONFIG_FUSE_PASSTHROUGH
void fuse_backing_put(struct fuse_backing *fb);
-struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, int backing_id);
+
#else
static inline void fuse_backing_put(struct fuse_backing *fb)
@@ -1278,7 +1289,9 @@ static inline void fuse_backing_put(struct fuse_backing *fb)
}
#endif
+struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, u64 backing_id);
void fuse_backing_files_init(struct fuse_conn *fc);
+void fuse_backing_files_init_64(struct fuse_conn *fc);
void fuse_backing_files_free(struct fuse_conn *fc);
/* passthrough.c */
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
index cbb10e19e7e8..b69c95ebb1ad 100644
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -1408,13 +1408,17 @@ static void process_init_reply(struct fuse_args *args, int error)
* them together.
*/
if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH) &&
- (flags & FUSE_PASSTHROUGH) &&
+ (flags & (FUSE_PASSTHROUGH | FUSE_PASSTHROUGH_V2)) &&
arg->max_stack_depth > 0 &&
arg->max_stack_depth <= FILESYSTEM_MAX_STACK_DEPTH &&
!(flags & FUSE_WRITEBACK_CACHE)) {
fc->passthrough = 1;
fc->max_stack_depth = arg->max_stack_depth;
fm->sb->s_stack_depth = arg->max_stack_depth;
+ if (flags & FUSE_PASSTHROUGH_V2) {
+ fc->backing_id_64 = true;
+ fuse_backing_files_init_64(fc);
+ }
}
if (flags & FUSE_NO_EXPORT_SUPPORT)
fm->sb->s_export_op = &fuse_export_fid_operations;
@@ -1500,7 +1504,7 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm)
if (fm->fc->auto_submounts)
flags |= FUSE_SUBMOUNTS;
if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
- flags |= FUSE_PASSTHROUGH;
+ flags |= FUSE_PASSTHROUGH | FUSE_PASSTHROUGH_V2;
/* Only offered to sufficiently privileged servers; see
* fuse_syncfs_enable().
*/
diff --git a/fs/fuse/notify.c b/fs/fuse/notify.c
index c0428e03d138..cc0afffc2c00 100644
--- a/fs/fuse/notify.c
+++ b/fs/fuse/notify.c
@@ -409,6 +409,31 @@ static int fuse_notify_prune(struct fuse_conn *fc, unsigned int size,
return 0;
}
+static int fuse_notify_backing_remove(struct fuse_conn *fc, unsigned int size,
+ struct fuse_copy_state *cs)
+{
+ struct fuse_notify_backing_remove_out outarg;
+ int err;
+
+ if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
+ return -EOPNOTSUPP;
+
+ if (size != sizeof(outarg))
+ return -EINVAL;
+
+ err = fuse_copy_one(cs, &outarg, sizeof(outarg));
+ if (err)
+ return err;
+
+ if (outarg.reserved)
+ return -EINVAL;
+
+ if (!fc->backing_id_64)
+ return -EINVAL;
+
+ return fuse_backing_close_64(fc, outarg.backing_id);
+}
+
int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
unsigned int size, struct fuse_copy_state *cs)
{
@@ -440,6 +465,9 @@ int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
case FUSE_NOTIFY_PRUNE:
return fuse_notify_prune(fc, size, cs);
+ case FUSE_NOTIFY_BACKING_REMOVE:
+ return fuse_notify_backing_remove(fc, size, cs);
+
default:
return -EINVAL;
}
diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c
index 489838462d1e..313c8d7ffc09 100644
--- a/fs/fuse/passthrough.c
+++ b/fs/fuse/passthrough.c
@@ -160,6 +160,9 @@ struct fuse_backing *fuse_passthrough_open(struct file *file, int backing_id)
struct fuse_backing *fb = NULL;
struct file *backing_file;
+ if (fc->backing_id_64)
+ return ERR_PTR(fuse_EIO("incompatible backing version"));
+
if (backing_id <= 0)
return ERR_PTR(fuse_EIO("invalid backing_id"));
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index 10a7f31c4bdf..37e9559783c3 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -251,6 +251,9 @@
*
* 7.47
* - add FUSE_HAS_SYNCFS opt-in flag for privileged userspace servers
+ * - add FUSE_PASSTHROUGH_V2
+ * - add FUSE_DEV_IOC_BACKING_CREATE, struct fuse_backing_create_in
+ * - add FUSE_NOTIFY_BACKING_REMOVE, struct fuse_notify_backing_remove_out
*/
#ifndef _LINUX_FUSE_H
@@ -473,6 +476,7 @@ struct fuse_file_lock {
* with CAP_SYS_ADMIN in the initial user namespace (the same
* privilege that mounting virtiofs or fuseblk requires).
* Insufficiently privileged servers ignore it.
+ * FUSE_PASSTHROUGH_V2: use 64 bit server allocated backing ID
*/
#define FUSE_ASYNC_READ (1 << 0)
#define FUSE_POSIX_LOCKS (1 << 1)
@@ -522,6 +526,7 @@ struct fuse_file_lock {
#define FUSE_REQUEST_TIMEOUT (1ULL << 42)
#define FUSE_HAS_IO_URING_BUFPOOL (1ULL << 43)
#define FUSE_HAS_SYNCFS (1ULL << 44)
+#define FUSE_PASSTHROUGH_V2 (1ULL << 45)
/**
* CUSE INIT request/reply flags
@@ -709,6 +714,7 @@ enum fuse_notify_code {
FUSE_NOTIFY_RESEND = 7,
FUSE_NOTIFY_INC_EPOCH = 8,
FUSE_NOTIFY_PRUNE = 9,
+ FUSE_NOTIFY_BACKING_REMOVE = 10,
};
/* The read buffer is required to be at least 8k, but may be much larger */
@@ -1159,13 +1165,20 @@ struct fuse_backing_map {
uint64_t padding;
};
+struct fuse_backing_create_in {
+ int32_t fd;
+ uint32_t padding;
+ uint64_t backing_id; /* Zero value is reserved */
+ uint64_t spare[2];
+};
+
/* Device ioctls: */
#define FUSE_DEV_IOC_MAGIC 229
#define FUSE_DEV_IOC_CLONE _IOR(FUSE_DEV_IOC_MAGIC, 0, uint32_t)
-#define FUSE_DEV_IOC_BACKING_OPEN _IOW(FUSE_DEV_IOC_MAGIC, 1, \
- struct fuse_backing_map)
+#define FUSE_DEV_IOC_BACKING_OPEN _IOW(FUSE_DEV_IOC_MAGIC, 1, struct fuse_backing_map)
#define FUSE_DEV_IOC_BACKING_CLOSE _IOW(FUSE_DEV_IOC_MAGIC, 2, uint32_t)
#define FUSE_DEV_IOC_SYNC_INIT _IO(FUSE_DEV_IOC_MAGIC, 3)
+#define FUSE_DEV_IOC_BACKING_CREATE _IOW(FUSE_DEV_IOC_MAGIC, 4, struct fuse_backing_create_in)
struct fuse_lseek_in {
uint64_t fh;
@@ -1193,6 +1206,11 @@ struct fuse_copy_file_range_out {
uint64_t bytes_copied;
};
+struct fuse_notify_backing_remove_out {
+ uint64_t backing_id;
+ uint64_t reserved;
+};
+
#define FUSE_SETUPMAPPING_FLAG_WRITE (1ull << 0)
#define FUSE_SETUPMAPPING_FLAG_READ (1ull << 1)
struct fuse_setupmapping_in {
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* Re: [PATCH v4 4/9] fuse: support 64 bit, server allocated backing ID
2026-10-08 11:19 ` [PATCH v4 4/9] fuse: support 64 bit, server allocated backing ID Miklos Szeredi
@ 2026-10-09 7:36 ` Amir Goldstein
0 siblings, 0 replies; 16+ messages in thread
From: Amir Goldstein @ 2026-10-09 7:36 UTC (permalink / raw)
To: Miklos Szeredi
Cc: fuse-devel, John Groves, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
On Thu, Oct 8, 2026 at 1:20 PM Miklos Szeredi <mszeredi@redhat.com> wrote:
>
> Add support for server allocated 64-bit backing IDs alongside the
> existing kernel allocated 32-bit IDs.
>
> Opened with FUSE_DEV_IOC_BACKING_CREATE, the backing ID sent via
> fuse_backing_create_in.backing_id.
>
> Closed with FUSE_NOTIFY_BACKING_REMOVE. Since close only provides the
> backing ID, not the file descriptor, it doesn't have to be done with an
> ioctl.
>
> Backing ID can take any 64 bit value other than zero.
>
> Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
> ---
> fs/file.c | 1 +
> fs/fuse/backing.c | 178 ++++++++++++++++++++++++++++----------
> fs/fuse/dev.c | 54 +++++++++---
> fs/fuse/dev.h | 3 +
> fs/fuse/fuse_i.h | 21 ++++-
> fs/fuse/inode.c | 8 +-
> fs/fuse/notify.c | 28 ++++++
> fs/fuse/passthrough.c | 3 +
> include/uapi/linux/fuse.h | 22 ++++-
> 9 files changed, 252 insertions(+), 66 deletions(-)
>
[...]
> int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map);
> int fuse_backing_close(struct fuse_conn *fc, int backing_id);
> +int fuse_backing_open_64(struct fuse_conn *fc, struct fuse_backing_create_in *map);
> +int fuse_backing_close_64(struct fuse_conn *fc, u64 backing_id);
>
> int fuse_copy_one(struct fuse_copy_state *cs, void *val, unsigned size);
> int fuse_copy_folio(struct fuse_copy_state *cs, struct folio **foliop,
> diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
> index d9ec88028d7b..b1e450f096f6 100644
> --- a/fs/fuse/fuse_i.h
> +++ b/fs/fuse/fuse_i.h
> @@ -32,6 +32,7 @@
> #include <linux/pid_namespace.h>
> #include <linux/refcount.h>
> #include <linux/user_namespace.h>
> +#include <linux/rhashtable-types.h>
>
> /** Default max number of pages that can be used in a single read request */
> #define FUSE_DEFAULT_MAX_PAGES_PER_REQ 32
> @@ -92,7 +93,8 @@ struct fuse_submount_lookup {
> struct fuse_backing {
> struct file *file;
> const struct cred *cred;
> -
> + u64 backing_id;
> + struct rhash_head hash_node;
> /* refcount */
> refcount_t count;
> struct rcu_head rcu;
> @@ -713,6 +715,9 @@ struct fuse_conn {
> /** @passthrough: Passthrough support for read/write IO */
> unsigned int passthrough:1;
>
> + /** @backing_id_64: Backing ID is 64 bit and allocated by the server */
> + bool backing_id_64:1;
> +
> /** @use_pages_for_kvec_io: Use pages instead of pointer for kernel I/O */
> unsigned int use_pages_for_kvec_io:1;
>
> @@ -770,8 +775,14 @@ struct fuse_conn {
> struct fuse_sync_bucket __rcu *curr_bucket;
>
> #ifdef CONFIG_FUSE_PASSTHROUGH
> - /** @backing_files_map: IDR for backing files ids */
> - struct idr backing_files_map;
> + /* Selected by backing_id_64 */
> + union {
> + /** @backing_files_map: IDR for backing files ids */
> + struct idr backing_files_map;
> +
> + /** @backing_64_ht: 64 bit ID lookup hash table */
> + struct rhashtable backing_64_ht;
> + };
> #endif
> };
>
> @@ -1270,7 +1281,7 @@ void fuse_file_release(struct inode *inode, struct fuse_file *ff,
> /* backing.c */
> #ifdef CONFIG_FUSE_PASSTHROUGH
> void fuse_backing_put(struct fuse_backing *fb);
> -struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, int backing_id);
> +
> #else
>
> static inline void fuse_backing_put(struct fuse_backing *fb)
> @@ -1278,7 +1289,9 @@ static inline void fuse_backing_put(struct fuse_backing *fb)
> }
> #endif
>
> +struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, u64 backing_id);
> void fuse_backing_files_init(struct fuse_conn *fc);
> +void fuse_backing_files_init_64(struct fuse_conn *fc);
> void fuse_backing_files_free(struct fuse_conn *fc);
>
> /* passthrough.c */
> diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
> index cbb10e19e7e8..b69c95ebb1ad 100644
> --- a/fs/fuse/inode.c
> +++ b/fs/fuse/inode.c
> @@ -1408,13 +1408,17 @@ static void process_init_reply(struct fuse_args *args, int error)
> * them together.
> */
> if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH) &&
> - (flags & FUSE_PASSTHROUGH) &&
> + (flags & (FUSE_PASSTHROUGH | FUSE_PASSTHROUGH_V2)) &&
> arg->max_stack_depth > 0 &&
> arg->max_stack_depth <= FILESYSTEM_MAX_STACK_DEPTH &&
> !(flags & FUSE_WRITEBACK_CACHE)) {
> fc->passthrough = 1;
> fc->max_stack_depth = arg->max_stack_depth;
> fm->sb->s_stack_depth = arg->max_stack_depth;
> + if (flags & FUSE_PASSTHROUGH_V2) {
> + fc->backing_id_64 = true;
> + fuse_backing_files_init_64(fc);
> + }
> }
> if (flags & FUSE_NO_EXPORT_SUPPORT)
> fm->sb->s_export_op = &fuse_export_fid_operations;
Much better, but I still think it is somewhat ugly to call
fuse_backing_files_init_64()
after fuse_backing_files_init() for an init of a union.
Don't you think this is cleaner?
Thanks,
Amir.
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
index 330655af3bd0f..9abf6b1afef5d 100644
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -996,9 +996,6 @@ void fuse_conn_init(struct fuse_conn *fc, struct
fuse_mount *fm,
fc->max_pages_limit = fuse_max_pages_limit;
fc->name_max = FUSE_NAME_LOW_MAX;
- if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
- fuse_backing_files_init(fc);
-
INIT_LIST_HEAD(&fc->mounts);
list_add(&fm->fc_entry, &fc->mounts);
fm->fc = fc;
@@ -1424,10 +1421,8 @@ static void process_init_reply(struct fuse_args
*args, int error)
fc->passthrough = 1;
fc->max_stack_depth = arg->max_stack_depth;
fm->sb->s_stack_depth = arg->max_stack_depth;
- if (flags & FUSE_PASSTHROUGH_V2) {
- fc->backing_id_64 = true;
- fuse_backing_files_init_64(fc);
- }
+ fc->backing_id_64 = !!(flags &
FUSE_PASSTHROUGH_V2);
+ fuse_backing_files_init(fc);
}
if (flags & FUSE_NO_EXPORT_SUPPORT)
fm->sb->s_export_op =
&fuse_export_fid_operations;
diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c
index c93e4a45dd6f6..6157a18b5d554 100644
--- a/fs/fuse/backing.c
+++ b/fs/fuse/backing.c
@@ -46,11 +46,6 @@ void fuse_backing_put(struct fuse_backing *fb)
fuse_backing_free(fb);
}
-void fuse_backing_files_init(struct fuse_conn *fc)
-{
- idr_init(&fc->backing_files_map);
-}
-
static int fuse_backing_id_alloc(struct fuse_conn *fc, struct fuse_backing *fb)
{
int id;
@@ -338,6 +333,9 @@ void fuse_backing_files_free(struct fuse_conn *fc)
struct rhashtable_iter iter;
struct fuse_backing *fb;
+ if (!fc->passthrough)
+ return;
+
if (!fc->backing_id_64) {
idr_for_each(&fc->backing_files_map,
fuse_backing_idr_free, NULL);
idr_destroy(&fc->backing_files_map);
@@ -369,7 +367,12 @@ void fuse_backing_files_free(struct fuse_conn *fc)
rhashtable_free_and_destroy(&fc->backing_64_ht,
fuse_backing_rht_free, NULL);
}
-void fuse_backing_files_init_64(struct fuse_conn *fc)
+void fuse_backing_files_init(struct fuse_conn *fc)
{
+ if (!fc->backing_id_64) {
+ idr_init(&fc->backing_files_map);
+ return;
+ }
+
rhashtable_init(&fc->backing_64_ht, &fuse_backing_prm);
}
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 064fed459b93c..4235b8fc669b0 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -1324,7 +1324,6 @@ struct fuse_backing *fuse_backing_lookup(struct
fuse_conn *fc, u64 backing_id);
int fuse_backing_add_64(struct fuse_conn *fc, struct fuse_backing *fb);
bool fuse_backing_is_dax(struct fuse_backing *fb);
void fuse_backing_files_init(struct fuse_conn *fc);
-void fuse_backing_files_init_64(struct fuse_conn *fc);
void fuse_backing_files_free(struct fuse_conn *fc);
^ permalink raw reply related [flat|nested] 16+ messages in thread
* [PATCH v4 5/9] fuse: support opening 64 bit backing ID
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
` (3 preceding siblings ...)
2026-10-08 11:19 ` [PATCH v4 4/9] fuse: support 64 bit, server allocated backing ID Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:19 ` [PATCH v4 6/9] fuse: add support for opening dax device as backing Miklos Szeredi
` (3 subsequent siblings)
8 siblings, 0 replies; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
Add backing_id_64 field to fuse_open_out to allow opening files with
64-bit server-allocated backing IDs.
If the connection was initialized with FUSE_PASSTHROUGH_V2, the kernel
reads the backing ID from open_out.backing_id_64 instead of
open_out.backing_id.
The open reply is now variable-length for backward compatibility with
servers that don't send the new backing_id_64 field.
Reviewed-by: Amir Goldstein <amir73il@gmail.com>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/fuse/dir.c | 3 ++-
fs/fuse/file.c | 6 +++++-
fs/fuse/fuse_i.h | 2 +-
fs/fuse/iomode.c | 35 +++++++++++++++++++++++++++++------
fs/fuse/passthrough.c | 28 ++++++----------------------
include/uapi/linux/fuse.h | 2 ++
6 files changed, 45 insertions(+), 31 deletions(-)
diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c
index f48fafccce4b..874ec7cfffb1 100644
--- a/fs/fuse/dir.c
+++ b/fs/fuse/dir.c
@@ -876,6 +876,7 @@ static int fuse_create_open(struct mnt_idmap *idmap, struct inode *dir,
args.out_args[0].value = &outentry;
/* Store outarg for fuse_finish_open() */
outopenp = &ff->args->open_outarg;
+ args.out_argvar = true; /* compat */
args.out_args[1].size = sizeof(*outopenp);
args.out_args[1].value = outopenp;
@@ -885,7 +886,7 @@ static int fuse_create_open(struct mnt_idmap *idmap, struct inode *dir,
err = fuse_simple_idmap_request(idmap, fm, &args);
free_ext_value(&args);
- if (err)
+ if (err < 0)
goto out_free_ff;
err = -EIO;
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
index bde07948d1f0..5273957f0399 100644
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -28,6 +28,7 @@ static int fuse_send_open(struct fuse_mount *fm, u64 nodeid,
{
struct fuse_open_in inarg;
FUSE_ARGS(args);
+ int res;
memset(&inarg, 0, sizeof(inarg));
inarg.flags = open_flags & ~(O_CREAT | O_EXCL | O_NOCTTY);
@@ -45,10 +46,13 @@ static int fuse_send_open(struct fuse_mount *fm, u64 nodeid,
args.in_args[0].size = sizeof(inarg);
args.in_args[0].value = &inarg;
args.out_numargs = 1;
+ args.out_argvar = true; /* compat */
args.out_args[0].size = sizeof(*outargp);
args.out_args[0].value = outargp;
- return fuse_simple_request(fm, &args);
+ res = fuse_simple_request(fm, &args);
+
+ return res < 0 ? res : 0;
}
struct fuse_file *fuse_file_alloc(struct fuse_mount *fm, bool release)
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index b1e450f096f6..89dedcb54ba7 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -1314,7 +1314,7 @@ static inline struct fuse_backing *fuse_inode_backing_set(struct fuse_inode *fi,
#endif
}
-struct fuse_backing *fuse_passthrough_open(struct file *file, int backing_id);
+int fuse_passthrough_open(struct file *file, struct fuse_backing *fb);
void fuse_passthrough_release(struct fuse_file *ff, struct fuse_backing *fb);
static inline bool fuse_is_passthrough(struct fuse_file *ff)
diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c
index 1b8b267d0c0d..8b4774c80b11 100644
--- a/fs/fuse/iomode.c
+++ b/fs/fuse/iomode.c
@@ -165,7 +165,9 @@ static int fuse_file_passthrough_open(struct inode *inode, struct file *file)
{
struct fuse_file *ff = file->private_data;
struct fuse_conn *fc = get_fuse_conn(inode);
+ struct fuse_open_out *outarg = &ff->args->open_outarg;
struct fuse_backing *fb;
+ u64 backing_id;
int err;
/* Check allowed conditions for file open in passthrough mode */
@@ -175,18 +177,39 @@ static int fuse_file_passthrough_open(struct inode *inode, struct file *file)
if (ff->open_flags & ~FOPEN_PASSTHROUGH_MASK)
return fuse_EIO("conflicting open flags");
- fb = fuse_passthrough_open(file, ff->args->open_outarg.backing_id);
- if (IS_ERR(fb))
- return PTR_ERR(fb);
+ if (!fc->backing_id_64) {
+ if (outarg->backing_id_64 != 0)
+ return fuse_EIO("64 bit backing ID set");
+
+ if (outarg->backing_id <= 0)
+ return fuse_EIO("invalid backing ID");
+
+ backing_id = outarg->backing_id;
+ } else {
+ if (outarg->backing_id != 0)
+ return fuse_EIO("32 bit backing ID set");
+
+ backing_id = outarg->backing_id_64;
+ }
+ fb = fuse_backing_lookup(fc, backing_id);
+ if (!fb)
+ return fuse_EIO("backing not found");
+
+ err = fuse_passthrough_open(file, fb);
+ if (err)
+ goto backing_put;
/* First passthrough file open denies caching inode io mode */
err = fuse_file_uncached_io_open(inode, ff, fb);
- if (!err)
- return 0;
+ if (err)
+ goto passthrough_release;
+ return 0;
+
+passthrough_release:
fuse_passthrough_release(ff, fb);
+backing_put:
fuse_backing_put(fb);
-
return err;
}
diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c
index 313c8d7ffc09..4894842ad6d0 100644
--- a/fs/fuse/passthrough.c
+++ b/fs/fuse/passthrough.c
@@ -150,41 +150,25 @@ ssize_t fuse_passthrough_mmap(struct file *file, struct vm_area_struct *vma)
/*
* Setup passthrough to a backing file.
- *
- * Returns an fb object with elevated refcount to be stored in fuse inode.
*/
-struct fuse_backing *fuse_passthrough_open(struct file *file, int backing_id)
+int fuse_passthrough_open(struct file *file, struct fuse_backing *fb)
{
struct fuse_file *ff = file->private_data;
- struct fuse_conn *fc = ff->fm->fc;
- struct fuse_backing *fb = NULL;
struct file *backing_file;
- if (fc->backing_id_64)
- return ERR_PTR(fuse_EIO("incompatible backing version"));
-
- if (backing_id <= 0)
- return ERR_PTR(fuse_EIO("invalid backing_id"));
-
- fb = fuse_backing_lookup(fc, backing_id);
- if (!fb)
- return ERR_PTR(fuse_EIO("backing not found"));
-
/* Allocate backing file per fuse file to store fuse path */
backing_file = backing_file_open(file, file->f_flags,
&fb->file->f_path, fb->cred);
- if (IS_ERR(backing_file)) {
- fuse_backing_put(fb);
- return ERR_PTR(fuse_EIO("failed to open backing file (%ld)", PTR_ERR(backing_file)));
- }
+ if (IS_ERR(backing_file))
+ return fuse_EIO("failed to open backing file (%ld)", PTR_ERR(backing_file));
ff->passthrough = backing_file;
ff->cred = get_cred(fb->cred);
- pr_debug("%s: backing_id=%d, fb=0x%p, backing_file=0x%p\n", __func__,
- backing_id, fb, ff->passthrough);
+ pr_debug("%s: backing_id=%llu, fb=0x%p, backing_file=0x%p\n", __func__,
+ fb->backing_id, fb, ff->passthrough);
- return fb;
+ return 0;
}
void fuse_passthrough_release(struct fuse_file *ff, struct fuse_backing *fb)
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index 37e9559783c3..4b3904b3acb7 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -254,6 +254,7 @@
* - add FUSE_PASSTHROUGH_V2
* - add FUSE_DEV_IOC_BACKING_CREATE, struct fuse_backing_create_in
* - add FUSE_NOTIFY_BACKING_REMOVE, struct fuse_notify_backing_remove_out
+ * - add backing_id_64 to fuse_open_out
*/
#ifndef _LINUX_FUSE_H
@@ -841,6 +842,7 @@ struct fuse_open_out {
uint64_t fh;
uint32_t open_flags;
int32_t backing_id;
+ uint64_t backing_id_64;
};
struct fuse_release_in {
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* [PATCH v4 6/9] fuse: add support for opening dax device as backing
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
` (4 preceding siblings ...)
2026-10-08 11:19 ` [PATCH v4 5/9] fuse: support opening 64 bit " Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:34 ` sashiko-bot
2026-10-08 11:19 ` [PATCH v4 7/9] fuse: add extent map data structure Miklos Szeredi
` (2 subsequent siblings)
8 siblings, 1 reply; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
This is only possible with FUSE_PASSTHROUGH_V2 enabled.
Mark the inode with S_DAX if FUSE_LOOKUP returns with FUSE_ATTR_DAX set.
This patch does not yet provide a way actually use the dax dev backing:
when such a backing ID is provided in reply to FUSE_OPEN with
FOPEN_PASSTHROUGH flag set, an error will be returned.
Move fuse_backing_put() from fuse_free_inode() to fuse_evict_inode() to
avoid sleeping in atomic context.
Originally-by: John Groves <john@groves.net>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/fuse/backing.c | 108 +++++++++++++++++++++++++++++++++---------
fs/fuse/file.c | 2 +-
fs/fuse/fuse_i.h | 33 +++++++++++--
fs/fuse/inode.c | 17 +++++--
fs/fuse/iomode.c | 10 ++--
fs/fuse/passthrough.c | 6 ++-
6 files changed, 140 insertions(+), 36 deletions(-)
diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c
index 15132150535a..9ded41a94fcc 100644
--- a/fs/fuse/backing.c
+++ b/fs/fuse/backing.c
@@ -9,6 +9,7 @@
#include "fuse_i.h"
#include <linux/file.h>
+#include <linux/dax.h>
#include <linux/rhashtable.h>
static struct fuse_backing *fuse_backing_get(struct fuse_backing *fb)
@@ -22,9 +23,16 @@ static void fuse_backing_free(struct fuse_backing *fb)
{
pr_debug("%s: fb=0x%p\n", __func__, fb);
- if (fb->file)
- fput(fb->file);
- put_cred(fb->cred);
+ switch (fb->type) {
+ case FUSE_BACKING_PATH:
+ path_put(&fb->path);
+ put_cred(fb->cred);
+ break;
+
+ case FUSE_BACKING_DAXDEV:
+ fs_put_dax(fb->dax_dev, fb);
+ break;
+ }
kfree_rcu(fb, rcu);
}
@@ -103,39 +111,93 @@ int fuse_backing_close_64(struct fuse_conn *fc, u64 backing_id)
return 0;
}
-static struct fuse_backing *fuse_backing_new(struct fuse_conn *fc, int fd)
+static int fuse_dax_notify_failure(struct dax_device *daxdev, u64 offset, u64 len, int mf_flags)
{
struct fuse_backing *fb;
- struct super_block *backing_sb;
- struct file *file;
- /* TODO: relax CAP_SYS_ADMIN once backing files are visible to lsof */
- if (!fc->passthrough || !capable(CAP_SYS_ADMIN))
- return ERR_PTR(-EPERM);
+ guard(rcu)();
- CLASS(fd_raw, f)(fd);
- if (fd_empty(f))
- return ERR_PTR(-EBADF);
+ fb = dax_holder(daxdev);
+ if (fb)
+ fb->dax_error = true;
+
+ return 0;
+}
+
+static const struct dax_holder_operations fuse_dax_holder_ops = {
+ .notify_failure = fuse_dax_notify_failure,
+};
+
+static int fuse_backing_open_file(struct fuse_conn *fc, struct fuse_backing *fb, struct file *file)
+{
+ struct inode *inode = file_inode(file);
+ struct dax_device *daxdev;
+ int err;
+
+ switch (inode->i_mode & S_IFMT) {
+ case S_IFREG:
+ /* TODO: relax CAP_SYS_ADMIN once backing files are visible to lsof */
+ if (!fc->passthrough || !capable(CAP_SYS_ADMIN))
+ return -EPERM;
+
+ if (inode->i_sb->s_stack_depth >= fc->max_stack_depth)
+ return -ELOOP;
+
+ fb->type = FUSE_BACKING_PATH;
+ fb->path = file->f_path;
+ path_get(&fb->path);
+ fb->cred = get_current_cred();
+ return 0;
+
+ case S_IFCHR:
+ if (!fc->passthrough || !fc->backing_id_64)
+ return -EINVAL;
+
+ if (!IS_ENABLED(CONFIG_DEV_DAX_FSDEV))
+ return -EINVAL;
+
+ daxdev = fsdev_dax_from_file(file);
+ if (!daxdev)
+ return -EINVAL;
+
+ err = -EPERM;
+ if (capable(CAP_SYS_RAWIO)) {
+ err = fs_dax_get(daxdev, fb, &fuse_dax_holder_ops);
+ if (!err) {
+ fb->type = FUSE_BACKING_DAXDEV;
+ fb->dax_dev = daxdev;
+ }
+ }
+ put_dax(daxdev);
+ return err;
- file = fd_file(f);
+ case S_IFDIR:
+ return -EISDIR;
- /* read/write/splice/mmap passthrough only relevant for regular files */
- if (!d_is_reg(file->f_path.dentry))
- return d_is_dir(file->f_path.dentry) ? ERR_PTR(-EISDIR) : ERR_PTR(-EINVAL);
+ default:
+ return -EINVAL;
+ }
+}
- backing_sb = file_inode(file)->i_sb;
- if (backing_sb->s_stack_depth >= fc->max_stack_depth)
- return ERR_PTR(-ELOOP);
+static struct fuse_backing *fuse_backing_new(struct fuse_conn *fc, int fd)
+{
+ struct fuse_backing *fb __free(kfree) = kzalloc_obj(*fb);
+ int err;
- fb = kmalloc_obj(struct fuse_backing);
if (!fb)
return ERR_PTR(-ENOMEM);
- fb->file = get_file(file);
- fb->cred = get_current_cred();
+ CLASS(fd_raw, f)(fd);
+ if (fd_empty(f))
+ return ERR_PTR(-EBADF);
+
+ err = fuse_backing_open_file(fc, fb, fd_file(f));
+ if (err)
+ return ERR_PTR(err);
+
refcount_set(&fb->count, 1);
- return fb;
+ return_ptr(fb);
}
int fuse_backing_open_64(struct fuse_conn *fc, struct fuse_backing_create_in *map)
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
index 5273957f0399..e0d72b9d3c4a 100644
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -297,7 +297,7 @@ static int fuse_open(struct inode *inode, struct file *file)
if (!err) {
if (is_truncate)
truncate_pagecache(inode, 0);
- else if (!(ff->open_flags & FOPEN_KEEP_CACHE))
+ else if (!(ff->open_flags & FOPEN_KEEP_CACHE) && !IS_DAX(inode))
invalidate_inode_pages2(inode->i_mapping);
}
out_unlock:
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 89dedcb54ba7..8c65b6b8c840 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -89,10 +89,25 @@ struct fuse_submount_lookup {
struct fuse_forget_link *forget;
};
+enum fuse_backing_type {
+ FUSE_BACKING_PATH,
+ FUSE_BACKING_DAXDEV,
+};
+
/* Container for data related to mapping to backing file */
struct fuse_backing {
- struct file *file;
- const struct cred *cred;
+ enum fuse_backing_type type;
+
+ union {
+ struct {
+ struct path path;
+ const struct cred *cred;
+ };
+ struct {
+ struct dax_device *dax_dev;
+ bool dax_error;
+ };
+ };
u64 backing_id;
struct rhash_head hash_node;
/* refcount */
@@ -1239,7 +1254,19 @@ void fuse_free_conn(struct fuse_conn *fc);
/* dax.c */
-#define FUSE_IS_VDAX(inode) (IS_ENABLED(CONFIG_FUSE_VDAX) && IS_DAX(inode))
+static inline bool fuse_inode_vdax(struct inode *inode)
+{
+#ifdef CONFIG_FUSE_VDAX
+ return get_fuse_inode(inode)->vdax;
+#else
+ return false;
+#endif
+}
+
+static inline bool FUSE_IS_VDAX(struct inode *inode)
+{
+ return fuse_inode_vdax(inode) && IS_DAX(inode);
+}
ssize_t fuse_vdax_read_iter(struct kiocb *iocb, struct iov_iter *to);
ssize_t fuse_vdax_write_iter(struct kiocb *iocb, struct iov_iter *from);
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
index b69c95ebb1ad..330655af3bd0 100644
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -124,9 +124,6 @@ static void fuse_free_inode(struct inode *inode)
#ifdef CONFIG_FUSE_VDAX
kfree(fi->vdax);
#endif
- if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
- fuse_backing_put(fuse_inode_backing(fi));
-
kmem_cache_free(fuse_inode_cachep, fi);
}
@@ -148,7 +145,7 @@ static void fuse_evict_inode(struct inode *inode)
/* Will write inode on close/munmap and in all other dirtiers */
WARN_ON(inode_state_read_once(inode) & I_DIRTY_INODE);
- if (FUSE_IS_VDAX(inode))
+ if (IS_DAX(inode))
dax_break_layout_final(inode);
truncate_inode_pages_final(&inode->i_data);
@@ -177,6 +174,9 @@ static void fuse_evict_inode(struct inode *inode)
if (inode->i_nlink > 0)
atomic64_inc(&fc->evict_ctr);
}
+ if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
+ fuse_backing_put(fuse_inode_backing(fi));
+
if (S_ISREG(inode->i_mode) && !fuse_is_bad(inode)) {
WARN_ON(fi->iocachectr != 0);
WARN_ON(!list_empty(&fi->write_files));
@@ -403,6 +403,10 @@ static void fuse_init_submount_lookup(struct fuse_submount_lookup *sl,
refcount_set(&sl->count, 1);
}
+static const struct address_space_operations fuse_dax_aops = {
+ .dirty_folio = noop_dirty_folio,
+};
+
static void fuse_init_inode(struct inode *inode, struct fuse_attr *attr,
struct fuse_conn *fc)
{
@@ -413,6 +417,11 @@ static void fuse_init_inode(struct inode *inode, struct fuse_attr *attr,
if (S_ISREG(inode->i_mode)) {
fuse_init_common(inode);
fuse_init_file_inode(inode, attr->flags);
+
+ if ((attr->flags & FUSE_ATTR_DAX) && !fuse_inode_vdax(inode)) {
+ inode->i_flags |= S_DAX;
+ inode->i_data.a_ops = &fuse_dax_aops;
+ }
} else if (S_ISDIR(inode->i_mode))
fuse_init_dir(inode);
else if (S_ISLNK(inode->i_mode))
diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c
index 8b4774c80b11..38afe1f238ef 100644
--- a/fs/fuse/iomode.c
+++ b/fs/fuse/iomode.c
@@ -230,10 +230,14 @@ int fuse_file_io_open(struct file *file, struct inode *inode)
* Server is expected to use FOPEN_PASSTHROUGH for all opens of an inode
* which is already open for passthrough. Using incorrect open mode is
* a server mistake, which results in user visible failure of open()
- * with EIO error.
+ * with EIO error. Same with DAX inodes.
*/
- if (fuse_inode_backing(fi) && !(ff->open_flags & FOPEN_PASSTHROUGH))
- return fuse_EIO("FOPEN_PASSTHROUGH expected");
+ if (!(ff->open_flags & FOPEN_PASSTHROUGH)) {
+ if (fuse_inode_backing(fi))
+ return fuse_EIO("FOPEN_PASSTHROUGH expected");
+ if (IS_DAX(inode))
+ return fuse_EIO("DAX inode without FOPEN_PASSTHROUGH");
+ }
/*
* FOPEN_PARALLEL_DIRECT_WRITES requires FOPEN_DIRECT_IO.
diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c
index 4894842ad6d0..e9ab1aea34e2 100644
--- a/fs/fuse/passthrough.c
+++ b/fs/fuse/passthrough.c
@@ -156,9 +156,11 @@ int fuse_passthrough_open(struct file *file, struct fuse_backing *fb)
struct fuse_file *ff = file->private_data;
struct file *backing_file;
+ if (fb->type != FUSE_BACKING_PATH)
+ return fuse_EIO("invalid backing type");
+
/* Allocate backing file per fuse file to store fuse path */
- backing_file = backing_file_open(file, file->f_flags,
- &fb->file->f_path, fb->cred);
+ backing_file = backing_file_open(file, file->f_flags, &fb->path, fb->cred);
if (IS_ERR(backing_file))
return fuse_EIO("failed to open backing file (%ld)", PTR_ERR(backing_file));
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* Re: [PATCH v4 6/9] fuse: add support for opening dax device as backing
2026-10-08 11:19 ` [PATCH v4 6/9] fuse: add support for opening dax device as backing Miklos Szeredi
@ 2026-10-08 11:34 ` sashiko-bot
0 siblings, 0 replies; 16+ messages in thread
From: sashiko-bot @ 2026-10-08 11:34 UTC (permalink / raw)
To: Miklos Szeredi; +Cc: linux-cxl, nvdimm
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Low] kconfig: FUSE_FS=y and DEV_DAX_FSDEV=m causes undefined reference to fsdev_dax_from_file
--
commit 0e6c3b76d8791b10c41a202ee1e4d5b108cb9e7d
Author: Miklos Szeredi <mszeredi@redhat.com>
fuse: add support for opening dax device as backing
This commit adds support for opening a dax device as a FUSE passthrough
backing file, enabling S_DAX on inodes when requested. It also updates
the cleanup sequence by moving fuse_backing_put() to fuse_evict_inode()
to avoid sleeping in atomic context.
> diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c
> index 15132150535a5..9ded41a94fcc6 100644
> --- a/fs/fuse/backing.c
> +++ b/fs/fuse/backing.c
[ ... ]
> +static int fuse_backing_open_file(struct fuse_conn *fc, struct fuse_backing *fb, struct file *file)
> +{
> + struct inode *inode = file_inode(file);
> + struct dax_device *daxdev;
> + int err;
> +
> + switch (inode->i_mode & S_IFMT) {
[ ... ]
> + case S_IFCHR:
> + if (!fc->passthrough || !fc->backing_id_64)
> + return -EINVAL;
> +
> + if (!IS_ENABLED(CONFIG_DEV_DAX_FSDEV))
> + return -EINVAL;
> +
> + daxdev = fsdev_dax_from_file(file);
[Severity: Low]
If FUSE_FS is built-in (y) and DEV_DAX_FSDEV is compiled as a loadable module
(m), will IS_ENABLED(CONFIG_DEV_DAX_FSDEV) evaluate to true here? Since this
code is built into vmlinux, wouldn't statically referencing the
fsdev_dax_from_file() symbol from the fsdev_dax.ko loadable module cause an
undefined reference linker error during the kernel build?
Could this use IS_REACHABLE(CONFIG_DEV_DAX_FSDEV) instead to prevent the build
failure?
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261008112004.1899560-1-mszeredi@redhat.com?part=6
^ permalink raw reply [flat|nested] 16+ messages in thread
* [PATCH v4 7/9] fuse: add extent map data structure
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
` (5 preceding siblings ...)
2026-10-08 11:19 ` [PATCH v4 6/9] fuse: add support for opening dax device as backing Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:19 ` [PATCH v4 8/9] fuse: add extent map I/O support Miklos Szeredi
2026-10-08 11:20 ` [PATCH v4 9/9] fuse: add support for striped backing Miklos Szeredi
8 siblings, 0 replies; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
Add support for creating and managing extent maps that map file regions to
dax device regions.
Introduce FUSE_NOTIFY_BACKING_MAP for populating extent maps from
userspace.
Extent maps are handled as a new type of backing and are assigned a 64 bit
backing ID by the server.
One extent consists of
- offset within the containing backing
- length of extent
- target backing ID
- offset into target backing
Extents must be non-overlapping, and can only refer to dax device backings
for now.
At this point only support creating and populating the extent map in one
operation, though the interface is not limited by this and later may be
changed, so that partial population of an extent map would be possible.
Originally-by: John Groves <john@groves.net>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/fuse/Makefile | 2 +-
fs/fuse/backing.c | 38 +++++++++--
fs/fuse/ext_map.c | 131 ++++++++++++++++++++++++++++++++++++++
fs/fuse/fuse_i.h | 14 +++-
fs/fuse/notify.c | 47 ++++++++++++++
include/uapi/linux/fuse.h | 26 ++++++++
6 files changed, 251 insertions(+), 7 deletions(-)
create mode 100644 fs/fuse/ext_map.c
diff --git a/fs/fuse/Makefile b/fs/fuse/Makefile
index 5858feafa916..da9e802d7ae1 100644
--- a/fs/fuse/Makefile
+++ b/fs/fuse/Makefile
@@ -15,7 +15,7 @@ fuse-y += dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o ioctl.o r
fuse-y += poll.o notify.o
fuse-y += iomode.o
fuse-$(CONFIG_FUSE_VDAX) += dax.o
-fuse-$(CONFIG_FUSE_PASSTHROUGH) += passthrough.o backing.o
+fuse-$(CONFIG_FUSE_PASSTHROUGH) += passthrough.o backing.o ext_map.o
fuse-$(CONFIG_SYSCTL) += sysctl.o
fuse-$(CONFIG_FUSE_IO_URING) += dev_uring.o
diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c
index 9ded41a94fcc..660d3e36a6e0 100644
--- a/fs/fuse/backing.c
+++ b/fs/fuse/backing.c
@@ -32,6 +32,10 @@ static void fuse_backing_free(struct fuse_backing *fb)
case FUSE_BACKING_DAXDEV:
fs_put_dax(fb->dax_dev, fb);
break;
+
+ case FUSE_BACKING_EXTMAP:
+ fuse_ext_map_destroy(&fb->extents);
+ break;
}
kfree_rcu(fb, rcu);
}
@@ -84,7 +88,7 @@ static const struct rhashtable_params fuse_backing_prm = {
.key_len = sizeof_field(struct fuse_backing, backing_id),
};
-static int fuse_backing_add_64(struct fuse_conn *fc, struct fuse_backing *fb)
+int fuse_backing_add_64(struct fuse_conn *fc, struct fuse_backing *fb)
{
return rhashtable_lookup_insert_fast(&fc->backing_64_ht, &fb->hash_node, fuse_backing_prm);
}
@@ -316,12 +320,38 @@ static void fuse_backing_rht_free(void *p, void *data)
void fuse_backing_files_free(struct fuse_conn *fc)
{
- if (fc->backing_id_64) {
- rhashtable_free_and_destroy(&fc->backing_64_ht, fuse_backing_rht_free, NULL);
- } else {
+ struct rhashtable_iter iter;
+ struct fuse_backing *fb;
+
+ if (!fc->backing_id_64) {
idr_for_each(&fc->backing_files_map, fuse_backing_idr_free, NULL);
idr_destroy(&fc->backing_files_map);
+ return;
}
+
+ /*
+ * extents are referencing other backings, put these refs before
+ * destroying the backings themselves
+ */
+ rhashtable_walk_enter(&fc->backing_64_ht, &iter);
+ rhashtable_walk_start(&iter);
+ while ((fb = rhashtable_walk_next(&iter))) {
+ if (IS_ERR(fb)) {
+ if (PTR_ERR(fb) == -EAGAIN)
+ continue;
+ break;
+ }
+ if (fb->type == FUSE_BACKING_EXTMAP && !RB_EMPTY_ROOT(&fb->extents)) {
+ rhashtable_walk_stop(&iter);
+ fuse_ext_map_destroy(&fb->extents);
+ fb->extents.rb_node = NULL;
+ rhashtable_walk_start(&iter);
+ }
+ }
+ rhashtable_walk_stop(&iter);
+ rhashtable_walk_exit(&iter);
+
+ rhashtable_free_and_destroy(&fc->backing_64_ht, fuse_backing_rht_free, NULL);
}
void fuse_backing_files_init_64(struct fuse_conn *fc)
diff --git a/fs/fuse/ext_map.c b/fs/fuse/ext_map.c
new file mode 100644
index 000000000000..afac5de51c80
--- /dev/null
+++ b/fs/fuse/ext_map.c
@@ -0,0 +1,131 @@
+// SPDX-License-Identifier: GPL-2.0-only
+
+#include "fuse_i.h"
+#include <linux/rbtree.h>
+#include <linux/pagemap.h>
+
+struct fuse_iext {
+ struct rb_node rb;
+ u64 start;
+ u64 end; /* exclusive */
+ struct fuse_backing *backing;
+ u64 backing_offset;
+};
+
+void fuse_ext_map_destroy(struct rb_root *extents)
+{
+ struct fuse_iext *fie, *tmp;
+
+ rbtree_postorder_for_each_entry_safe(fie, tmp, extents, rb) {
+ fuse_backing_put(fie->backing);
+ kfree(fie);
+ }
+}
+
+static int fuse_add_extent(struct fuse_conn *fc, struct rb_root *extents,
+ struct fuse_extent *ext)
+{
+ struct fuse_iext *new_fie __free(kfree) = kzalloc_obj(*new_fie);
+ struct rb_node *parent = NULL, **link = &extents->rb_node;
+ loff_t end;
+ u64 tmp;
+
+ if (!new_fie)
+ return -ENOMEM;
+
+ if (ext->reserved[0] || ext->reserved[1])
+ return fuse_EIO("reserved fields set");
+
+ if (!PAGE_ALIGNED(ext->offset) || !PAGE_ALIGNED(ext->length) || !PAGE_ALIGNED(ext->addr))
+ return fuse_EIO("not page aligned");
+
+ if (!ext->length)
+ return fuse_EIO("zero sized extent");
+
+ if (overflows_type(ext->offset, loff_t) ||
+ check_add_overflow(ext->offset, ext->length, &end) ||
+ check_add_overflow(ext->addr, ext->length, &tmp))
+ return fuse_EIO("offset overflow");
+
+ new_fie->start = ext->offset;
+ new_fie->end = end;
+ new_fie->backing_offset = ext->addr;
+
+ while (*link) {
+ struct fuse_iext *fie = rb_entry(*link, typeof(*fie), rb);
+
+ parent = *link;
+ if (new_fie->end <= fie->start)
+ link = &parent->rb_left;
+ else if (new_fie->start >= fie->end)
+ link = &parent->rb_right;
+ else
+ return fuse_EIO("overlap");
+ }
+
+ new_fie->backing = fuse_backing_lookup(fc, ext->backing_id);
+ if (!new_fie->backing)
+ return fuse_EIO("backing not found");
+
+ if (new_fie->backing->type != FUSE_BACKING_DAXDEV) {
+ fuse_backing_put(new_fie->backing);
+ return fuse_EIO("backing is not dax device");
+ }
+
+ rb_link_node(&new_fie->rb, parent, link);
+ rb_insert_color(&no_free_ptr(new_fie)->rb, extents);
+
+ return 0;
+}
+
+bool fuse_ext_map_is_dax(struct fuse_backing *fb)
+{
+ struct fuse_iext *fie;
+
+ if (WARN_ON(fb->type != FUSE_BACKING_EXTMAP || RB_EMPTY_ROOT(&fb->extents)))
+ return false;
+
+ fie = rb_entry(fb->extents.rb_node, typeof(*fie), rb);
+ return fie->backing->type == FUSE_BACKING_DAXDEV;
+}
+
+int fuse_ext_map_populate(struct fuse_conn *fc, struct fuse_notify_backing_map_out *arg,
+ struct fuse_extent *ext)
+{
+ struct fuse_backing *fb;
+ unsigned int i;
+ int err;
+
+ if (!arg->num_extents)
+ return fuse_EIO("no extents");
+
+ if (arg->flags & FUSE_BACKING_MAP_CREATE) {
+ fb = kzalloc_obj(*fb);
+ if (!fb)
+ return -ENOMEM;
+
+ refcount_set(&fb->count, 1);
+ fb->type = FUSE_BACKING_EXTMAP;
+ fb->backing_id = arg->backing_id;
+ } else {
+ /* Adding extents to an existing extmap backing is not yet supported */
+ return -EINVAL;
+ }
+
+ for (i = 0; i < arg->num_extents; i++) {
+ err = fuse_add_extent(fc, &fb->extents, &ext[i]);
+ if (err)
+ goto err_put;
+ }
+
+ err = 0;
+ if (arg->flags & FUSE_BACKING_MAP_CREATE) {
+ err = fuse_backing_add_64(fc, fb);
+ if (!err)
+ return 0;
+ }
+
+err_put:
+ fuse_backing_put(fb);
+ return err;
+}
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 8c65b6b8c840..512beacd4a44 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -24,7 +24,7 @@
#include <linux/backing-dev.h>
#include <linux/mutex.h>
#include <linux/rwsem.h>
-#include <linux/rbtree.h>
+#include <linux/rbtree_types.h>
#include <linux/poll.h>
#include <linux/workqueue.h>
#include <linux/kref.h>
@@ -92,6 +92,7 @@ struct fuse_submount_lookup {
enum fuse_backing_type {
FUSE_BACKING_PATH,
FUSE_BACKING_DAXDEV,
+ FUSE_BACKING_EXTMAP,
};
/* Container for data related to mapping to backing file */
@@ -107,6 +108,9 @@ struct fuse_backing {
struct dax_device *dax_dev;
bool dax_error;
};
+ struct {
+ struct rb_root extents;
+ };
};
u64 backing_id;
struct rhash_head hash_node;
@@ -1308,7 +1312,6 @@ void fuse_file_release(struct inode *inode, struct fuse_file *ff,
/* backing.c */
#ifdef CONFIG_FUSE_PASSTHROUGH
void fuse_backing_put(struct fuse_backing *fb);
-
#else
static inline void fuse_backing_put(struct fuse_backing *fb)
@@ -1317,6 +1320,7 @@ static inline void fuse_backing_put(struct fuse_backing *fb)
#endif
struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, u64 backing_id);
+int fuse_backing_add_64(struct fuse_conn *fc, struct fuse_backing *fb);
void fuse_backing_files_init(struct fuse_conn *fc);
void fuse_backing_files_init_64(struct fuse_conn *fc);
void fuse_backing_files_free(struct fuse_conn *fc);
@@ -1367,4 +1371,10 @@ extern void fuse_sysctl_unregister(void);
#define fuse_sysctl_unregister() do { } while (0)
#endif /* CONFIG_SYSCTL */
+/* ext_map.c */
+
+void fuse_ext_map_destroy(struct rb_root *extents);
+bool fuse_ext_map_is_dax(struct fuse_backing *fb);
+int fuse_ext_map_populate(struct fuse_conn *fc, struct fuse_notify_backing_map_out *arg,
+ struct fuse_extent *ext);
#endif /* _FS_FUSE_I_H */
diff --git a/fs/fuse/notify.c b/fs/fuse/notify.c
index cc0afffc2c00..90d1f5c053e6 100644
--- a/fs/fuse/notify.c
+++ b/fs/fuse/notify.c
@@ -434,6 +434,50 @@ static int fuse_notify_backing_remove(struct fuse_conn *fc, unsigned int size,
return fuse_backing_close_64(fc, outarg.backing_id);
}
+static int fuse_notify_map(struct fuse_conn *fc, unsigned int size,
+ struct fuse_copy_state *cs)
+{
+ struct fuse_notify_backing_map_out outarg;
+ struct fuse_extent *ext __free(kvfree) = NULL;
+ int err;
+
+ if (size < sizeof(outarg))
+ return -EINVAL;
+
+ err = fuse_copy_one(cs, &outarg, sizeof(outarg));
+ if (err)
+ return err;
+
+ if (!fc->backing_id_64)
+ return -EINVAL;
+
+ if (outarg.num_extents > FUSE_MAX_EXTENTS)
+ return -EINVAL;
+
+ size -= sizeof(outarg);
+ if (size != outarg.num_extents * sizeof(*ext))
+ return -EINVAL;
+
+ if (outarg.reserved[0] != 0 || outarg.reserved[1] != 0)
+ return -EINVAL;
+
+ if (outarg.flags & ~FUSE_BACKING_MAP_CREATE)
+ return -EINVAL;
+
+ if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
+ return -EOPNOTSUPP;
+
+ ext = kvmalloc_objs(*ext, outarg.num_extents);
+ if (!ext)
+ return -ENOMEM;
+
+ err = fuse_copy_one(cs, ext, size);
+ if (err)
+ return err;
+
+ return fuse_ext_map_populate(fc, &outarg, ext);
+}
+
int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
unsigned int size, struct fuse_copy_state *cs)
{
@@ -468,6 +512,9 @@ int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
case FUSE_NOTIFY_BACKING_REMOVE:
return fuse_notify_backing_remove(fc, size, cs);
+ case FUSE_NOTIFY_BACKING_MAP:
+ return fuse_notify_map(fc, size, cs);
+
default:
return -EINVAL;
}
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index 4b3904b3acb7..60060da98e8d 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -255,6 +255,7 @@
* - add FUSE_DEV_IOC_BACKING_CREATE, struct fuse_backing_create_in
* - add FUSE_NOTIFY_BACKING_REMOVE, struct fuse_notify_backing_remove_out
* - add backing_id_64 to fuse_open_out
+ * - add FUSE_NOTIFY_BACKING_MAP, fuse_notify_backing_map_out, fuse_extent, FUSE_BACKING_MAP_CREATE
*/
#ifndef _LINUX_FUSE_H
@@ -716,6 +717,7 @@ enum fuse_notify_code {
FUSE_NOTIFY_INC_EPOCH = 8,
FUSE_NOTIFY_PRUNE = 9,
FUSE_NOTIFY_BACKING_REMOVE = 10,
+ FUSE_NOTIFY_BACKING_MAP = 11,
};
/* The read buffer is required to be at least 8k, but may be much larger */
@@ -1213,6 +1215,30 @@ struct fuse_notify_backing_remove_out {
uint64_t reserved;
};
+/**
+ * notify_map flags
+ *
+ * FUSE_BACKING_MAP_CREATE: create backing with the supplied ID
+ */
+#define FUSE_BACKING_MAP_CREATE (1 << 0)
+
+struct fuse_notify_backing_map_out {
+ uint64_t backing_id;
+ uint32_t num_extents;
+ uint32_t flags;
+ uint64_t reserved[2];
+};
+
+#define FUSE_MAX_EXTENTS 1365 /* (1 << 16) / sizeof(struct fuse_extent) */
+
+struct fuse_extent {
+ uint64_t offset; /* offset of extent into parent backing */
+ uint64_t length; /* extent length */
+ uint64_t backing_id; /* target backing */
+ uint64_t addr; /* target offset within backing file/device */
+ uint64_t reserved[2];
+};
+
#define FUSE_SETUPMAPPING_FLAG_WRITE (1ull << 0)
#define FUSE_SETUPMAPPING_FLAG_READ (1ull << 1)
struct fuse_setupmapping_in {
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* [PATCH v4 8/9] fuse: add extent map I/O support
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
` (6 preceding siblings ...)
2026-10-08 11:19 ` [PATCH v4 7/9] fuse: add extent map data structure Miklos Szeredi
@ 2026-10-08 11:19 ` Miklos Szeredi
2026-10-08 11:36 ` sashiko-bot
2026-10-08 11:20 ` [PATCH v4 9/9] fuse: add support for striped backing Miklos Szeredi
8 siblings, 1 reply; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:19 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
Wire up read, write, splice and mmap operations for extent-mapped
files through the iomap/DAX infrastructure.
When a passthrough file is opened with an EXTMAP backing, I/O is
dispatched to the dax devices referenced by the extent map.
If an address is not mapped, EIO is returned on the I/O operation.
Originally-by: John Groves <john@groves.net>
Reviewed-by: Amir Goldstein <amir73il@gmail.com>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/fuse/backing.c | 15 ++++
fs/fuse/ext_map.c | 164 ++++++++++++++++++++++++++++++++++++++++++
fs/fuse/file.c | 11 +++
fs/fuse/fuse_i.h | 4 ++
fs/fuse/passthrough.c | 29 ++++++--
5 files changed, 217 insertions(+), 6 deletions(-)
diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c
index 660d3e36a6e0..c93e4a45dd6f 100644
--- a/fs/fuse/backing.c
+++ b/fs/fuse/backing.c
@@ -288,6 +288,21 @@ int fuse_backing_close(struct fuse_conn *fc, int backing_id)
return err;
}
+bool fuse_backing_is_dax(struct fuse_backing *fb)
+{
+ switch (fb->type) {
+ case FUSE_BACKING_PATH:
+ return false;
+ case FUSE_BACKING_DAXDEV:
+ return true;
+ case FUSE_BACKING_EXTMAP:
+ return fuse_ext_map_is_dax(fb);
+ default:
+ WARN_ON(1);
+ return false;
+ }
+}
+
struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, u64 backing_id)
{
struct fuse_backing *fb;
diff --git a/fs/fuse/ext_map.c b/fs/fuse/ext_map.c
index afac5de51c80..744940cbf36e 100644
--- a/fs/fuse/ext_map.c
+++ b/fs/fuse/ext_map.c
@@ -3,6 +3,8 @@
#include "fuse_i.h"
#include <linux/rbtree.h>
#include <linux/pagemap.h>
+#include <linux/iomap.h>
+#include <linux/dax.h>
struct fuse_iext {
struct rb_node rb;
@@ -22,6 +24,24 @@ void fuse_ext_map_destroy(struct rb_root *extents)
}
}
+static struct fuse_iext *fuse_find_extent(struct rb_root *extents, u64 offset)
+{
+ struct rb_node *node = extents->rb_node;
+
+ while (node) {
+ struct fuse_iext *fie = rb_entry(node, typeof(*fie), rb);
+
+ if (offset < fie->start)
+ node = node->rb_left;
+ else if (offset >= fie->end)
+ node = node->rb_right;
+ else
+ return fie;
+ }
+
+ return NULL;
+}
+
static int fuse_add_extent(struct fuse_conn *fc, struct rb_root *extents,
struct fuse_extent *ext)
{
@@ -78,6 +98,150 @@ static int fuse_add_extent(struct fuse_conn *fc, struct rb_root *extents,
return 0;
}
+static int fuse_ext_map_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
+ unsigned int flags, struct iomap *iomap, struct iomap *srcmap)
+{
+ struct fuse_backing *fb = fuse_inode_backing(get_fuse_inode(inode));
+ struct fuse_iext *fie;
+ loff_t ext_len;
+
+ if (!fb || fb->type != FUSE_BACKING_EXTMAP)
+ return fuse_EIO("missing or wrong type backing");
+
+ fie = fuse_find_extent(&fb->extents, offset);
+ if (!fie)
+ return fuse_EIO("missing mapping");
+
+ if (WARN_ON(fie->backing->type != FUSE_BACKING_DAXDEV))
+ return fuse_EIO("wrong type backing for extent");
+
+ if (fie->backing->dax_error) {
+ fuse_make_bad(inode);
+ return fuse_EIO("dax error");
+ }
+
+ ext_len = fie->end - fie->start;
+
+ iomap->offset = fie->start;
+ iomap->addr = fie->backing_offset;
+ iomap->length = ext_len;
+ iomap->dax_dev = fie->backing->dax_dev;
+ iomap->type = IOMAP_MAPPED;
+ iomap->flags = 0;
+
+ return 0;
+}
+
+static const struct iomap_ops fuse_ext_map_iomap_ops = {
+ .iomap_begin = fuse_ext_map_iomap_begin,
+};
+
+static vm_fault_t fuse_ext_map_huge_fault(struct vm_fault *vmf, unsigned int order)
+{
+ struct inode *inode = file_inode(vmf->vma->vm_file);
+ bool write_fault = (vmf->flags & FAULT_FLAG_WRITE) && (vmf->vma->vm_flags & VM_SHARED);
+ vm_fault_t ret;
+ unsigned long pfn;
+
+ if (!IS_ENABLED(CONFIG_FS_DAX))
+ return VM_FAULT_SIGBUS;
+
+ if (WARN_ON_ONCE(!IS_DAX(inode)))
+ return VM_FAULT_SIGBUS;
+
+ if (write_fault) {
+ sb_start_pagefault(inode->i_sb);
+ file_update_time(vmf->vma->vm_file);
+ }
+
+ filemap_invalidate_lock_shared(inode->i_mapping);
+
+ ret = dax_iomap_fault(vmf, order, &pfn, NULL, &fuse_ext_map_iomap_ops);
+ if (ret & VM_FAULT_NEEDDSYNC)
+ ret = dax_finish_sync_fault(vmf, order, pfn);
+
+ filemap_invalidate_unlock_shared(inode->i_mapping);
+
+ if (write_fault)
+ sb_end_pagefault(inode->i_sb);
+
+ return ret;
+}
+
+static vm_fault_t fuse_ext_map_fault(struct vm_fault *vmf)
+{
+ return fuse_ext_map_huge_fault(vmf, 0);
+}
+
+static const struct vm_operations_struct fuse_ext_map_vm_ops = {
+ .fault = fuse_ext_map_fault,
+ .huge_fault = fuse_ext_map_huge_fault,
+ .page_mkwrite = fuse_ext_map_fault,
+ .pfn_mkwrite = fuse_ext_map_fault,
+};
+
+static void fuse_rw_clamp(struct kiocb *iocb, struct iov_iter *ubuf)
+{
+ struct inode *inode = iocb->ki_filp->f_mapping->host;
+ loff_t i_size = i_size_read(inode);
+ loff_t max_count = iocb->ki_pos >= i_size ? 0 : i_size - iocb->ki_pos;
+
+ if (iov_iter_count(ubuf) > max_count)
+ iov_iter_truncate(ubuf, max_count);
+}
+
+ssize_t fuse_ext_map_read_iter(struct kiocb *iocb, struct iov_iter *to)
+{
+ struct inode *inode = iocb->ki_filp->f_mapping->host;
+ ssize_t res;
+
+ guard(rwsem_read)(&inode->i_rwsem);
+
+ fuse_rw_clamp(iocb, to);
+
+ if (!iov_iter_count(to))
+ return 0;
+
+ if (!IS_ENABLED(CONFIG_FS_DAX))
+ return -EIO;
+
+ res = dax_iomap_rw(iocb, to, &fuse_ext_map_iomap_ops);
+
+ file_accessed(iocb->ki_filp);
+ return res;
+}
+
+ssize_t fuse_ext_map_write_iter(struct kiocb *iocb, struct iov_iter *from)
+{
+ ssize_t res;
+
+ res = generic_write_checks(iocb, from);
+ if (res <= 0)
+ return res;
+
+ fuse_rw_clamp(iocb, from);
+
+ if (!iov_iter_count(from))
+ return -ENOSPC;
+
+ res = kiocb_modified(iocb);
+ if (res)
+ return res;
+
+ if (!IS_ENABLED(CONFIG_FS_DAX))
+ return -EIO;
+
+ return dax_iomap_rw(iocb, from, &fuse_ext_map_iomap_ops);
+}
+
+int fuse_ext_map_mmap(struct file *file, struct vm_area_struct *vma)
+{
+ file_accessed(file);
+ vma->vm_ops = &fuse_ext_map_vm_ops;
+ vm_flags_set(vma, VM_HUGEPAGE);
+ return 0;
+}
+
bool fuse_ext_map_is_dax(struct fuse_backing *fb)
{
struct fuse_iext *fie;
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
index e0d72b9d3c4a..48d01868e7a7 100644
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -551,6 +551,17 @@ static int fuse_fsync(struct file *file, loff_t start, loff_t end,
if (fuse_is_bad(inode))
return -EIO;
+ if (IS_DAX(inode) && !fuse_inode_vdax(inode)) {
+ /*
+ * Note: the DAX fault path calls this via dax_finish_sync_fault()
+ * and taking inode lock in that context is prohibited.
+ *
+ * FIXME: need to flush CPU caches for the DAX memory range
+ * FIXME: need to sync metadata to the server
+ */
+ return 0;
+ }
+
inode_lock(inode);
/*
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 512beacd4a44..8a9961c4c924 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -1321,6 +1321,7 @@ static inline void fuse_backing_put(struct fuse_backing *fb)
struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, u64 backing_id);
int fuse_backing_add_64(struct fuse_conn *fc, struct fuse_backing *fb);
+bool fuse_backing_is_dax(struct fuse_backing *fb);
void fuse_backing_files_init(struct fuse_conn *fc);
void fuse_backing_files_init_64(struct fuse_conn *fc);
void fuse_backing_files_free(struct fuse_conn *fc);
@@ -1374,6 +1375,9 @@ extern void fuse_sysctl_unregister(void);
/* ext_map.c */
void fuse_ext_map_destroy(struct rb_root *extents);
+ssize_t fuse_ext_map_write_iter(struct kiocb *iocb, struct iov_iter *from);
+ssize_t fuse_ext_map_read_iter(struct kiocb *iocb, struct iov_iter *to);
+int fuse_ext_map_mmap(struct file *file, struct vm_area_struct *vma);
bool fuse_ext_map_is_dax(struct fuse_backing *fb);
int fuse_ext_map_populate(struct fuse_conn *fc, struct fuse_notify_backing_map_out *arg,
struct fuse_extent *ext);
diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c
index e9ab1aea34e2..040817ad76e9 100644
--- a/fs/fuse/passthrough.c
+++ b/fs/fuse/passthrough.c
@@ -35,7 +35,6 @@ ssize_t fuse_passthrough_read_iter(struct kiocb *iocb, struct iov_iter *iter)
struct fuse_file *ff = file->private_data;
struct file *backing_file = fuse_file_passthrough(ff);
size_t count = iov_iter_count(iter);
- ssize_t ret;
struct backing_file_ctx ctx = {
.cred = ff->cred,
.accessed = fuse_file_accessed,
@@ -48,10 +47,10 @@ ssize_t fuse_passthrough_read_iter(struct kiocb *iocb, struct iov_iter *iter)
if (!count)
return 0;
- ret = backing_file_read_iter(backing_file, iter, iocb, iocb->ki_flags,
- &ctx);
+ if (!backing_file)
+ return fuse_ext_map_read_iter(iocb, iter);
- return ret;
+ return backing_file_read_iter(backing_file, iter, iocb, iocb->ki_flags, &ctx);
}
ssize_t fuse_passthrough_write_iter(struct kiocb *iocb,
@@ -74,10 +73,13 @@ ssize_t fuse_passthrough_write_iter(struct kiocb *iocb,
if (!count)
return 0;
- inode_lock(inode);
+ guard(rwsem_write)(&inode->i_rwsem);
+
+ if (!backing_file)
+ return fuse_ext_map_write_iter(iocb, iter);
+
ret = backing_file_write_iter(backing_file, iter, iocb, iocb->ki_flags,
&ctx);
- inode_unlock(inode);
return ret;
}
@@ -98,6 +100,9 @@ ssize_t fuse_passthrough_splice_read(struct file *in, loff_t *ppos,
pr_debug("%s: backing_file=0x%p, pos=%lld, len=%zu, flags=0x%x\n", __func__,
backing_file, *ppos, len, flags);
+ if (!backing_file)
+ return copy_splice_read(in, ppos, pipe, len, flags);
+
init_sync_kiocb(&iocb, in);
iocb.ki_pos = *ppos;
ret = backing_file_splice_read(backing_file, &iocb, pipe, len, flags, &ctx);
@@ -123,6 +128,9 @@ ssize_t fuse_passthrough_splice_write(struct pipe_inode_info *pipe,
pr_debug("%s: backing_file=0x%p, pos=%lld, len=%zu, flags=0x%x\n", __func__,
backing_file, *ppos, len, flags);
+ if (!backing_file)
+ return iter_file_splice_write(pipe, out, ppos, len, flags);
+
inode_lock(inode);
init_sync_kiocb(&iocb, out);
iocb.ki_pos = *ppos;
@@ -145,6 +153,9 @@ ssize_t fuse_passthrough_mmap(struct file *file, struct vm_area_struct *vma)
pr_debug("%s: backing_file=0x%p, start=%lu, end=%lu\n", __func__,
backing_file, vma->vm_start, vma->vm_end);
+ if (!backing_file)
+ return fuse_ext_map_mmap(file, vma);
+
return backing_file_mmap(backing_file, vma, &ctx);
}
@@ -156,6 +167,12 @@ int fuse_passthrough_open(struct file *file, struct fuse_backing *fb)
struct fuse_file *ff = file->private_data;
struct file *backing_file;
+ if (fb->type == FUSE_BACKING_EXTMAP) {
+ if (fuse_backing_is_dax(fb) != !!IS_DAX(file_inode(file)))
+ return fuse_EIO("dax mode mismatch");
+ return 0;
+ }
+
if (fb->type != FUSE_BACKING_PATH)
return fuse_EIO("invalid backing type");
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread* Re: [PATCH v4 8/9] fuse: add extent map I/O support
2026-10-08 11:19 ` [PATCH v4 8/9] fuse: add extent map I/O support Miklos Szeredi
@ 2026-10-08 11:36 ` sashiko-bot
0 siblings, 0 replies; 16+ messages in thread
From: sashiko-bot @ 2026-10-08 11:36 UTC (permalink / raw)
To: Miklos Szeredi; +Cc: linux-cxl, nvdimm
Thank you for your contribution! Sashiko AI review found 3 potential issue(s) to consider:
- [High] fuse: silent data loss on fsync for DAX extent-mapped files
- [High] fuse: ABBA deadlock between pipe_lock and inode lock in splice write
- [High] fuse: concurrent EXTMAP DAX faults and truncate/punch_hole race due to missing lock
--
commit 5f3e8f02d15211d77c4f619bc19ce42ddb86c547
Author: Miklos Szeredi <mszeredi@redhat.com>
fuse: add extent map I/O support
Wire up read, write, splice and mmap operations for extent-mapped files
through the iomap/DAX infrastructure. When a passthrough file is opened
with an EXTMAP backing, I/O is dispatched to the dax devices referenced
by the extent map.
> diff --git a/fs/fuse/ext_map.c b/fs/fuse/ext_map.c
> index afac5de51c802..744940cbf36e1 100644
> --- a/fs/fuse/ext_map.c
> +++ b/fs/fuse/ext_map.c
[ ... ]
> +static vm_fault_t fuse_ext_map_huge_fault(struct vm_fault *vmf, unsigned int order)
> +{
> + struct inode *inode = file_inode(vmf->vma->vm_file);
> + bool write_fault = (vmf->flags & FAULT_FLAG_WRITE) && (vmf->vma->vm_flags & VM_SHARED);
> + vm_fault_t ret;
> + unsigned long pfn;
> +
> + if (!IS_ENABLED(CONFIG_FS_DAX))
> + return VM_FAULT_SIGBUS;
> +
> + if (WARN_ON_ONCE(!IS_DAX(inode)))
> + return VM_FAULT_SIGBUS;
> +
> + if (write_fault) {
> + sb_start_pagefault(inode->i_sb);
> + file_update_time(vmf->vma->vm_file);
> + }
> +
> + filemap_invalidate_lock_shared(inode->i_mapping);
[Severity: High]
Does this allow concurrent truncate or punch_hole operations to race with page
faults?
The page fault handler acquires filemap_invalidate_lock_shared() here,
expecting writers to hold it exclusively during invalidation. However, writer
paths like fuse_do_setattr() and fuse_file_fallocate() check FUSE_IS_VDAX()
to decide whether to acquire the exclusive filemap_invalidate_lock. For EXTMAP
DAX files, FUSE_IS_VDAX() evaluates to false, causing the exclusive lock
acquisition to be skipped. Consequently, page faults are not blocked during
truncate_pagecache() or truncate_pagecache_range().
Can this leave stale DAX entries in the mapping tree, leading to
use-after-free of storage blocks when memory-mapped I/O accesses blocks that
have been freed or punched out?
> +
> + ret = dax_iomap_fault(vmf, order, &pfn, NULL, &fuse_ext_map_iomap_ops);
> + if (ret & VM_FAULT_NEEDDSYNC)
> + ret = dax_finish_sync_fault(vmf, order, pfn);
> +
> + filemap_invalidate_unlock_shared(inode->i_mapping);
> +
> + if (write_fault)
> + sb_end_pagefault(inode->i_sb);
> +
> + return ret;
> +}
[ ... ]
> diff --git a/fs/fuse/file.c b/fs/fuse/file.c
> index b5bef7b9e466e..77d7a54066e47 100644
> --- a/fs/fuse/file.c
> +++ b/fs/fuse/file.c
> @@ -551,6 +551,17 @@ static int fuse_fsync(struct file *file, loff_t start, loff_t end,
> if (fuse_is_bad(inode))
> return -EIO;
>
> + if (IS_DAX(inode) && !fuse_inode_vdax(inode)) {
> + /*
> + * Note: the DAX fault path calls this via dax_finish_sync_fault()
> + * and taking inode lock in that context is prohibited.
> + *
> + * FIXME: need to flush CPU caches for the DAX memory range
> + * FIXME: need to sync metadata to the server
> + */
> + return 0;
[Severity: High]
Can this cause silent data loss on fsync() for DAX extent-mapped files?
When an application calls fsync() on a FUSE DAX extent-mapped file, this
bypasses flushing the CPU caches for the DAX memory range or syncing metadata
to the server, while still returning 0 (success).
Because the operation falsely reports success, applications relying on fsync()
or MAP_SYNC (via dax_finish_sync_fault()) to ensure data durability will not
take fallback actions. Could this lead to silent data loss on a power failure
or crash?
> + }
> +
> inode_lock(inode);
>
> /*
> diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c
> index e9ab1aea34e27..040817ad76e94 100644
> --- a/fs/fuse/passthrough.c
> +++ b/fs/fuse/passthrough.c
[ ... ]
> @@ -74,10 +73,13 @@ ssize_t fuse_passthrough_write_iter(struct kiocb *iocb,
> if (!count)
> return 0;
>
> - inode_lock(inode);
> + guard(rwsem_write)(&inode->i_rwsem);
> +
> + if (!backing_file)
> + return fuse_ext_map_write_iter(iocb, iter);
[Severity: High]
Does this introduce an ABBA deadlock between pipe_lock and the inode lock?
For normal passthrough, fuse_passthrough_splice_write() acquires the inode
lock first, and then calls backing_file_splice_write(), which ultimately
acquires pipe_lock() via the backing file's splice_write callback.
However, the EXTMAP DAX path added below for splice_write() directly calls
iter_file_splice_write(), which acquires pipe_lock() first. That function then
invokes fuse_passthrough_write_iter() where the inode lock is acquired here
via guard(rwsem_write).
If a malicious FUSE server provides a normal passthrough backing for one open
and an EXTMAP backing for another open of the same inode, and two threads
perform splice writes using the same pipe and the two different file
descriptors, it results in a literal ABBA deadlock. Even operating on
different inodes would trigger a lockdep lock inversion splat.
> +
> ret = backing_file_write_iter(backing_file, iter, iocb, iocb->ki_flags,
> &ctx);
> - inode_unlock(inode);
[ ... ]
> @@ -123,6 +128,9 @@ ssize_t fuse_passthrough_splice_write(struct pipe_inode_info *pipe,
> pr_debug("%s: backing_file=0x%p, pos=%lld, len=%zu, flags=0x%x\n", __func__,
> backing_file, *ppos, len, flags);
>
> + if (!backing_file)
> + return iter_file_splice_write(pipe, out, ppos, len, flags);
> +
> inode_lock(inode);
> init_sync_kiocb(&iocb, out);
> iocb.ki_pos = *ppos;
--
Sashiko AI review · https://sashiko.dev/#/patchset/20261008112004.1899560-1-mszeredi@redhat.com?part=8
^ permalink raw reply [flat|nested] 16+ messages in thread
* [PATCH v4 9/9] fuse: add support for striped backing
2026-10-08 11:19 [PATCH v4 0/9] fuse: DAX device based extent maps (famfs) Miklos Szeredi
` (7 preceding siblings ...)
2026-10-08 11:19 ` [PATCH v4 8/9] fuse: add extent map I/O support Miklos Szeredi
@ 2026-10-08 11:20 ` Miklos Szeredi
8 siblings, 0 replies; 16+ messages in thread
From: Miklos Szeredi @ 2026-10-08 11:20 UTC (permalink / raw)
To: fuse-devel
Cc: John Groves, Amir Goldstein, Darrick J . Wong, Vishal Verma,
Dave Jiang, Alison Schofield, nvdimm, linux-cxl
Add the FUSE_BACKING_MAP_CYCLIC flag for FUSE_NOTIFY_BACKING_MAP
notifications to enable creating striped maps.
This is just a set of regular extents that repeats after the end of the
last extent using a striping pattern (i.e. on the target backings the
stripes have a contiguous footprint).
Two additional limits apply compared to non-cyclic extent map population:
- the size of the extents (chunk size) must be equal
- the extents must start from zero offset and follow each other without
gaps.
Originally-by: John Groves <john@groves.net>
Reviewed-by: Amir Goldstein <amir73il@gmail.com>
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
---
fs/fuse/ext_map.c | 30 +++++++++++++++++++++++++++---
fs/fuse/fuse_i.h | 1 +
fs/fuse/notify.c | 2 +-
include/uapi/linux/fuse.h | 5 ++++-
4 files changed, 33 insertions(+), 5 deletions(-)
diff --git a/fs/fuse/ext_map.c b/fs/fuse/ext_map.c
index 744940cbf36e..7adb78cf552d 100644
--- a/fs/fuse/ext_map.c
+++ b/fs/fuse/ext_map.c
@@ -103,12 +103,18 @@ static int fuse_ext_map_iomap_begin(struct inode *inode, loff_t offset, loff_t l
{
struct fuse_backing *fb = fuse_inode_backing(get_fuse_inode(inode));
struct fuse_iext *fie;
+ u64 ncycle = 0, seq_off = 0, addr, tmp;
loff_t ext_len;
if (!fb || fb->type != FUSE_BACKING_EXTMAP)
return fuse_EIO("missing or wrong type backing");
- fie = fuse_find_extent(&fb->extents, offset);
+ if (fb->cycle_length) {
+ ncycle = div64_u64(offset, fb->cycle_length);
+ seq_off = ncycle * fb->cycle_length;
+ }
+
+ fie = fuse_find_extent(&fb->extents, offset - seq_off);
if (!fie)
return fuse_EIO("missing mapping");
@@ -121,9 +127,15 @@ static int fuse_ext_map_iomap_begin(struct inode *inode, loff_t offset, loff_t l
}
ext_len = fie->end - fie->start;
+ addr = fie->backing_offset;
+ if (check_mul_overflow(ncycle, ext_len, &tmp) || check_add_overflow(addr, tmp, &addr))
+ return fuse_EIO("extent address overflow");
- iomap->offset = fie->start;
- iomap->addr = fie->backing_offset;
+ if (check_add_overflow(addr, ext_len, &tmp))
+ return fuse_EIO("extent end overflow");
+
+ iomap->offset = fie->start + seq_off;
+ iomap->addr = addr;
iomap->length = ext_len;
iomap->dax_dev = fie->backing->dax_dev;
iomap->type = IOMAP_MAPPED;
@@ -259,6 +271,7 @@ int fuse_ext_map_populate(struct fuse_conn *fc, struct fuse_notify_backing_map_o
struct fuse_backing *fb;
unsigned int i;
int err;
+ u64 chunk_size = 0;
if (!arg->num_extents)
return fuse_EIO("no extents");
@@ -276,7 +289,18 @@ int fuse_ext_map_populate(struct fuse_conn *fc, struct fuse_notify_backing_map_o
return -EINVAL;
}
+ if (arg->flags & FUSE_BACKING_MAP_CYCLIC) {
+ chunk_size = ext[0].length;
+ err = -EINVAL;
+ if (check_mul_overflow(chunk_size, arg->num_extents, &fb->cycle_length))
+ goto err_put;
+ }
+
for (i = 0; i < arg->num_extents; i++) {
+ err = -EINVAL;
+ if (chunk_size && (ext[i].offset != chunk_size * i || ext[i].length != chunk_size))
+ goto err_put;
+
err = fuse_add_extent(fc, &fb->extents, &ext[i]);
if (err)
goto err_put;
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 8a9961c4c924..064fed459b93 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -110,6 +110,7 @@ struct fuse_backing {
};
struct {
struct rb_root extents;
+ u64 cycle_length;
};
};
u64 backing_id;
diff --git a/fs/fuse/notify.c b/fs/fuse/notify.c
index 90d1f5c053e6..984e4945e3e3 100644
--- a/fs/fuse/notify.c
+++ b/fs/fuse/notify.c
@@ -461,7 +461,7 @@ static int fuse_notify_map(struct fuse_conn *fc, unsigned int size,
if (outarg.reserved[0] != 0 || outarg.reserved[1] != 0)
return -EINVAL;
- if (outarg.flags & ~FUSE_BACKING_MAP_CREATE)
+ if (outarg.flags & ~(FUSE_BACKING_MAP_CREATE | FUSE_BACKING_MAP_CYCLIC))
return -EINVAL;
if (!IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index 60060da98e8d..6b2fe68b4ab5 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -255,7 +255,8 @@
* - add FUSE_DEV_IOC_BACKING_CREATE, struct fuse_backing_create_in
* - add FUSE_NOTIFY_BACKING_REMOVE, struct fuse_notify_backing_remove_out
* - add backing_id_64 to fuse_open_out
- * - add FUSE_NOTIFY_BACKING_MAP, fuse_notify_backing_map_out, fuse_extent, FUSE_BACKING_MAP_CREATE
+ * - add FUSE_NOTIFY_BACKING_MAP, fuse_notify_backing_map_out, fuse_extent
+ * - add FUSE_BACKING_MAP_CREATE, FUSE_BACKING_MAP_CYCLIC
*/
#ifndef _LINUX_FUSE_H
@@ -1219,8 +1220,10 @@ struct fuse_notify_backing_remove_out {
* notify_map flags
*
* FUSE_BACKING_MAP_CREATE: create backing with the supplied ID
+ * FUSE_BACKING_MAP_CYCLIC: map repeats after last extent
*/
#define FUSE_BACKING_MAP_CREATE (1 << 0)
+#define FUSE_BACKING_MAP_CYCLIC (1 << 1)
struct fuse_notify_backing_map_out {
uint64_t backing_id;
--
2.54.0
^ permalink raw reply related [flat|nested] 16+ messages in thread