From: John Groves Virtio_fs now needs to determine if an inode is DAX && not famfs. This replaces the FUSE_IS_DAX() macro with FUSE_IS_VIRTIO_DAX(), in preparation for famfs in later commits. The dummy fuse_file_famfs() macro will be replaced with a working function. Reviewed-by: Joanne Koong Reviewed-by: Dave Jiang Signed-off-by: John Groves --- fs/fuse/dir.c | 2 +- fs/fuse/file.c | 16 ++++++++++------ fs/fuse/fuse_i.h | 9 ++++++++- fs/fuse/inode.c | 4 ++-- fs/fuse/iomode.c | 2 +- 5 files changed, 22 insertions(+), 11 deletions(-) diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index 0e2a1039fa43..850d624c2eee 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -2157,7 +2157,7 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, is_truncate = true; } - if (FUSE_IS_DAX(inode) && is_truncate) { + if (FUSE_IS_VIRTIO_DAX(fi) && is_truncate) { filemap_invalidate_lock(mapping); fault_blocked = true; err = fuse_dax_break_layouts(inode, 0, -1); diff --git a/fs/fuse/file.c b/fs/fuse/file.c index ceada75310b8..995e37c935c6 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -115,11 +115,12 @@ static void fuse_file_put(struct fuse_file *ff, bool sync) fuse_simple_request(ff->fm, args); fuse_release_end(args, 0); } else { + struct fuse_inode *fi = get_fuse_inode(ra->inode); /* * DAX inodes may need to issue a number of synchronous * request for clearing the mappings. */ - if (ra && ra->inode && FUSE_IS_DAX(ra->inode)) + if (ra && ra->inode && FUSE_IS_VIRTIO_DAX(fi)) args->may_block = true; args->end = fuse_release_end; if (fuse_simple_background(ff->fm, args, @@ -256,7 +257,7 @@ static int fuse_open(struct inode *inode, struct file *file) int err; bool is_truncate = (file->f_flags & O_TRUNC) && fc->atomic_o_trunc; bool is_wb_truncate = is_truncate && fc->writeback_cache; - bool dax_truncate = is_truncate && FUSE_IS_DAX(inode); + bool dax_truncate = is_truncate && FUSE_IS_VIRTIO_DAX(fi); if (fuse_is_bad(inode)) return -EIO; @@ -1839,11 +1840,12 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; struct inode *inode = file_inode(file); + struct fuse_inode *fi = get_fuse_inode(inode); if (fuse_is_bad(inode)) return -EIO; - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VIRTIO_DAX(fi)) return fuse_dax_read_iter(iocb, to); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ @@ -1860,11 +1862,12 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) struct file *file = iocb->ki_filp; struct fuse_file *ff = file->private_data; struct inode *inode = file_inode(file); + struct fuse_inode *fi = get_fuse_inode(inode); if (fuse_is_bad(inode)) return -EIO; - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VIRTIO_DAX(fi)) return fuse_dax_write_iter(iocb, from); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ @@ -2399,10 +2402,11 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) struct fuse_file *ff = file->private_data; struct fuse_conn *fc = ff->fm->fc; struct inode *inode = file_inode(file); + struct fuse_inode *fi = get_fuse_inode(inode); int rc; /* DAX mmap is superior to direct_io mmap */ - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VIRTIO_DAX(fi)) return fuse_dax_mmap(file, vma); /* @@ -2852,7 +2856,7 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, .mode = mode }; int err; - bool block_faults = FUSE_IS_DAX(inode) && + bool block_faults = FUSE_IS_VIRTIO_DAX(fi) && (!(mode & FALLOC_FL_KEEP_SIZE) || (mode & (FALLOC_FL_PUNCH_HOLE | FALLOC_FL_ZERO_RANGE))); diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 85f738c53122..f450194e877f 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1215,7 +1215,14 @@ void fuse_free_conn(struct fuse_conn *fc); /* dax.c */ -#define FUSE_IS_DAX(inode) (IS_ENABLED(CONFIG_FUSE_DAX) && IS_DAX(inode)) +static inline bool fuse_file_famfs(struct fuse_inode *fuse_inode) /* Will be superseded */ +{ + (void)fuse_inode; + return false; +} +#define FUSE_IS_VIRTIO_DAX(fuse_inode) (IS_ENABLED(CONFIG_FUSE_DAX) \ + && IS_DAX(&(fuse_inode)->inode) \ + && !fuse_file_famfs(fuse_inode)) ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to); ssize_t fuse_dax_write_iter(struct kiocb *iocb, struct iov_iter *from); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index d975073c6029..77c21b28b6fa 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -148,7 +148,7 @@ static void fuse_evict_inode(struct inode *inode) /* Will write inode on close/munmap and in all other dirtiers */ WARN_ON(inode_state_read_once(inode) & I_DIRTY_INODE); - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VIRTIO_DAX(fi)) dax_break_layout_final(inode); truncate_inode_pages_final(&inode->i_data); @@ -156,7 +156,7 @@ static void fuse_evict_inode(struct inode *inode) if (inode->i_sb->s_flags & SB_ACTIVE) { struct fuse_conn *fc = get_fuse_conn(inode); - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VIRTIO_DAX(fi)) fuse_dax_inode_cleanup(inode); if (fi->nlookup) { fuse_chan_queue_forget(fc->chan, fi->forget, fi->nodeid, diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c index 3728933188f3..31ee7f3304c6 100644 --- a/fs/fuse/iomode.c +++ b/fs/fuse/iomode.c @@ -203,7 +203,7 @@ int fuse_file_io_open(struct file *file, struct inode *inode) * io modes are not relevant with DAX and with server that does not * implement open. */ - if (FUSE_IS_DAX(inode) || !ff->args) + if (FUSE_IS_VIRTIO_DAX(fi) || !ff->args) return 0; /* -- 2.53.0 From: John Groves Add the minimal fuse plumbing that later famfs commits build on: - Kconfig: add FUSE_FAMFS_DAX to control compilation of famfs support within fuse. - uapi: add the FUSE_DAX_FMAP flag for the INIT request/reply, by which the server and kernel negotiate famfs (in-kernel fs-dax map) support. - fuse_conn->famfs_iomap to mark a famfs-enabled connection. Reviewed-by: Joanne Koong Reviewed-by: Dave Jiang Signed-off-by: John Groves --- fs/fuse/Kconfig | 13 +++++++++++++ fs/fuse/fuse_i.h | 3 +++ fs/fuse/inode.c | 6 ++++++ include/uapi/linux/fuse.h | 5 +++++ 4 files changed, 27 insertions(+) diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index 3a4ae632c94a..17fe1f490cbd 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -76,3 +76,16 @@ config FUSE_IO_URING If you want to allow fuse server/client communication through io-uring, answer Y + +config FUSE_FAMFS_DAX + bool "FUSE support for fs-dax filesystems backed by devdax" + depends on FUSE_FS + depends on DEV_DAX_FSDEV + default FUSE_FS + help + This enables the fabric-attached memory file system (famfs), + which enables formatting devdax memory as a file system. Famfs + is primarily intended for scale-out shared access to + disaggregated memory. + + To enable famfs or other fuse/fs-dax file systems, answer Y diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index f450194e877f..9c354118c931 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -711,6 +711,9 @@ struct fuse_conn { /** @sync_init: Is synchronous FUSE_INIT allowed? */ unsigned int sync_init:1; + /** @famfs_iomap: dev_dax_iomap support for famfs */ + unsigned int famfs_iomap:1; + /** @max_stack_depth: Maximum stack depth for passthrough backing files */ int max_stack_depth; diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 77c21b28b6fa..c347471d04b6 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1406,6 +1406,10 @@ static void process_init_reply(struct fuse_args *args, int error) if (flags & FUSE_REQUEST_TIMEOUT) timeout = arg->request_timeout; + + if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) && + flags & FUSE_DAX_FMAP) + fc->famfs_iomap = 1; } else { ra_pages = fc->max_read / PAGE_SIZE; fc->no_lock = 1; @@ -1473,6 +1477,8 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm) flags |= FUSE_SUBMOUNTS; if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) flags |= FUSE_PASSTHROUGH; + if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)) + flags |= FUSE_DAX_FMAP; /* * This is just an information flag for fuse server. No need to check diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index c13e1f9a2f12..25686f088e6a 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -240,6 +240,9 @@ * - add FUSE_COPY_FILE_RANGE_64 * - add struct fuse_copy_file_range_out * - add FUSE_NOTIFY_PRUNE + * + * 7.46 + * - Add FUSE_DAX_FMAP capability - ability to handle in-kernel fsdax maps */ #ifndef _LINUX_FUSE_H @@ -448,6 +451,7 @@ struct fuse_file_lock { * FUSE_OVER_IO_URING: Indicate that client supports io-uring * FUSE_REQUEST_TIMEOUT: kernel supports timing out requests. * init_out.request_timeout contains the timeout (in secs) + * FUSE_DAX_FMAP: kernel supports dev_dax_iomap (aka famfs) fmaps */ #define FUSE_ASYNC_READ (1 << 0) #define FUSE_POSIX_LOCKS (1 << 1) @@ -495,6 +499,7 @@ struct fuse_file_lock { #define FUSE_ALLOW_IDMAP (1ULL << 40) #define FUSE_OVER_IO_URING (1ULL << 41) #define FUSE_REQUEST_TIMEOUT (1ULL << 42) +#define FUSE_DAX_FMAP (1ULL << 43) /** * CUSE INIT request/reply flags -- 2.53.0 From: John Groves On completion of an OPEN in famfs mode, issue a GET_FMAP request to the server to retrieve the file's file-offset-to-dax map (fmap) and cache it on the fuse_inode (fi->famfs_meta). Once the map is cached, read, write and mmap are resolved directly to dax with no further upcalls. - uapi: add the FUSE_GET_FMAP opcode. - famfs.c: add fuse_get_fmap(), which retrieves the fmap into a kvmalloc'd buffer. The fmap size is not known in advance, so it uses a size probe: it starts with a PAGE_SIZE buffer and passes that size to the server (via fuse_getxattr_in.size). If the whole fmap does not fit, the server replies with just the header, whose fmap_size field reports the required size, and the kernel reallocates exactly that and retries once. The reply is bounded by FMAP_BUFSIZE_MAX (16 MiB); a larger fmap is rejected with -EFBIG. A famfs file is fixed-size, so a reply that reports a different size on the retry is rejected as a server bug. - file.c: hook the OPEN path to fetch the fmap for regular files on a famfs connection; failure is fatal to the open. - fuse_i.h/inode.c: add the fi->famfs_meta pointer and its init/free helpers. The retrieved map is parsed into its in-memory form in the following patch. Signed-off-by: John Groves --- MAINTAINERS | 7 ++ fs/fuse/Makefile | 1 + fs/fuse/famfs.c | 134 ++++++++++++++++++++++++++++++++++++++ fs/fuse/file.c | 14 +++- fs/fuse/fuse_i.h | 70 ++++++++++++++++++-- fs/fuse/inode.c | 8 ++- fs/fuse/iomode.c | 2 +- include/uapi/linux/fuse.h | 3 + 8 files changed, 231 insertions(+), 8 deletions(-) create mode 100644 fs/fuse/famfs.c diff --git a/MAINTAINERS b/MAINTAINERS index 806bd2d80d15..0d0fded4fddb 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -10713,6 +10713,13 @@ F: fs/fuse/backing.c F: fs/fuse/iomode.c F: fs/fuse/passthrough.c +FUSE FILESYSTEM [FAMFS Fabric-Attached Memory File System] +M: John Groves +L: linux-cxl@vger.kernel.org +L: linux-fsdevel@vger.kernel.org +S: Supported +F: fs/fuse/famfs.c + FUTEX SUBSYSTEM M: Thomas Gleixner M: Ingo Molnar diff --git a/fs/fuse/Makefile b/fs/fuse/Makefile index 245e67852b03..66507b9cfe1f 100644 --- a/fs/fuse/Makefile +++ b/fs/fuse/Makefile @@ -18,5 +18,6 @@ fuse-$(CONFIG_FUSE_DAX) += dax.o fuse-$(CONFIG_FUSE_PASSTHROUGH) += passthrough.o backing.o fuse-$(CONFIG_SYSCTL) += sysctl.o fuse-$(CONFIG_FUSE_IO_URING) += dev_uring.o +fuse-$(CONFIG_FUSE_FAMFS_DAX) += famfs.o virtiofs-y := virtio_fs.o diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c new file mode 100644 index 000000000000..80e6640ac970 --- /dev/null +++ b/fs/fuse/famfs.c @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * famfs - dax file system for shared fabric-attached memory + * + * Copyright 2023-2026 Micron Technology, Inc. + * + * This file system, originally based on ramfs the dax support from xfs, + * is intended to allow multiple host systems to mount a common file system + * view of dax files that map to shared memory. + */ + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "fuse_i.h" + + +#define FMAP_BUFSIZE_INIT PAGE_SIZE +/* + * Largest GET_FMAP reply buffer we will kvmalloc. Any fmap whose whole message + * fits in this buffer is handled; there is no separate extent-count cap, so the + * effective extent limit is just this size / sizeof(simple_ext) (~699k extents + * => ~1.3 TiB per striped file at a 2 MiB chunk). kvmalloc-backed, so it may + * exceed the contiguous kmalloc limit. Matches the server's reply-buffer cap. + */ +#define FMAP_BUFSIZE_MAX (16 * 1024 * 1024) + +int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + u64 nodeid = get_node_id(inode); + size_t bufsize = FMAP_BUFSIZE_INIT; + void *fmap_buf = NULL; + ssize_t fmap_size; + int attempt; + int rc; + + /* Don't retrieve if we already have the famfs metadata */ + if (fi->famfs_meta) + return 0; + + /* + * The fmap size is not known in advance. Start with a modest buffer and, + * if the server reports (via the returned header's fmap_size) that the + * whole fmap did not fit, reallocate exactly that size and retry once. + * The server learns our buffer size from the request's + * fuse_getxattr_in.size (GETXATTR-style size probe). + */ + for (attempt = 0; ; attempt++) { + struct fuse_getxattr_in in = { .size = bufsize }; + struct fuse_famfs_fmap_header *fmh; + u32 required; + + FUSE_ARGS(args); + + fmap_buf = kvmalloc(bufsize, GFP_KERNEL); + if (!fmap_buf) + return -ENOMEM; + + args.opcode = FUSE_GET_FMAP; + args.nodeid = nodeid; + args.in_numargs = 1; + args.in_args[0].size = sizeof(in); + args.in_args[0].value = ∈ + /* + * Variable-sized output buffer; fuse_simple_request() returns + * the size of the output payload. + */ + args.out_argvar = true; + args.out_numargs = 1; + args.out_args[0].size = bufsize; + args.out_args[0].value = fmap_buf; + + rc = fuse_simple_request(fm, &args); + if (rc < 0) { + pr_err("%s: err=%d from fuse_simple_request()\n", + __func__, rc); + kvfree(fmap_buf); + return rc; + } + fmap_size = rc; + + /* Need at least a header to learn the required size */ + if (fmap_size < (ssize_t)sizeof(*fmh)) { + pr_err("%s: short fmap reply %zd\n", __func__, fmap_size); + kvfree(fmap_buf); + return -EIO; + } + + fmh = fmap_buf; + required = fmh->fmap_size; + + /* Whole fmap fit in the buffer -> parse it */ + if (required <= bufsize) + break; + + /* Too small: server sent only the header. Grow and retry once. */ + kvfree(fmap_buf); + fmap_buf = NULL; + + if (required > FMAP_BUFSIZE_MAX) { + pr_err("%s: fmap size %u exceeds max %zu\n", + __func__, required, (size_t)FMAP_BUFSIZE_MAX); + return -EFBIG; + } + if (attempt >= 1) { + /* + * A famfs file is fixed-size, so the server must report + * the same fmap_size on the retry as on the first + * request. A larger value means the file's size/fmap + * changed between the two GET_FMAPs -- a server bug. + */ + pr_err("%s: fmap grew %zu -> %u across GET_FMAP retries; famfs file size must not change (server bug)\n", + __func__, bufsize, required); + return -EINVAL; + } + bufsize = required; + } + + /* We retrieved the "fmap" (the file's map to memory), but + * we haven't used it yet. A call to famfs_file_init_dax() will be added + * here in a subsequent patch, when we add the ability to attach + * fmaps to files. + */ + + kvfree(fmap_buf); + return 0; +} diff --git a/fs/fuse/file.c b/fs/fuse/file.c index 995e37c935c6..b4e7b6a64587 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -282,6 +282,16 @@ static int fuse_open(struct inode *inode, struct file *file) err = fuse_do_open(fm, get_node_id(inode), file, false); if (!err) { ff = file->private_data; + + if ((fm->fc->famfs_iomap) && (S_ISREG(inode->i_mode))) { + /* Get the famfs fmap - failure is fatal */ + err = fuse_get_fmap(fm, inode); + if (err) { + fuse_sync_release(fi, ff, file->f_flags); + goto out_nowrite; + } + } + err = fuse_finish_open(inode, file); if (err) fuse_sync_release(fi, ff, file->f_flags); @@ -289,12 +299,14 @@ static int fuse_open(struct inode *inode, struct file *file) fuse_truncate_update_attr(inode, file); } +out_nowrite: if (is_wb_truncate || dax_truncate) fuse_release_nowrite(inode); if (!err) { if (is_truncate) truncate_pagecache(inode, 0); - else if (!(ff->open_flags & FOPEN_KEEP_CACHE)) + else if (!(ff->open_flags & FOPEN_KEEP_CACHE) && + !fuse_file_famfs(fi)) invalidate_inode_pages2(inode->i_mapping); } if (dax_truncate) diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 9c354118c931..5bacc5098620 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -236,6 +236,14 @@ struct fuse_inode { * be modified, so preserve the blocksize specified by the server. */ u8 cached_i_blkbits; + +#if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) + /* Pointer to the file's famfs metadata. Primary content is the + * in-memory version of the fmap - the map from file's offset range + * to DAX memory + */ + void *famfs_meta; +#endif }; /** FUSE inode state bits */ @@ -1218,11 +1226,8 @@ void fuse_free_conn(struct fuse_conn *fc); /* dax.c */ -static inline bool fuse_file_famfs(struct fuse_inode *fuse_inode) /* Will be superseded */ -{ - (void)fuse_inode; - return false; -} +static inline int fuse_file_famfs(struct fuse_inode *fi); /* forward */ + #define FUSE_IS_VIRTIO_DAX(fuse_inode) (IS_ENABLED(CONFIG_FUSE_DAX) \ && IS_DAX(&(fuse_inode)->inode) \ && !fuse_file_famfs(fuse_inode)) @@ -1339,4 +1344,59 @@ extern void fuse_sysctl_unregister(void); #define fuse_sysctl_unregister() do { } while (0) #endif /* CONFIG_SYSCTL */ +/* famfs.c */ + +#if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) +void __famfs_meta_free(void *map); + +/* Set fi->famfs_meta = NULL regardless of prior value */ +static inline void famfs_meta_init(struct fuse_inode *fi) +{ + fi->famfs_meta = NULL; +} + +/* Set fi->famfs_meta iff the current value is NULL */ +static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi, + void *meta) +{ + return cmpxchg(&fi->famfs_meta, NULL, meta); +} + +static inline void famfs_meta_free(struct fuse_inode *fi) +{ + famfs_meta_set(fi, NULL); +} + +static inline int fuse_file_famfs(struct fuse_inode *fi) +{ + return (READ_ONCE(fi->famfs_meta) != NULL); +} + +int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode); + +#else /* !CONFIG_FUSE_FAMFS_DAX */ + +static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi, + void *meta) +{ + return NULL; +} + +static inline void famfs_meta_free(struct fuse_inode *fi) +{ +} + +static inline int fuse_file_famfs(struct fuse_inode *fi) +{ + return 0; +} + +static inline int +fuse_get_fmap(struct fuse_mount *fm, struct inode *inode) +{ + return 0; +} + +#endif /* CONFIG_FUSE_FAMFS_DAX */ + #endif /* _FS_FUSE_I_H */ diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index c347471d04b6..e030a302120f 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -106,6 +106,9 @@ static struct inode *fuse_alloc_inode(struct super_block *sb) if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) fuse_inode_backing_set(fi, NULL); + if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)) + famfs_meta_set(fi, NULL); + return &fi->inode; out_free_forget: @@ -127,6 +130,9 @@ static void fuse_free_inode(struct inode *inode) if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) fuse_backing_put(fuse_inode_backing(fi)); + if (S_ISREG(inode->i_mode) && fuse_file_famfs(fi)) + famfs_meta_free(fi); + kmem_cache_free(fuse_inode_cachep, fi); } @@ -148,7 +154,7 @@ static void fuse_evict_inode(struct inode *inode) /* Will write inode on close/munmap and in all other dirtiers */ WARN_ON(inode_state_read_once(inode) & I_DIRTY_INODE); - if (FUSE_IS_VIRTIO_DAX(fi)) + if (FUSE_IS_VIRTIO_DAX(fi) || fuse_file_famfs(fi)) dax_break_layout_final(inode); truncate_inode_pages_final(&inode->i_data); diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c index 31ee7f3304c6..948148316ef0 100644 --- a/fs/fuse/iomode.c +++ b/fs/fuse/iomode.c @@ -203,7 +203,7 @@ int fuse_file_io_open(struct file *file, struct inode *inode) * io modes are not relevant with DAX and with server that does not * implement open. */ - if (FUSE_IS_VIRTIO_DAX(fi) || !ff->args) + if (FUSE_IS_VIRTIO_DAX(fi) || fuse_file_famfs(fi) || !ff->args) return 0; /* diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index 25686f088e6a..d323c20e79bd 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -669,6 +669,9 @@ enum fuse_opcode { FUSE_STATX = 52, FUSE_COPY_FILE_RANGE_64 = 53, + /* Famfs / devdax opcodes */ + FUSE_GET_FMAP = 54, + /* CUSE specific operations */ CUSE_INIT = 4096, -- 2.53.0 From: John Groves Interpret the fmap retrieved by GET_FMAP into the in-memory metadata that later patches use to resolve read/write/mmap directly to dax. - uapi: add the fmap wire format -- struct fuse_famfs_fmap_header and struct fuse_famfs_simple_ext, plus enum fuse_famfs_file_type and enum famfs_ext_type. An fmap is a simple (linear) list of extents, each mapping a file-offset range to a range on a dax device. A striped file is expressed by unrolling it into a longer simple-extent list in userspace, so the kernel format stays simple-only. - famfs_kfmap.h: add the in-memory structures (struct famfs_file_meta, struct famfs_meta_simple_ext) that hang from fuse_inode->famfs_meta. - famfs.c: add famfs_fuse_meta_alloc()/famfs_file_init_dax() to parse and validate the fmap and build the in-memory form, plus __famfs_meta_free(). While parsing, detect a uniform (power-of-two) extent size and record it as meta->ext_shift, so the resolver can index the extent list by a shift (O(1)) instead of walking it; a non-uniform list falls back to a walk (ext_shift == 0). - inode.c: only allow famfs mode if the fuse server holds CAP_SYS_RAWIO. - Update MAINTAINERS for the new file. The dev_dax_iomap plumbing that consumes this metadata is added in a later patch. Signed-off-by: John Groves --- MAINTAINERS | 1 + fs/fuse/famfs.c | 240 +++++++++++++++++++++++++++++++++++++- fs/fuse/famfs_kfmap.h | 64 ++++++++++ fs/fuse/fuse_i.h | 8 +- fs/fuse/inode.c | 20 +++- include/uapi/linux/fuse.h | 40 +++++++ 6 files changed, 363 insertions(+), 10 deletions(-) create mode 100644 fs/fuse/famfs_kfmap.h diff --git a/MAINTAINERS b/MAINTAINERS index 0d0fded4fddb..07db3a5a2af7 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -10719,6 +10719,7 @@ L: linux-cxl@vger.kernel.org L: linux-fsdevel@vger.kernel.org S: Supported F: fs/fuse/famfs.c +F: fs/fuse/famfs_kfmap.h FUTEX SUBSYSTEM M: Thomas Gleixner diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c index 80e6640ac970..8f7ee7d6151b 100644 --- a/fs/fuse/famfs.c +++ b/fs/fuse/famfs.c @@ -14,13 +14,244 @@ #include #include #include +#include #include #include #include +#include "famfs_kfmap.h" #include "fuse_i.h" +/***************************************************************************/ + +void __famfs_meta_free(void *famfs_meta) +{ + struct famfs_file_meta *fmap = famfs_meta; + + if (!fmap) + return; + + kfree(fmap->se); + kfree(fmap); +} +DEFINE_FREE(__famfs_meta_free, void *, if (_T) __famfs_meta_free(_T)) + +static int +famfs_check_ext_alignment(struct famfs_meta_simple_ext *se) +{ + int errs = 0; + + if (se->dev_index != 0) + errs++; + + /* TODO: pass in alignment so we can support the other page sizes */ + if (!IS_ALIGNED(se->ext_offset, PMD_SIZE)) + errs++; + + if (!IS_ALIGNED(se->ext_len, PMD_SIZE)) + errs++; + + return errs; +} + +/** + * famfs_fuse_meta_alloc() - Allocate famfs file metadata + * @fmap_buf: fmap buffer from fuse server + * @fmap_buf_size: size of fmap buffer + * @metap: pointer where 'struct famfs_file_meta' is returned + * + * Returns: 0=success + * -errno=failure + */ +static int +famfs_fuse_meta_alloc( + void *fmap_buf, + size_t fmap_buf_size, + struct famfs_file_meta **metap) +{ + struct fuse_famfs_fmap_header *fmh; + size_t extent_total = 0; + size_t next_offset = 0; + int errs = 0; + int i; + + fmh = fmap_buf; + + /* Move past fmh in fmap_buf */ + next_offset += sizeof(*fmh); + if (next_offset > fmap_buf_size) { + pr_err("%s:%d: fmap_buf underflow offset/size %ld/%ld\n", + __func__, __LINE__, next_offset, fmap_buf_size); + return -EINVAL; + } + + if (fmh->nextents < 1) { + pr_err("%s: nextents %d < 1\n", __func__, fmh->nextents); + return -ERANGE; + } + + /* + * No separate upper cap on nextents: the reply buffer bounds it. The + * extent list is rejected below if it does not fit in fmap_buf_size, and + * fuse_get_fmap() already refused to kvmalloc a buffer larger than + * FMAP_BUFSIZE_MAX -- so anything that fits is small enough to handle. + */ + + struct famfs_file_meta *meta __free(__famfs_meta_free) = kzalloc(sizeof(*meta), GFP_KERNEL); + + if (!meta) + return -ENOMEM; + + meta->error = false; + meta->file_type = fmh->file_type; + meta->file_size = fmh->file_size; + + switch (fmh->ext_type) { + case FUSE_FAMFS_EXT_SIMPLE: { + struct fuse_famfs_simple_ext *se_in; + + se_in = fmap_buf + next_offset; + + /* Move past simple extents */ + next_offset += fmh->nextents * sizeof(*se_in); + if (next_offset > fmap_buf_size) { + pr_err("%s:%d: fmap_buf underflow offset/size %ld/%ld\n", + __func__, __LINE__, next_offset, fmap_buf_size); + return -EINVAL; + } + + meta->fm_nextents = fmh->nextents; + + meta->se = kcalloc(meta->fm_nextents, sizeof(*(meta->se)), + GFP_KERNEL); + if (!meta->se) + return -ENOMEM; + + for (i = 0; i < fmh->nextents; i++) { + meta->se[i].dev_index = se_in[i].se_devindex; + meta->se[i].ext_offset = se_in[i].se_offset; + meta->se[i].ext_len = se_in[i].se_len; + + /* Record bitmap of referenced daxdev indices */ + meta->dev_bitmap |= BIT_ULL(meta->se[i].dev_index); + + errs += famfs_check_ext_alignment(&meta->se[i]); + + extent_total += meta->se[i].ext_len; + } + + /* + * Detect a uniform extent size so the resolver can index se[] + * by a shift rather than walking the list. This requires every + * extent but the last to be the same power-of-2 size, with the + * last no larger. ext_shift stays 0 (walk) for a single extent, + * or a non-uniform / non-power-of-2 list. + */ + meta->ext_shift = 0; + if (meta->fm_nextents > 1) { + u64 esz = meta->se[0].ext_len; + bool uniform = is_power_of_2(esz); + + for (i = 1; uniform && i < meta->fm_nextents - 1; i++) + if (meta->se[i].ext_len != esz) + uniform = false; + + if (uniform && meta->se[meta->fm_nextents - 1].ext_len > esz) + uniform = false; + + if (uniform) + meta->ext_shift = ilog2(esz); + } + break; + } + + default: + pr_err("%s: invalid ext_type %d\n", __func__, fmh->ext_type); + return -EINVAL; + } + + if (errs > 0) { + pr_err("%s: %d alignment errors found\n", __func__, errs); + return -EINVAL; + } + + /* More sanity checks */ + if (extent_total < meta->file_size) { + pr_err("%s: file size %ld larger than map size %ld\n", + __func__, meta->file_size, extent_total); + return -EINVAL; + } + + if (cmpxchg(metap, NULL, meta) != NULL) { + pr_debug("%s: fmap race detected\n", __func__); + return 0; /* fmap already installed */ + } + retain_and_null_ptr(meta); + + return 0; +} + +/** + * famfs_file_init_dax() - init famfs dax file metadata + * + * @fm: fuse_mount + * @inode: the inode + * @fmap_buf: fmap response message + * @fmap_size: Size of the fmap message + * + * Initialize famfs metadata for a file, based on the contents of the GET_FMAP + * response + * + * Return: 0=success + * -errno=failure + */ +int +famfs_file_init_dax( + struct fuse_mount *fm, + struct inode *inode, + void *fmap_buf, + size_t fmap_size) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct famfs_file_meta *meta = NULL; + int rc; + + if (fi->famfs_meta) { + pr_notice("%s: i_no=%llu fmap_size=%zu ALREADY INITIALIZED\n", + __func__, + inode->i_ino, fmap_size); + return 0; + } + + rc = famfs_fuse_meta_alloc(fmap_buf, fmap_size, &meta); + if (rc) + goto errout; + + /* Publish the famfs metadata on fi->famfs_meta */ + inode_lock(inode); + + if (famfs_meta_set(fi, meta) == NULL) { + i_size_write(inode, meta->file_size); + inode->i_flags |= S_DAX; + } else { + pr_debug("%s: file already had metadata\n", __func__); + __famfs_meta_free(meta); + /* rc is 0 - the file is valid */ + } + + inode_unlock(inode); + return 0; + +errout: + if (rc) + __famfs_meta_free(meta); + + return rc; +} + +#define FMAP_BUFSIZE PAGE_SIZE + #define FMAP_BUFSIZE_INIT PAGE_SIZE /* * Largest GET_FMAP reply buffer we will kvmalloc. Any fmap whose whole message @@ -123,12 +354,9 @@ int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode) bufsize = required; } - /* We retrieved the "fmap" (the file's map to memory), but - * we haven't used it yet. A call to famfs_file_init_dax() will be added - * here in a subsequent patch, when we add the ability to attach - * fmaps to files. - */ + /* Convert fmap into in-memory format and hang from inode */ + rc = famfs_file_init_dax(fm, inode, fmap_buf, fmap_size); kvfree(fmap_buf); - return 0; + return rc; } diff --git a/fs/fuse/famfs_kfmap.h b/fs/fuse/famfs_kfmap.h new file mode 100644 index 000000000000..d87b065e8ac8 --- /dev/null +++ b/fs/fuse/famfs_kfmap.h @@ -0,0 +1,64 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +/* + * famfs - dax file system for shared fabric-attached memory + * + * Copyright 2023-2026 Micron Technology, Inc. + */ +#ifndef FAMFS_KFMAP_H +#define FAMFS_KFMAP_H + +/* KABI version 43 (aka v2) fmap structures + * + * The location of the memory backing for a famfs file is described by + * the response to the GET_FMAP fuse message (defined in + * include/uapi/linux/fuse.h). + * + * A famfs file is described by a list of simple extents: (devindex, offset, + * length) tuples, where devindex references a devdax device that has been + * registered with the kernel. The extent list must cover at least file_size. + * Striping/interleaving is expressed by unrolling into a longer simple-extent + * list in userspace; the kernel handles only simple extents. + */ + +/* + * The structures below are the in-memory metadata format for famfs files. + * Metadata retrieved via the GET_FMAP response is converted to this format + * for use in resolving file mapping faults. + * + * The GET_FMAP response contains the same information, but in a more + * message-and-versioning-friendly format. Those structs can be found in the + * famfs section of include/uapi/linux/fuse.h (aka fuse_kernel.h in libfuse) + */ + +enum famfs_file_type { + FAMFS_REG, + FAMFS_SUPERBLOCK, + FAMFS_LOG, +}; + +struct famfs_meta_simple_ext { + u64 dev_index; + u64 ext_offset; + u64 ext_len; +}; + +/* + * Each famfs dax file has this hanging from its fuse_inode->famfs_meta + */ +struct famfs_file_meta { + bool error; + enum famfs_file_type file_type; + size_t file_size; + u64 dev_bitmap; /* bitmap of referenced daxdevs by index */ + size_t fm_nextents; + /* + * If every extent but the last is the same power-of-2 size, ext_shift + * is ilog2(that size) and the resolver indexes se[] by a shift of the + * file offset (O(1)). Otherwise ext_shift is 0 and the resolver walks + * the list (single-extent or non-uniform files). + */ + u32 ext_shift; + struct famfs_meta_simple_ext *se; +}; + +#endif /* FAMFS_KFMAP_H */ diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 5bacc5098620..46a7040b38dc 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1347,6 +1347,9 @@ extern void fuse_sysctl_unregister(void); /* famfs.c */ #if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) +int famfs_file_init_dax(struct fuse_mount *fm, + struct inode *inode, void *fmap_buf, + size_t fmap_size); void __famfs_meta_free(void *map); /* Set fi->famfs_meta = NULL regardless of prior value */ @@ -1364,7 +1367,10 @@ static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi, static inline void famfs_meta_free(struct fuse_inode *fi) { - famfs_meta_set(fi, NULL); + if (fi->famfs_meta != NULL) { + __famfs_meta_free(fi->famfs_meta); + WRITE_ONCE(fi->famfs_meta, NULL); + } } static inline int fuse_file_famfs(struct fuse_inode *fi) diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index e030a302120f..78ffc5fd50d0 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -7,6 +7,7 @@ #include "dev.h" #include "fuse_i.h" +#include #include #include #include @@ -1414,8 +1415,21 @@ static void process_init_reply(struct fuse_args *args, int error) timeout = arg->request_timeout; if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) && - flags & FUSE_DAX_FMAP) - fc->famfs_iomap = 1; + flags & FUSE_DAX_FMAP) { + /* famfs_iomap is only allowed if the fuse + * server has CAP_SYS_RAWIO. This was checked + * in fuse_send_init, and FUSE_DAX_IOMAP was + * set in in_flags if so. Only allow enablement + * if we find it there. This function is + * normally not running in fuse server context, + * so we can't do the capability check here... + */ + u64 in_flags = FIELD_PREP(GENMASK_ULL(63, 32), ia->in.flags2) + | ia->in.flags; + + if (in_flags & FUSE_DAX_FMAP) + fc->famfs_iomap = 1; + } } else { ra_pages = fc->max_read / PAGE_SIZE; fc->no_lock = 1; @@ -1483,7 +1497,7 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm) flags |= FUSE_SUBMOUNTS; if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) flags |= FUSE_PASSTHROUGH; - if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)) + if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) && capable(CAP_SYS_RAWIO)) flags |= FUSE_DAX_FMAP; /* diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index d323c20e79bd..4b84a58a8f1c 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -243,6 +243,12 @@ * * 7.46 * - Add FUSE_DAX_FMAP capability - ability to handle in-kernel fsdax maps + * - Add the following structures for the GET_FMAP message reply components: + * - struct fuse_famfs_simple_ext + * - struct fuse_famfs_fmap_header + * - Add the following enumerated types + * - enum fuse_famfs_file_type + * - enum famfs_ext_type */ #ifndef _LINUX_FUSE_H @@ -1316,4 +1322,38 @@ struct fuse_uring_cmd_req { uint8_t padding[6]; }; +/* Famfs fmap message components */ + +#define FAMFS_FMAP_VERSION 1 + +#define FAMFS_FMAP_MAX 32768 /* Largest supported fmap message */ + +enum fuse_famfs_file_type { + FUSE_FAMFS_FILE_REG, + FUSE_FAMFS_FILE_SUPERBLOCK, + FUSE_FAMFS_FILE_LOG, +}; + +enum famfs_ext_type { + FUSE_FAMFS_EXT_SIMPLE = 0, +}; + +struct fuse_famfs_simple_ext { + uint32_t se_devindex; + uint32_t reserved; + uint64_t se_offset; + uint64_t se_len; +}; + +struct fuse_famfs_fmap_header { + uint8_t file_type; /* enum fuse_famfs_file_type */ + uint8_t reserved; + uint16_t fmap_version; + uint32_t ext_type; /* enum famfs_ext_type */ + uint32_t nextents; + uint32_t fmap_size; /* Inclusive of this header */ + uint64_t file_size; + uint64_t reserved1; +}; + #endif /* _LINUX_FUSE_H */ -- 2.53.0 From: John Groves Add a dedicated ioctl, FUSE_DEV_IOC_DAXDEV_OPEN, by which a fuse server registers the devdax devices that back an fs-dax (famfs) filesystem: the server hands the kernel an fd to a /dev/daxN.N device plus its cluster-invariant famfs index. A dedicated ioctl, rather than overloading FUSE_DEV_IOC_BACKING_OPEN, avoids a dependency on CONFIG_FUSE_PASSTHROUGH and the backing-file machinery, which famfs does not use. - struct fuse_backing_map is reused as the argument; the index rides on the reserved 'padding' field (daxdev_index). - fuse_dev_ioctl_daxdev_open() copies the map and calls famfs_daxdev_open(), which resolves the fd to a dax device by its inode i_rdev. The daxdev table store is added in the following patch. Access control: - The ioctl is gated on famfs mode (enabled at FUSE_INIT only when the server held CAP_SYS_RAWIO), and additionally re-checks capable(CAP_SYS_RAWIO) on each call. The famfs-mode flag attests only to the session founder's privilege; the fuse device fd may be inherited across fork() or SCM_RIGHTS-passed to a less-privileged task, so the capability is re-checked against the task actually performing the registration. - famfs_daxdev_open() resolves the fd with fget(), not fget_raw(), so O_PATH fds are rejected. A successful lookup then proves the caller holds an fd from a real open() of the daxdev -- i.e. it already passed may_open_dev() and the device node's DAC checks -- rather than an O_PATH reference that bypasses them. Signed-off-by: John Groves --- fs/fuse/dev.c | 31 ++++ fs/fuse/dev.h | 1 + fs/fuse/famfs.c | 330 ++++++++++++++++++++++++++++++++++++++ fs/fuse/famfs_kfmap.h | 27 ++++ fs/fuse/fuse_i.h | 28 ++++ fs/fuse/inode.c | 7 +- include/uapi/linux/fuse.h | 7 +- 7 files changed, 429 insertions(+), 2 deletions(-) diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c index 5763a7cd3b37..3e6aa15e0346 100644 --- a/fs/fuse/dev.c +++ b/fs/fuse/dev.c @@ -2330,6 +2330,34 @@ static long fuse_dev_ioctl_backing_close(struct file *file, __u32 __user *argp) return fuse_backing_close(fud->chan->conn, backing_id); } +static long fuse_dev_ioctl_daxdev_open(struct file *file, + struct fuse_backing_map __user *argp) +{ + struct fuse_dev *fud = fuse_get_dev(file); + struct fuse_backing_map map; + + if (IS_ERR(fud)) + return PTR_ERR(fud); + + if (!IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)) + return -EOPNOTSUPP; + + /* + * The famfs-mode gate (fc->famfs_iomap) lives in famfs_daxdev_open(), + * which has the full fuse_conn definition. Here, re-check CAP_SYS_RAWIO + * against the task performing the registration: famfs mode being enabled + * only attests that the session founder held it at FUSE_INIT, and the + * fuse device fd may have been passed to a less-privileged process. + */ + if (!capable(CAP_SYS_RAWIO)) + return -EPERM; + + if (copy_from_user(&map, argp, sizeof(map))) + return -EFAULT; + + return famfs_daxdev_open(fud->chan->conn, &map); +} + static long fuse_dev_ioctl_sync_init(struct file *file) { struct fuse_dev *fud = fuse_file_to_fud(file); @@ -2359,6 +2387,9 @@ static long fuse_dev_ioctl(struct file *file, unsigned int cmd, case FUSE_DEV_IOC_SYNC_INIT: return fuse_dev_ioctl_sync_init(file); + case FUSE_DEV_IOC_DAXDEV_OPEN: + return fuse_dev_ioctl_daxdev_open(file, argp); + default: return -ENOTTY; } diff --git a/fs/fuse/dev.h b/fs/fuse/dev.h index aed69fd14c41..545940b635cc 100644 --- a/fs/fuse/dev.h +++ b/fs/fuse/dev.h @@ -87,6 +87,7 @@ int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code, int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map); int fuse_backing_close(struct fuse_conn *fc, int backing_id); +int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map); int fuse_copy_one(struct fuse_copy_state *cs, void *val, unsigned size); int fuse_copy_folio(struct fuse_copy_state *cs, struct folio **foliop, diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c index 8f7ee7d6151b..a2a7dd631dc0 100644 --- a/fs/fuse/famfs.c +++ b/fs/fuse/famfs.c @@ -11,6 +11,7 @@ #include #include +#include #include #include #include @@ -22,6 +23,331 @@ #include "famfs_kfmap.h" #include "fuse_i.h" +static void famfs_set_daxdev_err( + struct fuse_conn *fc, struct dax_device *dax_devp); + +static int +famfs_dax_notify_failure(struct dax_device *dax_devp, u64 offset, + u64 len, int mf_flags) +{ + struct fuse_conn *fc = dax_holder(dax_devp); + + famfs_set_daxdev_err(fc, dax_devp); + + return 0; +} + +static const struct dax_holder_operations famfs_fuse_dax_holder_ops = { + .notify_failure = famfs_dax_notify_failure, +}; + +/*****************************************************************************/ + +/* + * famfs_teardown() + * + * Deallocate famfs metadata for a fuse_conn + */ +void +famfs_teardown(struct fuse_conn *fc) +{ + struct famfs_dax_devlist *devlist __free(kfree) = NULL; + int i; + + /* + * Detach the table under the same lock famfs_set_daxdev_err() takes, so + * a notify_failure racing teardown either runs first against the live + * table or observes dax_devlist == NULL and bails, rather than + * dereferencing it after we clear it. The daxdev holders are dropped + * below, after which no further notify_failure can arrive. + */ + scoped_guard(rwsem_write, &fc->famfs_devlist_sem) { + devlist = fc->dax_devlist; + fc->dax_devlist = NULL; + } + + if (!devlist) + return; + + if (!devlist->devlist) + return; + + /* Close & release all the daxdevs in our table */ + for (i = 0; i < devlist->nslots; i++) { + struct famfs_daxdev *dd = &devlist->devlist[i]; + + if (!dd->valid) + continue; + + /* Only call fs_put_dax if fs_dax_get succeeded */ + if (dd->devp) { + if (!dd->dax_err) + fs_put_dax(dd->devp, fc); + put_dax(dd->devp); + } + + kfree(dd->name); + } + kfree(devlist->devlist); +} + +/* Allocate the daxdev table on first use (idempotent via cmpxchg) */ +static int famfs_devlist_alloc(struct fuse_conn *fc) +{ + struct famfs_dax_devlist *devlist; + + if (fc->dax_devlist) + return 0; + + devlist = kcalloc(1, sizeof(*devlist), GFP_KERNEL); + if (!devlist) + return -ENOMEM; + + devlist->nslots = MAX_DAXDEVS; + devlist->devlist = kcalloc(MAX_DAXDEVS, sizeof(struct famfs_daxdev), + GFP_KERNEL); + if (!devlist->devlist) { + kfree(devlist); + return -ENOMEM; + } + + /* If another thread allocated it first, drop ours */ + if (cmpxchg(&fc->dax_devlist, NULL, devlist) != NULL) { + kfree(devlist->devlist); + kfree(devlist); + } + + return 0; +} + +/* + * famfs_install_daxdev() - exclusively acquire a resolved daxdev and publish + * it in the table at @index. Shared by the GET_DAXDEV (pull) and + * DAXDEV_OPEN (push) registration paths. + * + * Serializes with concurrent installers under famfs_devlist_sem and rechecks + * ->valid. A daxdev is entered in the table only once it has been exclusively + * acquired via fs_dax_get(); on failure the dax_dev_get() reference is + * released and the slot is left invalid, so the referencing fmap is rejected + * rather than mapped without an exclusive holder. @name may be NULL (the push + * path passes no pathname). + */ +static int famfs_install_daxdev(struct fuse_conn *fc, u64 index, dev_t devno, + const char *name) +{ + struct famfs_daxdev *daxdev; + int rc = 0; + + if (index >= fc->dax_devlist->nslots) { + pr_err("%s: index(%llu) >= nslots(%d)\n", + __func__, index, fc->dax_devlist->nslots); + return -EINVAL; + } + + scoped_guard(rwsem_write, &fc->famfs_devlist_sem) { + daxdev = &fc->dax_devlist->devlist[index]; + + /* Installed already by a concurrent push/pull */ + if (daxdev->valid) + return 0; + + /* + * A prior attempt already determined this daxdev cannot be + * exclusively acquired (see the fs_dax_get() failure handling + * below). Don't thrash on GET_DAXDEV/fs_dax_get(); fail fast. + */ + if (daxdev->dax_err) + return -EIO; + + /* + * Temporary: dax_dev_get() is the exported upstream lookup, but + * unlike dax_dev_find() it allocates for any dev_t and does not + * reject non-dax devices. Restore dax_dev_find() (and that + * rejection) once it is upstream. + */ + daxdev->devp = dax_dev_get(devno); + if (!daxdev->devp) { + pr_warn("%s: device %u:%u not found or not dax\n", + __func__, MAJOR(devno), MINOR(devno)); + return -ENODEV; + } + + rc = fs_dax_get(daxdev->devp, fc, &famfs_fuse_dax_holder_ops); + if (rc) { + /* + * Distinguish a lost race from a real failure. -EBUSY + * with the daxdev already held by *this* fuse_conn + * means a concurrent acquire won and will publish the + * slot valid: not an error, and must not be cached as + * dax_err. Any other failure (foreign holder, not a dax + * device, wrong driver type) is permanent for this + * connection, so record dax_err to stop re-fetching and + * re-acquiring it. + */ + if (!(rc == -EBUSY && dax_holder(daxdev->devp) == fc)) { + pr_err("%s: fs_dax_get(%u:%u) failed rc=%d\n", + __func__, MAJOR(devno), MINOR(devno), rc); + daxdev->dax_err = true; + } + put_dax(daxdev->devp); + daxdev->devp = NULL; + return rc; + } + + daxdev->devno = devno; + if (name) { + daxdev->name = kstrdup(name, GFP_KERNEL); + if (!daxdev->name) { + fs_put_dax(daxdev->devp, fc); + put_dax(daxdev->devp); + daxdev->devp = NULL; + return -ENOMEM; + } + } + + wmb(); /* All other fields must be visible before valid */ + daxdev->valid = 1; + } + + return 0; +} + +/** + * famfs_daxdev_open() - Register a daxdev via FUSE_DEV_IOC_DAXDEV_OPEN + * @fc: fuse_conn + * @map: fuse_backing_map; @map->fd is an fd to the devdax device and + * @map->daxdev_index is the (cluster-invariant) famfs index. + * + * The server pushes a daxdev to the kernel by reference (an fd), rather than + * the kernel pulling it by name via GET_DAXDEV. The resolved daxdev is + * exclusively acquired and entered in the table at @map->daxdev_index. + * + * Return: 0=success + * -errno=failure + */ +int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map) +{ + struct inode *inode; + struct file *file; + dev_t devno; + int rc; + + /* Only fs-dax (famfs) mode accepts daxdev registration */ + if (!fc->famfs_iomap) + return -EOPNOTSUPP; + + file = fget(map->fd); + if (!file) + return -EBADF; + + inode = file_inode(file); + if (!S_ISCHR(inode->i_mode)) { + fput(file); + return -EINVAL; + } + devno = inode->i_rdev; + fput(file); + + rc = famfs_devlist_alloc(fc); + if (rc) + return rc; + + rc = famfs_install_daxdev(fc, map->daxdev_index, devno, NULL); + if (rc) + pr_err("%s: failed to install daxdev\n", __func__); + + return rc; +} + +/** + * famfs_check_daxdev_table() - Verify an fmap's referenced daxdevs are installed + * @fm: fuse_mount + * @meta: famfs_file_meta, in-memory format, built from a GET_FMAP response + * + * Called for each new file fmap. Every daxdev the fmap references must already + * be installed in the table, having been pushed in via FUSE_DEV_IOC_DAXDEV_OPEN + * before any file that uses it is accessed. If any referenced daxdev is not + * present, the fmap is rejected so the file is never mapped against a daxdev + * that has no exclusive holder. + * + * Return: 0=success (all referenced daxdevs present) + * <0=a referenced daxdev is missing from the table + */ +static int +famfs_check_daxdev_table( + struct fuse_mount *fm, + const struct famfs_file_meta *meta) +{ + struct fuse_conn *fc = fm->fc; + int nmissing = 0; + int err; + + err = famfs_devlist_alloc(fc); + if (err) + return err; + + /* Count missing daxdevs while holding the reader lock */ + scoped_guard(rwsem_read, &fc->famfs_devlist_sem) { + unsigned long i; + + for_each_set_bit(i, (unsigned long *)&meta->dev_bitmap, + MAX_DAXDEVS) { + struct famfs_daxdev *dd = &fc->dax_devlist->devlist[i]; + + /* + * Skip daxdevs already installed (valid) or already + * known to be unusable (dax_err). Re-fetching either + * just thrashes on GET_DAXDEV and fs_dax_get(). + */ + if (!dd->valid && !dd->dax_err) + nmissing++; + } + } + + if (nmissing > 0) { + /* this file referenced at least one daxdev that is not in + * the table. Daxdevs must be known before any file that + * uses them is accessed + */ + pr_err("%s: %d missing daxdev(s)\n", __func__, nmissing); + return -ENODEV; + } + + return 0; +} + +static void +famfs_set_daxdev_err( + struct fuse_conn *fc, + struct dax_device *dax_devp) +{ + int i; + + /* + * Search the list by dax_devp under the write lock: we set dd->error, + * and it serializes against famfs_teardown() clearing the table. + */ + scoped_guard(rwsem_write, &fc->famfs_devlist_sem) { + if (!fc->dax_devlist) + return; + for (i = 0; i < fc->dax_devlist->nslots; i++) { + if (fc->dax_devlist->devlist[i].valid) { + struct famfs_daxdev *dd; + + dd = &fc->dax_devlist->devlist[i]; + if (dd->devp != dax_devp) + continue; + + dd->error = true; + + pr_err("%s: memory error on daxdev %s (%d)\n", + __func__, dd->name, i); + return; + } + } + } + pr_err("%s: memory err on unrecognized daxdev\n", __func__); +} /***************************************************************************/ @@ -228,6 +554,10 @@ famfs_file_init_dax( if (rc) goto errout; + /* Make sure this fmap doesn't reference any unknown daxdevs */ + if (famfs_check_daxdev_table(fm, meta)) + meta->error = true; + /* Publish the famfs metadata on fi->famfs_meta */ inode_lock(inode); diff --git a/fs/fuse/famfs_kfmap.h b/fs/fuse/famfs_kfmap.h index d87b065e8ac8..6b78e78325c8 100644 --- a/fs/fuse/famfs_kfmap.h +++ b/fs/fuse/famfs_kfmap.h @@ -61,4 +61,31 @@ struct famfs_file_meta { struct famfs_meta_simple_ext *se; }; +/* + * famfs_daxdev - tracking struct for a daxdev within a famfs file system + * + * This is the in-memory daxdev metadata that is populated by parsing + * the responses to GET_FMAP messages + */ +struct famfs_daxdev { + /* Include dev uuid? */ + bool valid; + bool error; /* Dax has reported a memory error (probably poison) */ + bool dax_err; /* fs_dax_get() failed */ + dev_t devno; + struct dax_device *devp; + char *name; +}; + +#define MAX_DAXDEVS 24 + +/* + * famfs_dax_devlist - list of famfs_daxdev's + */ +struct famfs_dax_devlist { + int nslots; + int ndevs; + struct famfs_daxdev *devlist; +}; + #endif /* FAMFS_KFMAP_H */ diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 46a7040b38dc..5394aae9dbac 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -773,6 +773,11 @@ struct fuse_conn { /** @backing_files_map: IDR for backing files ids */ struct idr backing_files_map; #endif + +#if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) + struct rw_semaphore famfs_devlist_sem; + struct famfs_dax_devlist *dax_devlist; +#endif }; /* @@ -1352,6 +1357,8 @@ int famfs_file_init_dax(struct fuse_mount *fm, size_t fmap_size); void __famfs_meta_free(void *map); +void famfs_teardown(struct fuse_conn *fc); + /* Set fi->famfs_meta = NULL regardless of prior value */ static inline void famfs_meta_init(struct fuse_inode *fi) { @@ -1373,6 +1380,11 @@ static inline void famfs_meta_free(struct fuse_inode *fi) } } +static inline void famfs_init_devlist_sem(struct fuse_conn *fc) +{ + init_rwsem(&fc->famfs_devlist_sem); +} + static inline int fuse_file_famfs(struct fuse_inode *fi) { return (READ_ONCE(fi->famfs_meta) != NULL); @@ -1380,8 +1392,14 @@ static inline int fuse_file_famfs(struct fuse_inode *fi) int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode); +int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map); + #else /* !CONFIG_FUSE_FAMFS_DAX */ +static inline void famfs_teardown(struct fuse_conn *fc) +{ +} + static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi, void *meta) { @@ -1392,6 +1410,10 @@ static inline void famfs_meta_free(struct fuse_inode *fi) { } +static inline void famfs_init_devlist_sem(struct fuse_conn *fc) +{ +} + static inline int fuse_file_famfs(struct fuse_inode *fi) { return 0; @@ -1403,6 +1425,12 @@ fuse_get_fmap(struct fuse_mount *fm, struct inode *inode) return 0; } +static inline int +famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map) +{ + return -EOPNOTSUPP; +} + #endif /* CONFIG_FUSE_FAMFS_DAX */ #endif /* _FS_FUSE_I_H */ diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index 78ffc5fd50d0..9fc37015fb11 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -1021,6 +1021,9 @@ void fuse_conn_put(struct fuse_conn *fc) WARN_ON(atomic_read(&bucket->count) != 1); kfree(bucket); } + if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)) + famfs_teardown(fc); + if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) fuse_backing_files_free(fc); call_rcu(&fc->rcu, delayed_release); @@ -1427,8 +1430,10 @@ static void process_init_reply(struct fuse_args *args, int error) u64 in_flags = FIELD_PREP(GENMASK_ULL(63, 32), ia->in.flags2) | ia->in.flags; - if (in_flags & FUSE_DAX_FMAP) + if (in_flags & FUSE_DAX_FMAP) { + famfs_init_devlist_sem(fc); fc->famfs_iomap = 1; + } } } else { ra_pages = fc->max_read / PAGE_SIZE; diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h index 4b84a58a8f1c..a143e6818416 100644 --- a/include/uapi/linux/fuse.h +++ b/include/uapi/linux/fuse.h @@ -1143,7 +1143,10 @@ struct fuse_notify_prune_out { struct fuse_backing_map { int32_t fd; uint32_t flags; - uint64_t padding; + union { + uint64_t padding; + uint64_t daxdev_index; /* FUSE_DEV_IOC_DAXDEV_OPEN */ + }; }; /* Device ioctls: */ @@ -1153,6 +1156,8 @@ struct fuse_backing_map { struct fuse_backing_map) #define FUSE_DEV_IOC_BACKING_CLOSE _IOW(FUSE_DEV_IOC_MAGIC, 2, uint32_t) #define FUSE_DEV_IOC_SYNC_INIT _IO(FUSE_DEV_IOC_MAGIC, 3) +#define FUSE_DEV_IOC_DAXDEV_OPEN _IOW(FUSE_DEV_IOC_MAGIC, 4, \ + struct fuse_backing_map) struct fuse_lseek_in { uint64_t fh; -- 2.53.0 From: John Groves Fill in read/write/mmap handling for famfs files using the dev_dax_iomap interface, the same path xfs uses in fs-dax mode. - Read/write go through famfs_fuse_[read|write]_iter() via dax_iomap_rw() to fsdev_dax. - Mmap is handled by famfs_fuse_mmap(). - Faults are handled by famfs_filemap_fault() via dax_iomap_fault() to fsdev_dax. - File-offset-to-dax-offset resolution is handled by famfs_fuse_iomap_begin(), which uses the file's fmap to resolve a (file, offset) to an offset on a dax device via famfs_fileofs_to_daxofs(). Signed-off-by: John Groves --- fs/fuse/famfs.c | 337 ++++++++++++++++++++++++++++++++++++++++++++++- fs/fuse/file.c | 18 ++- fs/fuse/fuse_i.h | 19 +++ 3 files changed, 371 insertions(+), 3 deletions(-) diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c index a2a7dd631dc0..ac56317944d9 100644 --- a/fs/fuse/famfs.c +++ b/fs/fuse/famfs.c @@ -580,7 +580,342 @@ famfs_file_init_dax( return rc; } -#define FMAP_BUFSIZE PAGE_SIZE +/********************************************************************* + * iomap_operations + * + * This stuff uses the iomap (dax-related) helpers to resolve file offsets to + * offsets within a dax device. + */ + +static int famfs_file_bad(struct inode *inode); + +/** + * famfs_fileofs_to_daxofs() - Resolve (file, offset, len) to (daxdev, offset, len) + * + * This function is called by famfs_fuse_iomap_begin() to resolve an offset in a + * file to an offset in a dax device. This is upcalled from dax from calls to + * both * dax_iomap_fault() and dax_iomap_rw(). Dax finishes the job resolving + * a fault to a specific physical page (the fault case) or doing a memcpy + * variant (the rw case) + * + * Pages can be PTE (4k), PMD (2MiB) or (theoretically) PuD (1GiB) + * (these sizes are for X86; may vary on other cpu architectures + * + * @inode: The file where the fault occurred + * @iomap: To be filled in to indicate where to find the right memory, + * relative to a dax device. + * @file_offset: Within the file where the fault occurred (will be page boundary) + * @len: The length of the faulted mapping (will be a page multiple) + * (will be trimmed in *iomap if it's disjoint in the extent list) + * @flags: flags passed to famfs_fuse_iomap_begin(), and sent back via + * struct iomap + * + * Return values: 0 on success, with the result in the modified @iomap struct. + * -EIO if (file_offset, len) cannot be resolved to a dax extent, + * which includes access past EOF (the caller turns this into a + * short read/write or a SIGBUS). + */ +static int +famfs_fileofs_to_daxofs(struct inode *inode, struct iomap *iomap, + loff_t file_offset, off_t len, unsigned int flags) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct famfs_file_meta *meta = fi->famfs_meta; + struct fuse_conn *fc = get_fuse_conn(inode); + loff_t local_offset = file_offset; + u64 i; + + if (!fc->dax_devlist) { + pr_err("%s: null dax_devlist\n", __func__); + goto err_out; + } + + if (famfs_file_bad(inode)) + goto err_out; + + iomap->offset = file_offset; + + if (meta->ext_shift) { + /* + * Uniform extent size (detected at parse time): the containing + * extent index is a shift of the file offset, so resolve in + * O(1) instead of walking the extent list. + */ + u64 ext_size = 1ULL << meta->ext_shift; + + i = file_offset >> meta->ext_shift; + local_offset = file_offset & (ext_size - 1); + } else { + /* Single extent or non-uniform list: walk it */ + for (i = 0; i < meta->fm_nextents; i++) { + if (local_offset < meta->se[i].ext_len) + break; + local_offset -= meta->se[i].ext_len; + } + } + + /* local_offset is now the offset within extent i (if i is in range) */ + if (i < meta->fm_nextents && local_offset < meta->se[i].ext_len) { + loff_t dax_ext_offset = meta->se[i].ext_offset; + loff_t dax_ext_len = meta->se[i].ext_len; + u64 daxdev_idx = meta->se[i].dev_index; + loff_t ext_len_remainder = dax_ext_len - local_offset; + struct famfs_daxdev *dd; + + if (daxdev_idx >= fc->dax_devlist->nslots) { + pr_err("%s: daxdev_idx %llu >= nslots %d\n", + __func__, daxdev_idx, fc->dax_devlist->nslots); + goto err_out; + } + + dd = &fc->dax_devlist->devlist[daxdev_idx]; + + iomap->addr = dax_ext_offset + local_offset; + iomap->offset = file_offset; + iomap->length = min_t(loff_t, len, ext_len_remainder); + iomap->dax_dev = dd->devp; + iomap->type = IOMAP_MAPPED; + iomap->flags = flags; + return 0; + } + + err_out: + /* + * We fell out the end of the extent list, i.e. the access is past EOF. + * This is a normal, unprivileged-reachable condition (e.g. a fault in + * the tail of a mapping that extends past EOF), so log at debug level. + * The specific error cases above (null dax_devlist, bad daxdev index) + * have already logged at error level before jumping here. + */ + pr_debug("%s: could not resolve file_offset %lld (past EOF?)\n", + __func__, (long long)file_offset); + + /* + * Return a zero-length mapping and -EIO. dax turns this into a short + * read/write or a SIGBUS rather than touching dax memory. + */ + iomap->addr = 0; /* there is no valid dax device offset */ + iomap->offset = file_offset; /* file offset */ + iomap->length = 0; /* this had better result in no access to dax mem */ + iomap->dax_dev = NULL; + iomap->type = IOMAP_MAPPED; + iomap->flags = flags; + + return -EIO; +} + +/** + * famfs_fuse_iomap_begin() - Handler for iomap_begin upcall from dax + * + * This function is pretty simple because files are + * * never partially allocated + * * never have holes (never sparse) + * * never "allocate on write" + * + * @inode: inode for the file being accessed + * @offset: offset within the file + * @length: Length being accessed at offset + * @flags: flags to be retured via struct iomap + * @iomap: iomap struct to be filled in, resolving (offset, length) to + * (daxdev, offset, len) + * @srcmap: source mapping if it is a COW operation (which it is not here) + */ +static int +famfs_fuse_iomap_begin(struct inode *inode, loff_t offset, loff_t length, + unsigned int flags, struct iomap *iomap, struct iomap *srcmap) +{ + return famfs_fileofs_to_daxofs(inode, iomap, offset, length, flags); +} + +/* Note: We never need a special set of write_iomap_ops because famfs never + * performs allocation on write. + */ +const struct iomap_ops famfs_iomap_ops = { + .iomap_begin = famfs_fuse_iomap_begin, +}; + +/********************************************************************* + * vm_operations + */ +static vm_fault_t +__famfs_fuse_filemap_fault(struct vm_fault *vmf, unsigned int order, + bool write_fault) +{ + struct inode *inode = file_inode(vmf->vma->vm_file); + vm_fault_t ret; + unsigned long pfn; + + if (!IS_DAX(file_inode(vmf->vma->vm_file))) { + pr_err("%s: file not marked IS_DAX!!\n", __func__); + return VM_FAULT_SIGBUS; + } + + if (write_fault) { + sb_start_pagefault(inode->i_sb); + file_update_time(vmf->vma->vm_file); + } + + ret = dax_iomap_fault(vmf, order, &pfn, NULL, &famfs_iomap_ops); + if (ret & VM_FAULT_NEEDDSYNC) + ret = dax_finish_sync_fault(vmf, order, pfn); + + if (write_fault) + sb_end_pagefault(inode->i_sb); + + return ret; +} + +static inline bool +famfs_is_write_fault(struct vm_fault *vmf) +{ + return (vmf->flags & FAULT_FLAG_WRITE) && + (vmf->vma->vm_flags & VM_SHARED); +} + +static vm_fault_t +famfs_filemap_fault(struct vm_fault *vmf) +{ + return __famfs_fuse_filemap_fault(vmf, 0, famfs_is_write_fault(vmf)); +} + +static vm_fault_t +famfs_filemap_huge_fault(struct vm_fault *vmf, unsigned int order) +{ + return __famfs_fuse_filemap_fault(vmf, order, + famfs_is_write_fault(vmf)); +} + +static vm_fault_t +famfs_filemap_mkwrite(struct vm_fault *vmf) +{ + return __famfs_fuse_filemap_fault(vmf, 0, true); +} + +const struct vm_operations_struct famfs_file_vm_ops = { + .fault = famfs_filemap_fault, + .huge_fault = famfs_filemap_huge_fault, + .map_pages = filemap_map_pages, + .page_mkwrite = famfs_filemap_mkwrite, + .pfn_mkwrite = famfs_filemap_mkwrite, +}; + +/********************************************************************* + * file_operations + */ + +/** + * famfs_file_bad() - Check for files that aren't in a valid state + * + * @inode: inode + * + * Returns: 0=success + * -errno=failure + */ +static int +famfs_file_bad(struct inode *inode) +{ + struct fuse_inode *fi = get_fuse_inode(inode); + struct famfs_file_meta *meta = fi->famfs_meta; + size_t i_size = i_size_read(inode); + + if (!meta) { + pr_err("%s: un-initialized famfs file\n", __func__); + return -EIO; + } + if (meta->error) { + pr_debug("%s: previously detected metadata errors\n", __func__); + return -EIO; + } + if (i_size != meta->file_size) { + pr_warn("%s: i_size overwritten from %ld to %ld\n", + __func__, meta->file_size, i_size); + meta->error = true; + return -ENXIO; + } + if (!IS_DAX(inode)) { + pr_debug("%s: inode %llx IS_DAX is false\n", + __func__, (u64)inode); + return -ENXIO; + } + return 0; +} + +static ssize_t +famfs_fuse_rw_prep(struct kiocb *iocb, struct iov_iter *ubuf) +{ + struct inode *inode = iocb->ki_filp->f_mapping->host; + size_t i_size = i_size_read(inode); + size_t count = iov_iter_count(ubuf); + size_t max_count; + ssize_t rc; + + rc = famfs_file_bad(inode); + if (rc) + return (ssize_t)rc; + + /* Avoid unsigned underflow if position is past EOF */ + if (iocb->ki_pos >= i_size) + max_count = 0; + else + max_count = i_size - iocb->ki_pos; + + if (count > max_count) + iov_iter_truncate(ubuf, max_count); + + if (!iov_iter_count(ubuf)) + return 0; + + return rc; +} + +ssize_t +famfs_fuse_read_iter(struct kiocb *iocb, struct iov_iter *to) +{ + ssize_t rc; + + rc = famfs_fuse_rw_prep(iocb, to); + if (rc) + return rc; + + if (!iov_iter_count(to)) + return 0; + + rc = dax_iomap_rw(iocb, to, &famfs_iomap_ops); + + file_accessed(iocb->ki_filp); + return rc; +} + +ssize_t +famfs_fuse_write_iter(struct kiocb *iocb, struct iov_iter *from) +{ + ssize_t rc; + + rc = famfs_fuse_rw_prep(iocb, from); + if (rc) + return rc; + + if (!iov_iter_count(from)) + return 0; + + return dax_iomap_rw(iocb, from, &famfs_iomap_ops); +} + +int +famfs_fuse_mmap(struct file *file, struct vm_area_struct *vma) +{ + struct inode *inode = file_inode(file); + ssize_t rc; + + rc = famfs_file_bad(inode); + if (rc) + return rc; + + file_accessed(file); + vma->vm_ops = &famfs_file_vm_ops; + vm_flags_set(vma, VM_HUGEPAGE); + return 0; +} #define FMAP_BUFSIZE_INIT PAGE_SIZE /* diff --git a/fs/fuse/file.c b/fs/fuse/file.c index b4e7b6a64587..2435a79cbb4a 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1859,6 +1859,8 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) if (FUSE_IS_VIRTIO_DAX(fi)) return fuse_dax_read_iter(iocb, to); + if (fuse_file_famfs(fi)) + return famfs_fuse_read_iter(iocb, to); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ if (ff->open_flags & FOPEN_DIRECT_IO) @@ -1881,6 +1883,8 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) if (FUSE_IS_VIRTIO_DAX(fi)) return fuse_dax_write_iter(iocb, from); + if (fuse_file_famfs(fi)) + return famfs_fuse_write_iter(iocb, from); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ if (ff->open_flags & FOPEN_DIRECT_IO) @@ -1896,9 +1900,13 @@ static ssize_t fuse_splice_read(struct file *in, loff_t *ppos, unsigned int flags) { struct fuse_file *ff = in->private_data; + struct inode *inode = file_inode(in); + struct fuse_inode *fi = get_fuse_inode(inode); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ - if (fuse_file_passthrough(ff) && !(ff->open_flags & FOPEN_DIRECT_IO)) + if (fuse_file_famfs(fi)) + return -EIO; /* famfs does not use the page cache... */ + else if (fuse_file_passthrough(ff) && !(ff->open_flags & FOPEN_DIRECT_IO)) return fuse_passthrough_splice_read(in, ppos, pipe, len, flags); else return filemap_splice_read(in, ppos, pipe, len, flags); @@ -1908,9 +1916,13 @@ static ssize_t fuse_splice_write(struct pipe_inode_info *pipe, struct file *out, loff_t *ppos, size_t len, unsigned int flags) { struct fuse_file *ff = out->private_data; + struct inode *inode = file_inode(out); + struct fuse_inode *fi = get_fuse_inode(inode); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ - if (fuse_file_passthrough(ff) && !(ff->open_flags & FOPEN_DIRECT_IO)) + if (fuse_file_famfs(fi)) + return -EIO; /* famfs does not use the page cache... */ + else if (fuse_file_passthrough(ff) && !(ff->open_flags & FOPEN_DIRECT_IO)) return fuse_passthrough_splice_write(pipe, out, ppos, len, flags); else return iter_file_splice_write(pipe, out, ppos, len, flags); @@ -2420,6 +2432,8 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) /* DAX mmap is superior to direct_io mmap */ if (FUSE_IS_VIRTIO_DAX(fi)) return fuse_dax_mmap(file, vma); + if (fuse_file_famfs(fi)) + return famfs_fuse_mmap(file, vma); /* * If inode is in passthrough io mode, because it has some file open diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 5394aae9dbac..7281fb8b6402 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1355,6 +1355,9 @@ extern void fuse_sysctl_unregister(void); int famfs_file_init_dax(struct fuse_mount *fm, struct inode *inode, void *fmap_buf, size_t fmap_size); +ssize_t famfs_fuse_write_iter(struct kiocb *iocb, struct iov_iter *from); +ssize_t famfs_fuse_read_iter(struct kiocb *iocb, struct iov_iter *to); +int famfs_fuse_mmap(struct file *file, struct vm_area_struct *vma); void __famfs_meta_free(void *map); void famfs_teardown(struct fuse_conn *fc); @@ -1400,6 +1403,22 @@ static inline void famfs_teardown(struct fuse_conn *fc) { } +static inline ssize_t famfs_fuse_write_iter(struct kiocb *iocb, + struct iov_iter *to) +{ + return -ENODEV; +} +static inline ssize_t famfs_fuse_read_iter(struct kiocb *iocb, + struct iov_iter *to) +{ + return -ENODEV; +} +static inline int famfs_fuse_mmap(struct file *file, + struct vm_area_struct *vma) +{ + return -ENODEV; +} + static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi, void *meta) { -- 2.53.0 From: John Groves Gate the iomap resolution path on the state of the daxdev backing each referenced extent. famfs_dax_err() returns an error if the daxdev slot is invalid (-EIO), was flagged dax_err (-EIO), or has reported a memory error via notify_failure (-EHWPOISON). famfs_fileofs_to_daxofs() calls it and, on error, marks the file (meta->error) and stops allowing access. Memory errors are at least somewhat more likely on disaggregated memory than on-board memory. In general the recovery is to unmount and re-initialize the memory, though degraded modes may be possible in the future when famfs supports file systems backed by more than one daxdev (data on a working daxdev can still be accessed). For now, return errors for any file that has touched an invalid or errored daxdev. Signed-off-by: John Groves --- fs/fuse/famfs.c | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c index ac56317944d9..8d13282e8949 100644 --- a/fs/fuse/famfs.c +++ b/fs/fuse/famfs.c @@ -589,6 +589,26 @@ famfs_file_init_dax( static int famfs_file_bad(struct inode *inode); +static int famfs_dax_err(struct famfs_daxdev *dd) +{ + if (!dd->valid) { + pr_err("%s: daxdev=%s invalid\n", + __func__, dd->name); + return -EIO; + } + if (dd->dax_err) { + pr_err("%s: daxdev=%s dax_err\n", + __func__, dd->name); + return -EIO; + } + if (dd->error) { + pr_err("%s: daxdev=%s memory error\n", + __func__, dd->name); + return -EHWPOISON; + } + return 0; +} + /** * famfs_fileofs_to_daxofs() - Resolve (file, offset, len) to (daxdev, offset, len) * @@ -661,6 +681,7 @@ famfs_fileofs_to_daxofs(struct inode *inode, struct iomap *iomap, u64 daxdev_idx = meta->se[i].dev_index; loff_t ext_len_remainder = dax_ext_len - local_offset; struct famfs_daxdev *dd; + int rc; if (daxdev_idx >= fc->dax_devlist->nslots) { pr_err("%s: daxdev_idx %llu >= nslots %d\n", @@ -670,6 +691,13 @@ famfs_fileofs_to_daxofs(struct inode *inode, struct iomap *iomap, dd = &fc->dax_devlist->devlist[daxdev_idx]; + rc = famfs_dax_err(dd); + if (rc) { + /* Shut down access to this file */ + meta->error = true; + return rc; + } + iomap->addr = dax_ext_offset + local_offset; iomap->offset = file_offset; iomap->length = min_t(loff_t, len, ext_len_remainder); -- 2.53.0 From: John Groves Famfs is memory-backed; there is no place to write back to, and no reason to mark pages dirty at all. Provide an address_space_operations using noop_dirty_folio so the dax paths have the ops they expect. Signed-off-by: John Groves --- fs/fuse/famfs.c | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c index 8d13282e8949..aea0bceef774 100644 --- a/fs/fuse/famfs.c +++ b/fs/fuse/famfs.c @@ -16,6 +16,7 @@ #include #include #include +#include #include #include #include @@ -41,6 +42,15 @@ static const struct dax_holder_operations famfs_fuse_dax_holder_ops = { .notify_failure = famfs_dax_notify_failure, }; +/* + * DAX address_space_operations for famfs. + * famfs doesn't need dirty tracking - writes go directly to + * memory with no writeback required. + */ +static const struct address_space_operations famfs_dax_aops = { + .dirty_folio = noop_dirty_folio, +}; + /*****************************************************************************/ /* @@ -564,6 +574,7 @@ famfs_file_init_dax( if (famfs_meta_set(fi, meta) == NULL) { i_size_write(inode, meta->file_size); inode->i_flags |= S_DAX; + inode->i_data.a_ops = &famfs_dax_aops; } else { pr_debug("%s: file already had metadata\n", __func__); __famfs_meta_free(meta); -- 2.53.0 From: John Groves Add Documentation/filesystems/famfs.rst and update MAINTAINERS Reviewed-by: Randy Dunlap Tested-by: Randy Dunlap Reviewed-by: Jonathan Cameron Signed-off-by: John Groves --- Documentation/filesystems/famfs.rst | 143 ++++++++++++++++++++++++++++ Documentation/filesystems/index.rst | 1 + MAINTAINERS | 1 + 3 files changed, 145 insertions(+) create mode 100644 Documentation/filesystems/famfs.rst diff --git a/Documentation/filesystems/famfs.rst b/Documentation/filesystems/famfs.rst new file mode 100644 index 000000000000..2b4a2269ef55 --- /dev/null +++ b/Documentation/filesystems/famfs.rst @@ -0,0 +1,143 @@ +.. SPDX-License-Identifier: GPL-2.0 + +.. _famfs_index: + +================================================================== +famfs: The fabric-attached memory file system +================================================================== + +- Copyright (C) 2024-2026 Micron Technology, Inc. + +Introduction +============ +Compute Express Link (CXL) provides a mechanism for disaggregated or +fabric-attached memory (FAM). This creates opportunities for data sharing; +clustered apps that would otherwise have to shard or replicate data can +share one copy in disaggregated memory. + +Famfs, which is not CXL-specific in any way, provides a mechanism for +multiple hosts to concurrently access data in shared memory, by giving it +a file system interface. With famfs, any app that understands files can +access data sets in shared memory. Although famfs supports read and write, +the real point is to support mmap, which provides direct (dax) access to +the memory - either writable or read-only. + +Shared memory can pose complex coherency and synchronization issues, but +there are also simple cases. Two simple and eminently useful patterns that +occur frequently in data analytics and AI are: + +* Serial Sharing - Only one host or process at a time has access to a file +* Read-only Sharing - Multiple hosts or processes share read-only access + to a file + +The famfs fuse file system is part of the famfs framework; user space +components [1] handle metadata allocation and distribution, and provide a +low-level fuse server to expose files that map directly to [presumably +shared] memory. + +The famfs framework manages coherency of its own metadata and structures, +but does not attempt to manage coherency for applications. + +Famfs also provides data isolation between files. That is, even though +the host has access to an entire memory "device" (as a devdax device), apps +cannot write to memory for which the file is read-only, and mapping one +file provides isolation from the memory of all other files. This is pretty +basic, but some experimental shared memory usage patterns provide no such +isolation. + +Principles of Operation +======================= + +Famfs is a file system with one or more devdax devices as a first-class +backing device(s). Metadata maintenance and query operations happen +entirely in user space. + +The famfs low-level fuse server daemon provides file maps (fmaps) and +devdax device info to the fuse/famfs kernel component so that +read/write/mapping faults can be handled without up-calls for all active +files. + +The famfs user space is responsible for maintaining and distributing +consistent metadata. This is currently handled via an append-only +metadata log within the memory, but this is orthogonal to the fuse/famfs +kernel code. + +Once instantiated, "the same file" on each host points to the same shared +memory, but in-memory metadata (inodes, etc.) is ephemeral on each host +that has a famfs instance mounted. Use cases are free to allow or not +allow mutations to data on a file-by-file basis. + +When an app accesses a data object in a famfs file, there is no page cache +involvement. The CPU cache is loaded directly from the shared memory. In +some use cases, this is an enormous reduction in read amplification +compared to loading an entire page into the page cache. + + +Famfs is Not a Conventional File System +--------------------------------------- + +Famfs files can be accessed by conventional means, but there are +limitations. The kernel component of fuse/famfs is not involved in the +allocation of backing memory for files at all; the famfs user space +creates files and responds as a low-level fuse server with fmaps and +devdax device info upon request. + +Famfs differs in some important ways from conventional file systems: + +* Files must be pre-allocated by the famfs framework; allocation is never + performed on (or after) write. +* Any operation that changes a file's size is considered to put the file + in an invalid state, disabling access to the data. It may be possible to + revisit this in the future. (Typically the famfs user space can restore + files to a valid state by replaying the famfs metadata log.) + +Famfs exists to apply the existing file system abstractions to shared +memory so applications and workflows can more easily adapt to an +environment with disaggregated shared memory. + +Memory Error Handling +===================== + +Possible memory errors include timeouts, poison, and unexpected +reconfiguration of an underlying dax device. In all of these cases, famfs +receives a call from the devdax layer via its +dax_holder_operations->notify_failure() function. If any memory errors have +been detected, access to the affected +daxdev is disabled to avoid further errors or corruption. + +In all known cases, famfs can be unmounted cleanly. In most cases errors +can be cleared by re-initializing the memory - at which point a new famfs +file system can be created. + +Key Requirements +================ + +The primary requirements for famfs are: + +1. Must support a file system abstraction backed by sharable devdax memory +2. Files must efficiently handle VMA faults +3. Must support metadata distribution in a sharable way +4. Must handle clients with a stale copy of metadata + +The famfs kernel component takes care of 1-2 above by caching each file's +mapping metadata in the kernel. + +Requirements 3 and 4 are handled by the user space components, and are +largely orthogonal to the functionality of the famfs kernel module. + +Requirements 3 and 4 cannot be met by conventional fs-dax file systems +(e.g. xfs) because they use write-back metadata; it is not valid to mount +such a file system on two hosts from the same in-memory image. + + +Famfs Usage +=========== + +Famfs usage is documented at [1]. + + +References +========== + +- [1] Famfs user space repository and documentation + https://github.com/cxl-micron-reskit/famfs diff --git a/Documentation/filesystems/index.rst b/Documentation/filesystems/index.rst index 1f71cf159547..97a9a1f1e96a 100644 --- a/Documentation/filesystems/index.rst +++ b/Documentation/filesystems/index.rst @@ -91,6 +91,7 @@ Documentation for filesystem implementations. ext3 ext4/index f2fs + famfs gfs2/index hfs hfsplus diff --git a/MAINTAINERS b/MAINTAINERS index 07db3a5a2af7..1477b66f4450 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -10718,6 +10718,7 @@ M: John Groves L: linux-cxl@vger.kernel.org L: linux-fsdevel@vger.kernel.org S: Supported +F: Documentation/filesystems/famfs.rst F: fs/fuse/famfs.c F: fs/fuse/famfs_kfmap.h -- 2.53.0