[PATCH V12 07/12] famfs: MAP_CREATE ioctl and fmap ingest (ABI 44)
From: John Groves
Date: Sun Aug 02 2026 - 22:34:28 EST
From: John Groves <John@xxxxxxxxxx>
Add the famfs file ioctl handler (FAMFSIOC_NOP, FAMFSIOC_MAP_CREATE) and
the KABI-44 self-describing fmap message: the wire ABI in famfs_ioctl.h
(famfs_ioc_fmap_header plus the simple and interleaved extent structs), the
in-core famfs_file_meta, and famfs_file_init_dax(), which copies the
message in, parses both the simple-extent and interleaved (striped) wire
forms into inode->i_private, and sets S_DAX.
Resolving those mappings to dax-device offsets (iomap_begin) is added in
the following commit; the read/write/fault paths keep their NULL iomap_ops
stub until then.
Also add famfs ioctls to ioctl-number.rst
Signed-off-by: John Groves <john@xxxxxxxxxx>
---
.../userspace-api/ioctl/ioctl-number.rst | 1 +
fs/famfs/famfs_file.c | 326 +++++++++++++++++-
fs/famfs/famfs_inode.c | 1 +
fs/famfs/famfs_internal.h | 46 +++
include/uapi/linux/famfs_ioctl.h | 91 +++++
5 files changed, 462 insertions(+), 3 deletions(-)
create mode 100644 include/uapi/linux/famfs_ioctl.h
diff --git a/Documentation/userspace-api/ioctl/ioctl-number.rst b/Documentation/userspace-api/ioctl/ioctl-number.rst
index 3f0ef1e27eb0..5e244dec1b98 100644
--- a/Documentation/userspace-api/ioctl/ioctl-number.rst
+++ b/Documentation/userspace-api/ioctl/ioctl-number.rst
@@ -299,6 +299,7 @@ Code Seq# Include File Comments
'u' 00-2F linux/ublk_cmd.h conflict!
'u' 20-3F linux/uvcvideo.h USB video class host driver
'u' 40-4f linux/udmabuf.h userspace dma-buf misc device
+'u' 50-5F linux/famfs_ioctl.h famfs shared memory file system
'v' 00-1F linux/ext2_fs.h conflict!
'v' 00-1F linux/fs.h conflict!
'v' 00-0F linux/sonypi.h conflict!
diff --git a/fs/famfs/famfs_file.c b/fs/famfs/famfs_file.c
index 678f2035fd5f..d710c8a0c923 100644
--- a/fs/famfs/famfs_file.c
+++ b/fs/famfs/famfs_file.c
@@ -13,9 +13,313 @@
#include <linux/mm.h>
#include <linux/dax.h>
#include <linux/iomap.h>
+#include <linux/capability.h>
+#include <linux/famfs_ioctl.h>
#include "famfs_internal.h"
+/* Expose famfs kernel abi version as a read-only module parameter */
+static int famfs_kabi_version = FAMFS_KABI_VERSION;
+module_param(famfs_kabi_version, int, 0444);
+MODULE_PARM_DESC(famfs_kabi_version, "famfs kernel abi version");
+
+void
+famfs_meta_free(struct famfs_file_meta *map)
+{
+ if (map) {
+ switch (map->fm_extent_type) {
+ case FAMFS_IOC_EXT_SIMPLE:
+ kfree(map->se);
+ break;
+ case FAMFS_IOC_EXT_INTERLEAVE:
+ if (map->ie) {
+ u32 i;
+
+ for (i = 0; i < map->fm_niext; i++)
+ kfree(map->ie[i].ie_strips);
+ }
+ kfree(map->ie);
+ break;
+ default:
+ break;
+ }
+ }
+ kfree(map);
+}
+
+/**
+ * famfs_file_init_dax() - FAMFSIOC_MAP_CREATE ioctl handler
+ * @file: the un-initialized file
+ * @arg: user pointer to a self-describing fmap message
+ *
+ * The map-create ioctl carries the fmap as a self-describing message: a
+ * struct famfs_ioc_fmap_header followed by an extent list. The message is
+ * copied in, parsed into a famfs_file_meta, and published on inode->i_private.
+ * Both the simple-extent and the interleaved (striped) wire forms are handled.
+ * The wire layout byte-matches the fmap carried in a fuse famfs GET_FMAP reply.
+ */
+static int
+famfs_check_ext_alignment(struct famfs_meta_simple_ext *se)
+{
+ int errs = 0;
+
+ if (!IS_ALIGNED(se->ext_offset, PMD_SIZE))
+ errs++;
+ if (!IS_ALIGNED(se->ext_len, PMD_SIZE))
+ errs++;
+
+ return errs;
+}
+
+static int
+famfs_file_init_dax(struct file *file, void __user *arg)
+{
+ struct famfs_ioc_fmap_header fmh;
+ struct famfs_file_meta *meta = NULL;
+ struct famfs_fs_info *fsi;
+ struct super_block *sb;
+ struct inode *inode;
+ void *fmap_buf = NULL;
+ size_t extent_total = 0;
+ size_t next_offset;
+ int errs = 0;
+ int rc;
+ u32 i, j;
+
+ inode = file_inode(file);
+ if (!inode)
+ return -EBADF;
+ if (inode->i_private)
+ return -EEXIST;
+
+ sb = inode->i_sb;
+ fsi = sb->s_fs_info;
+ if (fsi->deverror)
+ return -ENODEV;
+ if (!famfs_opt_enabled(fsi, FAMFS_OPT_MAP_CREATE))
+ return -EPERM;
+
+ if (copy_from_user(&fmh, arg, sizeof(fmh)))
+ return -EFAULT;
+
+ if (fmh.fmap_version != FAMFS_FMAP_VERSION)
+ return -EINVAL;
+ if (fmh.fmap_size < sizeof(fmh))
+ return -EINVAL;
+ if (fmh.fmap_size > FAMFS_FMAP_MSG_MAX)
+ return -EFBIG;
+ if (fmh.nextents < 1)
+ return -EINVAL;
+
+ fmap_buf = kvmalloc(fmh.fmap_size, GFP_KERNEL);
+ if (!fmap_buf)
+ return -ENOMEM;
+
+ if (copy_from_user(fmap_buf, arg, fmh.fmap_size)) {
+ rc = -EFAULT;
+ goto out;
+ }
+ next_offset = sizeof(fmh); /* start of the extent list */
+
+ meta = kzalloc_obj(*meta, GFP_KERNEL);
+ if (!meta) {
+ rc = -ENOMEM;
+ goto out;
+ }
+
+ meta->error = false;
+ meta->file_type = fmh.file_type;
+ meta->file_size = fmh.file_size;
+ meta->fm_extent_type = fmh.ext_type;
+
+ switch (fmh.ext_type) {
+ case FAMFS_IOC_EXT_SIMPLE: {
+ struct famfs_ioc_simple_ext *se_in = fmap_buf + next_offset;
+
+ next_offset += (size_t)fmh.nextents * sizeof(*se_in);
+ if (next_offset > fmh.fmap_size) {
+ rc = -EINVAL;
+ goto out;
+ }
+
+ meta->fm_nextents = fmh.nextents;
+ meta->se = kcalloc(meta->fm_nextents, sizeof(*meta->se),
+ GFP_KERNEL);
+ if (!meta->se) {
+ rc = -ENOMEM;
+ goto out;
+ }
+
+ for (i = 0; i < fmh.nextents; i++) {
+ meta->se[i].dev_index = se_in[i].se_devindex;
+ meta->se[i].ext_offset = se_in[i].se_offset;
+ meta->se[i].ext_len = se_in[i].se_len;
+
+ if (meta->se[i].dev_index >= FAMFS_MAX_DAXDEVS) {
+ rc = -EINVAL;
+ goto out;
+ }
+ meta->dev_bitmap |= BIT_ULL(meta->se[i].dev_index);
+ errs += famfs_check_ext_alignment(&meta->se[i]);
+ extent_total += meta->se[i].ext_len;
+ }
+ break;
+ }
+
+ case FAMFS_IOC_EXT_INTERLEAVE: {
+ s64 size_remainder = meta->file_size;
+ u32 niext = fmh.nextents;
+
+ meta->fm_niext = niext;
+ meta->ie = kcalloc(niext, sizeof(*meta->ie), GFP_KERNEL);
+ if (!meta->ie) {
+ rc = -ENOMEM;
+ goto out;
+ }
+
+ /* Outer loop is over the separate interleaved extents */
+ for (i = 0; i < niext; i++) {
+ struct famfs_ioc_iext *ie_in = fmap_buf + next_offset;
+ struct famfs_ioc_simple_ext *sie_in;
+ u64 nstrips;
+
+ next_offset += sizeof(*ie_in);
+ if (next_offset > fmh.fmap_size) {
+ rc = -EINVAL;
+ goto out;
+ }
+
+ if (ie_in->ie_chunk_size == 0 ||
+ !IS_ALIGNED(ie_in->ie_chunk_size, PMD_SIZE)) {
+ rc = -EINVAL;
+ goto out;
+ }
+ if (ie_in->ie_nbytes == 0) {
+ rc = -EINVAL;
+ goto out;
+ }
+
+ nstrips = ie_in->ie_nstrips;
+ if (nstrips < 1) {
+ rc = -EINVAL;
+ goto out;
+ }
+
+ meta->ie[i].fie_chunk_size = ie_in->ie_chunk_size;
+ meta->ie[i].fie_nstrips = ie_in->ie_nstrips;
+ meta->ie[i].fie_nbytes = ie_in->ie_nbytes;
+
+ /* The strip extents follow the interleaved-ext header */
+ sie_in = fmap_buf + next_offset;
+ next_offset += nstrips * sizeof(*sie_in);
+ if (next_offset > fmh.fmap_size) {
+ rc = -EINVAL;
+ goto out;
+ }
+
+ meta->ie[i].ie_strips =
+ kcalloc(nstrips, sizeof(meta->ie[i].ie_strips[0]),
+ GFP_KERNEL);
+ if (!meta->ie[i].ie_strips) {
+ rc = -ENOMEM;
+ goto out;
+ }
+
+ /* Inner loop is over the strips */
+ for (j = 0; j < nstrips; j++) {
+ struct famfs_meta_simple_ext *so =
+ &meta->ie[i].ie_strips[j];
+
+ so->dev_index = sie_in[j].se_devindex;
+ so->ext_offset = sie_in[j].se_offset;
+ so->ext_len = sie_in[j].se_len;
+
+ if (so->dev_index >= FAMFS_MAX_DAXDEVS) {
+ rc = -EINVAL;
+ goto out;
+ }
+ meta->dev_bitmap |= BIT_ULL(so->dev_index);
+ errs += famfs_check_ext_alignment(so);
+ extent_total += so->ext_len;
+ size_remainder -= so->ext_len;
+ }
+ }
+
+ if (size_remainder > 0) {
+ /* Strips do not cover the whole file */
+ rc = -EINVAL;
+ goto out;
+ }
+ break;
+ }
+
+ default:
+ rc = -EINVAL;
+ goto out;
+ }
+
+ if (errs > 0) {
+ rc = -EINVAL;
+ goto out;
+ }
+ if (extent_total < meta->file_size) {
+ rc = -EINVAL;
+ goto out;
+ }
+
+ /* Publish the famfs metadata on inode->i_private */
+ inode_lock(inode);
+ if (inode->i_private) {
+ rc = -EEXIST; /* file already has famfs metadata */
+ } else {
+ inode->i_private = meta;
+ i_size_write(inode, meta->file_size);
+ inode->i_flags |= S_DAX;
+ meta = NULL; /* owned by the inode now */
+ rc = 0;
+ }
+ inode_unlock(inode);
+
+out:
+ kvfree(fmap_buf);
+ if (meta)
+ famfs_meta_free(meta);
+ return rc;
+}
+
+/**
+ * famfs_file_ioctl() - Top-level famfs file ioctl handler
+ * @file: the file
+ * @cmd: ioctl opcode
+ * @arg: ioctl opcode argument (if any)
+ */
+static long
+famfs_file_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
+{
+ struct inode *inode = file_inode(file);
+ struct famfs_fs_info *fsi = inode->i_sb->s_fs_info;
+ long rc;
+
+ if (fsi->deverror && (cmd != FAMFSIOC_NOP))
+ return -ENODEV;
+
+ switch (cmd) {
+ case FAMFSIOC_NOP:
+ rc = 0;
+ break;
+
+ case FAMFSIOC_MAP_CREATE:
+ rc = famfs_file_init_dax(file, (void __user *)arg);
+ break;
+
+ default:
+ rc = -ENOTTY;
+ break;
+ }
+
+ return rc;
+}
+
/*********************************************************************
* vm_operations
*/
@@ -93,9 +397,25 @@ const struct vm_operations_struct famfs_file_vm_ops = {
static ssize_t
famfs_file_invalid(struct inode *inode)
{
+ struct famfs_file_meta *meta = inode->i_private;
+ size_t i_size = i_size_read(inode);
+
+ if (!meta) {
+ pr_debug("%s: un-initialized famfs file\n", __func__);
+ return -EIO;
+ }
+ if (meta->error) {
+ pr_debug("%s: previously detected metadata errors\n", __func__);
+ return -EIO;
+ }
+ if (i_size != meta->file_size) {
+ pr_warn("%s: i_size overwritten from %ld to %ld\n",
+ __func__, meta->file_size, i_size);
+ meta->error = true;
+ return -ENXIO;
+ }
if (!IS_DAX(inode)) {
- pr_debug("%s: inode %llx IS_DAX is false\n",
- __func__, (u64)inode);
+ pr_debug("%s: inode %llx IS_DAX is false\n", __func__, (u64)inode);
return -ENXIO;
}
return 0;
@@ -222,7 +542,7 @@ const struct file_operations famfs_file_operations = {
/* Custom famfs operations */
.write_iter = famfs_dax_write_iter,
.read_iter = famfs_dax_read_iter,
- .unlocked_ioctl = NULL /*famfs_file_ioctl*/,
+ .unlocked_ioctl = famfs_file_ioctl,
.mmap = famfs_file_mmap,
/* Force PMD alignment for mmap */
diff --git a/fs/famfs/famfs_inode.c b/fs/famfs/famfs_inode.c
index 910a143dad30..a6c3b4574e69 100644
--- a/fs/famfs/famfs_inode.c
+++ b/fs/famfs/famfs_inode.c
@@ -300,6 +300,7 @@ static int famfs_show_options(struct seq_file *m, struct dentry *root)
static void famfs_evict_inode(struct inode *inode)
{
+ famfs_meta_free((struct famfs_file_meta *)inode->i_private);
inode->i_private = NULL;
dax_break_layout_final(inode);
truncate_inode_pages_final(&inode->i_data);
diff --git a/fs/famfs/famfs_internal.h b/fs/famfs/famfs_internal.h
index 26f5abda96dc..b5f9c8d0349f 100644
--- a/fs/famfs/famfs_internal.h
+++ b/fs/famfs/famfs_internal.h
@@ -15,8 +15,52 @@
#include <linux/bits.h>
#include <linux/build_bug.h>
+#include <linux/famfs_ioctl.h>
+
extern const struct file_operations famfs_file_operations;
+/*
+ * Internal sanity bound on a FAMFSIOC_MAP_CREATE fmap message. The ABI does
+ * not advertise a maximum (the message is self-describing); this only guards
+ * the copy-in against an unreasonable allocation. Oversize is rejected with
+ * -EFBIG.
+ */
+#define FAMFS_FMAP_MSG_MAX (4 * 1024 * 1024)
+
+struct famfs_meta_simple_ext {
+ u64 dev_index;
+ u64 ext_offset;
+ u64 ext_len;
+};
+
+struct famfs_meta_interleaved_ext {
+ u64 fie_nstrips;
+ u64 fie_chunk_size;
+ u64 fie_nbytes;
+ struct famfs_meta_simple_ext *ie_strips;
+};
+
+/*
+ * Each famfs dax file has this hanging from its inode->i_private.
+ */
+struct famfs_file_meta {
+ bool error;
+ enum famfs_file_type file_type;
+ size_t file_size;
+ enum famfs_ioc_ext_type fm_extent_type;
+ u64 dev_bitmap; /* referenced daxdev indices */
+ union { /* This will make code a bit more readable */
+ struct {
+ size_t fm_nextents;
+ struct famfs_meta_simple_ext *se;
+ };
+ struct {
+ size_t fm_niext;
+ struct famfs_meta_interleaved_ext *ie;
+ };
+ };
+};
+
struct famfs_mount_opts {
umode_t mode;
};
@@ -83,4 +127,6 @@ int famfs_devlist_alloc(struct famfs_fs_info *fsi);
int famfs_install_daxdev(struct famfs_fs_info *fsi, struct super_block *sb,
u64 index, dev_t devno, const char *name);
+void famfs_meta_free(struct famfs_file_meta *map);
+
#endif /* FAMFS_INTERNAL_H */
diff --git a/include/uapi/linux/famfs_ioctl.h b/include/uapi/linux/famfs_ioctl.h
new file mode 100644
index 000000000000..b4eb373c1ade
--- /dev/null
+++ b/include/uapi/linux/famfs_ioctl.h
@@ -0,0 +1,91 @@
+/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */
+/*
+ * famfs - dax file system for shared fabric-attached memory
+ *
+ * Copyright 2023-2024 Micron Technology, Inc.
+ *
+ * This file system, originally based on ramfs the dax support from xfs,
+ * is intended to allow multiple host systems to mount a common file system
+ * view of dax files that map to shared memory.
+ */
+#ifndef FAMFS_IOCTL_H
+#define FAMFS_IOCTL_H
+
+#include <linux/ioctl.h>
+#include <linux/uuid.h>
+
+#define FAMFS_KABI_VERSION 44
+
+enum famfs_file_type {
+ FAMFS_REG,
+ FAMFS_SUPERBLOCK,
+ FAMFS_LOG,
+};
+
+/*
+ * Extent type in a famfs fmap message, and of the in-core map
+ * (famfs_file_meta.fm_extent_type).
+ */
+enum famfs_ioc_ext_type {
+ FAMFS_IOC_EXT_SIMPLE,
+ FAMFS_IOC_EXT_INTERLEAVE,
+};
+
+/*
+ * The FAMFSIOC_MAP_CREATE payload is a self-describing fmap message: a
+ * struct famfs_ioc_fmap_header immediately followed by @nextents extent
+ * records. @fmap_size gives the total message length, so a reader is
+ * self-delimiting.
+ *
+ * For ext_type == FAMFS_IOC_EXT_SIMPLE the records are an array of
+ * @nextents famfs_ioc_simple_ext. For ext_type == FAMFS_IOC_EXT_INTERLEAVE
+ * each of the @nextents records is a famfs_ioc_iext header immediately
+ * followed by ie_nstrips famfs_ioc_simple_ext strip extents.
+ *
+ * This wire layout is byte-identical to the fmap carried in a fuse famfs
+ * GET_FMAP reply, so the same userspace serializer emits both.
+ *
+ * The message is self-describing (@fmap_size bounds it), so neither the extent
+ * and strip counts nor the total size are capped by this ABI. The kernel
+ * applies an internal sanity limit to the copy-in and returns -EFBIG for a
+ * message larger than it will accept.
+ */
+#define FAMFS_FMAP_VERSION 1
+
+struct famfs_ioc_simple_ext {
+ __u32 se_devindex;
+ __u32 reserved;
+ __u64 se_offset;
+ __u64 se_len;
+};
+
+struct famfs_ioc_iext { /* interleaved (striped) extent */
+ __u32 ie_nstrips;
+ __u32 ie_chunk_size;
+ __u64 ie_nbytes; /* total bytes mapped by this interleaved extent */
+ __u64 reserved;
+};
+
+struct famfs_ioc_fmap_header {
+ __u8 file_type; /* enum famfs_file_type */
+ __u8 reserved;
+ __u16 fmap_version; /* FAMFS_FMAP_VERSION */
+ __u32 ext_type; /* enum famfs_ioc_ext_type */
+ __u32 nextents;
+ __u32 fmap_size; /* total message bytes, including this header */
+ __u64 file_size;
+ __u64 reserved1;
+};
+
+#define FAMFSIOC_MAGIC 'u'
+
+/* famfs file ioctl opcodes */
+#define FAMFSIOC_NOP _IO(FAMFSIOC_MAGIC, 0x50)
+
+/*
+ * MAP_CREATE carries the self-describing fmap message - struct
+ * famfs_ioc_fmap_header followed by the extent list (see above).
+ */
+#define FAMFSIOC_MAP_CREATE _IOW(FAMFSIOC_MAGIC, 0x51, struct famfs_ioc_fmap_header)
+
+#endif /* FAMFS_IOCTL_H */
--
2.53.0