From: John Groves <[email protected]>

Add the famfs file ioctl handler (FAMFSIOC_NOP, FAMFSIOC_MAP_CREATE) and
the KABI-44 self-describing fmap message: the wire ABI in famfs_ioctl.h
(famfs_ioc_fmap_header plus the simple and interleaved extent structs), the
in-core famfs_file_meta, and famfs_file_init_dax(), which copies the
message in, parses both the simple-extent and interleaved (striped) wire
forms into inode->i_private, and sets S_DAX.

Resolving those mappings to dax-device offsets (iomap_begin) is added in
the following commit; the read/write/fault paths keep their NULL iomap_ops
stub until then.

Also add famfs ioctls to ioctl-number.rst

Signed-off-by: John Groves <[email protected]>
---
 .../userspace-api/ioctl/ioctl-number.rst      |   1 +
 fs/famfs/famfs_file.c                         | 326 +++++++++++++++++-
 fs/famfs/famfs_inode.c                        |   1 +
 fs/famfs/famfs_internal.h                     |  46 +++
 include/uapi/linux/famfs_ioctl.h              |  91 +++++
 5 files changed, 462 insertions(+), 3 deletions(-)
 create mode 100644 include/uapi/linux/famfs_ioctl.h

diff --git a/Documentation/userspace-api/ioctl/ioctl-number.rst 
b/Documentation/userspace-api/ioctl/ioctl-number.rst
index 3f0ef1e27eb0..5e244dec1b98 100644
--- a/Documentation/userspace-api/ioctl/ioctl-number.rst
+++ b/Documentation/userspace-api/ioctl/ioctl-number.rst
@@ -299,6 +299,7 @@ Code  Seq#    Include File                                  
           Comments
 'u'   00-2F  linux/ublk_cmd.h                                          
conflict!
 'u'   20-3F  linux/uvcvideo.h                                          USB 
video class host driver
 'u'   40-4f  linux/udmabuf.h                                           
userspace dma-buf misc device
+'u'   50-5F  linux/famfs_ioctl.h                                       famfs 
shared memory file system
 'v'   00-1F  linux/ext2_fs.h                                           
conflict!
 'v'   00-1F  linux/fs.h                                                
conflict!
 'v'   00-0F  linux/sonypi.h                                            
conflict!
diff --git a/fs/famfs/famfs_file.c b/fs/famfs/famfs_file.c
index 678f2035fd5f..d710c8a0c923 100644
--- a/fs/famfs/famfs_file.c
+++ b/fs/famfs/famfs_file.c
@@ -13,9 +13,313 @@
 #include <linux/mm.h>
 #include <linux/dax.h>
 #include <linux/iomap.h>
+#include <linux/capability.h>
 
+#include <linux/famfs_ioctl.h>
 #include "famfs_internal.h"
 
+/* Expose famfs kernel abi version as a read-only module parameter */
+static int famfs_kabi_version = FAMFS_KABI_VERSION;
+module_param(famfs_kabi_version, int, 0444);
+MODULE_PARM_DESC(famfs_kabi_version, "famfs kernel abi version");
+
+void
+famfs_meta_free(struct famfs_file_meta *map)
+{
+       if (map) {
+               switch (map->fm_extent_type) {
+               case FAMFS_IOC_EXT_SIMPLE:
+                       kfree(map->se);
+                       break;
+               case FAMFS_IOC_EXT_INTERLEAVE:
+                       if (map->ie) {
+                               u32 i;
+
+                               for (i = 0; i < map->fm_niext; i++)
+                                       kfree(map->ie[i].ie_strips);
+                       }
+                       kfree(map->ie);
+                       break;
+               default:
+                       break;
+               }
+       }
+       kfree(map);
+}
+
+/**
+ * famfs_file_init_dax() - FAMFSIOC_MAP_CREATE ioctl handler
+ * @file: the un-initialized file
+ * @arg:  user pointer to a self-describing fmap message
+ *
+ * The map-create ioctl carries the fmap as a self-describing message: a
+ * struct famfs_ioc_fmap_header followed by an extent list. The message is
+ * copied in, parsed into a famfs_file_meta, and published on inode->i_private.
+ * Both the simple-extent and the interleaved (striped) wire forms are handled.
+ * The wire layout byte-matches the fmap carried in a fuse famfs GET_FMAP 
reply.
+ */
+static int
+famfs_check_ext_alignment(struct famfs_meta_simple_ext *se)
+{
+       int errs = 0;
+
+       if (!IS_ALIGNED(se->ext_offset, PMD_SIZE))
+               errs++;
+       if (!IS_ALIGNED(se->ext_len, PMD_SIZE))
+               errs++;
+
+       return errs;
+}
+
+static int
+famfs_file_init_dax(struct file *file, void __user *arg)
+{
+       struct famfs_ioc_fmap_header fmh;
+       struct famfs_file_meta *meta = NULL;
+       struct famfs_fs_info *fsi;
+       struct super_block *sb;
+       struct inode *inode;
+       void *fmap_buf = NULL;
+       size_t extent_total = 0;
+       size_t next_offset;
+       int errs = 0;
+       int rc;
+       u32 i, j;
+
+       inode = file_inode(file);
+       if (!inode)
+               return -EBADF;
+       if (inode->i_private)
+               return -EEXIST;
+
+       sb  = inode->i_sb;
+       fsi = sb->s_fs_info;
+       if (fsi->deverror)
+               return -ENODEV;
+       if (!famfs_opt_enabled(fsi, FAMFS_OPT_MAP_CREATE))
+               return -EPERM;
+
+       if (copy_from_user(&fmh, arg, sizeof(fmh)))
+               return -EFAULT;
+
+       if (fmh.fmap_version != FAMFS_FMAP_VERSION)
+               return -EINVAL;
+       if (fmh.fmap_size < sizeof(fmh))
+               return -EINVAL;
+       if (fmh.fmap_size > FAMFS_FMAP_MSG_MAX)
+               return -EFBIG;
+       if (fmh.nextents < 1)
+               return -EINVAL;
+
+       fmap_buf = kvmalloc(fmh.fmap_size, GFP_KERNEL);
+       if (!fmap_buf)
+               return -ENOMEM;
+
+       if (copy_from_user(fmap_buf, arg, fmh.fmap_size)) {
+               rc = -EFAULT;
+               goto out;
+       }
+       next_offset = sizeof(fmh);      /* start of the extent list */
+
+       meta = kzalloc_obj(*meta, GFP_KERNEL);
+       if (!meta) {
+               rc = -ENOMEM;
+               goto out;
+       }
+
+       meta->error = false;
+       meta->file_type = fmh.file_type;
+       meta->file_size = fmh.file_size;
+       meta->fm_extent_type = fmh.ext_type;
+
+       switch (fmh.ext_type) {
+       case FAMFS_IOC_EXT_SIMPLE: {
+               struct famfs_ioc_simple_ext *se_in = fmap_buf + next_offset;
+
+               next_offset += (size_t)fmh.nextents * sizeof(*se_in);
+               if (next_offset > fmh.fmap_size) {
+                       rc = -EINVAL;
+                       goto out;
+               }
+
+               meta->fm_nextents = fmh.nextents;
+               meta->se = kcalloc(meta->fm_nextents, sizeof(*meta->se),
+                                  GFP_KERNEL);
+               if (!meta->se) {
+                       rc = -ENOMEM;
+                       goto out;
+               }
+
+               for (i = 0; i < fmh.nextents; i++) {
+                       meta->se[i].dev_index  = se_in[i].se_devindex;
+                       meta->se[i].ext_offset = se_in[i].se_offset;
+                       meta->se[i].ext_len    = se_in[i].se_len;
+
+                       if (meta->se[i].dev_index >= FAMFS_MAX_DAXDEVS) {
+                               rc = -EINVAL;
+                               goto out;
+                       }
+                       meta->dev_bitmap |= BIT_ULL(meta->se[i].dev_index);
+                       errs += famfs_check_ext_alignment(&meta->se[i]);
+                       extent_total += meta->se[i].ext_len;
+               }
+               break;
+       }
+
+       case FAMFS_IOC_EXT_INTERLEAVE: {
+               s64 size_remainder = meta->file_size;
+               u32 niext = fmh.nextents;
+
+               meta->fm_niext = niext;
+               meta->ie = kcalloc(niext, sizeof(*meta->ie), GFP_KERNEL);
+               if (!meta->ie) {
+                       rc = -ENOMEM;
+                       goto out;
+               }
+
+               /* Outer loop is over the separate interleaved extents */
+               for (i = 0; i < niext; i++) {
+                       struct famfs_ioc_iext *ie_in = fmap_buf + next_offset;
+                       struct famfs_ioc_simple_ext *sie_in;
+                       u64 nstrips;
+
+                       next_offset += sizeof(*ie_in);
+                       if (next_offset > fmh.fmap_size) {
+                               rc = -EINVAL;
+                               goto out;
+                       }
+
+                       if (ie_in->ie_chunk_size == 0 ||
+                           !IS_ALIGNED(ie_in->ie_chunk_size, PMD_SIZE)) {
+                               rc = -EINVAL;
+                               goto out;
+                       }
+                       if (ie_in->ie_nbytes == 0) {
+                               rc = -EINVAL;
+                               goto out;
+                       }
+
+                       nstrips = ie_in->ie_nstrips;
+                       if (nstrips < 1) {
+                               rc = -EINVAL;
+                               goto out;
+                       }
+
+                       meta->ie[i].fie_chunk_size = ie_in->ie_chunk_size;
+                       meta->ie[i].fie_nstrips    = ie_in->ie_nstrips;
+                       meta->ie[i].fie_nbytes     = ie_in->ie_nbytes;
+
+                       /* The strip extents follow the interleaved-ext header 
*/
+                       sie_in = fmap_buf + next_offset;
+                       next_offset += nstrips * sizeof(*sie_in);
+                       if (next_offset > fmh.fmap_size) {
+                               rc = -EINVAL;
+                               goto out;
+                       }
+
+                       meta->ie[i].ie_strips =
+                               kcalloc(nstrips, 
sizeof(meta->ie[i].ie_strips[0]),
+                                       GFP_KERNEL);
+                       if (!meta->ie[i].ie_strips) {
+                               rc = -ENOMEM;
+                               goto out;
+                       }
+
+                       /* Inner loop is over the strips */
+                       for (j = 0; j < nstrips; j++) {
+                               struct famfs_meta_simple_ext *so =
+                                       &meta->ie[i].ie_strips[j];
+
+                               so->dev_index  = sie_in[j].se_devindex;
+                               so->ext_offset = sie_in[j].se_offset;
+                               so->ext_len    = sie_in[j].se_len;
+
+                               if (so->dev_index >= FAMFS_MAX_DAXDEVS) {
+                                       rc = -EINVAL;
+                                       goto out;
+                               }
+                               meta->dev_bitmap |= BIT_ULL(so->dev_index);
+                               errs += famfs_check_ext_alignment(so);
+                               extent_total += so->ext_len;
+                               size_remainder -= so->ext_len;
+                       }
+               }
+
+               if (size_remainder > 0) {
+                       /* Strips do not cover the whole file */
+                       rc = -EINVAL;
+                       goto out;
+               }
+               break;
+       }
+
+       default:
+               rc = -EINVAL;
+               goto out;
+       }
+
+       if (errs > 0) {
+               rc = -EINVAL;
+               goto out;
+       }
+       if (extent_total < meta->file_size) {
+               rc = -EINVAL;
+               goto out;
+       }
+
+       /* Publish the famfs metadata on inode->i_private */
+       inode_lock(inode);
+       if (inode->i_private) {
+               rc = -EEXIST; /* file already has famfs metadata */
+       } else {
+               inode->i_private = meta;
+               i_size_write(inode, meta->file_size);
+               inode->i_flags |= S_DAX;
+               meta = NULL; /* owned by the inode now */
+               rc = 0;
+       }
+       inode_unlock(inode);
+
+out:
+       kvfree(fmap_buf);
+       if (meta)
+               famfs_meta_free(meta);
+       return rc;
+}
+
+/**
+ * famfs_file_ioctl() - Top-level famfs file ioctl handler
+ * @file: the file
+ * @cmd:  ioctl opcode
+ * @arg:  ioctl opcode argument (if any)
+ */
+static long
+famfs_file_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
+{
+       struct inode *inode = file_inode(file);
+       struct famfs_fs_info *fsi = inode->i_sb->s_fs_info;
+       long rc;
+
+       if (fsi->deverror && (cmd != FAMFSIOC_NOP))
+               return -ENODEV;
+
+       switch (cmd) {
+       case FAMFSIOC_NOP:
+               rc = 0;
+               break;
+
+       case FAMFSIOC_MAP_CREATE:
+               rc = famfs_file_init_dax(file, (void __user *)arg);
+               break;
+
+       default:
+               rc = -ENOTTY;
+               break;
+       }
+
+       return rc;
+}
+
 /*********************************************************************
  * vm_operations
  */
@@ -93,9 +397,25 @@ const struct vm_operations_struct famfs_file_vm_ops = {
 static ssize_t
 famfs_file_invalid(struct inode *inode)
 {
+       struct famfs_file_meta *meta = inode->i_private;
+       size_t i_size = i_size_read(inode);
+
+       if (!meta) {
+               pr_debug("%s: un-initialized famfs file\n", __func__);
+               return -EIO;
+       }
+       if (meta->error) {
+               pr_debug("%s: previously detected metadata errors\n", __func__);
+               return -EIO;
+       }
+       if (i_size != meta->file_size) {
+               pr_warn("%s: i_size overwritten from %ld to %ld\n",
+                      __func__, meta->file_size, i_size);
+               meta->error = true;
+               return -ENXIO;
+       }
        if (!IS_DAX(inode)) {
-               pr_debug("%s: inode %llx IS_DAX is false\n",
-                        __func__, (u64)inode);
+               pr_debug("%s: inode %llx IS_DAX is false\n", __func__, 
(u64)inode);
                return -ENXIO;
        }
        return 0;
@@ -222,7 +542,7 @@ const struct file_operations famfs_file_operations = {
        /* Custom famfs operations */
        .write_iter        = famfs_dax_write_iter,
        .read_iter         = famfs_dax_read_iter,
-       .unlocked_ioctl    = NULL /*famfs_file_ioctl*/,
+       .unlocked_ioctl    = famfs_file_ioctl,
        .mmap              = famfs_file_mmap,
 
        /* Force PMD alignment for mmap */
diff --git a/fs/famfs/famfs_inode.c b/fs/famfs/famfs_inode.c
index 910a143dad30..a6c3b4574e69 100644
--- a/fs/famfs/famfs_inode.c
+++ b/fs/famfs/famfs_inode.c
@@ -300,6 +300,7 @@ static int famfs_show_options(struct seq_file *m, struct 
dentry *root)
 
 static void famfs_evict_inode(struct inode *inode)
 {
+       famfs_meta_free((struct famfs_file_meta *)inode->i_private);
        inode->i_private = NULL;
        dax_break_layout_final(inode);
        truncate_inode_pages_final(&inode->i_data);
diff --git a/fs/famfs/famfs_internal.h b/fs/famfs/famfs_internal.h
index 26f5abda96dc..b5f9c8d0349f 100644
--- a/fs/famfs/famfs_internal.h
+++ b/fs/famfs/famfs_internal.h
@@ -15,8 +15,52 @@
 #include <linux/bits.h>
 #include <linux/build_bug.h>
 
+#include <linux/famfs_ioctl.h>
+
 extern const struct file_operations famfs_file_operations;
 
+/*
+ * Internal sanity bound on a FAMFSIOC_MAP_CREATE fmap message. The ABI does
+ * not advertise a maximum (the message is self-describing); this only guards
+ * the copy-in against an unreasonable allocation. Oversize is rejected with
+ * -EFBIG.
+ */
+#define FAMFS_FMAP_MSG_MAX (4 * 1024 * 1024)
+
+struct famfs_meta_simple_ext {
+       u64 dev_index;
+       u64 ext_offset;
+       u64 ext_len;
+};
+
+struct famfs_meta_interleaved_ext {
+       u64 fie_nstrips;
+       u64 fie_chunk_size;
+       u64 fie_nbytes;
+       struct famfs_meta_simple_ext *ie_strips;
+};
+
+/*
+ * Each famfs dax file has this hanging from its inode->i_private.
+ */
+struct famfs_file_meta {
+       bool                   error;
+       enum famfs_file_type   file_type;
+       size_t                 file_size;
+       enum famfs_ioc_ext_type fm_extent_type;
+       u64                    dev_bitmap; /* referenced daxdev indices */
+       union { /* This will make code a bit more readable */
+               struct {
+                       size_t         fm_nextents;
+                       struct famfs_meta_simple_ext  *se;
+               };
+               struct {
+                       size_t         fm_niext;
+                       struct famfs_meta_interleaved_ext *ie;
+               };
+       };
+};
+
 struct famfs_mount_opts {
        umode_t mode;
 };
@@ -83,4 +127,6 @@ int famfs_devlist_alloc(struct famfs_fs_info *fsi);
 int famfs_install_daxdev(struct famfs_fs_info *fsi, struct super_block *sb,
                         u64 index, dev_t devno, const char *name);
 
+void famfs_meta_free(struct famfs_file_meta *map);
+
 #endif /* FAMFS_INTERNAL_H */
diff --git a/include/uapi/linux/famfs_ioctl.h b/include/uapi/linux/famfs_ioctl.h
new file mode 100644
index 000000000000..b4eb373c1ade
--- /dev/null
+++ b/include/uapi/linux/famfs_ioctl.h
@@ -0,0 +1,91 @@
+/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */
+/*
+ * famfs - dax file system for shared fabric-attached memory
+ *
+ * Copyright 2023-2024 Micron Technology, Inc.
+ *
+ * This file system, originally based on ramfs the dax support from xfs,
+ * is intended to allow multiple host systems to mount a common file system
+ * view of dax files that map to shared memory.
+ */
+#ifndef FAMFS_IOCTL_H
+#define FAMFS_IOCTL_H
+
+#include <linux/ioctl.h>
+#include <linux/uuid.h>
+
+#define FAMFS_KABI_VERSION 44
+
+enum famfs_file_type {
+       FAMFS_REG,
+       FAMFS_SUPERBLOCK,
+       FAMFS_LOG,
+};
+
+/*
+ * Extent type in a famfs fmap message, and of the in-core map
+ * (famfs_file_meta.fm_extent_type).
+ */
+enum famfs_ioc_ext_type {
+       FAMFS_IOC_EXT_SIMPLE,
+       FAMFS_IOC_EXT_INTERLEAVE,
+};
+
+/*
+ * The FAMFSIOC_MAP_CREATE payload is a self-describing fmap message: a
+ * struct famfs_ioc_fmap_header immediately followed by @nextents extent
+ * records. @fmap_size gives the total message length, so a reader is
+ * self-delimiting.
+ *
+ * For ext_type == FAMFS_IOC_EXT_SIMPLE the records are an array of
+ * @nextents famfs_ioc_simple_ext. For ext_type == FAMFS_IOC_EXT_INTERLEAVE
+ * each of the @nextents records is a famfs_ioc_iext header immediately
+ * followed by ie_nstrips famfs_ioc_simple_ext strip extents.
+ *
+ * This wire layout is byte-identical to the fmap carried in a fuse famfs
+ * GET_FMAP reply, so the same userspace serializer emits both.
+ *
+ * The message is self-describing (@fmap_size bounds it), so neither the extent
+ * and strip counts nor the total size are capped by this ABI. The kernel
+ * applies an internal sanity limit to the copy-in and returns -EFBIG for a
+ * message larger than it will accept.
+ */
+#define FAMFS_FMAP_VERSION 1
+
+struct famfs_ioc_simple_ext {
+       __u32 se_devindex;
+       __u32 reserved;
+       __u64 se_offset;
+       __u64 se_len;
+};
+
+struct famfs_ioc_iext {                /* interleaved (striped) extent */
+       __u32 ie_nstrips;
+       __u32 ie_chunk_size;
+       __u64 ie_nbytes;        /* total bytes mapped by this interleaved 
extent */
+       __u64 reserved;
+};
+
+struct famfs_ioc_fmap_header {
+       __u8  file_type;        /* enum famfs_file_type */
+       __u8  reserved;
+       __u16 fmap_version;     /* FAMFS_FMAP_VERSION */
+       __u32 ext_type;         /* enum famfs_ioc_ext_type */
+       __u32 nextents;
+       __u32 fmap_size;        /* total message bytes, including this header */
+       __u64 file_size;
+       __u64 reserved1;
+};
+
+#define FAMFSIOC_MAGIC 'u'
+
+/* famfs file ioctl opcodes */
+#define FAMFSIOC_NOP           _IO(FAMFSIOC_MAGIC,   0x50)
+
+/*
+ * MAP_CREATE carries the self-describing fmap message - struct
+ * famfs_ioc_fmap_header followed by the extent list (see above).
+ */
+#define FAMFSIOC_MAP_CREATE    _IOW(FAMFSIOC_MAGIC,  0x51, struct 
famfs_ioc_fmap_header)
+
+#endif /* FAMFS_IOCTL_H */
-- 
2.53.0



Reply via email to