On Mon, Aug 03, 2026 at 02:28:47AM +0000, John Groves wrote:
> From: John Groves <[email protected]>
> 
> Famfs file systems can span multiple dax devices, and daxdevs are stored
> in the daxdev_table. This adds the basic table structure, primtives and
> serialization code. Famfs file extents reference daxdevs by index, which
> is a cluster invariant maintained by user space.
> 
> We also add dax_holder_operations and a notify_failure handler, which
> is necessary to properly "open" a famfs-mode daxdev.

So you have to "mount /dev/somedaxthing0 /mnt" but you can then add more
dax devices later through some ioctl?

Can you remove daxdevs?  (I gather not...)

> Signed-off-by: John Groves <[email protected]>
> ---
>  fs/famfs/famfs_inode.c    | 234 ++++++++++++++++++++++++++++++++++++++
>  fs/famfs/famfs_internal.h |  46 ++++++++
>  2 files changed, 280 insertions(+)
> 
> diff --git a/fs/famfs/famfs_inode.c b/fs/famfs/famfs_inode.c
> index c299a90912a5..ad71e5e7a8e3 100644
> --- a/fs/famfs/famfs_inode.c
> +++ b/fs/famfs/famfs_inode.c
> @@ -75,6 +75,225 @@ static struct inode *famfs_get_inode(
>  /*
>   * famfs dax_operations (for famfs-mode dax)
>   */
> +static void famfs_set_daxdev_err(struct famfs_fs_info *fsi,
> +                              struct dax_device *dax_devp);
> +
> +static int
> +famfs_dax_notify_failure(
> +             struct dax_device *dax_dev, u64 offset,
> +             u64 len, int mf_flags)
> +{
> +     struct super_block *sb = dax_holder(dax_dev);
> +     struct famfs_fs_info *fsi = sb->s_fs_info;
> +
> +     pr_err("%s: offset=%lld len=%llu flags=%x\n", __func__,
> +            offset, len, mf_flags);
> +
> +     /*
> +      * Record the error on the specific daxdev and, near-term, shut the
> +      * mount down: famfs_set_daxdev_err() also sets fsi->deverror so
> +      * subsequent famfs operations fail. The resolver's per-daxdev
> +      * famfs_dax_err() check remains and can make this finer later.
> +      */
> +     famfs_set_daxdev_err(fsi, dax_dev);
> +
> +     return 0;
> +}
> +
> +static const struct dax_holder_operations famfs_dax_holder_ops = {
> +     .notify_failure         = famfs_dax_notify_failure,
> +};
> +
> +/*
> + * Allocate the daxdev table on first use (idempotent via cmpxchg).
> + */
> +int famfs_devlist_alloc(struct famfs_fs_info *fsi)
> +{
> +     struct famfs_dax_devlist *devlist;
> +
> +     if (fsi->dax_devlist)
> +             return 0;
> +
> +     devlist = kcalloc(1, sizeof(*devlist), GFP_KERNEL);
> +     if (!devlist)
> +             return -ENOMEM;
> +
> +     devlist->nslots = FAMFS_MAX_DAXDEVS;
> +     devlist->devlist = kcalloc(FAMFS_MAX_DAXDEVS, sizeof(struct 
> famfs_daxdev),
> +                                GFP_KERNEL);
> +     if (!devlist->devlist) {
> +             kfree(devlist);
> +             return -ENOMEM;
> +     }
> +
> +     /* If another thread allocated it first, drop ours */
> +     if (cmpxchg(&fsi->dax_devlist, NULL, devlist) != NULL) {
> +             kfree(devlist->devlist);
> +             kfree(devlist);
> +     }
> +
> +     return 0;
> +}
> +
> +/*
> + * famfs_install_daxdev() - exclusively acquire a resolved daxdev and publish
> + * it in the table at @index. Slot 0 is the mount primary; slots 1..n come 
> from
> + * the daxdev-open ioctl.
> + *
> + * Serializes with concurrent installers under devlist_sem and rechecks
> + * ->valid, so re-registering an already-installed slot is idempotent. A 
> daxdev
> + * is entered in the table only once it has been exclusively acquired via
> + * fs_dax_get() (with the super_block as the holder); on failure the
> + * dax_dev_find() reference is released and the slot is left invalid. @name 
> may
> + * be NULL (the ioctl path passes no pathname).
> + */
> +int famfs_install_daxdev(
> +             struct famfs_fs_info *fsi,
> +             struct super_block *sb,
> +             u64 index,
> +             dev_t devno,
> +             const char *name)
> +{
> +     struct famfs_daxdev *daxdev;
> +     int rc = 0;
> +
> +     if (index >= fsi->dax_devlist->nslots) {
> +             pr_debug("%s: index(%llu) >= nslots(%d)\n",
> +                    __func__, index, fsi->dax_devlist->nslots);
> +             return -EINVAL;
> +     }
> +
> +     scoped_guard(rwsem_write, &fsi->devlist_sem) {
> +             daxdev = &fsi->dax_devlist->devlist[index];
> +
> +             /* Installed already by a concurrent (or repeated) open */
> +             if (daxdev->valid)
> +                     return 0;
> +
> +             /*
> +              * A prior attempt already determined this daxdev cannot be
> +              * exclusively acquired (see the fs_dax_get() failure handling
> +              * below). Don't thrash on fs_dax_get(); fail fast.
> +              */
> +             if (daxdev->dax_err)
> +                     return -EIO;
> +
> +             daxdev->devp = dax_dev_find(devno);
> +             if (!daxdev->devp) {
> +                     pr_debug("%s: device %u:%u not found or not dax\n",
> +                             __func__, MAJOR(devno), MINOR(devno));
> +                     return -ENODEV;
> +             }
> +
> +             rc = fs_dax_get(daxdev->devp, sb, &famfs_dax_holder_ops);
> +             if (rc) {
> +                     /*
> +                      * Distinguish a lost race from a real failure. -EBUSY
> +                      * with the daxdev already held by *this* super_block
> +                      * means a concurrent acquire won and will publish the
> +                      * slot valid: not an error, and must not be cached as
> +                      * dax_err. Any other failure is permanent for this
> +                      * mount, so record dax_err to stop re-acquiring it.
> +                      */
> +                     if (!(rc == -EBUSY && dax_holder(daxdev->devp) == sb)) {
> +                             pr_debug("%s: fs_dax_get(%u:%u) failed rc=%d\n",
> +                                    __func__, MAJOR(devno), MINOR(devno), 
> rc);
> +                             daxdev->dax_err = true;
> +                     }
> +                     put_dax(daxdev->devp);
> +                     daxdev->devp = NULL;
> +                     return rc;
> +             }
> +
> +             daxdev->devno = devno;
> +             if (name) {
> +                     daxdev->name = kstrdup(name, GFP_KERNEL);
> +                     if (!daxdev->name) {
> +                             fs_put_dax(daxdev->devp, sb);
> +                             put_dax(daxdev->devp);
> +                             daxdev->devp = NULL;
> +                             return -ENOMEM;
> +                     }
> +             }
> +
> +             wmb(); /* All other fields must be visible before valid */
> +             daxdev->valid = 1;
> +     }
> +
> +     return 0;
> +}
> +
> +/*
> + * Release every daxdev in the table and free it. Detach the table under
> + * devlist_sem so a notify_failure racing teardown either runs first against
> + * the live table or observes dax_devlist == NULL and bails.
> + */
> +static void famfs_devlist_free(
> +                     struct famfs_fs_info *fsi,
> +                     struct super_block *sb)
> +{
> +     struct famfs_dax_devlist *devlist __free(kfree) = NULL;
> +     int i;
> +
> +     scoped_guard(rwsem_write, &fsi->devlist_sem) {
> +             devlist = fsi->dax_devlist;
> +             fsi->dax_devlist = NULL;
> +     }
> +
> +     if (!devlist || !devlist->devlist)
> +             return;
> +
> +     for (i = 0; i < devlist->nslots; i++) {
> +             struct famfs_daxdev *dd = &devlist->devlist[i];
> +
> +             if (!dd->valid)
> +                     continue;
> +
> +             if (dd->devp) {
> +                     if (!dd->dax_err)
> +                             fs_put_dax(dd->devp, sb);
> +                     put_dax(dd->devp);
> +             }
> +             kfree(dd->name);
> +     }
> +     kfree(devlist->devlist);
> +}
> +
> +/*
> + * Record a memory error on the daxdev matching @dax_devp. Searches the table
> + * under the write lock (which serializes against famfs_devlist_free()).
> + */
> +static void famfs_set_daxdev_err(
> +                     struct famfs_fs_info *fsi,
> +                     struct dax_device *dax_devp)
> +{
> +     int i;
> +
> +     scoped_guard(rwsem_write, &fsi->devlist_sem) {
> +             if (!fsi->dax_devlist)
> +                     return;
> +             for (i = 0; i < fsi->dax_devlist->nslots; i++) {
> +                     struct famfs_daxdev *dd = &fsi->dax_devlist->devlist[i];
> +
> +                     if (!dd->valid || dd->devp != dax_devp)
> +                             continue;
> +
> +                     dd->error = true;
> +                     /*
> +                      * Near-term policy: any daxdev memory error shuts down
> +                      * the whole mount. Finer per-daxdev handling (via
> +                      * famfs_dax_err() in the resolver) already exists and
> +                      * can supersede this later.
> +                      */
> +                     fsi->deverror = true;

Hmm, I guess this is memory where really bad stuff happens ...?

I would have thought that you'd just kill the files on that device, but
you can relax this later if desirable.

> +                     pr_err("%s: memory error on daxdev %s (%d)\n",
> +                            __func__, dd->name, i);
> +                     return;
> +             }
> +     }
> +     pr_debug("%s: memory error on unrecognized daxdev\n", __func__);
> +}
> +
>  
> /*****************************************************************************
>   * fs_context_operations
>   */
> @@ -158,6 +377,18 @@ famfs_get_tree(struct fs_context *fc)
>               famfs_fill_super(sb, fc);
>       }
>  
> +     /* Install the primary daxdev (from the mount device) at slot 0 */
> +     err = famfs_devlist_alloc(fsi);
> +     if (err)
> +             goto deactivate_out;
> +
> +     err = famfs_install_daxdev(fsi, sb, 0, daxdevno, fc->source);
> +     if (err) {
> +             pr_err("%s: failed to install primary daxdev %s\n",
> +                    __func__, fc->source);
> +             goto deactivate_out;
> +     }
> +
>       inode = famfs_get_inode(sb, NULL, S_IFDIR | fsi->mount_opts.mode, 0);
>       sb->s_root = d_make_root(inode);
>       if (!sb->s_root) {
> @@ -241,6 +472,7 @@ static int famfs_init_fs_context(struct fs_context *fc)
>       if (!fsi)
>               return -ENOMEM;
>  
> +     init_rwsem(&fsi->devlist_sem);
>       fsi->mount_opts.mode = FAMFS_DEFAULT_MODE;
>       fc->s_fs_info        = fsi;
>       fc->ops              = &famfs_context_ops;
> @@ -251,6 +483,8 @@ static void famfs_kill_sb(struct super_block *sb)
>  {
>       struct famfs_fs_info *fsi = sb->s_fs_info;
>  
> +     famfs_devlist_free(fsi, sb);
> +
>       kill_char_super(sb);
>  
>       kfree(fsi);
> diff --git a/fs/famfs/famfs_internal.h b/fs/famfs/famfs_internal.h
> index 6f481d79ed88..ebb9c499cf69 100644
> --- a/fs/famfs/famfs_internal.h
> +++ b/fs/famfs/famfs_internal.h
> @@ -11,22 +11,68 @@
>  #ifndef FAMFS_INTERNAL_H
>  #define FAMFS_INTERNAL_H
>  
> +#include <linux/rwsem.h>
> +#include <linux/bits.h>
> +#include <linux/build_bug.h>
> +
>  struct famfs_mount_opts {
>       umode_t mode;
>  };
>  
> +/*
> + * famfs_daxdev - one entry in the per-superblock daxdev table
> + *
> + * @valid:   slot is populated and the daxdev has been exclusively acquired
> + * @error:   dax reported a memory error (probably poison) via notify_failure
> + * @dax_err: fs_dax_get() failed for this daxdev
> + * @devno:   dax device dev_t
> + * @devp:    the acquired dax_device
> + * @name:    dax device path (may be NULL for ioctl-registered daxdevs)
> + */
> +struct famfs_daxdev {
> +     bool valid;
> +     bool error;
> +     bool dax_err;
> +     dev_t devno;
> +     struct dax_device *devp;
> +     char *name;
> +};
> +
> +/*
> + * The daxdev index space (and thus this table) is capped at 64 so the set of
> + * daxdev indices referenced by a file's fmap fits in a u64 bitmap.
> + */
> +#define FAMFS_MAX_DAXDEVS 64
> +static_assert(BITS_PER_TYPE(u64) >= FAMFS_MAX_DAXDEVS);
> +
> +/*
> + * famfs_dax_devlist - the per-superblock table of famfs_daxdev's. Slot 0 is
> + * the primary daxdev supplied at mount; slots 1..n are registered via ioctl.
> + */
> +struct famfs_dax_devlist {
> +     int nslots;
> +     struct famfs_daxdev *devlist;
> +};
> +
>  /**
>   * @famfs_fs_info
>   *
>   * @mount_opts:  The mount options
>   * @deverror:    True if the dax device has called our notify_failure entry
>   *               point, or if other "shutdown" conditions exist
> + * @dax_devlist: Table of backing daxdevs (slot 0 is the mount primary)
> + * @devlist_sem: Serializes installs into, and teardown of, @dax_devlist
>   */
>  struct famfs_fs_info {
>       struct famfs_mount_opts   mount_opts;
>       bool                      deverror;
> +     struct famfs_dax_devlist *dax_devlist;
> +     struct rw_semaphore       devlist_sem;
>  };
>  
>  int lookup_daxdev(const char *pathname, dev_t *devno);
> +int famfs_devlist_alloc(struct famfs_fs_info *fsi);
> +int famfs_install_daxdev(struct famfs_fs_info *fsi, struct super_block *sb,
> +                      u64 index, dev_t devno, const char *name);
>  
>  #endif /* FAMFS_INTERNAL_H */
> -- 
> 2.53.0
> 
> 
> 

Reply via email to