On Thu, Sep 24, 2026 at 12:15 PM Eugenio Pérez <[email protected]> wrote:
>
> This header will be used by the Vhost library.
>
> Signed-off-by: Eugenio Pérez <[email protected]>
> ---
>  kernel/linux/uapi/linux/vduse.h | 115 ++++++++++++++++++++++++++++++--
>  kernel/linux/uapi/linux/vfio.h  |  91 ++++++++++++++++++++++++-
>  kernel/linux/uapi/version       |   2 +-
>  3 files changed, 199 insertions(+), 9 deletions(-)
>
> diff --git a/kernel/linux/uapi/linux/vduse.h b/kernel/linux/uapi/linux/vduse.h
> index f46269af349a..bab47129db63 100644
> --- a/kernel/linux/uapi/linux/vduse.h
> +++ b/kernel/linux/uapi/linux/vduse.h
> @@ -10,6 +10,16 @@
>
>  #define VDUSE_API_VERSION      0
>
> +/* VQ groups and ASID support */
> +
> +#define VDUSE_API_VERSION_1    1
> +
> +/* The VDUSE instance expects a request for vq ready */
> +#define VDUSE_F_QUEUE_READY    0
> +
> +/* The VDUSE instance expects a request for suspend */
> +#define VDUSE_F_SUSPEND                1
> +
>  /*
>   * Get the version of VDUSE API that kernel supported (VDUSE_API_VERSION).
>   * This is used for future extension.
> @@ -27,6 +37,8 @@
>   * @features: virtio features
>   * @vq_num: the number of virtqueues
>   * @vq_align: the allocation alignment of virtqueue's metadata
> + * @ngroups: number of vq groups that VDUSE device declares
> + * @nas: number of address spaces that VDUSE device declares
>   * @reserved: for future use, needs to be initialized to zero
>   * @config_size: the size of the configuration space
>   * @config: the buffer of the configuration space
> @@ -41,7 +53,9 @@ struct vduse_dev_config {
>         __u64 features;
>         __u32 vq_num;
>         __u32 vq_align;
> -       __u32 reserved[13];
> +       __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> +       __u32 nas; /* if VDUSE_API_VERSION >= 1 */
> +       __u32 reserved[11];
>         __u32 config_size;
>         __u8 config[];
>  };
> @@ -55,6 +69,12 @@ struct vduse_dev_config {
>   */
>  #define VDUSE_DESTROY_DEV      _IOW(VDUSE_BASE, 0x03, char[VDUSE_NAME_MAX])
>
> +/* Get the VDUSE supported features */
> +#define VDUSE_GET_FEATURES     _IOR(VDUSE_BASE, 0x04, __u64)
> +
> +/* Set the VDUSE features */
> +#define VDUSE_SET_FEATURES     _IOW(VDUSE_BASE, 0x05, __u64)
> +
>  /* The ioctls for VDUSE device (/dev/vduse/$NAME) */
>
>  /**
> @@ -118,14 +138,18 @@ struct vduse_config_data {
>   * struct vduse_vq_config - basic configuration of a virtqueue
>   * @index: virtqueue index
>   * @max_size: the max size of virtqueue
> - * @reserved: for future use, needs to be initialized to zero
> + * @reserved1: for future use, needs to be initialized to zero
> + * @group: virtqueue group
> + * @reserved2: for future use, needs to be initialized to zero
>   *
>   * Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
>   */
>  struct vduse_vq_config {
>         __u32 index;
>         __u16 max_size;
> -       __u16 reserved[13];
> +       __u16 reserved1;
> +       __u32 group;
> +       __u16 reserved2[10];
>  };
>
>  /*
> @@ -156,6 +180,16 @@ struct vduse_vq_state_packed {
>         __u16 last_used_idx;
>  };
>
> +/**
> + * struct vduse_vq_group_asid - virtqueue group ASID
> + * @group: Index of the virtqueue group
> + * @asid: Address space ID of the group
> + */
> +struct vduse_vq_group_asid {
> +       __u32 group;
> +       __u32 asid;
> +};
> +
>  /**
>   * struct vduse_vq_info - information of a virtqueue
>   * @index: virtqueue index
> @@ -215,6 +249,7 @@ struct vduse_vq_eventfd {
>   * @uaddr: start address of userspace memory, it must be aligned to page size
>   * @iova: start of the IOVA region
>   * @size: size of the IOVA region
> + * @asid: Address space ID of the IOVA region
>   * @reserved: for future use, needs to be initialized to zero
>   *
>   * Structure used by VDUSE_IOTLB_REG_UMEM and VDUSE_IOTLB_DEREG_UMEM
> @@ -224,7 +259,8 @@ struct vduse_iova_umem {
>         __u64 uaddr;
>         __u64 iova;
>         __u64 size;
> -       __u64 reserved[3];
> +       __u32 asid;
> +       __u32 reserved[5];
>  };
>
>  /* Register userspace memory for IOVA regions */
> @@ -237,7 +273,8 @@ struct vduse_iova_umem {
>   * struct vduse_iova_info - information of one IOVA region
>   * @start: start of the IOVA region
>   * @last: last of the IOVA region
> - * @capability: capability of the IOVA regsion
> + * @capability: capability of the IOVA region
> + * @asid: Address space ID of the IOVA region, only if device API version >= 
> 1
>   * @reserved: for future use, needs to be initialized to zero
>   *
>   * Structure used by VDUSE_IOTLB_GET_INFO ioctl to get information of
> @@ -248,7 +285,8 @@ struct vduse_iova_info {
>         __u64 last;
>  #define VDUSE_IOVA_CAP_UMEM (1 << 0)
>         __u64 capability;
> -       __u64 reserved[3];
> +       __u32 asid; /* Only if device API version >= 1 */
> +       __u32 reserved[5];
>  };
>
>  /*
> @@ -257,6 +295,32 @@ struct vduse_iova_info {
>   */
>  #define VDUSE_IOTLB_GET_INFO   _IOWR(VDUSE_BASE, 0x1a, struct 
> vduse_iova_info)
>
> +/**
> + * struct vduse_iotlb_entry_v2 - entry of IOTLB to describe one IOVA region
> + *
> + * @v1: the original vduse_iotlb_entry
> + * @asid: address space ID of the IOVA region
> + * @reserved: for future use, needs to be initialized to zero
> + *
> + * Structure used by VDUSE_IOTLB_GET_FD2 ioctl to find an overlapped IOVA 
> region.
> + */
> +struct vduse_iotlb_entry_v2 {
> +       __u64 offset;
> +       __u64 start;
> +       __u64 last;
> +       __u8 perm;
> +       __u8 padding[7];
> +       __u32 asid;
> +       __u32 reserved[11];
> +};
> +
> +/*
> + * Same as VDUSE_IOTLB_GET_FD but with vduse_iotlb_entry_v2 argument that
> + * support extra fields.
> + */
> +#define VDUSE_IOTLB_GET_FD2    _IOWR(VDUSE_BASE, 0x1b, struct 
> vduse_iotlb_entry_v2)
> +
> +
>  /* The control messages definition for read(2)/write(2) on /dev/vduse/$NAME 
> */
>
>  /**
> @@ -265,11 +329,16 @@ struct vduse_iova_info {
>   * @VDUSE_SET_STATUS: set the device status
>   * @VDUSE_UPDATE_IOTLB: Notify userspace to update the memory mapping for
>   *                      specified IOVA range via VDUSE_IOTLB_GET_FD ioctl
> + * @VDUSE_SET_VQ_GROUP_ASID: Notify userspace to update the address space of 
> a
> + *                           virtqueue group.
>   */
>  enum vduse_req_type {
>         VDUSE_GET_VQ_STATE,
>         VDUSE_SET_STATUS,
>         VDUSE_UPDATE_IOTLB,
> +       VDUSE_SET_VQ_GROUP_ASID,
> +       VDUSE_SET_VQ_READY,
> +       VDUSE_SUSPEND,
>  };
>
>  /**
> @@ -304,6 +373,28 @@ struct vduse_iova_range {
>         __u64 last;
>  };
>
> +/**
> + * struct vduse_iova_range_v2 - IOVA range [start, last] if API_VERSION >= 1
> + * @start: start of the IOVA range
> + * @last: last of the IOVA range
> + * @asid: address space ID of the IOVA range
> + */
> +struct vduse_iova_range_v2 {
> +       __u64 start;
> +       __u64 last;
> +       __u32 asid;
> +       __u32 padding;
> +};
> +
> +/**
> + * struct vduse_vq_ready - Virtqueue ready request message
> + * @num: Virtqueue number
> + */
> +struct vduse_vq_ready {
> +       __u32 num;
> +       __u32 ready;
> +};
> +
>  /**
>   * struct vduse_dev_request - control request
>   * @type: request type
> @@ -312,6 +403,9 @@ struct vduse_iova_range {
>   * @vq_state: virtqueue state, only index field is available
>   * @s: device status
>   * @iova: IOVA range for updating
> + * @iova_v2: IOVA range for updating if API_VERSION >= 1
> + * @vq_group_asid: ASID of a virtqueue group
> + * @vq_ready: Virtqueue ready request
>   * @padding: padding
>   *
>   * Structure used by read(2) on /dev/vduse/$NAME.
> @@ -324,6 +418,15 @@ struct vduse_dev_request {
>                 struct vduse_vq_state vq_state;
>                 struct vduse_dev_status s;
>                 struct vduse_iova_range iova;
> +               /* Following members but padding exist only if vduse api
> +                * version >= 1
> +                */
> +               struct vduse_iova_range_v2 iova_v2;
> +               struct vduse_vq_group_asid vq_group_asid;
> +
> +               /* Only if VDUSE_F_QUEUE_READY is negotiated */
> +               struct vduse_vq_ready vq_ready;
> +
>                 __u32 padding[32];
>         };
>  };
> diff --git a/kernel/linux/uapi/linux/vfio.h b/kernel/linux/uapi/linux/vfio.h
> index 79bf8c0cc5e4..c85dcbfe302b 100644
> --- a/kernel/linux/uapi/linux/vfio.h
> +++ b/kernel/linux/uapi/linux/vfio.h
> @@ -14,6 +14,7 @@
>
>  #include <linux/types.h>
>  #include <linux/ioctl.h>
> +#include <linux/stddef.h>
>
>  #define VFIO_API_VERSION       0
>
> @@ -140,7 +141,7 @@ struct vfio_info_cap_header {
>   *
>   * Retrieve information about the group.  Fills in provided
>   * struct vfio_group_info.  Caller sets argsz.
> - * Return: 0 on succes, -errno on failure.
> + * Return: 0 on success, -errno on failure.
>   * Availability: Always
>   */
>  struct vfio_group_status {
> @@ -905,10 +906,12 @@ struct vfio_device_feature {
>   * VFIO_DEVICE_BIND_IOMMUFD - _IOR(VFIO_TYPE, VFIO_BASE + 18,
>   *                                struct vfio_device_bind_iommufd)
>   * @argsz:      User filled size of this data.
> - * @flags:      Must be 0.
> + * @flags:      Must be 0 or a bit flags of VFIO_DEVICE_BIND_*
>   * @iommufd:    iommufd to bind.
>   * @out_devid:  The device id generated by this bind. devid is a handle for
>   *              this device/iommufd bond and can be used in IOMMUFD commands.
> + * @token_uuid_ptr: Valid if VFIO_DEVICE_BIND_FLAG_TOKEN. Points to a 16 byte
> + *                  UUID in the same format as 
> VFIO_DEVICE_FEATURE_PCI_VF_TOKEN.
>   *
>   * Bind a vfio_device to the specified iommufd.
>   *
> @@ -917,13 +920,21 @@ struct vfio_device_feature {
>   *
>   * Unbind is automatically conducted when device fd is closed.
>   *
> + * A token is sometimes required to open the device, unless this is known to 
> be
> + * needed VFIO_DEVICE_BIND_FLAG_TOKEN should not be set and token_uuid_ptr is
> + * ignored. The only case today is a PF/VF relationship where the VF bind 
> must
> + * be provided the same token as VFIO_DEVICE_FEATURE_PCI_VF_TOKEN provided to
> + * the PF.
> + *
>   * Return: 0 on success, -errno on failure.
>   */
>  struct vfio_device_bind_iommufd {
>         __u32           argsz;
>         __u32           flags;
> +#define VFIO_DEVICE_BIND_FLAG_TOKEN (1 << 0)
>         __s32           iommufd;
>         __u32           out_devid;
> +       __aligned_u64   token_uuid_ptr;
>  };
>
>  #define VFIO_DEVICE_BIND_IOMMUFD       _IO(VFIO_TYPE, VFIO_BASE + 18)
> @@ -953,6 +964,10 @@ struct vfio_device_bind_iommufd {
>   * hwpt corresponding to the given pt_id.
>   *
>   * Return: 0 on success, -errno on failure.
> + *
> + * When a device is resetting, -EBUSY will be returned to reject any 
> concurrent
> + * attachment to the resetting device itself or any sibling device in the 
> IOMMU
> + * group having the resetting device.
>   */
>  struct vfio_device_attach_iommufd_pt {
>         __u32   argsz;
> @@ -1251,6 +1266,19 @@ enum vfio_device_mig_state {
>   * The initial_bytes field indicates the amount of initial precopy
>   * data available from the device. This field should have a non-zero initial
>   * value and decrease as migration data is read from the device.
> + * The presence of the VFIO_PRECOPY_INFO_REINIT output flag indicates
> + * that new initial data is present on the stream.
> + * The new initial data may result, for example, from device reconfiguration
> + * during migration that requires additional initialization data.
> + * In that case initial_bytes may report a non-zero value irrespective of
> + * any previously reported values, which progresses towards zero as precopy
> + * data is read from the data stream. dirty_bytes is also reset
> + * to zero and represents the state change of the device relative to the new
> + * initial_bytes.
> + * VFIO_PRECOPY_INFO_REINIT can be reported only after userspace opts in to
> + * VFIO_DEVICE_FEATURE_MIG_PRECOPY_INFOv2. Without this opt-in, the flags 
> field
> + * of struct vfio_precopy_info is reserved for bug-compatibility reasons.
> + *
>   * It is recommended to leave PRE_COPY for STOP_COPY only after this field
>   * reaches zero. Leaving PRE_COPY earlier might make things slower.
>   *
> @@ -1286,6 +1314,7 @@ enum vfio_device_mig_state {
>  struct vfio_precopy_info {
>         __u32 argsz;
>         __u32 flags;
> +#define VFIO_PRECOPY_INFO_REINIT (1 << 0) /* output - new initial data is 
> present */
>         __aligned_u64 initial_bytes;
>         __aligned_u64 dirty_bytes;
>  };
> @@ -1468,6 +1497,64 @@ struct vfio_device_feature_bus_master {
>  };
>  #define VFIO_DEVICE_FEATURE_BUS_MASTER 10
>
> +/**
> + * Upon VFIO_DEVICE_FEATURE_GET create a dma_buf fd for the
> + * regions selected.
> + *
> + * open_flags are the typical flags passed to open(2), eg O_RDWR, O_CLOEXEC,
> + * etc. offset/length specify a slice of the region to create the dmabuf 
> from.
> + * nr_ranges is the total number of (P2P DMA) ranges that comprise the 
> dmabuf.
> + *
> + * flags should be 0.
> + *
> + * Return: The fd number on success, -1 and errno is set on failure.
> + */
> +#define VFIO_DEVICE_FEATURE_DMA_BUF 11
> +
> +struct vfio_region_dma_range {
> +       __u64 offset;
> +       __u64 length;
> +};
> +
> +struct vfio_device_feature_dma_buf {
> +       __u32   region_index;
> +       __u32   open_flags;
> +       __u32   flags;
> +       __u32   nr_ranges;
> +       struct vfio_region_dma_range dma_ranges[] __counted_by(nr_ranges);

Not all gcc versions of CI like the __counted_by argument. Should I
not import this header?

> +};
> +
> +/*
> + * Enables the migration precopy_info_v2 behaviour.
> + *
> + * VFIO_DEVICE_FEATURE_MIG_PRECOPY_INFOv2.
> + *
> + * On SET, enables the v2 pre_copy_info behaviour, where the
> + * vfio_precopy_info.flags is a valid output field.
> + */
> +#define VFIO_DEVICE_FEATURE_MIG_PRECOPY_INFOv2  12
> +
> +/**
> + * VFIO_DEVICE_FEATURE_ZPCI_ERROR feature provides PCI error information to
> + * userspace for vfio-pci devices on s390. On s390, PCI error recovery
> + * involves platform firmware and notification to operating systems is done
> + * by architecture specific mechanism. Exposing this information to
> + * userspace allows it to take appropriate actions to handle an
> + * error on the device.
> + *
> + * Userspace provides an opaque buffer of fixed length, and the kernel
> + * fills it with the zpci_ccdf_err data structure. The length of
> + * zpci_ccdf_err is provided to userspace via the
> + * VFIO_DEVICE_INFO_CAP_ZPCI_BASE capability.
> + *
> + * The ioctl returns -ENOMSG if there are no pending PCI errors.
> + */
> +struct vfio_device_feature_zpci_err {
> +       __aligned_u64 data;
> +};
> +
> +#define VFIO_DEVICE_FEATURE_ZPCI_ERROR 13
> +
>  /* -------- API for Type1 VFIO IOMMU -------- */
>
>  /**
> diff --git a/kernel/linux/uapi/version b/kernel/linux/uapi/version
> index 966a9983019b..4e528e220b99 100644
> --- a/kernel/linux/uapi/version
> +++ b/kernel/linux/uapi/version
> @@ -1 +1 @@
> -v6.16
> +v7.3-rc3
> --
> 2.55.0
>

Reply via email to