On Thu, Sep 24, 2026 at 12:15 PM Eugenio Pérez <[email protected]> wrote:
>
> This header will be used by the Vhost library.
>
> Signed-off-by: Eugenio Pérez <[email protected]>
> ---
> kernel/linux/uapi/linux/vduse.h | 115 ++++++++++++++++++++++++++++++--
> kernel/linux/uapi/linux/vfio.h | 91 ++++++++++++++++++++++++-
> kernel/linux/uapi/version | 2 +-
> 3 files changed, 199 insertions(+), 9 deletions(-)
>
> diff --git a/kernel/linux/uapi/linux/vduse.h b/kernel/linux/uapi/linux/vduse.h
> index f46269af349a..bab47129db63 100644
> --- a/kernel/linux/uapi/linux/vduse.h
> +++ b/kernel/linux/uapi/linux/vduse.h
> @@ -10,6 +10,16 @@
>
> #define VDUSE_API_VERSION 0
>
> +/* VQ groups and ASID support */
> +
> +#define VDUSE_API_VERSION_1 1
> +
> +/* The VDUSE instance expects a request for vq ready */
> +#define VDUSE_F_QUEUE_READY 0
> +
> +/* The VDUSE instance expects a request for suspend */
> +#define VDUSE_F_SUSPEND 1
> +
> /*
> * Get the version of VDUSE API that kernel supported (VDUSE_API_VERSION).
> * This is used for future extension.
> @@ -27,6 +37,8 @@
> * @features: virtio features
> * @vq_num: the number of virtqueues
> * @vq_align: the allocation alignment of virtqueue's metadata
> + * @ngroups: number of vq groups that VDUSE device declares
> + * @nas: number of address spaces that VDUSE device declares
> * @reserved: for future use, needs to be initialized to zero
> * @config_size: the size of the configuration space
> * @config: the buffer of the configuration space
> @@ -41,7 +53,9 @@ struct vduse_dev_config {
> __u64 features;
> __u32 vq_num;
> __u32 vq_align;
> - __u32 reserved[13];
> + __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> + __u32 nas; /* if VDUSE_API_VERSION >= 1 */
> + __u32 reserved[11];
> __u32 config_size;
> __u8 config[];
> };
> @@ -55,6 +69,12 @@ struct vduse_dev_config {
> */
> #define VDUSE_DESTROY_DEV _IOW(VDUSE_BASE, 0x03, char[VDUSE_NAME_MAX])
>
> +/* Get the VDUSE supported features */
> +#define VDUSE_GET_FEATURES _IOR(VDUSE_BASE, 0x04, __u64)
> +
> +/* Set the VDUSE features */
> +#define VDUSE_SET_FEATURES _IOW(VDUSE_BASE, 0x05, __u64)
> +
> /* The ioctls for VDUSE device (/dev/vduse/$NAME) */
>
> /**
> @@ -118,14 +138,18 @@ struct vduse_config_data {
> * struct vduse_vq_config - basic configuration of a virtqueue
> * @index: virtqueue index
> * @max_size: the max size of virtqueue
> - * @reserved: for future use, needs to be initialized to zero
> + * @reserved1: for future use, needs to be initialized to zero
> + * @group: virtqueue group
> + * @reserved2: for future use, needs to be initialized to zero
> *
> * Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
> */
> struct vduse_vq_config {
> __u32 index;
> __u16 max_size;
> - __u16 reserved[13];
> + __u16 reserved1;
> + __u32 group;
> + __u16 reserved2[10];
> };
>
> /*
> @@ -156,6 +180,16 @@ struct vduse_vq_state_packed {
> __u16 last_used_idx;
> };
>
> +/**
> + * struct vduse_vq_group_asid - virtqueue group ASID
> + * @group: Index of the virtqueue group
> + * @asid: Address space ID of the group
> + */
> +struct vduse_vq_group_asid {
> + __u32 group;
> + __u32 asid;
> +};
> +
> /**
> * struct vduse_vq_info - information of a virtqueue
> * @index: virtqueue index
> @@ -215,6 +249,7 @@ struct vduse_vq_eventfd {
> * @uaddr: start address of userspace memory, it must be aligned to page size
> * @iova: start of the IOVA region
> * @size: size of the IOVA region
> + * @asid: Address space ID of the IOVA region
> * @reserved: for future use, needs to be initialized to zero
> *
> * Structure used by VDUSE_IOTLB_REG_UMEM and VDUSE_IOTLB_DEREG_UMEM
> @@ -224,7 +259,8 @@ struct vduse_iova_umem {
> __u64 uaddr;
> __u64 iova;
> __u64 size;
> - __u64 reserved[3];
> + __u32 asid;
> + __u32 reserved[5];
> };
>
> /* Register userspace memory for IOVA regions */
> @@ -237,7 +273,8 @@ struct vduse_iova_umem {
> * struct vduse_iova_info - information of one IOVA region
> * @start: start of the IOVA region
> * @last: last of the IOVA region
> - * @capability: capability of the IOVA regsion
> + * @capability: capability of the IOVA region
> + * @asid: Address space ID of the IOVA region, only if device API version >=
> 1
> * @reserved: for future use, needs to be initialized to zero
> *
> * Structure used by VDUSE_IOTLB_GET_INFO ioctl to get information of
> @@ -248,7 +285,8 @@ struct vduse_iova_info {
> __u64 last;
> #define VDUSE_IOVA_CAP_UMEM (1 << 0)
> __u64 capability;
> - __u64 reserved[3];
> + __u32 asid; /* Only if device API version >= 1 */
> + __u32 reserved[5];
> };
>
> /*
> @@ -257,6 +295,32 @@ struct vduse_iova_info {
> */
> #define VDUSE_IOTLB_GET_INFO _IOWR(VDUSE_BASE, 0x1a, struct
> vduse_iova_info)
>
> +/**
> + * struct vduse_iotlb_entry_v2 - entry of IOTLB to describe one IOVA region
> + *
> + * @v1: the original vduse_iotlb_entry
> + * @asid: address space ID of the IOVA region
> + * @reserved: for future use, needs to be initialized to zero
> + *
> + * Structure used by VDUSE_IOTLB_GET_FD2 ioctl to find an overlapped IOVA
> region.
> + */
> +struct vduse_iotlb_entry_v2 {
> + __u64 offset;
> + __u64 start;
> + __u64 last;
> + __u8 perm;
> + __u8 padding[7];
> + __u32 asid;
> + __u32 reserved[11];
> +};
> +
> +/*
> + * Same as VDUSE_IOTLB_GET_FD but with vduse_iotlb_entry_v2 argument that
> + * support extra fields.
> + */
> +#define VDUSE_IOTLB_GET_FD2 _IOWR(VDUSE_BASE, 0x1b, struct
> vduse_iotlb_entry_v2)
> +
> +
> /* The control messages definition for read(2)/write(2) on /dev/vduse/$NAME
> */
>
> /**
> @@ -265,11 +329,16 @@ struct vduse_iova_info {
> * @VDUSE_SET_STATUS: set the device status
> * @VDUSE_UPDATE_IOTLB: Notify userspace to update the memory mapping for
> * specified IOVA range via VDUSE_IOTLB_GET_FD ioctl
> + * @VDUSE_SET_VQ_GROUP_ASID: Notify userspace to update the address space of
> a
> + * virtqueue group.
> */
> enum vduse_req_type {
> VDUSE_GET_VQ_STATE,
> VDUSE_SET_STATUS,
> VDUSE_UPDATE_IOTLB,
> + VDUSE_SET_VQ_GROUP_ASID,
> + VDUSE_SET_VQ_READY,
> + VDUSE_SUSPEND,
> };
>
> /**
> @@ -304,6 +373,28 @@ struct vduse_iova_range {
> __u64 last;
> };
>
> +/**
> + * struct vduse_iova_range_v2 - IOVA range [start, last] if API_VERSION >= 1
> + * @start: start of the IOVA range
> + * @last: last of the IOVA range
> + * @asid: address space ID of the IOVA range
> + */
> +struct vduse_iova_range_v2 {
> + __u64 start;
> + __u64 last;
> + __u32 asid;
> + __u32 padding;
> +};
> +
> +/**
> + * struct vduse_vq_ready - Virtqueue ready request message
> + * @num: Virtqueue number
> + */
> +struct vduse_vq_ready {
> + __u32 num;
> + __u32 ready;
> +};
> +
> /**
> * struct vduse_dev_request - control request
> * @type: request type
> @@ -312,6 +403,9 @@ struct vduse_iova_range {
> * @vq_state: virtqueue state, only index field is available
> * @s: device status
> * @iova: IOVA range for updating
> + * @iova_v2: IOVA range for updating if API_VERSION >= 1
> + * @vq_group_asid: ASID of a virtqueue group
> + * @vq_ready: Virtqueue ready request
> * @padding: padding
> *
> * Structure used by read(2) on /dev/vduse/$NAME.
> @@ -324,6 +418,15 @@ struct vduse_dev_request {
> struct vduse_vq_state vq_state;
> struct vduse_dev_status s;
> struct vduse_iova_range iova;
> + /* Following members but padding exist only if vduse api
> + * version >= 1
> + */
> + struct vduse_iova_range_v2 iova_v2;
> + struct vduse_vq_group_asid vq_group_asid;
> +
> + /* Only if VDUSE_F_QUEUE_READY is negotiated */
> + struct vduse_vq_ready vq_ready;
> +
> __u32 padding[32];
> };
> };
> diff --git a/kernel/linux/uapi/linux/vfio.h b/kernel/linux/uapi/linux/vfio.h
> index 79bf8c0cc5e4..c85dcbfe302b 100644
> --- a/kernel/linux/uapi/linux/vfio.h
> +++ b/kernel/linux/uapi/linux/vfio.h
> @@ -14,6 +14,7 @@
>
> #include <linux/types.h>
> #include <linux/ioctl.h>
> +#include <linux/stddef.h>
>
> #define VFIO_API_VERSION 0
>
> @@ -140,7 +141,7 @@ struct vfio_info_cap_header {
> *
> * Retrieve information about the group. Fills in provided
> * struct vfio_group_info. Caller sets argsz.
> - * Return: 0 on succes, -errno on failure.
> + * Return: 0 on success, -errno on failure.
> * Availability: Always
> */
> struct vfio_group_status {
> @@ -905,10 +906,12 @@ struct vfio_device_feature {
> * VFIO_DEVICE_BIND_IOMMUFD - _IOR(VFIO_TYPE, VFIO_BASE + 18,
> * struct vfio_device_bind_iommufd)
> * @argsz: User filled size of this data.
> - * @flags: Must be 0.
> + * @flags: Must be 0 or a bit flags of VFIO_DEVICE_BIND_*
> * @iommufd: iommufd to bind.
> * @out_devid: The device id generated by this bind. devid is a handle for
> * this device/iommufd bond and can be used in IOMMUFD commands.
> + * @token_uuid_ptr: Valid if VFIO_DEVICE_BIND_FLAG_TOKEN. Points to a 16 byte
> + * UUID in the same format as
> VFIO_DEVICE_FEATURE_PCI_VF_TOKEN.
> *
> * Bind a vfio_device to the specified iommufd.
> *
> @@ -917,13 +920,21 @@ struct vfio_device_feature {
> *
> * Unbind is automatically conducted when device fd is closed.
> *
> + * A token is sometimes required to open the device, unless this is known to
> be
> + * needed VFIO_DEVICE_BIND_FLAG_TOKEN should not be set and token_uuid_ptr is
> + * ignored. The only case today is a PF/VF relationship where the VF bind
> must
> + * be provided the same token as VFIO_DEVICE_FEATURE_PCI_VF_TOKEN provided to
> + * the PF.
> + *
> * Return: 0 on success, -errno on failure.
> */
> struct vfio_device_bind_iommufd {
> __u32 argsz;
> __u32 flags;
> +#define VFIO_DEVICE_BIND_FLAG_TOKEN (1 << 0)
> __s32 iommufd;
> __u32 out_devid;
> + __aligned_u64 token_uuid_ptr;
> };
>
> #define VFIO_DEVICE_BIND_IOMMUFD _IO(VFIO_TYPE, VFIO_BASE + 18)
> @@ -953,6 +964,10 @@ struct vfio_device_bind_iommufd {
> * hwpt corresponding to the given pt_id.
> *
> * Return: 0 on success, -errno on failure.
> + *
> + * When a device is resetting, -EBUSY will be returned to reject any
> concurrent
> + * attachment to the resetting device itself or any sibling device in the
> IOMMU
> + * group having the resetting device.
> */
> struct vfio_device_attach_iommufd_pt {
> __u32 argsz;
> @@ -1251,6 +1266,19 @@ enum vfio_device_mig_state {
> * The initial_bytes field indicates the amount of initial precopy
> * data available from the device. This field should have a non-zero initial
> * value and decrease as migration data is read from the device.
> + * The presence of the VFIO_PRECOPY_INFO_REINIT output flag indicates
> + * that new initial data is present on the stream.
> + * The new initial data may result, for example, from device reconfiguration
> + * during migration that requires additional initialization data.
> + * In that case initial_bytes may report a non-zero value irrespective of
> + * any previously reported values, which progresses towards zero as precopy
> + * data is read from the data stream. dirty_bytes is also reset
> + * to zero and represents the state change of the device relative to the new
> + * initial_bytes.
> + * VFIO_PRECOPY_INFO_REINIT can be reported only after userspace opts in to
> + * VFIO_DEVICE_FEATURE_MIG_PRECOPY_INFOv2. Without this opt-in, the flags
> field
> + * of struct vfio_precopy_info is reserved for bug-compatibility reasons.
> + *
> * It is recommended to leave PRE_COPY for STOP_COPY only after this field
> * reaches zero. Leaving PRE_COPY earlier might make things slower.
> *
> @@ -1286,6 +1314,7 @@ enum vfio_device_mig_state {
> struct vfio_precopy_info {
> __u32 argsz;
> __u32 flags;
> +#define VFIO_PRECOPY_INFO_REINIT (1 << 0) /* output - new initial data is
> present */
> __aligned_u64 initial_bytes;
> __aligned_u64 dirty_bytes;
> };
> @@ -1468,6 +1497,64 @@ struct vfio_device_feature_bus_master {
> };
> #define VFIO_DEVICE_FEATURE_BUS_MASTER 10
>
> +/**
> + * Upon VFIO_DEVICE_FEATURE_GET create a dma_buf fd for the
> + * regions selected.
> + *
> + * open_flags are the typical flags passed to open(2), eg O_RDWR, O_CLOEXEC,
> + * etc. offset/length specify a slice of the region to create the dmabuf
> from.
> + * nr_ranges is the total number of (P2P DMA) ranges that comprise the
> dmabuf.
> + *
> + * flags should be 0.
> + *
> + * Return: The fd number on success, -1 and errno is set on failure.
> + */
> +#define VFIO_DEVICE_FEATURE_DMA_BUF 11
> +
> +struct vfio_region_dma_range {
> + __u64 offset;
> + __u64 length;
> +};
> +
> +struct vfio_device_feature_dma_buf {
> + __u32 region_index;
> + __u32 open_flags;
> + __u32 flags;
> + __u32 nr_ranges;
> + struct vfio_region_dma_range dma_ranges[] __counted_by(nr_ranges);
Not all gcc versions of CI like the __counted_by argument. Should I
not import this header?
> +};
> +
> +/*
> + * Enables the migration precopy_info_v2 behaviour.
> + *
> + * VFIO_DEVICE_FEATURE_MIG_PRECOPY_INFOv2.
> + *
> + * On SET, enables the v2 pre_copy_info behaviour, where the
> + * vfio_precopy_info.flags is a valid output field.
> + */
> +#define VFIO_DEVICE_FEATURE_MIG_PRECOPY_INFOv2 12
> +
> +/**
> + * VFIO_DEVICE_FEATURE_ZPCI_ERROR feature provides PCI error information to
> + * userspace for vfio-pci devices on s390. On s390, PCI error recovery
> + * involves platform firmware and notification to operating systems is done
> + * by architecture specific mechanism. Exposing this information to
> + * userspace allows it to take appropriate actions to handle an
> + * error on the device.
> + *
> + * Userspace provides an opaque buffer of fixed length, and the kernel
> + * fills it with the zpci_ccdf_err data structure. The length of
> + * zpci_ccdf_err is provided to userspace via the
> + * VFIO_DEVICE_INFO_CAP_ZPCI_BASE capability.
> + *
> + * The ioctl returns -ENOMSG if there are no pending PCI errors.
> + */
> +struct vfio_device_feature_zpci_err {
> + __aligned_u64 data;
> +};
> +
> +#define VFIO_DEVICE_FEATURE_ZPCI_ERROR 13
> +
> /* -------- API for Type1 VFIO IOMMU -------- */
>
> /**
> diff --git a/kernel/linux/uapi/version b/kernel/linux/uapi/version
> index 966a9983019b..4e528e220b99 100644
> --- a/kernel/linux/uapi/version
> +++ b/kernel/linux/uapi/version
> @@ -1 +1 @@
> -v6.16
> +v7.3-rc3
> --
> 2.55.0
>