vfio: Enable cdev noiommu mode under iommufd

Now that devices under noiommu mode can bind with IOMMUFD and perform
IOAS operations, lift restrictions on cdev from VFIO side.
Use cases are documented in Documentation/driver-api/vfio.rst

Link: https://patch.msgid.link/r/fdac3395ecc52f87b722b54a055ced66733dc48f.1783360051.git.jacob.pan@linux.microsoft.com
Reviewed-by: Kevin Tian <kevin.tian@intel.com>
Reviewed-by: Pranjal Shrivastava <praan@google.com>
Reviewed-by: Yi Liu <yi.l.liu@intel.com>
Reviewed-by: Alex Williamson <alex@shazbot.org>
Signed-off-by: Jacob Pan <jacob.pan@linux.microsoft.com>
Fixes: 2c6cf6ab1564 ("iommufd: Allow binding to a noiommu device")
Signed-off-by: Jason Gunthorpe <jgg@nvidia.com>
This commit is contained in:
Jacob Pan
2026-07-14 12:05:01 -03:00
committed by Jason Gunthorpe
parent 60decab325
commit e8efdf02d3
6 changed files with 56 additions and 22 deletions
+4 -3
View File
@@ -22,8 +22,7 @@ config VFIO_DEVICE_CDEV
The VFIO device cdev is another way for userspace to get device
access. Userspace gets device fd by opening device cdev under
/dev/vfio/devices/vfioX, and then bind the device fd with an iommufd
to set up secure DMA context for device access. This interface does
not support noiommu.
to set up secure DMA context for device access.
If you don't know what to do here, say N.
@@ -62,7 +61,9 @@ endif
config VFIO_NOIOMMU
bool "VFIO No-IOMMU support"
depends on VFIO_GROUP
depends on VFIO_GROUP || (VFIO_DEVICE_CDEV && !GENERIC_ATOMIC64)
depends on !VFIO_GROUP || VFIO_CONTAINER || IOMMUFD_VFIO_CONTAINER
select IOMMUFD_NOIOMMU if VFIO_DEVICE_CDEV && !GENERIC_ATOMIC64
help
VFIO is built on the ability to isolate devices using the IOMMU.
Only with an IOMMU can userspace access to DMA capable devices be
+9
View File
@@ -11,6 +11,10 @@ static dev_t device_devt;
void vfio_init_device_cdev(struct vfio_device *device)
{
if (vfio_device_is_noiommu(device) &&
!IS_ENABLED(CONFIG_IOMMUFD_NOIOMMU))
return;
device->device.devt = MKDEV(MAJOR(device_devt), device->index);
cdev_init(&device->cdev, &vfio_device_fops);
device->cdev.owner = THIS_MODULE;
@@ -31,6 +35,11 @@ int vfio_device_fops_cdev_open(struct inode *inode, struct file *filep)
if (!vfio_device_try_get_registration(device))
return -ENODEV;
if (vfio_device_is_noiommu(device) && !capable(CAP_SYS_RAWIO)) {
ret = -EPERM;
goto err_put_registration;
}
df = vfio_allocate_device_file(device);
if (IS_ERR(df)) {
ret = PTR_ERR(df);
+8 -4
View File
@@ -25,8 +25,8 @@ int vfio_df_iommufd_bind(struct vfio_device_file *df)
lockdep_assert_held(&vdev->dev_set->lock);
/* Returns 0 to permit device opening under noiommu mode */
if (vfio_device_is_noiommu(vdev))
/* Group noiommu via iommufd compat needs no device binding */
if (df->group && vfio_device_is_noiommu(vdev))
return 0;
return vdev->ops->bind_iommufd(vdev, ictx, &df->devid);
@@ -40,7 +40,11 @@ int vfio_iommufd_compat_attach_ioas(struct vfio_device *vdev,
lockdep_assert_held(&vdev->dev_set->lock);
/* compat noiommu does not need to do ioas attach */
/*
* Compat noiommu does not need to do ioas attach. This helper is
* only called from the legacy group/iommufd compat path, so no
* explicit df->group check is needed.
*/
if (vfio_device_is_noiommu(vdev))
return 0;
@@ -58,7 +62,7 @@ void vfio_df_iommufd_unbind(struct vfio_device_file *df)
lockdep_assert_held(&vdev->dev_set->lock);
if (vfio_device_is_noiommu(vdev))
if (df->group && vfio_device_is_noiommu(vdev))
return;
if (vdev->ops->unbind_iommufd)
+9 -14
View File
@@ -112,11 +112,6 @@ bool vfio_device_has_container(struct vfio_device *device);
int __init vfio_group_init(void);
void vfio_group_cleanup(void);
static inline bool vfio_device_is_noiommu(struct vfio_device *vdev)
{
return IS_ENABLED(CONFIG_VFIO_NOIOMMU) &&
vdev->group->type == VFIO_NO_IOMMU;
}
#else
struct vfio_group;
@@ -188,11 +183,17 @@ static inline void vfio_group_cleanup(void)
{
}
#endif /* CONFIG_VFIO_GROUP */
static inline bool vfio_device_is_noiommu(struct vfio_device *vdev)
{
return false;
#if IS_ENABLED(CONFIG_VFIO_GROUP)
if (vdev->group && vdev->group->type == VFIO_NO_IOMMU)
return true;
#endif
return IS_ENABLED(CONFIG_IOMMUFD_NOIOMMU) && vdev->noiommu;
}
#endif /* CONFIG_VFIO_GROUP */
#if IS_ENABLED(CONFIG_VFIO_CONTAINER)
/**
@@ -358,19 +359,13 @@ void vfio_init_device_cdev(struct vfio_device *device);
static inline int vfio_device_add(struct vfio_device *device)
{
/* cdev does not support noiommu device */
if (vfio_device_is_noiommu(device))
return device_add(&device->device);
vfio_init_device_cdev(device);
return cdev_device_add(&device->cdev, &device->device);
}
static inline void vfio_device_del(struct vfio_device *device)
{
if (vfio_device_is_noiommu(device))
device_del(&device->device);
else
cdev_device_del(&device->cdev, &device->device);
cdev_device_del(&device->cdev, &device->device);
}
int vfio_device_fops_cdev_open(struct inode *inode, struct file *filep);
+25 -1
View File
@@ -321,6 +321,24 @@ out_inode:
return ret;
}
static int vfio_device_set_noiommu_and_name(struct vfio_device *device, enum vfio_group_type type)
{
if (IS_ENABLED(CONFIG_IOMMUFD_NOIOMMU) && vfio_noiommu &&
!device->dev->iommu && type == VFIO_IOMMU)
device->noiommu = true;
/*
* device->noiommu records no-IOMMU support for the standalone cdev
* interface. VFIO_NOIOMMU enables both group and cdev no-IOMMU; when
* cdev no-IOMMU is available, device->noiommu is set before
* vfio_device_set_group(), so the cdev is named noiommu-vfio%d up
* front. If IOMMUFD_NOIOMMU is unavailable, no-IOMMU devices are
* limited to the group interface and do not receive a device cdev.
*/
return dev_set_name(&device->device, "%svfio%d",
device->noiommu ? "noiommu-" : "", device->index);
}
static int __vfio_register_dev(struct vfio_device *device,
enum vfio_group_type type)
{
@@ -340,7 +358,7 @@ static int __vfio_register_dev(struct vfio_device *device,
if (!device->dev_set)
vfio_assign_device_set(device, device);
ret = dev_set_name(&device->device, "vfio%d", device->index);
ret = vfio_device_set_noiommu_and_name(device, type);
if (ret)
return ret;
@@ -348,6 +366,12 @@ static int __vfio_register_dev(struct vfio_device *device,
if (ret)
return ret;
if (vfio_device_is_noiommu(device) && IS_ENABLED(CONFIG_IOMMUFD_NOIOMMU)) {
add_taint(TAINT_USER, LOCKDEP_STILL_OK);
dev_warn(device->dev,
"Adding kernel taint for vfio-noiommu cdev\n");
}
/*
* VFIO always sets IOMMU_CACHE because we offer no way for userspace to
* restore cache coherency. It has to be checked here because it is only
+1
View File
@@ -74,6 +74,7 @@ struct vfio_device {
u8 iommufd_attached:1;
#endif
u8 cdev_opened:1;
u8 noiommu:1;
/*
* debug_root is a static property of the vfio_device
* which must be set prior to registering the vfio_device.