iommufd: Support a HWPT without an iommu driver for noiommu

Create just a little part of a real iommu driver, enough to
slot in under the dev_iommu_ops() and allow iommufd to call
domain_alloc_paging_flags() and fail everything else.

This allows explicitly creating a HWPT under an IOAS.

A new Kconfig option IOMMUFD_NOIOMMU is introduced to differentiate
from the VFIO group/container based noiommu mode.

Link: https://patch.msgid.link/r/8d1ad0b90db0381d4139eda5c126cc8d168d89f8.1783360051.git.jacob.pan@linux.microsoft.com
Reviewed-by: Lu Baolu <baolu.lu@linux.intel.com>
Reviewed-by: Samiullah Khawaja <skhawaja@google.com>
Reviewed-by: Kevin Tian <kevin.tian@intel.com>
Reviewed-by: Pranjal Shrivastava <praan@google.com>
Signed-off-by: Jacob Pan <jacob.pan@linux.microsoft.com>
Fixes: 2c6cf6ab1564 ("iommufd: Allow binding to a noiommu device")
Reviewed-by: Yi Liu <yi.l.liu@intel.com>
Signed-off-by: Jason Gunthorpe <jgg@nvidia.com>
This commit is contained in:
Jason Gunthorpe 2026-07-06 11:48:29 -07:00
parent 8062148046
commit 17ea9f74bb
7 changed files with 158 additions and 6 deletions

View File

@ -16,6 +16,18 @@ config IOMMUFD
If you don't know what to do here, say N.
if IOMMUFD
config IOMMUFD_NOIOMMU
bool
depends on !GENERIC_ATOMIC64 # IOMMU_PT_AMDV1 requires cmpxchg64
select GENERIC_PT
select IOMMU_PT
select IOMMU_PT_AMDV1
help
Provides a SW-only IO page table for devices without hardware
IOMMU backing. This uses the AMDV1 page table format for
IOVA-to-PA lookups only, not for hardware DMA translation.
To be selected by VFIO_NOIOMMU when VFIO_DEVICE_CDEV is enabled.
config IOMMUFD_VFIO_CONTAINER
bool "IOMMUFD provides the VFIO container /dev/vfio/vfio"
depends on VFIO_GROUP && !VFIO_CONTAINER

View File

@ -10,6 +10,7 @@ iommufd-y := \
vfio_compat.o \
viommu.o
iommufd-$(CONFIG_IOMMUFD_NOIOMMU) += hwpt_noiommu.o
iommufd-$(CONFIG_IOMMUFD_TEST) += selftest.o
obj-$(CONFIG_IOMMUFD) += iommufd.o

View File

@ -8,6 +8,15 @@
#include "../iommu-priv.h"
#include "iommufd_private.h"
static const struct iommu_ops *get_iommu_ops(struct iommufd_device *idev)
{
if (IS_ENABLED(CONFIG_IOMMUFD_NOIOMMU) && !idev->igroup->group)
return &iommufd_noiommu_ops;
if (WARN_ON_ONCE(!idev->dev->iommu))
return NULL;
return dev_iommu_ops(idev->dev);
}
static void __iommufd_hwpt_destroy(struct iommufd_hw_pagetable *hwpt)
{
if (hwpt->domain)
@ -114,11 +123,13 @@ iommufd_hwpt_paging_alloc(struct iommufd_ctx *ictx, struct iommufd_ioas *ioas,
IOMMU_HWPT_ALLOC_DIRTY_TRACKING |
IOMMU_HWPT_FAULT_ID_VALID |
IOMMU_HWPT_ALLOC_PASID;
const struct iommu_ops *ops = dev_iommu_ops(idev->dev);
const struct iommu_ops *ops = get_iommu_ops(idev);
struct iommufd_hwpt_paging *hwpt_paging;
struct iommufd_hw_pagetable *hwpt;
int rc;
if (!ops)
return ERR_PTR(-ENODEV);
lockdep_assert_held(&ioas->mutex);
if ((flags || user_data) && !ops->domain_alloc_paging_flags)
@ -229,7 +240,7 @@ iommufd_hwpt_nested_alloc(struct iommufd_ctx *ictx,
struct iommufd_device *idev, u32 flags,
const struct iommu_user_data *user_data)
{
const struct iommu_ops *ops = dev_iommu_ops(idev->dev);
const struct iommu_ops *ops = get_iommu_ops(idev);
struct iommufd_hwpt_nested *hwpt_nested;
struct iommufd_hw_pagetable *hwpt;
int rc;
@ -389,10 +400,12 @@ int iommufd_hwpt_alloc(struct iommufd_ucmd *ucmd)
hwpt = &hwpt_nested->common;
} else if (pt_obj->type == IOMMUFD_OBJ_VIOMMU) {
struct iommufd_hwpt_nested *hwpt_nested;
struct iommu_device *iommu_dev;
struct iommufd_viommu *viommu;
viommu = container_of(pt_obj, struct iommufd_viommu, obj);
if (viommu->iommu_dev != __iommu_get_iommu_dev(idev->dev)) {
iommu_dev = iommufd_device_get_iommu_dev(idev);
if (!iommu_dev || viommu->iommu_dev != iommu_dev) {
rc = -EINVAL;
goto out_unlock;
}

View File

@ -0,0 +1,105 @@
// SPDX-License-Identifier: GPL-2.0-only
/*
* Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES
*/
#include <linux/generic_pt/iommu.h>
#include <linux/iommu.h>
#include "../iommu-pages.h"
#include "iommufd_private.h"
static const struct iommu_domain_ops noiommu_amdv1_ops;
struct noiommu_domain {
union {
struct iommu_domain domain;
struct pt_iommu_amdv1 amdv1;
};
spinlock_t lock;
};
PT_IOMMU_CHECK_DOMAIN(struct noiommu_domain, amdv1.iommu, domain);
static void noiommu_change_top(struct pt_iommu *iommu_table,
phys_addr_t top_paddr, unsigned int top_level)
{
}
static spinlock_t *noiommu_get_top_lock(struct pt_iommu *iommupt)
{
struct noiommu_domain *domain =
container_of(iommupt, struct noiommu_domain, amdv1.iommu);
return &domain->lock;
}
static const struct pt_iommu_driver_ops noiommu_driver_ops = {
.get_top_lock = noiommu_get_top_lock,
.change_top = noiommu_change_top,
};
static struct iommu_domain *
noiommu_alloc_paging_flags(struct device *dev, u32 flags,
const struct iommu_user_data *user_data)
{
struct pt_iommu_amdv1_cfg cfg = {};
struct noiommu_domain *dom;
int rc;
if (flags || user_data)
return ERR_PTR(-EOPNOTSUPP);
cfg.common.hw_max_vasz_lg2 = 64;
cfg.common.hw_max_oasz_lg2 = 52;
cfg.starting_level = 2;
cfg.common.features =
(BIT(PT_FEAT_DYNAMIC_TOP) | BIT(PT_FEAT_AMDV1_ENCRYPT_TABLES) |
BIT(PT_FEAT_AMDV1_FORCE_COHERENCE));
dom = kzalloc(sizeof(*dom), GFP_KERNEL);
if (!dom)
return ERR_PTR(-ENOMEM);
spin_lock_init(&dom->lock);
dom->amdv1.iommu.nid = NUMA_NO_NODE;
dom->amdv1.iommu.driver_ops = &noiommu_driver_ops;
dom->domain.ops = &noiommu_amdv1_ops;
/* Use SW-only page table which is based on AMDV1 */
rc = pt_iommu_amdv1_init(&dom->amdv1, &cfg, GFP_KERNEL);
if (rc) {
kfree(dom);
return ERR_PTR(rc);
}
return &dom->domain;
}
static void noiommu_domain_free(struct iommu_domain *iommu_domain)
{
struct noiommu_domain *domain =
container_of(iommu_domain, struct noiommu_domain, domain);
pt_iommu_deinit(&domain->amdv1.iommu);
kfree(domain);
}
static void noiommu_iotlb_sync(struct iommu_domain *domain,
struct iommu_iotlb_gather *gather)
{
iommu_put_pages_list(&gather->freelist);
}
/*
* Domain ops for iommufd no-IOMMU mode. Uses AMDV1 format as a
* SW-only IOPT because it has the best multi-page size options
* of all the formats. IOVAs serve only for IOVA-to-PA lookups,
* not for hardware DMA translation.
*/
static const struct iommu_domain_ops noiommu_amdv1_ops = {
IOMMU_PT_DOMAIN_OPS(amdv1),
.iotlb_sync = noiommu_iotlb_sync,
.free = noiommu_domain_free,
};
const struct iommu_ops iommufd_noiommu_ops = {
.domain_alloc_paging_flags = noiommu_alloc_paging_flags,
};

View File

@ -465,6 +465,8 @@ static inline void iommufd_hw_pagetable_put(struct iommufd_ctx *ictx,
refcount_dec(&hwpt->obj.users);
}
extern const struct iommu_ops iommufd_noiommu_ops;
struct iommufd_attach;
struct iommufd_group {
@ -502,6 +504,16 @@ iommufd_get_device(struct iommufd_ucmd *ucmd, u32 id)
struct iommufd_device, obj);
}
static inline struct iommu_device *
iommufd_device_get_iommu_dev(struct iommufd_device *idev)
{
if (IS_ENABLED(CONFIG_IOMMUFD_NOIOMMU) && !idev->igroup->group)
return NULL;
if (WARN_ON_ONCE(!idev->dev->iommu))
return NULL;
return __iommu_get_iommu_dev(idev->dev);
}
void iommufd_device_pre_destroy(struct iommufd_object *obj);
void iommufd_device_destroy(struct iommufd_object *obj);
int iommufd_get_hw_info(struct iommufd_ucmd *ucmd);

View File

@ -824,5 +824,6 @@ MODULE_ALIAS("devname:vfio/vfio");
MODULE_IMPORT_NS("IOMMUFD_INTERNAL");
MODULE_IMPORT_NS("IOMMUFD");
MODULE_IMPORT_NS("DMA_BUF");
MODULE_IMPORT_NS("GENERIC_PT_IOMMU");
MODULE_DESCRIPTION("I/O Address Space Management for passthrough devices");
MODULE_LICENSE("GPL");

View File

@ -25,6 +25,7 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
struct iommufd_hwpt_paging *hwpt_paging;
struct iommufd_viommu *viommu;
struct iommufd_device *idev;
struct iommu_device *iommu_dev;
const struct iommu_ops *ops;
size_t viommu_size;
int rc;
@ -36,7 +37,12 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
if (IS_ERR(idev))
return PTR_ERR(idev);
ops = dev_iommu_ops(idev->dev);
iommu_dev = iommufd_device_get_iommu_dev(idev);
if (!iommu_dev) {
rc = -EOPNOTSUPP;
goto out_put_idev;
}
ops = iommu_dev->ops;
if (!ops->get_viommu_size || !ops->viommu_init) {
rc = -EOPNOTSUPP;
goto out_put_idev;
@ -87,7 +93,7 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
* pluggable IOMMU instance (if exists) is responsible for refcounting
* on its own.
*/
viommu->iommu_dev = __iommu_get_iommu_dev(idev->dev);
viommu->iommu_dev = iommu_dev;
rc = ops->viommu_init(viommu, hwpt_paging->common.domain,
user_data.len ? &user_data : NULL);
@ -146,6 +152,7 @@ int iommufd_vdevice_alloc_ioctl(struct iommufd_ucmd *ucmd)
struct iommufd_vdevice *vdev, *curr;
size_t vdev_size = sizeof(*vdev);
struct iommufd_viommu *viommu;
struct iommu_device *iommu_dev;
struct iommufd_device *idev;
u64 virt_id = cmd->virt_id;
int rc = 0;
@ -164,7 +171,8 @@ int iommufd_vdevice_alloc_ioctl(struct iommufd_ucmd *ucmd)
goto out_put_viommu;
}
if (viommu->iommu_dev != __iommu_get_iommu_dev(idev->dev)) {
iommu_dev = iommufd_device_get_iommu_dev(idev);
if (!iommu_dev || viommu->iommu_dev != iommu_dev) {
rc = -EINVAL;
goto out_put_idev;
}