* [PATCH v7 01/16] KVM: Introduce file_to_kvm_<arch>() infrastructure
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 02/16] KVM: Add file back-pointer to struct kvm Aneesh Kumar K.V (Arm)
` (14 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Steffen Eiden, Sean Christopherson,
Janosch Frank
From: Steffen Eiden <seiden@linux.ibm.com>
Add a macro mechanism that generates an opt-in and arch-namespaced
file_to_kvm_<arch>() to convert a file reference to a kvm object if the
file handle represents a kvm handle.
Activate for x86 and for s390 (native KVM only).
Suggested-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Steffen Eiden <seiden@linux.ibm.com>
Acked-by: Sean Christopherson <seanjc@google.com>
Acked-by: Janosch Frank <frankja@linux.ibm.com>
---
arch/s390/include/asm/kvm_host_s390.h | 2 ++
arch/x86/include/asm/kvm_host.h | 2 ++
arch/x86/kvm/Makefile | 1 +
include/linux/kvm_host.h | 12 ++++++++++++
virt/kvm/kvm_main.c | 11 +++++++++++
5 files changed, 28 insertions(+)
diff --git a/arch/s390/include/asm/kvm_host_s390.h b/arch/s390/include/asm/kvm_host_s390.h
index cd692f8fb764..8a7eed5847e1 100644
--- a/arch/s390/include/asm/kvm_host_s390.h
+++ b/arch/s390/include/asm/kvm_host_s390.h
@@ -27,6 +27,8 @@
#include <asm/isc.h>
#include <asm/guarded_storage.h>
+#define kvm_file_to_kvm_arch s390
+
#define KVM_HAVE_MMU_RWLOCK
#define KVM_MAX_VCPUS 255
diff --git a/arch/x86/include/asm/kvm_host.h b/arch/x86/include/asm/kvm_host.h
index 683bb8bf43a9..e460073ff176 100644
--- a/arch/x86/include/asm/kvm_host.h
+++ b/arch/x86/include/asm/kvm_host.h
@@ -44,6 +44,8 @@
#include <hyperv/hvhdk.h>
+#define kvm_file_to_kvm_arch x86
+
#define __KVM_HAVE_ARCH_VCPU_DEBUGFS
/*
diff --git a/arch/x86/kvm/Makefile b/arch/x86/kvm/Makefile
index 0474604ab8a1..e6d395ede9e2 100644
--- a/arch/x86/kvm/Makefile
+++ b/arch/x86/kvm/Makefile
@@ -61,6 +61,7 @@ exports_grep_trailer := --include='*.[ch]' -nrw $(srctree)/virt/kvm $(srctree)/a
-e kvm_page_track_unregister_notifier \
-e kvm_write_track_add_gfn \
-e kvm_write_track_remove_gfn \
+ -e kvm_file_to_kvm_fn \
-e kvm_get_kvm \
-e kvm_get_kvm_safe \
-e kvm_put_kvm
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index cf7fe835c4ad..bc3abfde8393 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -1082,6 +1082,18 @@ void kvm_get_kvm(struct kvm *kvm);
bool kvm_get_kvm_safe(struct kvm *kvm);
void kvm_put_kvm(struct kvm *kvm);
bool file_is_kvm(struct file *file);
+
+/*
+ * Architectures define kvm_file_to_kvm_arch to <arch>
+ * to get a typed, arch-namespaced helper:
+ *
+ * struct kvm *file_to_kvm_<arch>(struct file *file)
+ */
+#ifdef kvm_file_to_kvm_arch
+#define kvm_file_to_kvm_fn CONCATENATE(file_to_kvm_, kvm_file_to_kvm_arch)
+struct kvm *kvm_file_to_kvm_fn(struct file *file);
+#endif /* kvm_file_to_kvm_arch */
+
void kvm_put_kvm_no_destroy(struct kvm *kvm);
static inline struct kvm_memslots *__kvm_memslots(struct kvm *kvm, int as_id)
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index 85f42289748d..07fc874a38e1 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -5519,6 +5519,17 @@ bool file_is_kvm(struct file *file)
}
EXPORT_SYMBOL_FOR_KVM_INTERNAL(file_is_kvm);
+#ifdef kvm_file_to_kvm_arch
+struct kvm *kvm_file_to_kvm_fn(struct file *file)
+{
+ if (!file || file->f_op != &kvm_vm_fops)
+ return NULL;
+
+ return file->private_data;
+}
+EXPORT_SYMBOL_GPL(kvm_file_to_kvm_fn);
+#endif
+
static int kvm_dev_ioctl_create_vm(unsigned long type)
{
char fdname[ITOA_MAX_LEN + 1];
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 02/16] KVM: Add file back-pointer to struct kvm
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 01/16] KVM: Introduce file_to_kvm_<arch>() infrastructure Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 03/16] KVM: x86: Use file_to_kvm_x86() in SEV Aneesh Kumar K.V (Arm)
` (13 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Steffen Eiden, Sean Christopherson,
Janosch Frank
From: Steffen Eiden <seiden@linux.ibm.com>
Add kvm->file as a no-reference back-pointer to the VM file in struct
kvm. The pointer is set at VM creation time and cleared under
WRITE_ONCE() in kvm_vm_release() before the file is freed.
Callers storing the file must take their own reference.
Co-developed-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Steffen Eiden <seiden@linux.ibm.com>
Acked-by: Sean Christopherson <seanjc@google.com>
Acked-by: Janosch Frank <frankja@linux.ibm.com>
---
include/linux/kvm_host.h | 6 ++++++
virt/kvm/kvm_main.c | 4 ++++
2 files changed, 10 insertions(+)
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index bc3abfde8393..3c5a389361c6 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -784,6 +784,12 @@ struct kvm {
* kvm_swap_active_memslots().
*/
struct mutex slots_arch_lock;
+ /*
+ * Back-reference to the VM file for subsystems (e.g., VFIO). Holds no
+ * reference to avoid pinning the VM. Callers storing the file must
+ * take their own reference.
+ */
+ struct file *file;
struct mm_struct *mm; /* userspace tied to this vm */
unsigned long nr_memslot_pages;
/* The two memslot sets - active and inactive (per address space) */
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index 07fc874a38e1..8e49ea809528 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -1352,6 +1352,8 @@ static int kvm_vm_release(struct inode *inode, struct file *filp)
kvm_irqfd_release(kvm);
+ WRITE_ONCE(kvm->file, NULL);
+
kvm_put_kvm(kvm);
return 0;
}
@@ -5555,6 +5557,8 @@ static int kvm_dev_ioctl_create_vm(unsigned long type)
goto put_kvm;
}
+ kvm->file = file;
+
/*
* Don't call kvm_put_kvm anymore at this point; file->f_op is
* already set, with ->release() being kvm_vm_release(). In error
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 03/16] KVM: x86: Use file_to_kvm_x86() in SEV
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 01/16] KVM: Introduce file_to_kvm_<arch>() infrastructure Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 02/16] KVM: Add file back-pointer to struct kvm Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 04/16] KVM/vfio: Use file-based reference counting for KVM Aneesh Kumar K.V (Arm)
` (12 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Steffen Eiden, Sean Christopherson,
Janosch Frank
From: Steffen Eiden <seiden@linux.ibm.com>
Use the new arch-namespaced helper instead of open-coding the file check
and private_data cast separately.
Suggested-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Steffen Eiden <seiden@linux.ibm.com>
Acked-by: Sean Christopherson <seanjc@google.com>
Acked-by: Janosch Frank <frankja@linux.ibm.com>
---
arch/x86/kvm/svm/sev.c | 8 ++++----
1 file changed, 4 insertions(+), 4 deletions(-)
diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c
index 63eb2155a774..db7ed11e9056 100644
--- a/arch/x86/kvm/svm/sev.c
+++ b/arch/x86/kvm/svm/sev.c
@@ -2151,10 +2151,10 @@ int sev_vm_move_enc_context_from(struct kvm *kvm, unsigned int source_fd)
if (fd_empty(f))
return -EBADF;
- if (!file_is_kvm(fd_file(f)))
+ source_kvm = file_to_kvm_x86(fd_file(f));
+ if (!source_kvm)
return -EBADF;
- source_kvm = fd_file(f)->private_data;
ret = sev_lock_two_vms(kvm, source_kvm);
if (ret)
return ret;
@@ -2876,10 +2876,10 @@ int sev_vm_copy_enc_context_from(struct kvm *kvm, unsigned int source_fd)
if (fd_empty(f))
return -EBADF;
- if (!file_is_kvm(fd_file(f)))
+ source_kvm = file_to_kvm_x86(fd_file(f));
+ if (!source_kvm)
return -EBADF;
- source_kvm = fd_file(f)->private_data;
ret = sev_lock_two_vms(kvm, source_kvm);
if (ret)
return ret;
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 04/16] KVM/vfio: Use file-based reference counting for KVM
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (2 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 03/16] KVM: x86: Use file_to_kvm_x86() in SEV Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 05/16] KVM: Restrict kvm_get_kvm/kvm_put_kvm export to internal KVM modules Aneesh Kumar K.V (Arm)
` (11 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Steffen Eiden, Jason Gunthorpe,
Sean Christopherson, Janosch Frank
From: Steffen Eiden <seiden@linux.ibm.com>
Replace manual module reference counting with file-based reference
counting for KVM integration. Previously, VFIO used symbol_get() to
obtain function pointers for kvm_get_kvm_safe() and kvm_put_kvm(),
then manually tracked module references through these symbols. This
approach required storing the put_kvm function pointer in each device
and carefully managing symbol references.
Pass struct file pointers instead of struct kvm pointers throughout the
VFIO-KVM interface, leveraging the kernel's existing file reference
counting via get_file()/get_file_active() and fput(). Convert the x86
page-track API and update s390 vfio to use file_to_kvm_<arch>(). This
simplifies the code.
Suggested-by: Jason Gunthorpe <jgg@nvidia.com>
Co-developed-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Steffen Eiden <seiden@linux.ibm.com>
Acked-by: Sean Christopherson <seanjc@google.com>
Acked-by: Janosch Frank <frankja@linux.ibm.com>
---
arch/s390/include/asm/kvm_host_s390.h | 2 +-
arch/s390/kvm/s390/pci.c | 9 ++++-
arch/x86/include/asm/kvm_page_track.h | 10 +++--
arch/x86/kvm/mmu/page_track.c | 22 +++++++----
drivers/s390/crypto/vfio_ap_ops.c | 20 +++++++---
drivers/vfio/group.c | 11 +++++-
drivers/vfio/vfio.h | 12 +++---
drivers/vfio/vfio_main.c | 57 +++++++++------------------
include/linux/vfio.h | 5 +--
virt/kvm/vfio.c | 13 ++++--
10 files changed, 87 insertions(+), 74 deletions(-)
diff --git a/arch/s390/include/asm/kvm_host_s390.h b/arch/s390/include/asm/kvm_host_s390.h
index 8a7eed5847e1..9519b8028b10 100644
--- a/arch/s390/include/asm/kvm_host_s390.h
+++ b/arch/s390/include/asm/kvm_host_s390.h
@@ -722,7 +722,7 @@ static inline void kvm_arch_vcpu_unblocking(struct kvm_vcpu *vcpu) {}
void kvm_arch_free_vm(struct kvm *kvm);
struct zpci_kvm_hook {
- int (*kvm_register)(void *opaque, struct kvm *kvm);
+ int (*kvm_register)(void *opaque, struct file *kvm_file);
void (*kvm_unregister)(void *opaque);
};
diff --git a/arch/s390/kvm/s390/pci.c b/arch/s390/kvm/s390/pci.c
index 82892e1e03d9..d79bd3bdc68e 100644
--- a/arch/s390/kvm/s390/pci.c
+++ b/arch/s390/kvm/s390/pci.c
@@ -498,17 +498,22 @@ static void kvm_s390_pci_dev_release(struct zpci_dev *zdev)
* available, enable them and let userspace indicate whether or not they will
* be used (specify SHM bit to disable).
*/
-static int kvm_s390_pci_register_kvm(void *opaque, struct kvm *kvm)
+static int kvm_s390_pci_register_kvm(void *opaque, struct file *kvm_file)
{
struct zpci_dev *zdev = opaque;
+ struct kvm *kvm;
int rc;
if (!zdev)
return -EINVAL;
+ kvm = file_to_kvm_s390(kvm_file);
+ if (!kvm)
+ return -ENOENT;
+
mutex_lock(&zdev->kzdev_lock);
- if (zdev->kzdev || zdev->gisa != 0 || !kvm) {
+ if (zdev->kzdev || zdev->gisa != 0) {
mutex_unlock(&zdev->kzdev_lock);
return -EINVAL;
}
diff --git a/arch/x86/include/asm/kvm_page_track.h b/arch/x86/include/asm/kvm_page_track.h
index 3d040741044b..ea885dd5c9af 100644
--- a/arch/x86/include/asm/kvm_page_track.h
+++ b/arch/x86/include/asm/kvm_page_track.h
@@ -44,13 +44,15 @@ struct kvm_page_track_notifier_node {
struct kvm_page_track_notifier_node *node);
};
-int kvm_page_track_register_notifier(struct kvm *kvm,
+struct file;
+
+int kvm_page_track_register_notifier(struct file *file,
struct kvm_page_track_notifier_node *n);
-void kvm_page_track_unregister_notifier(struct kvm *kvm,
+void kvm_page_track_unregister_notifier(struct file *file,
struct kvm_page_track_notifier_node *n);
-int kvm_write_track_add_gfn(struct kvm *kvm, gfn_t gfn);
-int kvm_write_track_remove_gfn(struct kvm *kvm, gfn_t gfn);
+int kvm_write_track_add_gfn(struct file *file, gfn_t gfn);
+int kvm_write_track_remove_gfn(struct file *file, gfn_t gfn);
#else
/*
* Allow defining a node in a structure even if page tracking is disabled, e.g.
diff --git a/arch/x86/kvm/mmu/page_track.c b/arch/x86/kvm/mmu/page_track.c
index 7e8195a311bb..f12558dfcd81 100644
--- a/arch/x86/kvm/mmu/page_track.c
+++ b/arch/x86/kvm/mmu/page_track.c
@@ -16,6 +16,8 @@
#include <linux/kvm_host.h>
#include <linux/rculist.h>
+#include <asm/kvm_page_track.h>
+
#include "mmu.h"
#include "mmu_internal.h"
#include "page_track.h"
@@ -237,10 +239,11 @@ static int kvm_enable_external_write_tracking(struct kvm *kvm)
* register the notifier so that event interception for the tracked guest
* pages can be received.
*/
-int kvm_page_track_register_notifier(struct kvm *kvm,
+int kvm_page_track_register_notifier(struct file *file,
struct kvm_page_track_notifier_node *n)
{
struct kvm_page_track_notifier_head *head;
+ struct kvm *kvm = file_to_kvm_x86(file);
int r;
if (!kvm || kvm->mm != current->mm)
@@ -252,7 +255,7 @@ int kvm_page_track_register_notifier(struct kvm *kvm,
return r;
}
- kvm_get_kvm(kvm);
+ get_file(file);
head = &kvm->arch.track_notifier_head;
@@ -267,10 +270,11 @@ EXPORT_SYMBOL_GPL(kvm_page_track_register_notifier);
* stop receiving the event interception. It is the opposed operation of
* kvm_page_track_register_notifier().
*/
-void kvm_page_track_unregister_notifier(struct kvm *kvm,
+void kvm_page_track_unregister_notifier(struct file *file,
struct kvm_page_track_notifier_node *n)
{
struct kvm_page_track_notifier_head *head;
+ struct kvm *kvm = file_to_kvm_x86(file);
head = &kvm->arch.track_notifier_head;
@@ -279,7 +283,7 @@ void kvm_page_track_unregister_notifier(struct kvm *kvm,
write_unlock(&kvm->mmu_lock);
synchronize_srcu(&head->track_srcu);
- kvm_put_kvm(kvm);
+ fput(file);
}
EXPORT_SYMBOL_GPL(kvm_page_track_unregister_notifier);
@@ -336,11 +340,12 @@ void kvm_page_track_delete_slot(struct kvm *kvm, struct kvm_memory_slot *slot)
* add guest page to the tracking pool so that corresponding access on that
* page will be intercepted.
*
- * @kvm: the guest instance we are interested in.
+ * @file: the VM file of the guest instance we are interested in.
* @gfn: the guest page.
*/
-int kvm_write_track_add_gfn(struct kvm *kvm, gfn_t gfn)
+int kvm_write_track_add_gfn(struct file *file, gfn_t gfn)
{
+ struct kvm *kvm = file_to_kvm_x86(file);
struct kvm_memory_slot *slot;
int idx;
@@ -366,11 +371,12 @@ EXPORT_SYMBOL_GPL(kvm_write_track_add_gfn);
* remove the guest page from the tracking pool which stops the interception
* of corresponding access on that page.
*
- * @kvm: the guest instance we are interested in.
+ * @file: the VM file of the guest instance we are interested in.
* @gfn: the guest page.
*/
-int kvm_write_track_remove_gfn(struct kvm *kvm, gfn_t gfn)
+int kvm_write_track_remove_gfn(struct file *file, gfn_t gfn)
{
+ struct kvm *kvm = file_to_kvm_x86(file);
struct kvm_memory_slot *slot;
int idx;
diff --git a/drivers/s390/crypto/vfio_ap_ops.c b/drivers/s390/crypto/vfio_ap_ops.c
index 4db878c18f41..373ce574c431 100644
--- a/drivers/s390/crypto/vfio_ap_ops.c
+++ b/drivers/s390/crypto/vfio_ap_ops.c
@@ -1822,17 +1822,27 @@ static const struct attribute_group *vfio_ap_mdev_attr_groups[] = {
/**
* vfio_ap_mdev_set_kvm - sets all data for @matrix_mdev that are needed
- * to manage AP resources for the guest whose state is represented by @kvm
+ * to manage AP resources for the guest whose state is represented by
+ * @kvm_file
*
* @matrix_mdev: a mediated matrix device
- * @kvm: reference to KVM instance
+ * @kvm_file: the KVM VM file this vfio device is associated with
*
- * Return: 0 if no other mediated matrix device has a reference to @kvm;
+ * Return: 0 if no other mediated matrix device has a reference to the VM;
* otherwise, returns an -EPERM.
*/
static int vfio_ap_mdev_set_kvm(struct ap_matrix_mdev *matrix_mdev,
- struct kvm *kvm)
+ struct file *kvm_file)
{
+ struct kvm *kvm;
+
+ if (!kvm_file)
+ return -ENOENT;
+
+ kvm = file_to_kvm_s390(kvm_file);
+ if (!kvm)
+ return -ENOENT;
+
if (kvm->arch.crypto.crycbd) {
get_update_locks_for_kvm(kvm);
if (kvm->arch.crypto.pqap_hook) {
@@ -1841,7 +1851,6 @@ static int vfio_ap_mdev_set_kvm(struct ap_matrix_mdev *matrix_mdev,
}
kvm->arch.crypto.pqap_hook = &matrix_mdev->pqap_hook;
- kvm_get_kvm(kvm);
matrix_mdev->kvm = kvm;
vfio_ap_mdev_update_guest_apcb(matrix_mdev);
release_update_locks_for_kvm(kvm);
@@ -1894,7 +1903,6 @@ static void vfio_ap_mdev_unset_kvm(struct ap_matrix_mdev *matrix_mdev)
matrix_mdev->kvm = NULL;
release_update_locks_for_kvm(kvm);
- kvm_put_kvm(kvm);
}
}
diff --git a/drivers/vfio/group.c b/drivers/vfio/group.c
index b2299e5bc6df..5bf8cbdff377 100644
--- a/drivers/vfio/group.c
+++ b/drivers/vfio/group.c
@@ -860,11 +860,20 @@ bool vfio_group_enforced_coherent(struct vfio_group *group)
return ret;
}
-void vfio_group_set_kvm(struct vfio_group *group, struct kvm *kvm)
+void vfio_group_set_kvm(struct vfio_group *group, struct file *kvm)
{
+ struct file *old;
+
+ if (kvm)
+ get_file(kvm);
+
spin_lock(&group->kvm_ref_lock);
+ old = group->kvm;
group->kvm = kvm;
spin_unlock(&group->kvm_ref_lock);
+
+ if (old)
+ fput(old);
}
/**
diff --git a/drivers/vfio/vfio.h b/drivers/vfio/vfio.h
index 7728bc99b63d..9b619951a5d0 100644
--- a/drivers/vfio/vfio.h
+++ b/drivers/vfio/vfio.h
@@ -23,7 +23,7 @@ struct vfio_device_file {
u8 access_granted;
u32 devid; /* only valid when iommufd is valid */
spinlock_t kvm_ref_lock; /* protect kvm field */
- struct kvm *kvm;
+ struct file *kvm;
struct iommufd_ctx *iommufd; /* protected by struct vfio_device_set::lock */
};
@@ -88,7 +88,7 @@ struct vfio_group {
#endif
enum vfio_group_type type;
struct mutex group_lock;
- struct kvm *kvm;
+ struct file *kvm;
struct file *opened_file;
struct iommufd_ctx *iommufd;
spinlock_t kvm_ref_lock;
@@ -107,7 +107,7 @@ void vfio_device_group_unuse_iommu(struct vfio_device *device);
void vfio_df_group_close(struct vfio_device_file *df);
struct vfio_group *vfio_group_from_file(struct file *file);
bool vfio_group_enforced_coherent(struct vfio_group *group);
-void vfio_group_set_kvm(struct vfio_group *group, struct kvm *kvm);
+void vfio_group_set_kvm(struct vfio_group *group, struct file *kvm);
bool vfio_device_has_container(struct vfio_device *device);
int __init vfio_group_init(void);
void vfio_group_cleanup(void);
@@ -165,7 +165,7 @@ static inline bool vfio_group_enforced_coherent(struct vfio_group *group)
return true;
}
-static inline void vfio_group_set_kvm(struct vfio_group *group, struct kvm *kvm)
+static inline void vfio_group_set_kvm(struct vfio_group *group, struct file *kvm)
{
}
@@ -429,11 +429,11 @@ static inline void vfio_virqfd_exit(void)
#endif
#if IS_ENABLED(CONFIG_KVM)
-void vfio_device_get_kvm_safe(struct vfio_device *device, struct kvm *kvm);
+void vfio_device_get_kvm_safe(struct vfio_device *device, struct file *kvm);
void vfio_device_put_kvm(struct vfio_device *device);
#else
static inline void vfio_device_get_kvm_safe(struct vfio_device *device,
- struct kvm *kvm)
+ struct file *kvm)
{
}
diff --git a/drivers/vfio/vfio_main.c b/drivers/vfio/vfio_main.c
index 423ead48aafe..ed96acfa8635 100644
--- a/drivers/vfio/vfio_main.c
+++ b/drivers/vfio/vfio_main.c
@@ -472,36 +472,14 @@ void vfio_unregister_group_dev(struct vfio_device *device)
EXPORT_SYMBOL_GPL(vfio_unregister_group_dev);
#if IS_ENABLED(CONFIG_KVM)
-void vfio_device_get_kvm_safe(struct vfio_device *device, struct kvm *kvm)
+void vfio_device_get_kvm_safe(struct vfio_device *device, struct file *kvm)
{
- void (*pfn)(struct kvm *kvm);
- bool (*fn)(struct kvm *kvm);
- bool ret;
-
lockdep_assert_held(&device->dev_set->lock);
if (!kvm)
return;
- pfn = symbol_get(kvm_put_kvm);
- if (WARN_ON(!pfn))
- return;
-
- fn = symbol_get(kvm_get_kvm_safe);
- if (WARN_ON(!fn)) {
- symbol_put(kvm_put_kvm);
- return;
- }
-
- ret = fn(kvm);
- symbol_put(kvm_get_kvm_safe);
- if (!ret) {
- symbol_put(kvm_put_kvm);
- return;
- }
-
- device->put_kvm = pfn;
- device->kvm = kvm;
+ device->kvm = get_file(kvm);
}
void vfio_device_put_kvm(struct vfio_device *device)
@@ -511,14 +489,7 @@ void vfio_device_put_kvm(struct vfio_device *device)
if (!device->kvm)
return;
- if (WARN_ON(!device->put_kvm))
- goto clear;
-
- device->put_kvm(device->kvm);
- device->put_kvm = NULL;
- symbol_put(kvm_put_kvm);
-
-clear:
+ fput(device->kvm);
device->kvm = NULL;
}
#endif
@@ -1544,9 +1515,13 @@ bool vfio_file_enforced_coherent(struct file *file)
}
EXPORT_SYMBOL_GPL(vfio_file_enforced_coherent);
-static void vfio_device_file_set_kvm(struct file *file, struct kvm *kvm)
+static void vfio_device_file_set_kvm(struct file *file, struct file *kvm)
{
struct vfio_device_file *df = file->private_data;
+ struct file *old;
+
+ if (kvm)
+ get_file(kvm);
/*
* The kvm is first recorded in the vfio_device_file, and will
@@ -1554,28 +1529,32 @@ static void vfio_device_file_set_kvm(struct file *file, struct kvm *kvm)
* iommufd successfully in the vfio device cdev path.
*/
spin_lock(&df->kvm_ref_lock);
+ old = df->kvm;
df->kvm = kvm;
spin_unlock(&df->kvm_ref_lock);
+
+ if (old)
+ fput(old);
}
/**
* vfio_file_set_kvm - Link a kvm with VFIO drivers
- * @file: VFIO group file or VFIO device file
- * @kvm: KVM to link
+ * @vfio_file: VFIO group file or VFIO device file
+ * @kvm: KVM file to link
*
* When a VFIO device is first opened the KVM will be available in
* device->kvm if one was associated with the file.
*/
-void vfio_file_set_kvm(struct file *file, struct kvm *kvm)
+void vfio_file_set_kvm(struct file *vfio_file, struct file *kvm)
{
struct vfio_group *group;
- group = vfio_group_from_file(file);
+ group = vfio_group_from_file(vfio_file);
if (group)
vfio_group_set_kvm(group, kvm);
- if (vfio_device_from_file(file))
- vfio_device_file_set_kvm(file, kvm);
+ if (vfio_device_from_file(vfio_file))
+ vfio_device_file_set_kvm(vfio_file, kvm);
}
EXPORT_SYMBOL_GPL(vfio_file_set_kvm);
diff --git a/include/linux/vfio.h b/include/linux/vfio.h
index 45f08986359e..0cc91c6f96d2 100644
--- a/include/linux/vfio.h
+++ b/include/linux/vfio.h
@@ -54,7 +54,7 @@ struct vfio_device {
struct list_head dev_set_list;
unsigned int migration_flags;
u8 precopy_info_v2;
- struct kvm *kvm;
+ struct file *kvm;
/* Members below here are private, not for driver use */
unsigned int index;
@@ -66,7 +66,6 @@ struct vfio_device {
unsigned int open_count;
struct completion comp;
struct iommufd_access *iommufd_access;
- void (*put_kvm)(struct kvm *kvm);
struct inode *inode;
#if IS_ENABLED(CONFIG_IOMMUFD)
struct iommufd_device *iommufd_device;
@@ -378,7 +377,7 @@ static inline bool vfio_file_has_dev(struct file *file, struct vfio_device *devi
#endif
bool vfio_file_is_valid(struct file *file);
bool vfio_file_enforced_coherent(struct file *file);
-void vfio_file_set_kvm(struct file *file, struct kvm *kvm);
+void vfio_file_set_kvm(struct file *vfio_file, struct file *kvm);
#define VFIO_PIN_PAGES_MAX_ENTRIES (PAGE_SIZE/sizeof(unsigned long))
diff --git a/virt/kvm/vfio.c b/virt/kvm/vfio.c
index 6cdc4e9a333a..19548a430942 100644
--- a/virt/kvm/vfio.c
+++ b/virt/kvm/vfio.c
@@ -35,15 +35,15 @@ struct kvm_vfio {
bool noncoherent;
};
-static void kvm_vfio_file_set_kvm(struct file *file, struct kvm *kvm)
+static void kvm_vfio_file_set_kvm(struct file *vfio_file, struct file *kvm)
{
- void (*fn)(struct file *file, struct kvm *kvm);
+ void (*fn)(struct file *vfio_file, struct file *kvm);
fn = symbol_get(vfio_file_set_kvm);
if (!fn)
return;
- fn(file, kvm);
+ fn(vfio_file, kvm);
symbol_put(vfio_file_set_kvm);
}
@@ -144,6 +144,7 @@ static int kvm_vfio_file_add(struct kvm_device *dev, unsigned int fd)
{
struct kvm_vfio *kv = dev->private;
struct kvm_vfio_file *kvf;
+ struct file *kvm_file __free(fput) = NULL;
struct file *filp __free(fput) = NULL;
filp = fget(fd);
@@ -154,6 +155,10 @@ static int kvm_vfio_file_add(struct kvm_device *dev, unsigned int fd)
if (!kvm_vfio_file_is_valid(filp))
return -EINVAL;
+ kvm_file = get_file_active(&dev->kvm->file);
+ if (!kvm_file)
+ return -ENOENT;
+
guard(mutex)(&kv->lock);
list_for_each_entry(kvf, &kv->file_list, node) {
@@ -168,7 +173,7 @@ static int kvm_vfio_file_add(struct kvm_device *dev, unsigned int fd)
kvf->file = get_file(filp);
list_add_tail(&kvf->node, &kv->file_list);
- kvm_vfio_file_set_kvm(kvf->file, dev->kvm);
+ kvm_vfio_file_set_kvm(kvf->file, kvm_file);
kvm_vfio_update_coherency(dev);
return 0;
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 05/16] KVM: Restrict kvm_get_kvm/kvm_put_kvm export to internal KVM modules
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (3 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 04/16] KVM/vfio: Use file-based reference counting for KVM Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 06/16] KVM: Remove unused file_is_kvm Aneesh Kumar K.V (Arm)
` (10 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Steffen Eiden, Sean Christopherson,
Janosch Frank
From: Steffen Eiden <seiden@linux.ibm.com>
Switch kvm_get_kvm, kvm_get_kvm_safe, and kvm_put_kvm from
EXPORT_SYMBOL_GPL to EXPORT_SYMBOL_FOR_KVM_INTERNAL. These symbols are
now only used within KVM itself. No external module needs them now that
VFIO uses file-based reference counting. Remove the corresponding
exemptions from the x86 KVM Makefile exports check.
Suggested-by: Sean Christopherson <seanjc@google.com>
Signed-off-by: Steffen Eiden <seiden@linux.ibm.com>
Acked-by: Sean Christopherson <seanjc@google.com>
Acked-by: Janosch Frank <frankja@linux.ibm.com>
---
arch/x86/kvm/Makefile | 5 +----
virt/kvm/kvm_main.c | 6 +++---
2 files changed, 4 insertions(+), 7 deletions(-)
diff --git a/arch/x86/kvm/Makefile b/arch/x86/kvm/Makefile
index e6d395ede9e2..c6bf463f5032 100644
--- a/arch/x86/kvm/Makefile
+++ b/arch/x86/kvm/Makefile
@@ -61,10 +61,7 @@ exports_grep_trailer := --include='*.[ch]' -nrw $(srctree)/virt/kvm $(srctree)/a
-e kvm_page_track_unregister_notifier \
-e kvm_write_track_add_gfn \
-e kvm_write_track_remove_gfn \
- -e kvm_file_to_kvm_fn \
- -e kvm_get_kvm \
- -e kvm_get_kvm_safe \
- -e kvm_put_kvm
+ -e kvm_file_to_kvm_fn
# Force grep to emit a goofy group separator that can in turn be replaced with
# the above newline macro (newlines in Make are a nightmare). Note, grep only
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index 8e49ea809528..ff4fab9753ca 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -1314,7 +1314,7 @@ void kvm_get_kvm(struct kvm *kvm)
{
refcount_inc(&kvm->users_count);
}
-EXPORT_SYMBOL_GPL(kvm_get_kvm);
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_get_kvm);
/*
* Make sure the vm is not during destruction, which is a safe version of
@@ -1324,14 +1324,14 @@ bool kvm_get_kvm_safe(struct kvm *kvm)
{
return refcount_inc_not_zero(&kvm->users_count);
}
-EXPORT_SYMBOL_GPL(kvm_get_kvm_safe);
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_get_kvm_safe);
void kvm_put_kvm(struct kvm *kvm)
{
if (refcount_dec_and_test(&kvm->users_count))
kvm_destroy_vm(kvm);
}
-EXPORT_SYMBOL_GPL(kvm_put_kvm);
+EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_put_kvm);
/*
* Used to put a reference that was taken on behalf of an object associated
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 06/16] KVM: Remove unused file_is_kvm
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (4 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 05/16] KVM: Restrict kvm_get_kvm/kvm_put_kvm export to internal KVM modules Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 07/16] iommufd/device: Associate KVM file pointer with iommufd_device Aneesh Kumar K.V (Arm)
` (9 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Steffen Eiden, Sean Christopherson,
Janosch Frank
From: Steffen Eiden <seiden@linux.ibm.com>
Remove the now unused file_is_kvm function.
Signed-off-by: Steffen Eiden <seiden@linux.ibm.com>
Acked-by: Sean Christopherson <seanjc@google.com>
Acked-by: Janosch Frank <frankja@linux.ibm.com>
---
include/linux/kvm_host.h | 1 -
virt/kvm/kvm_main.c | 6 ------
2 files changed, 7 deletions(-)
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index 3c5a389361c6..586f6e8bb81b 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -1087,7 +1087,6 @@ void kvm_exit(void);
void kvm_get_kvm(struct kvm *kvm);
bool kvm_get_kvm_safe(struct kvm *kvm);
void kvm_put_kvm(struct kvm *kvm);
-bool file_is_kvm(struct file *file);
/*
* Architectures define kvm_file_to_kvm_arch to <arch>
diff --git a/virt/kvm/kvm_main.c b/virt/kvm/kvm_main.c
index ff4fab9753ca..60e614795a36 100644
--- a/virt/kvm/kvm_main.c
+++ b/virt/kvm/kvm_main.c
@@ -5515,12 +5515,6 @@ static struct file_operations kvm_vm_fops = {
KVM_COMPAT(kvm_vm_compat_ioctl),
};
-bool file_is_kvm(struct file *file)
-{
- return file && file->f_op == &kvm_vm_fops;
-}
-EXPORT_SYMBOL_FOR_KVM_INTERNAL(file_is_kvm);
-
#ifdef kvm_file_to_kvm_arch
struct kvm *kvm_file_to_kvm_fn(struct file *file)
{
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 07/16] iommufd/device: Associate KVM file pointer with iommufd_device
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (5 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 06/16] KVM: Remove unused file_is_kvm Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 08/16] iommufd/viommu: Keep a reference to the KVM file Aneesh Kumar K.V (Arm)
` (8 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Jason Gunthorpe
From: Shameer Kolothum <shameerali.kolothum.thodi@huawei.com>
TSM vDevice support needs access to the KVM associated with a VFIO device
after the device has been bound to iommufd.
Extend iommufd_device_bind() to accept the device's KVM file and store it
in the iommufd_device. The KVM file reference is owned by VFIO and is
already held for the duration of the device open path.
Signed-off-by: Shameer Kolothum <shameerali.kolothum.thodi@huawei.com>
Reviewed-by: Jason Gunthorpe <jgg@nvidia.com>
[nicolinc: fix build error in iommufd_test_mock_domain()]
Signed-off-by: Nicolin Chen <nicolinc@nvidia.com>
[aneesh.kumar: Switch to use kvm_file]
Cc: jgg@ziepe.ca
Cc: kevin.tian@intel.com
Cc: joro@8bytes.org
Cc: will@kernel.org
Cc: robin.murphy@arm.com
Cc: alex@shazbot.org
Cc: kvm@vger.kernel.org
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/iommu/iommufd/device.c | 7 ++++++-
drivers/iommu/iommufd/iommufd_private.h | 2 ++
drivers/iommu/iommufd/selftest.c | 2 +-
drivers/vfio/iommufd.c | 3 ++-
include/linux/iommufd.h | 4 +++-
5 files changed, 14 insertions(+), 4 deletions(-)
diff --git a/drivers/iommu/iommufd/device.c b/drivers/iommu/iommufd/device.c
index a664c70a6fe7..f99f7bee2e77 100644
--- a/drivers/iommu/iommufd/device.c
+++ b/drivers/iommu/iommufd/device.c
@@ -295,6 +295,7 @@ static int iommufd_bind_noiommu(struct iommufd_device *idev)
* iommufd_device_bind - Bind a physical device to an iommu fd
* @ictx: iommufd file descriptor
* @dev: Pointer to a physical device struct
+ * @kvm_file: VM file if device belongs to a KVM VM
* @id: Output ID number to return to userspace for this device
*
* A successful bind establishes an ownership over the device and returns
@@ -308,7 +309,9 @@ static int iommufd_bind_noiommu(struct iommufd_device *idev)
* The caller must undo this with iommufd_device_unbind()
*/
struct iommufd_device *iommufd_device_bind(struct iommufd_ctx *ictx,
- struct device *dev, u32 *id)
+ struct device *dev,
+ struct file *kvm_file,
+ u32 *id)
{
struct iommufd_device *idev;
int rc;
@@ -319,6 +322,8 @@ struct iommufd_device *iommufd_device_bind(struct iommufd_ctx *ictx,
idev->ictx = ictx;
idev->dev = dev;
+ /* VFIO holds the file reference while the device is open. */
+ idev->kvm_file = kvm_file;
if (!iommufd_device_is_noiommu(idev))
rc = iommufd_bind_iommu(idev);
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index eb2e85b27e42..2560b4faa6f9 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -509,6 +509,8 @@ struct iommufd_device {
struct list_head group_item;
/* always the physical device */
struct device *dev;
+ /* ..and the VM file if available */
+ struct file *kvm_file;
bool enforce_cache_coherency;
struct iommufd_vdevice *vdev;
bool destroying;
diff --git a/drivers/iommu/iommufd/selftest.c b/drivers/iommu/iommufd/selftest.c
index ee706f18f7e9..e9c825b24356 100644
--- a/drivers/iommu/iommufd/selftest.c
+++ b/drivers/iommu/iommufd/selftest.c
@@ -1074,7 +1074,7 @@ static int iommufd_test_mock_domain(struct iommufd_ucmd *ucmd,
goto out_sobj;
}
- idev = iommufd_device_bind(ucmd->ictx, &sobj->idev.mock_dev->dev,
+ idev = iommufd_device_bind(ucmd->ictx, &sobj->idev.mock_dev->dev, NULL,
&idev_id);
if (IS_ERR(idev)) {
rc = PTR_ERR(idev);
diff --git a/drivers/vfio/iommufd.c b/drivers/vfio/iommufd.c
index e9893d34d07b..3beb1b1bae36 100644
--- a/drivers/vfio/iommufd.c
+++ b/drivers/vfio/iommufd.c
@@ -123,7 +123,8 @@ int vfio_iommufd_physical_bind(struct vfio_device *vdev,
{
struct iommufd_device *idev;
- idev = iommufd_device_bind(ictx, vdev->dev, out_device_id);
+ idev = iommufd_device_bind(ictx, vdev->dev, vdev->kvm,
+ out_device_id);
if (IS_ERR(idev))
return PTR_ERR(idev);
vdev->iommufd_device = idev;
diff --git a/include/linux/iommufd.h b/include/linux/iommufd.h
index 6e7efe83bc5d..0a0bb4abfbd2 100644
--- a/include/linux/iommufd.h
+++ b/include/linux/iommufd.h
@@ -59,7 +59,9 @@ struct iommufd_object {
};
struct iommufd_device *iommufd_device_bind(struct iommufd_ctx *ictx,
- struct device *dev, u32 *id);
+ struct device *dev,
+ struct file *kvm_file,
+ u32 *id);
void iommufd_device_unbind(struct iommufd_device *idev);
int iommufd_device_attach(struct iommufd_device *idev, ioasid_t pasid,
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 08/16] iommufd/viommu: Keep a reference to the KVM file
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (6 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 07/16] iommufd/device: Associate KVM file pointer with iommufd_device Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 09/16] tsm: Remove the device from lookup before PCI teardown Aneesh Kumar K.V (Arm)
` (7 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
From: Nicolin Chen <nicolinc@nvidia.com>
The TSM vDevice operations need access to the KVM associated with the
device's vIOMMU. Save the device's KVM file in the iommufd_viommu when the
vIOMMU is allocated, and take a file reference so it remains valid for the
lifetime of the vIOMMU.
Release the reference when the vIOMMU is destroyed.
Based on an original patch by
Shameer Kolothum <shameerali.kolothum.thodi@huawei.com>
[nicolinc: hold kvm's users_count]
Signed-off-by: Nicolin Chen <nicolinc@nvidia.com>
[aneesh.kumar: Switch to use kvm_file]
Cc: jgg@ziepe.ca
Cc: kevin.tian@intel.com
Cc: joro@8bytes.org
Cc: will@kernel.org
Cc: robin.murphy@arm.com
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/iommu/iommufd/viommu.c | 5 +++++
include/linux/iommufd.h | 1 +
2 files changed, 6 insertions(+)
diff --git a/drivers/iommu/iommufd/viommu.c b/drivers/iommu/iommufd/viommu.c
index f7951057a1e5..b7489ed259bd 100644
--- a/drivers/iommu/iommufd/viommu.c
+++ b/drivers/iommu/iommufd/viommu.c
@@ -1,6 +1,7 @@
// SPDX-License-Identifier: GPL-2.0-only
/* Copyright (c) 2024, NVIDIA CORPORATION & AFFILIATES
*/
+#include <linux/file.h>
#include "iommufd_private.h"
void iommufd_viommu_destroy(struct iommufd_object *obj)
@@ -11,6 +12,8 @@ void iommufd_viommu_destroy(struct iommufd_object *obj)
if (viommu->ops && viommu->ops->destroy)
viommu->ops->destroy(viommu);
refcount_dec(&viommu->hwpt->common.obj.users);
+ if (viommu->kvm_file)
+ fput(viommu->kvm_file);
xa_destroy(&viommu->vdevs);
}
@@ -82,6 +85,8 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
}
xa_init(&viommu->vdevs);
+ if (idev->kvm_file)
+ viommu->kvm_file = get_file(idev->kvm_file);
viommu->type = cmd->type;
viommu->ictx = ucmd->ictx;
viommu->hwpt = hwpt_paging;
diff --git a/include/linux/iommufd.h b/include/linux/iommufd.h
index 0a0bb4abfbd2..3267717f676d 100644
--- a/include/linux/iommufd.h
+++ b/include/linux/iommufd.h
@@ -103,6 +103,7 @@ struct iommufd_viommu {
struct iommufd_ctx *ictx;
struct iommu_device *iommu_dev;
struct iommufd_hwpt_paging *hwpt;
+ struct file *kvm_file;
const struct iommufd_viommu_ops *ops;
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 09/16] tsm: Remove the device from lookup before PCI teardown
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (7 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 08/16] iommufd/viommu: Keep a reference to the KVM file Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 10/16] iommufd: Add the vdevice TSM request ioctl Aneesh Kumar K.V (Arm)
` (6 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
PCI connect looks up a TSM by ID under pci_tsm_rwsem. Previously,
tsm_unregister() released that lock after tearing down PCI state while
the TSM was still in the class lookup. A racing connect could then
attach a new PCI context that the teardown would never see.
Remove the device from the class lookup first, then take the PCI write
lock to drain in-flight operations and destroy their contexts. Drop the
device reference only after PCI teardown completes.
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/virt/coco/tsm-core.c | 22 ++++++++++++++--------
1 file changed, 14 insertions(+), 8 deletions(-)
diff --git a/drivers/virt/coco/tsm-core.c b/drivers/virt/coco/tsm-core.c
index e784993353d8..f79135986102 100644
--- a/drivers/virt/coco/tsm-core.c
+++ b/drivers/virt/coco/tsm-core.c
@@ -56,26 +56,25 @@ static struct tsm_dev *alloc_tsm_dev(struct device *parent)
return no_free_ptr(tsm_dev);
}
-static struct tsm_dev *tsm_register_pci_or_reset(struct tsm_dev *tsm_dev,
- struct pci_tsm_ops *pci_ops)
+static int tsm_register_pci(struct tsm_dev *tsm_dev, struct pci_tsm_ops *pci_ops)
{
int rc;
if (!pci_ops)
- return tsm_dev;
+ return 0;
tsm_dev->pci_ops = pci_ops;
rc = pci_tsm_register(tsm_dev);
if (rc) {
+ tsm_dev->pci_ops = NULL;
dev_err(tsm_dev->dev.parent,
"PCI/TSM registration failure: %d\n", rc);
- device_unregister(&tsm_dev->dev);
- return ERR_PTR(rc);
+ return rc;
}
/* Notify TSM userspace that PCI/TSM operations are now possible */
kobject_uevent(&tsm_dev->dev.kobj, KOBJ_CHANGE);
- return tsm_dev;
+ return 0;
}
struct tsm_dev *tsm_register(struct device *parent, struct pci_tsm_ops *pci_ops)
@@ -96,15 +95,22 @@ struct tsm_dev *tsm_register(struct device *parent, struct pci_tsm_ops *pci_ops)
if (rc)
return ERR_PTR(rc);
- return tsm_register_pci_or_reset(no_free_ptr(tsm_dev), pci_ops);
+ rc = tsm_register_pci(tsm_dev, pci_ops);
+ if (rc) {
+ device_del(dev);
+ return ERR_PTR(rc);
+ }
+ return no_free_ptr(tsm_dev);
}
EXPORT_SYMBOL_GPL(tsm_register);
void tsm_unregister(struct tsm_dev *tsm_dev)
{
+ /* Remove the class lookup first. */
+ device_del(&tsm_dev->dev);
if (tsm_dev->pci_ops)
pci_tsm_unregister(tsm_dev);
- device_unregister(&tsm_dev->dev);
+ put_device(&tsm_dev->dev);
}
EXPORT_SYMBOL_GPL(tsm_unregister);
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 10/16] iommufd: Add the vdevice TSM request ioctl
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (8 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 09/16] tsm: Remove the device from lookup before PCI teardown Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 11/16] PCI/TSM: Remove the legacy guest request interface Aneesh Kumar K.V (Arm)
` (5 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
A VMM needs to forward guest-originated TSM commands to the TSM
implementation managing an assigned device. The IOMMUFD vdevice
identifies that device within its vIOMMU and provides the appropriate
dispatch point.
Add IOMMU_VDEVICE_TSM_REQ to send an opaque request to a vdevice and
optionally return a response. Define common operation and guest
architecture identifiers for CCA, SEV and TDX requests.
Let vdevice providers declare the TVM architecture and supported operations
in a u64 mask. Check these capabilities before dispatch. Define a
one-past-last operation value and assert that every operation fits in the
mask; reject out-of-range userspace values before indexing it.
The ioctl return value reports errors or the number of unused buffer bytes,
while tsm_code carries the TSM-specific result. Use userspace pointers for
the internal request and response buffers, and u32 lengths matching the
UAPI. Reject lengths above INT_MAX so successful residues fit in the
ioctl's int return value.
Cc: jgg@ziepe.ca
Cc: kevin.tian@intel.com
Cc: joro@8bytes.org
Cc: will@kernel.org
Cc: robin.murphy@arm.com
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/iommu/iommufd/Makefile | 2 +
drivers/iommu/iommufd/iommufd_private.h | 8 +++
drivers/iommu/iommufd/main.c | 3 +
drivers/iommu/iommufd/tsm.c | 80 ++++++++++++++++++++++++
drivers/iommu/iommufd/viommu.c | 3 +
include/linux/iommufd.h | 9 +++
include/linux/tsm.h | 23 +++++++
include/uapi/linux/iommufd.h | 82 +++++++++++++++++++++++++
8 files changed, 210 insertions(+)
create mode 100644 drivers/iommu/iommufd/tsm.c
diff --git a/drivers/iommu/iommufd/Makefile b/drivers/iommu/iommufd/Makefile
index 67207914bb6e..f90efe2f9d10 100644
--- a/drivers/iommu/iommufd/Makefile
+++ b/drivers/iommu/iommufd/Makefile
@@ -11,6 +11,8 @@ iommufd-y := \
viommu.o
iommufd-$(CONFIG_IOMMUFD_NOIOMMU) += hwpt_noiommu.o
+iommufd-$(CONFIG_TSM) += tsm.o
+
iommufd-$(CONFIG_IOMMUFD_TEST) += selftest.o
obj-$(CONFIG_IOMMUFD) += iommufd.o
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index 2560b4faa6f9..64e0d9e4da16 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -730,6 +730,14 @@ void iommufd_vdevice_destroy(struct iommufd_object *obj);
void iommufd_vdevice_abort(struct iommufd_object *obj);
int iommufd_hw_queue_alloc_ioctl(struct iommufd_ucmd *ucmd);
void iommufd_hw_queue_destroy(struct iommufd_object *obj);
+#ifdef CONFIG_TSM
+int iommufd_vdevice_tsm_req_ioctl(struct iommufd_ucmd *ucmd);
+#else
+static inline int iommufd_vdevice_tsm_req_ioctl(struct iommufd_ucmd *ucmd)
+{
+ return -EOPNOTSUPP;
+}
+#endif
#ifdef CONFIG_IOMMUFD_TEST
int iommufd_test(struct iommufd_ucmd *ucmd);
diff --git a/drivers/iommu/iommufd/main.c b/drivers/iommu/iommufd/main.c
index 9a921b153162..577eef2760a3 100644
--- a/drivers/iommu/iommufd/main.c
+++ b/drivers/iommu/iommufd/main.c
@@ -453,6 +453,7 @@ union ucmd_buffer {
struct iommu_veventq_alloc veventq;
struct iommu_vfio_ioas vfio_ioas;
struct iommu_viommu_alloc viommu;
+ struct iommu_vdevice_tsm_req tsm_req;
#ifdef CONFIG_IOMMUFD_TEST
struct iommu_test_cmd test;
#endif
@@ -516,6 +517,8 @@ static const struct iommufd_ioctl_op iommufd_ioctl_ops[] = {
__reserved),
IOCTL_OP(IOMMU_VIOMMU_ALLOC, iommufd_viommu_alloc_ioctl,
struct iommu_viommu_alloc, out_viommu_id),
+ IOCTL_OP(IOMMU_VDEVICE_TSM_REQ, iommufd_vdevice_tsm_req_ioctl,
+ struct iommu_vdevice_tsm_req, out_tsm_code),
#ifdef CONFIG_IOMMUFD_TEST
IOCTL_OP(IOMMU_TEST_CMD, iommufd_test, struct iommu_test_cmd, last),
#endif
diff --git a/drivers/iommu/iommufd/tsm.c b/drivers/iommu/iommufd/tsm.c
new file mode 100644
index 000000000000..b28ff8640345
--- /dev/null
+++ b/drivers/iommu/iommufd/tsm.c
@@ -0,0 +1,80 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Copyright (C) 2026 ARM Ltd.
+ */
+
+#include <linux/bits.h>
+#include <linux/build_bug.h>
+#include <linux/limits.h>
+#include <linux/tsm.h>
+#include "iommufd_private.h"
+
+/**
+ * iommufd_vdevice_tsm_req_ioctl - Forward TSM requests
+ * @ucmd: user command data for IOMMU_VDEVICE_TSM_REQ
+ *
+ * Resolve @iommu_vdevice_tsm_req::vdevice_id to a vdevice and pass the
+ * request/response buffers to its vIOMMU implementation.
+ *
+ * Return:
+ * -errno on error.
+ * positive residue if response/request bytes were left unconsumed.
+ * if response buffer is provided, residue indicates the number of bytes
+ * not used in response buffer
+ * if there is no response buffer, residue indicates the number of bytes
+ * not consumed in req buffer
+ * 0 otherwise.
+ */
+int iommufd_vdevice_tsm_req_ioctl(struct iommufd_ucmd *ucmd)
+{
+ int ret;
+ struct iommufd_object *obj;
+ struct iommufd_vdevice *vdev;
+ struct iommu_vdevice_tsm_req *cmd = ucmd->cmd;
+ struct tsm_guest_req_info info = {
+ .op = cmd->op,
+ .tvm_arch = cmd->tvm_arch,
+ .req = u64_to_user_ptr(cmd->req_uptr),
+ .req_len = cmd->req_len,
+ .resp = u64_to_user_ptr(cmd->resp_uptr),
+ .resp_len = cmd->resp_len,
+ };
+
+ /* Every defined operation must fit in the vdevice's u64 mask. */
+ BUILD_BUG_ON(BITS_PER_TYPE(u64) < TSM_REQ_MAX);
+
+ /* Successful residues must fit in the ioctl's int return value. */
+ if (cmd->req_len > INT_MAX || cmd->resp_len > INT_MAX)
+ return -EINVAL;
+
+ obj = iommufd_get_object(ucmd->ictx, cmd->vdevice_id,
+ IOMMUFD_OBJ_VDEVICE);
+ if (IS_ERR(obj))
+ return PTR_ERR(obj);
+ vdev = container_of(obj, struct iommufd_vdevice, obj);
+
+ cmd->out_tsm_code = 0;
+ if (!vdev->viommu->ops || !vdev->viommu->ops->vdevice_tsm_req ||
+ !vdev->tsm_req_op_mask) {
+ ret = -EOPNOTSUPP;
+ goto out_respond;
+ }
+ /* Bounds-check the user-supplied shift before BIT_ULL(). */
+ if (cmd->tvm_arch != vdev->tsm_tvm_arch ||
+ cmd->op >= TSM_REQ_MAX ||
+ !(vdev->tsm_req_op_mask & BIT_ULL(cmd->op))) {
+ ret = -EINVAL;
+ goto out_put_object;
+ }
+ ret = vdev->viommu->ops->vdevice_tsm_req(vdev, &info,
+ &cmd->out_tsm_code);
+
+out_respond:
+ /* Always copy the tsm_code as response */
+ if (iommufd_ucmd_respond(ucmd, sizeof(*cmd)))
+ ret = -EFAULT;
+
+out_put_object:
+ iommufd_put_object(ucmd->ictx, obj);
+ return ret;
+}
diff --git a/drivers/iommu/iommufd/viommu.c b/drivers/iommu/iommufd/viommu.c
index b7489ed259bd..f3d5b5a7eb4a 100644
--- a/drivers/iommu/iommufd/viommu.c
+++ b/drivers/iommu/iommufd/viommu.c
@@ -2,6 +2,9 @@
/* Copyright (c) 2024, NVIDIA CORPORATION & AFFILIATES
*/
#include <linux/file.h>
+#include <linux/cleanup.h>
+#include <linux/tsm.h>
+
#include "iommufd_private.h"
void iommufd_viommu_destroy(struct iommufd_object *obj)
diff --git a/include/linux/iommufd.h b/include/linux/iommufd.h
index 3267717f676d..cfcf53b7c9e8 100644
--- a/include/linux/iommufd.h
+++ b/include/linux/iommufd.h
@@ -24,6 +24,7 @@ struct iommufd_ctx;
struct iommufd_device;
struct iommufd_viommu_ops;
struct page;
+struct tsm_guest_req_info;
enum iommufd_object_type {
IOMMUFD_OBJ_NONE,
@@ -125,6 +126,10 @@ struct iommufd_vdevice {
*/
u64 virt_id;
+ /* Guest TSM requests accepted by this vdevice; set by vdevice_init(). */
+ u64 tsm_req_op_mask;
+ u32 tsm_tvm_arch;
+
/* Clean up all driver-specific parts of an iommufd_vdevice */
void (*destroy)(struct iommufd_vdevice *vdev);
};
@@ -167,6 +172,7 @@ struct iommufd_hw_queue {
* include/uapi/linux/iommufd.h)
* If driver has a deinit function to revert what vdevice_init op
* does, it should set it to the @vdev->destroy function pointer
+ * @vdevice_tsm_req: Forward a guest TSM request to a driver-owned vDEVICE
* @get_hw_queue_size: Get the size of a driver-defined HW queue structure for a
* given @viommu corresponding to @queue_type. Driver should
* return 0 if HW queue aren't supported accordingly. It is
@@ -193,6 +199,9 @@ struct iommufd_viommu_ops {
struct iommu_user_data_array *array);
const size_t vdevice_size;
int (*vdevice_init)(struct iommufd_vdevice *vdev);
+ ssize_t (*vdevice_tsm_req)(struct iommufd_vdevice *vdev,
+ struct tsm_guest_req_info *info,
+ u64 *tsm_code);
size_t (*get_hw_queue_size)(struct iommufd_viommu *viommu,
enum iommu_hw_queue_type queue_type);
/* AMD's HW will add hw_queue_init simply using @hw_queue->base_addr */
diff --git a/include/linux/tsm.h b/include/linux/tsm.h
index 381c53244c83..4bce4c9fa6f0 100644
--- a/include/linux/tsm.h
+++ b/include/linux/tsm.h
@@ -6,6 +6,7 @@
#include <linux/types.h>
#include <linux/uuid.h>
#include <linux/device.h>
+#include <uapi/linux/iommufd.h>
#define TSM_REPORT_INBLOB_MAX 64
#define TSM_REPORT_OUTBLOB_MAX SZ_16M
@@ -123,4 +124,26 @@ int tsm_report_unregister(const struct tsm_report_ops *ops);
struct tsm_dev *tsm_register(struct device *parent, struct pci_tsm_ops *ops);
void tsm_unregister(struct tsm_dev *tsm_dev);
struct tsm_dev *find_tsm_dev(int id);
+
+#ifdef CONFIG_TSM
+/**
+ * struct tsm_guest_req_info - parameters for a guest-initiated TSM request
+ * @op: operation for the guest-initiated request
+ * @tvm_arch: guest TVM architecture
+ * @req: userspace buffer containing the guest request
+ * @req_len: request size in bytes, at most INT_MAX
+ * @resp: userspace buffer for the response
+ * @resp_len: response buffer capacity in bytes, at most INT_MAX
+ */
+struct tsm_guest_req_info {
+ enum iommu_vdevice_tsm_guest_req_op op;
+ enum iommu_vdevice_tsm_guest_tvm_arch tvm_arch;
+ const void __user *req;
+ u32 req_len;
+ void __user *resp;
+ u32 resp_len;
+};
+#else
+struct tsm_guest_req_info;
+#endif
#endif /* __TSM_H */
diff --git a/include/uapi/linux/iommufd.h b/include/uapi/linux/iommufd.h
index 206fa667c782..9920138f9eda 100644
--- a/include/uapi/linux/iommufd.h
+++ b/include/uapi/linux/iommufd.h
@@ -58,6 +58,7 @@ enum {
IOMMUFD_CMD_VEVENTQ_ALLOC = 0x93,
IOMMUFD_CMD_HW_QUEUE_ALLOC = 0x94,
IOMMUFD_CMD_IOAS_NOIOMMU_GET_PA = 0x95,
+ IOMMUFD_CMD_VDEVICE_TSM_REQ = 0x96,
};
/**
@@ -1390,4 +1391,85 @@ struct iommu_hw_queue_alloc {
__aligned_u64 length;
};
#define IOMMU_HW_QUEUE_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_HW_QUEUE_ALLOC)
+
+/**
+ * enum iommu_vdevice_tsm_guest_tvm_arch - guest TVM architecture
+ * @IOMMU_VDEVICE_TSM_TVM_ARCH_CCA: Arm CCA TVM
+ * @IOMMU_VDEVICE_TSM_TVM_ARCH_SEV: AMD SEV TVM
+ * @IOMMU_VDEVICE_TSM_TVM_ARCH_TDX: Intel TDX TVM
+ */
+enum iommu_vdevice_tsm_guest_tvm_arch {
+ IOMMU_VDEVICE_TSM_TVM_ARCH_CCA = 1,
+ IOMMU_VDEVICE_TSM_TVM_ARCH_SEV,
+ IOMMU_VDEVICE_TSM_TVM_ARCH_TDX,
+};
+
+/**
+ * enum iommu_vdevice_tsm_guest_req_op - Guest TSM request operations
+ * @TSM_REQ_VALIDATE_MMIO: Validate and map a guest MMIO range for the trusted
+ * device.
+ * @TSM_REQ_SET_TDI_STATE: Transition the trusted device to the unlocked,
+ * locked, or running state.
+ * @TSM_REQ_SEV_ENABLE_DMA: Enable DMA for an SEV device.
+ * @TSM_REQ_SEV_DISABLE_DMA: Disable DMA for an SEV device.
+ * @TSM_REQ_READ_OBJECT: Read bytes from a device object, such as a certificate,
+ * measurement, or interface report.
+ * @TSM_REQ_REGEN_OBJECT: Regenerate a device object, such as a measurement or
+ * interface report.
+ * @TSM_REQ_OBJECT_INFO: Query the size of a device object.
+ * @TSM_REQ_MAX: One past the last request operation; not a command.
+ *
+ * The request payload and response format depend on the TVM architecture.
+ */
+enum iommu_vdevice_tsm_guest_req_op {
+ TSM_REQ_VALIDATE_MMIO = 1,
+ TSM_REQ_SET_TDI_STATE,
+ TSM_REQ_SEV_ENABLE_DMA,
+ TSM_REQ_SEV_DISABLE_DMA,
+ TSM_REQ_READ_OBJECT,
+ TSM_REQ_REGEN_OBJECT,
+ TSM_REQ_OBJECT_INFO,
+ TSM_REQ_MAX,
+};
+
+/**
+ * struct iommu_vdevice_tsm_req - ioctl(IOMMU_VDEVICE_TSM_REQ)
+ * @size: sizeof(struct iommu_vdevice_tsm_req)
+ * @vdevice_id: vDevice ID the guest request is for
+ * @op: One of enum iommu_vdevice_tsm_guest_req_op, excluding TSM_REQ_MAX
+ * @tvm_arch: One of enum iommu_vdevice_tsm_guest_tvm_arch
+ * @req_len: Size in bytes of the input payload at @req_uptr, at most INT_MAX
+ * @resp_len: Size in bytes of the output buffer at @resp_uptr, at most INT_MAX
+ * @req_uptr: Userspace pointer to the guest-provided request payload
+ * @resp_uptr: Userspace pointer to the guest response buffer
+ * @out_tsm_code: TSM-specific result code returned by the TSM implementation
+ *
+ * Forward a TSM request to the TSM bound vDevice. This is intended for
+ * guest TSM/TDISP message transport where the host kernel only marshals
+ * bytes between userspace and the TSM implementation.
+ *
+ * The request operation is guest initiated. The vDEVICE implementation
+ * declares the TVM architecture and operations it accepts; iommufd checks
+ * both before forwarding the request.
+ *
+ * The request payload is read from @req_uptr/@req_len. If a response is
+ * expected, userspace provides @resp_uptr/@resp_len as writable storage for
+ * response bytes returned by the TSM path.
+ *
+ * The ioctl is only suitable for commands and results that the host kernel
+ * has no use, the host is only facilitating guest to TSM communication.
+ */
+struct iommu_vdevice_tsm_req {
+ __u32 size;
+ __u32 vdevice_id;
+ __u32 op;
+ __u32 tvm_arch;
+ __u32 req_len;
+ __u32 resp_len;
+ __aligned_u64 req_uptr;
+ __aligned_u64 resp_uptr;
+ __aligned_u64 out_tsm_code;
+};
+
+#define IOMMU_VDEVICE_TSM_REQ _IO(IOMMUFD_TYPE, IOMMUFD_CMD_VDEVICE_TSM_REQ)
#endif
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 11/16] PCI/TSM: Remove the legacy guest request interface
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (9 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 10/16] iommufd: Add the vdevice TSM request ioctl Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 12/16] PCI/TSM: Add vIOMMU-bound contexts for vdevices Aneesh Kumar K.V (Arm)
` (4 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
Guest TSM requests can now be dispatched through the IOMMUFD vdevice
operation. Remove the PCI-device-based guest request entry point, its
scope enum and the corresponding PCI/TSM driver callback.
Cc: bhelgaas@google.com
Cc: linux-pci@vger.kernel.org
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/pci/tsm.c | 60 ----------------------------------------
include/linux/pci-tsm.h | 61 +----------------------------------------
2 files changed, 1 insertion(+), 120 deletions(-)
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index 5fdcd7f2e820..1423c64dc9c8 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -375,66 +375,6 @@ int pci_tsm_bind(struct pci_dev *pdev, struct kvm *kvm, u32 tdi_id)
}
EXPORT_SYMBOL_GPL(pci_tsm_bind);
-/**
- * pci_tsm_guest_req() - helper to marshal guest requests to the TSM driver
- * @pdev: @pdev representing a bound tdi
- * @scope: caller asserts this passthrough request is limited to TDISP operations
- * @req_in: Input payload forwarded from the guest
- * @in_len: Length of @req_in
- * @req_out: Output payload buffer response to the guest
- * @out_len: Length of @req_out on input, bytes filled in @req_out on output
- * @tsm_code: Optional TSM arch specific result code for the guest TSM
- *
- * This is a common entry point for requests triggered by userspace KVM-exit
- * service handlers responding to TDI information or state change requests. The
- * scope parameter limits requests to TDISP state management, or limited debug.
- * This path is only suitable for commands and results that are the host kernel
- * has no use, the host is only facilitating guest to TSM communication.
- *
- * Returns 0 on success and -error on failure and positive "residue" on success
- * but @req_out is filled with less then @out_len, or @req_out is NULL and a
- * residue number of bytes were not consumed from @req_in. On success or
- * failure @tsm_code may be populated with a TSM implementation specific result
- * code for the guest to consume.
- *
- * Context: Caller is responsible for calling this within the pci_tsm_bind()
- * state of the TDI.
- */
-ssize_t pci_tsm_guest_req(struct pci_dev *pdev, enum pci_tsm_req_scope scope,
- sockptr_t req_in, size_t in_len, sockptr_t req_out,
- size_t out_len, u64 *tsm_code)
-{
- struct pci_tsm_pf0 *tsm_pf0;
- struct pci_tdi *tdi;
- int rc;
-
- /* Forbid requests that are not directly related to TDISP operations */
- if (scope > PCI_TSM_REQ_STATE_CHANGE)
- return -EINVAL;
-
- ACQUIRE(rwsem_read_intr, lock)(&pci_tsm_rwsem);
- if ((rc = ACQUIRE_ERR(rwsem_read_intr, &lock)))
- return rc;
-
- if (!pdev->tsm)
- return -ENXIO;
-
- if (!is_link_tsm(pdev->tsm->tsm_dev))
- return -ENXIO;
-
- tsm_pf0 = to_pci_tsm_pf0(pdev->tsm);
- ACQUIRE(mutex_intr, ops_lock)(&tsm_pf0->lock);
- if ((rc = ACQUIRE_ERR(mutex_intr, &ops_lock)))
- return rc;
-
- tdi = pdev->tsm->tdi;
- if (!tdi)
- return -ENXIO;
- return to_pci_tsm_ops(pdev->tsm)->guest_req(tdi, scope, req_in, in_len,
- req_out, out_len, tsm_code);
-}
-EXPORT_SYMBOL_GPL(pci_tsm_guest_req);
-
static void pci_tsm_unbind_all(struct pci_dev *pdev)
{
pci_tsm_walk_fns_reverse(pdev, __pci_tsm_unbind, NULL);
diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
index a6435aba03f9..6fe3d0ea875f 100644
--- a/include/linux/pci-tsm.h
+++ b/include/linux/pci-tsm.h
@@ -3,12 +3,10 @@
#define __PCI_TSM_H
#include <linux/mutex.h>
#include <linux/pci.h>
-#include <linux/sockptr.h>
struct pci_tsm;
struct tsm_dev;
struct kvm;
-enum pci_tsm_req_scope;
/*
* struct pci_tsm_ops - manage confidential links and security state
@@ -34,14 +32,13 @@ struct pci_tsm_ops {
* @disconnect: teardown the secure link
* @bind: bind a TDI in preparation for it to be accepted by a TVM
* @unbind: remove a TDI from secure operation with a TVM
- * @guest_req: marshal TVM information and state change requests
*
* Context: @probe, @remove, @connect, and @disconnect run under
* pci_tsm_rwsem held for write to sync with TSM unregistration and
* mutual exclusion of @connect and @disconnect. @connect and
* @disconnect additionally run under the DSM lock (struct
* pci_tsm_pf0::lock) as well as @probe and @remove of the subfunctions.
- * @bind, @unbind, and @guest_req run under pci_tsm_rwsem held for read
+ * @bind and @unbind run under pci_tsm_rwsem held for read
* and the DSM lock.
*/
struct_group_tagged(pci_tsm_link_ops, link_ops,
@@ -53,11 +50,6 @@ struct pci_tsm_ops {
struct pci_tdi *(*bind)(struct pci_dev *pdev,
struct kvm *kvm, u32 tdi_id);
void (*unbind)(struct pci_tdi *tdi);
- ssize_t (*guest_req)(struct pci_tdi *tdi,
- enum pci_tsm_req_scope scope,
- sockptr_t req_in, size_t in_len,
- sockptr_t req_out, size_t out_len,
- u64 *tsm_code);
);
/*
@@ -159,46 +151,6 @@ static inline bool is_pci_tsm_pf0(struct pci_dev *pdev)
return PCI_FUNC(pdev->devfn) == 0;
}
-/**
- * enum pci_tsm_req_scope - Scope of guest requests to be validated by TSM
- *
- * Guest requests are a transport for a TVM to communicate with a TSM + DSM for
- * a given TDI. A TSM driver is responsible for maintaining the kernel security
- * model and limit commands that may affect the host, or are otherwise outside
- * the typical TDISP operational model.
- */
-enum pci_tsm_req_scope {
- /**
- * @PCI_TSM_REQ_INFO: Read-only, without side effects, request for
- * typical TDISP collateral information like Device Interface Reports.
- * No device secrets are permitted, and no device state is changed.
- */
- PCI_TSM_REQ_INFO = 0,
- /**
- * @PCI_TSM_REQ_STATE_CHANGE: Request to change the TDISP state from
- * UNLOCKED->LOCKED, LOCKED->RUN, or other architecture specific state
- * changes to support those transitions for a TDI. No other (unrelated
- * to TDISP) device / host state, configuration, or data change is
- * permitted.
- */
- PCI_TSM_REQ_STATE_CHANGE = 1,
- /**
- * @PCI_TSM_REQ_DEBUG_READ: Read-only request for debug information
- *
- * A method to facilitate TVM information retrieval outside of typical
- * TDISP operational requirements. No device secrets are permitted.
- */
- PCI_TSM_REQ_DEBUG_READ = 2,
- /**
- * @PCI_TSM_REQ_DEBUG_WRITE: Device state changes for debug purposes
- *
- * The request may affect the operational state of the device outside of
- * the TDISP operational model. If allowed, requires CAP_SYS_RAW_IO, and
- * will taint the kernel.
- */
- PCI_TSM_REQ_DEBUG_WRITE = 3,
-};
-
#ifdef CONFIG_PCI_TSM
int pci_tsm_register(struct tsm_dev *tsm_dev);
void pci_tsm_unregister(struct tsm_dev *tsm_dev);
@@ -213,9 +165,6 @@ int pci_tsm_bind(struct pci_dev *pdev, struct kvm *kvm, u32 tdi_id);
void pci_tsm_unbind(struct pci_dev *pdev);
void pci_tsm_tdi_constructor(struct pci_dev *pdev, struct pci_tdi *tdi,
struct kvm *kvm, u32 tdi_id);
-ssize_t pci_tsm_guest_req(struct pci_dev *pdev, enum pci_tsm_req_scope scope,
- sockptr_t req_in, size_t in_len, sockptr_t req_out,
- size_t out_len, u64 *tsm_code);
#else
static inline int pci_tsm_register(struct tsm_dev *tsm_dev)
{
@@ -231,13 +180,5 @@ static inline int pci_tsm_bind(struct pci_dev *pdev, struct kvm *kvm, u64 tdi_id
static inline void pci_tsm_unbind(struct pci_dev *pdev)
{
}
-static inline ssize_t pci_tsm_guest_req(struct pci_dev *pdev,
- enum pci_tsm_req_scope scope,
- sockptr_t req_in, size_t in_len,
- sockptr_t req_out, size_t out_len,
- u64 *tsm_code)
-{
- return -ENXIO;
-}
#endif
#endif /*__PCI_TSM_H */
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 12/16] PCI/TSM: Add vIOMMU-bound contexts for vdevices
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (10 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 11/16] PCI/TSM: Remove the legacy guest request interface Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 13/16] iommufd/viommu: Select vIOMMU operations before allocation Aneesh Kumar K.V (Arm)
` (3 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
An IOMMUFD vdevice must use the same TEE Security Manager as its vIOMMU.
Pinning the vIOMMU backend keeps its tsm_dev registered, but does not
preserve the connection between the PF0 PCI device and that TSM.
Add a per-vdevice PCI TSM context. Verify that the PCI device and the
vIOMMU use the same tsm_dev, and account active contexts on PF0. Reject
disconnect with -EBUSY until all dependent vdevices release their
contexts.
With vdevice binding handled by the backend, PCI/TSM no longer has
generic bind state to report. Remove the legacy bind and unbind
interface. The sysfs "bound" attribute reported the state maintained by
that interface. Remove the attribute and its ABI documentation as well.
Cc: bhelgaas@google.com
Cc: linux-pci@vger.kernel.org
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
Documentation/ABI/testing/sysfs-bus-pci | 17 +-
drivers/pci/tsm.c | 230 ++++++++++--------------
include/linux/pci-tsm.h | 71 +++++---
3 files changed, 137 insertions(+), 181 deletions(-)
diff --git a/Documentation/ABI/testing/sysfs-bus-pci b/Documentation/ABI/testing/sysfs-bus-pci
index 55ea1db749a1..4c64b30544cd 100644
--- a/Documentation/ABI/testing/sysfs-bus-pci
+++ b/Documentation/ABI/testing/sysfs-bus-pci
@@ -654,6 +654,9 @@ Contact: linux-coco@lists.linux.dev
Description:
(WO) Write the name of the TSM device that was specified
to 'connect' to teardown the connection.
+ The write fails with EBUSY while any vdevice depends on the
+ connection. Userspace must destroy those vdevices before
+ disconnecting the link.
What: /sys/bus/pci/devices/.../tsm/dsm
Contact: linux-coco@lists.linux.dev
@@ -671,20 +674,6 @@ Description: (RO) Return PCI device name of this device's DSM (Device
PCIe switch port. This is a "link" TSM attribute, see
Documentation/ABI/testing/sysfs-class-tsm.
-What: /sys/bus/pci/devices/.../tsm/bound
-Contact: linux-coco@lists.linux.dev
-Description: (RO) Return the device name of the TSM when the device is in a
- TDISP (TEE Device Interface Security Protocol) operational state
- (LOCKED, RUN, or ERROR, not UNLOCKED). Bound devices consume
- platform TSM resources and depend on the device's configuration
- (e.g. BME (Bus Master Enable) and MSE (Memory Space Enable)
- among other settings) to remain stable for the duration of the
- bound state. This attribute is only visible for devices that
- support TDISP operation, and it is only populated after
- successful connect and TSM bind. The TSM bind operation is
- initiated by VFIO/IOMMUFD. This is a "link" TSM attribute, see
- Documentation/ABI/testing/sysfs-class-tsm.
-
What: /sys/bus/pci/devices/.../authenticated
Contact: linux-pci@vger.kernel.org
Description:
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index 1423c64dc9c8..dca9634f8d5e 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -12,6 +12,7 @@
#include <linux/pci.h>
#include <linux/pci-doe.h>
#include <linux/pci-tsm.h>
+#include <linux/slab.h>
#include <linux/sysfs.h>
#include <linux/tsm.h>
#include <linux/xarray.h>
@@ -292,96 +293,94 @@ static int remove_fn(struct pci_dev *pdev, void *data)
return 0;
}
-/*
- * Note, this helper only returns an error code and takes an argument for
- * compatibility with the pci_walk_bus() callback prototype. pci_tsm_unbind()
- * always succeeds.
- */
-static int __pci_tsm_unbind(struct pci_dev *pdev, void *data)
+bool pci_tsm_is_configured(struct pci_dev *pdev)
{
- struct pci_tdi *tdi;
- struct pci_tsm_pf0 *tsm_pf0;
-
- lockdep_assert_held(&pci_tsm_rwsem);
-
- if (!pdev->tsm)
- return 0;
-
- tsm_pf0 = to_pci_tsm_pf0(pdev->tsm);
- guard(mutex)(&tsm_pf0->lock);
-
- tdi = pdev->tsm->tdi;
- if (!tdi)
- return 0;
-
- to_pci_tsm_ops(pdev->tsm)->unbind(tdi);
- pdev->tsm->tdi = NULL;
+ guard(rwsem_read)(&pci_tsm_rwsem);
- return 0;
+ return !!pdev->tsm;
}
+EXPORT_SYMBOL_GPL(pci_tsm_is_configured);
-void pci_tsm_unbind(struct pci_dev *pdev)
-{
- guard(rwsem_read)(&pci_tsm_rwsem);
- __pci_tsm_unbind(pdev, NULL);
-}
-EXPORT_SYMBOL_GPL(pci_tsm_unbind);
+struct pci_tsm_context {
+ struct pci_tsm_pf0 *pf0;
+ struct pci_dev *pdev;
+ struct pci_dev *dsm_dev;
+};
-/**
- * pci_tsm_bind() - Bind @pdev as a TDI for @kvm
- * @pdev: PCI device function to bind
- * @kvm: Private memory attach context
- * @tdi_id: Identifier (virtual BDF) for the TDI as referenced by the TSM and DSM
- *
- * Returns 0 on success, or a negative error code on failure.
- *
- * Context: Caller is responsible for constraining the bind lifetime to the
- * registered state of the device. For example, pci_tsm_bind() /
- * pci_tsm_unbind() limited to the VFIO driver bound state of the device.
- */
-int pci_tsm_bind(struct pci_dev *pdev, struct kvm *kvm, u32 tdi_id)
+struct pci_tsm_context *pci_tsm_context_get(struct pci_dev *pdev,
+ struct tsm_dev *viommu_tsm_dev)
{
- struct pci_tsm_pf0 *tsm_pf0;
- struct pci_tdi *tdi;
-
- if (!kvm)
- return -EINVAL;
+ struct pci_tsm_context *context;
+ struct pci_tsm_pf0 *pf0;
guard(rwsem_read)(&pci_tsm_rwsem);
+ if (!pdev->tsm || !is_link_tsm(pdev->tsm->tsm_dev))
+ return ERR_PTR(-EOPNOTSUPP);
+ /*
+ * The vdevice can use a different PCI function from VIOMMU_ALLOC. The
+ * module pin keeps the vIOMMU's TSM registered, but does not freeze this
+ * function's link. Match its current TSM before pinning the link below.
+ */
+ if (pdev->tsm->tsm_dev != viommu_tsm_dev)
+ return ERR_PTR(-EXDEV);
- if (!pdev->tsm)
- return -EINVAL;
+ pf0 = to_pci_tsm_pf0(pdev->tsm);
+ if (!pf0)
+ return ERR_PTR(-ENXIO);
- if (!is_link_tsm(pdev->tsm->tsm_dev))
- return -ENXIO;
+ context = kzalloc_obj(*context);
+ if (!context)
+ return ERR_PTR(-ENOMEM);
- tsm_pf0 = to_pci_tsm_pf0(pdev->tsm);
- guard(mutex)(&tsm_pf0->lock);
+ guard(mutex)(&pf0->lock);
+ pf0->context_users++;
+ context->pf0 = pf0;
+ context->pdev = pci_dev_get(pdev);
+ context->dsm_dev = pci_dev_get(pf0->base_tsm.pdev);
+ return context;
+}
+EXPORT_SYMBOL_GPL(pci_tsm_context_get);
- /* Resolve races to bind a TDI */
- if (pdev->tsm->tdi) {
- if (pdev->tsm->tdi->kvm != kvm)
- return -EBUSY;
- return 0;
- }
+void pci_tsm_context_put(struct pci_tsm_context *context)
+{
+ struct pci_tsm_pf0 *pf0 = context->pf0;
- tdi = to_pci_tsm_ops(pdev->tsm)->bind(pdev, kvm, tdi_id);
- if (IS_ERR(tdi))
- return PTR_ERR(tdi);
+ down_read(&pci_tsm_rwsem);
+ mutex_lock(&pf0->lock);
+ if (!WARN_ON(!pf0->context_users))
+ pf0->context_users--;
+ mutex_unlock(&pf0->lock);
+ up_read(&pci_tsm_rwsem);
- pdev->tsm->tdi = tdi;
+ pci_dev_put(context->pdev);
+ pci_dev_put(context->dsm_dev);
+ kfree(context);
+}
+EXPORT_SYMBOL_GPL(pci_tsm_context_put);
- return 0;
+struct pci_tsm_pf0 *pci_tsm_context_pf0(struct pci_tsm_context *context)
+{
+ return context->pf0;
}
-EXPORT_SYMBOL_GPL(pci_tsm_bind);
+EXPORT_SYMBOL_GPL(pci_tsm_context_pf0);
-static void pci_tsm_unbind_all(struct pci_dev *pdev)
+struct pci_dev *pci_tsm_context_dsm_dev(struct pci_tsm_context *context)
{
- pci_tsm_walk_fns_reverse(pdev, __pci_tsm_unbind, NULL);
- __pci_tsm_unbind(pdev, NULL);
+ return context->dsm_dev;
+}
+EXPORT_SYMBOL_GPL(pci_tsm_context_dsm_dev);
+
+bool pci_tsm_context_match_device(struct pci_tsm_context *context,
+ struct pci_dev *pdev)
+{
+ guard(rwsem_read)(&pci_tsm_rwsem);
+
+ return pdev->tsm && is_link_tsm(pdev->tsm->tsm_dev) &&
+ to_pci_tsm_pf0(pdev->tsm) == context->pf0;
}
+EXPORT_SYMBOL_GPL(pci_tsm_context_match_device);
-static void __pci_tsm_disconnect(struct pci_dev *pdev)
+static int __pci_tsm_disconnect(struct pci_dev *pdev)
{
struct pci_tsm_pf0 *tsm_pf0 = to_pci_tsm_pf0(pdev->tsm);
const struct pci_tsm_ops *ops = to_pci_tsm_ops(pdev->tsm);
@@ -389,21 +388,29 @@ static void __pci_tsm_disconnect(struct pci_dev *pdev)
/* disconnect() mutually exclusive with subfunction pci_tsm_init() */
lockdep_assert_held_write(&pci_tsm_rwsem);
- pci_tsm_unbind_all(pdev);
-
/*
- * disconnect() is uninterruptible as it may be called for device
- * teardown
+ * A vdevice holds a context for its lifetime. Refuse to tear down the
+ * link until userspace destroys all dependent vdevices.
+ *
+ * disconnect() is uninterruptible as it may also be called for device
+ * teardown.
*/
guard(mutex)(&tsm_pf0->lock);
+ if (tsm_pf0->context_users)
+ return -EBUSY;
pci_tsm_walk_fns_reverse(pdev, remove_fn, NULL);
ops->disconnect(pdev);
+ return 0;
}
-static void pci_tsm_disconnect(struct pci_dev *pdev)
+static int pci_tsm_disconnect(struct pci_dev *pdev)
{
- __pci_tsm_disconnect(pdev);
+ int ret = __pci_tsm_disconnect(pdev);
+
+ if (ret)
+ return ret;
tsm_remove(pdev->tsm);
+ return 0;
}
static ssize_t disconnect_store(struct device *dev,
@@ -425,38 +432,13 @@ static ssize_t disconnect_store(struct device *dev,
if (!sysfs_streq(buf, dev_name(&tsm_dev->dev)))
return -EINVAL;
- pci_tsm_disconnect(pdev);
+ rc = pci_tsm_disconnect(pdev);
+ if (rc)
+ return rc;
return len;
}
static DEVICE_ATTR_WO(disconnect);
-static ssize_t bound_show(struct device *dev,
- struct device_attribute *attr, char *buf)
-{
- struct pci_dev *pdev = to_pci_dev(dev);
- struct pci_tsm_pf0 *tsm_pf0;
- struct pci_tsm *tsm;
- int rc;
-
- ACQUIRE(rwsem_read_intr, lock)(&pci_tsm_rwsem);
- if ((rc = ACQUIRE_ERR(rwsem_read_intr, &lock)))
- return rc;
-
- tsm = pdev->tsm;
- if (!tsm)
- return sysfs_emit(buf, "\n");
- tsm_pf0 = to_pci_tsm_pf0(tsm);
-
- ACQUIRE(mutex_intr, ops_lock)(&tsm_pf0->lock);
- if ((rc = ACQUIRE_ERR(mutex_intr, &ops_lock)))
- return rc;
-
- if (!tsm->tdi)
- return sysfs_emit(buf, "\n");
- return sysfs_emit(buf, "%s\n", dev_name(&tsm->tsm_dev->dev));
-}
-static DEVICE_ATTR_RO(bound);
-
static ssize_t dsm_show(struct device *dev, struct device_attribute *attr,
char *buf)
{
@@ -511,13 +493,6 @@ static umode_t pci_tsm_attr_visible(struct kobject *kobj,
if (pci_tsm_link_group_visible(kobj)) {
struct pci_dev *pdev = to_pci_dev(kobj_to_dev(kobj));
- if (attr == &dev_attr_bound.attr) {
- if (is_pci_tsm_pf0(pdev) && has_tee(pdev))
- return attr->mode;
- if (pdev->tsm && has_tee(pdev->tsm->dsm_dev))
- return attr->mode;
- }
-
if (attr == &dev_attr_dsm.attr) {
if (is_pci_tsm_pf0(pdev))
return attr->mode;
@@ -544,7 +519,6 @@ DEFINE_SYSFS_GROUP_VISIBLE(pci_tsm);
static struct attribute *pci_tsm_attrs[] = {
&dev_attr_connect.attr,
&dev_attr_disconnect.attr,
- &dev_attr_bound.attr,
&dev_attr_dsm.attr,
NULL
};
@@ -634,22 +608,6 @@ static struct pci_dev *find_dsm_dev(struct pci_dev *pdev)
return NULL;
}
-/**
- * pci_tsm_tdi_constructor() - base 'struct pci_tdi' initialization for link TSMs
- * @pdev: PCI device function representing the TDI
- * @tdi: context to initialize
- * @kvm: Private memory attach context
- * @tdi_id: Identifier (virtual BDF) for the TDI as referenced by the TSM and DSM
- */
-void pci_tsm_tdi_constructor(struct pci_dev *pdev, struct pci_tdi *tdi,
- struct kvm *kvm, u32 tdi_id)
-{
- tdi->pdev = pdev;
- tdi->kvm = kvm;
- tdi->tdi_id = tdi_id;
-}
-EXPORT_SYMBOL_GPL(pci_tsm_tdi_constructor);
-
/**
* pci_tsm_link_constructor() - base 'struct pci_tsm' initialization for link TSMs
* @pdev: The PCI device
@@ -729,12 +687,6 @@ int pci_tsm_register(struct tsm_dev *tsm_dev)
return 0;
}
-static void pci_tsm_fn_exit(struct pci_dev *pdev)
-{
- __pci_tsm_unbind(pdev, NULL);
- tsm_remove(pdev->tsm);
-}
-
/**
* __pci_tsm_destroy() - destroy the TSM context for @pdev
* @pdev: device to cleanup
@@ -768,10 +720,12 @@ static void __pci_tsm_destroy(struct pci_dev *pdev, struct tsm_dev *tsm_dev)
else if (tsm_dev != tsm->tsm_dev)
return;
- if (is_link_tsm(tsm_dev) && is_pci_tsm_pf0(pdev))
- pci_tsm_disconnect(pdev);
- else
- pci_tsm_fn_exit(pdev);
+ if (is_link_tsm(tsm_dev) && is_pci_tsm_pf0(pdev)) {
+ if (pci_tsm_disconnect(pdev))
+ pci_warn(pdev, "TSM connection is still in use\n");
+ } else {
+ tsm_remove(pdev->tsm);
+ }
}
void pci_tsm_destroy(struct pci_dev *pdev)
diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
index 6fe3d0ea875f..eaf0199d01f7 100644
--- a/include/linux/pci-tsm.h
+++ b/include/linux/pci-tsm.h
@@ -5,8 +5,8 @@
#include <linux/pci.h>
struct pci_tsm;
+struct pci_tsm_context;
struct tsm_dev;
-struct kvm;
/*
* struct pci_tsm_ops - manage confidential links and security state
@@ -30,16 +30,12 @@ struct pci_tsm_ops {
* @connect: establish / validate a secure connection (e.g. IDE)
* with the device
* @disconnect: teardown the secure link
- * @bind: bind a TDI in preparation for it to be accepted by a TVM
- * @unbind: remove a TDI from secure operation with a TVM
*
* Context: @probe, @remove, @connect, and @disconnect run under
* pci_tsm_rwsem held for write to sync with TSM unregistration and
* mutual exclusion of @connect and @disconnect. @connect and
* @disconnect additionally run under the DSM lock (struct
* pci_tsm_pf0::lock) as well as @probe and @remove of the subfunctions.
- * @bind and @unbind run under pci_tsm_rwsem held for read
- * and the DSM lock.
*/
struct_group_tagged(pci_tsm_link_ops, link_ops,
struct pci_tsm *(*probe)(struct tsm_dev *tsm_dev,
@@ -47,9 +43,6 @@ struct pci_tsm_ops {
void (*remove)(struct pci_tsm *tsm);
int (*connect)(struct pci_dev *pdev);
void (*disconnect)(struct pci_dev *pdev);
- struct pci_tdi *(*bind)(struct pci_dev *pdev,
- struct kvm *kvm, u32 tdi_id);
- void (*unbind)(struct pci_tdi *tdi);
);
/*
@@ -69,25 +62,12 @@ struct pci_tsm_ops {
);
};
-/**
- * struct pci_tdi - Core TEE I/O Device Interface (TDI) context
- * @pdev: host side representation of guest-side TDI
- * @kvm: TEE VM context of bound TDI
- * @tdi_id: Identifier (virtual BDF) for the TDI as referenced by the TSM and DSM
- */
-struct pci_tdi {
- struct pci_dev *pdev;
- struct kvm *kvm;
- u32 tdi_id;
-};
-
/**
* struct pci_tsm - Core TSM context for a given PCIe endpoint
* @pdev: Back ref to device function, distinguishes type of pci_tsm context
* @dsm_dev: PCI Device Security Manager for link operations on @pdev
* @tsm_dev: PCI TEE Security Manager device for Link Confidentiality or Device
* Function Security operations
- * @tdi: TDI context established by the @bind link operation
*
* This structure is wrapped by low level TSM driver data and returned by
* probe()/lock(), it is freed by the corresponding remove()/unlock().
@@ -103,18 +83,20 @@ struct pci_tsm {
struct pci_dev *pdev;
struct pci_dev *dsm_dev;
struct tsm_dev *tsm_dev;
- struct pci_tdi *tdi;
};
/**
* struct pci_tsm_pf0 - Physical Function 0 TDISP link context
* @base_tsm: generic core "tsm" context
* @lock: mutual exclustion for pci_tsm_ops invocation
+ * @context_users: live per-function contexts on this PF0; a nonzero count
+ * blocks link disconnect
* @doe_mb: PCIe Data Object Exchange mailbox
*/
struct pci_tsm_pf0 {
struct pci_tsm base_tsm;
struct mutex lock;
+ unsigned int context_users;
struct pci_doe_mb *doe_mb;
};
@@ -161,10 +143,14 @@ int pci_tsm_pf0_constructor(struct pci_dev *pdev, struct pci_tsm_pf0 *tsm,
void pci_tsm_pf0_destructor(struct pci_tsm_pf0 *tsm);
int pci_tsm_doe_transfer(struct pci_dev *pdev, u8 type, const void *req,
size_t req_sz, void *resp, size_t resp_sz);
-int pci_tsm_bind(struct pci_dev *pdev, struct kvm *kvm, u32 tdi_id);
-void pci_tsm_unbind(struct pci_dev *pdev);
-void pci_tsm_tdi_constructor(struct pci_dev *pdev, struct pci_tdi *tdi,
- struct kvm *kvm, u32 tdi_id);
+bool pci_tsm_is_configured(struct pci_dev *pdev);
+struct pci_tsm_context *pci_tsm_context_get(struct pci_dev *pdev,
+ struct tsm_dev *viommu_tsm_dev);
+void pci_tsm_context_put(struct pci_tsm_context *context);
+struct pci_tsm_pf0 *pci_tsm_context_pf0(struct pci_tsm_context *context);
+struct pci_dev *pci_tsm_context_dsm_dev(struct pci_tsm_context *context);
+bool pci_tsm_context_match_device(struct pci_tsm_context *context,
+ struct pci_dev *pdev);
#else
static inline int pci_tsm_register(struct tsm_dev *tsm_dev)
{
@@ -173,12 +159,39 @@ static inline int pci_tsm_register(struct tsm_dev *tsm_dev)
static inline void pci_tsm_unregister(struct tsm_dev *tsm_dev)
{
}
-static inline int pci_tsm_bind(struct pci_dev *pdev, struct kvm *kvm, u64 tdi_id)
+
+static inline bool pci_tsm_is_configured(struct pci_dev *pdev)
{
- return -ENXIO;
+ return false;
}
-static inline void pci_tsm_unbind(struct pci_dev *pdev)
+
+static inline struct pci_tsm_context *
+pci_tsm_context_get(struct pci_dev *pdev, struct tsm_dev *viommu_tsm_dev)
+{
+ return ERR_PTR(-EOPNOTSUPP);
+}
+
+static inline void pci_tsm_context_put(struct pci_tsm_context *context)
+{
+}
+
+static inline struct pci_tsm_pf0 *
+pci_tsm_context_pf0(struct pci_tsm_context *context)
+{
+ return NULL;
+}
+
+static inline struct pci_dev *
+pci_tsm_context_dsm_dev(struct pci_tsm_context *context)
+{
+ return NULL;
+}
+
+static inline bool
+pci_tsm_context_match_device(struct pci_tsm_context *context,
+ struct pci_dev *pdev)
{
+ return false;
}
#endif
#endif /*__PCI_TSM_H */
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 13/16] iommufd/viommu: Select vIOMMU operations before allocation
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (11 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 12/16] PCI/TSM: Add vIOMMU-bound contexts for vdevices Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 14/16] iommufd/viommu: Allow PCI TSM backends to provide vIOMMU operations Aneesh Kumar K.V (Arm)
` (2 subsequent siblings)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
Physical IOMMU drivers currently provide the vIOMMU size and
initialization callbacks through iommu_ops. The initialization callback
then selects and installs the corresponding iommufd_viommu_ops.
Add iommu_ops::get_viommu_ops() to select the operations from the
physical device and requested vIOMMU type before allocation. Move the
size and initialization callbacks into iommufd_viommu_ops. The core can
then validate the selected operations, allocate the driver structure,
initialize it, and install the operations.
Convert AMD, Arm SMMUv3, Tegra CMDQV, and the selftest backend to the
new interface.
Selecting the operations early allows subsequent changes to use
implementation-specific vIOMMU properties, including when validating the
parent HWPT.
Cc: joro@8bytes.org
Cc: suravee.suthikulpanit@amd.com
Cc: vasant.hegde@amd.com
Cc: will@kernel.org
Cc: robin.murphy@arm.com
Cc: thierry.reding@kernel.org
Cc: vdumpa@nvidia.com
Cc: jonathanh@nvidia.com
Cc: jgg@ziepe.ca
Cc: kevin.tian@intel.com
Cc: linux-arm-kernel@lists.infradead.org
Cc: linux-tegra@vger.kernel.org
Based on original patch by Jason Gunthorpe <jgg@nvidia.com>
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/iommu/amd/iommu.c | 3 +-
drivers/iommu/amd/iommufd.c | 18 ++++++---
drivers/iommu/amd/iommufd.h | 11 ++++--
drivers/iommu/amd/nested.c | 4 +-
.../arm/arm-smmu-v3/arm-smmu-v3-iommufd.c | 37 +++++++++++--------
drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c | 7 ++--
drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h | 16 ++++----
.../iommu/arm/arm-smmu-v3/tegra241-cmdqv.c | 24 ++++++++----
drivers/iommu/iommufd/selftest.c | 31 +++++++++++-----
drivers/iommu/iommufd/viommu.c | 32 ++++++++--------
include/linux/iommu.h | 20 +++-------
include/linux/iommufd.h | 14 +++++++
12 files changed, 129 insertions(+), 88 deletions(-)
diff --git a/drivers/iommu/amd/iommu.c b/drivers/iommu/amd/iommu.c
index 56262f6b1f70..a3c58dd39549 100644
--- a/drivers/iommu/amd/iommu.c
+++ b/drivers/iommu/amd/iommu.c
@@ -3211,8 +3211,7 @@ const struct iommu_ops amd_iommu_ops = {
.is_attach_deferred = amd_iommu_is_attach_deferred,
.def_domain_type = amd_iommu_def_domain_type,
.page_response = amd_iommu_page_response,
- .get_viommu_size = amd_iommufd_get_viommu_size,
- .viommu_init = amd_iommufd_viommu_init,
+ .get_viommu_ops = amd_iommufd_get_viommu_ops,
};
#ifdef CONFIG_IRQ_REMAP
diff --git a/drivers/iommu/amd/iommufd.c b/drivers/iommu/amd/iommufd.c
index 52300b867c1f..c1e132382adc 100644
--- a/drivers/iommu/amd/iommufd.c
+++ b/drivers/iommu/amd/iommufd.c
@@ -32,13 +32,21 @@ void *amd_iommufd_hw_info(struct device *dev, u32 *length, enum iommu_hw_info_ty
return hwinfo;
}
-size_t amd_iommufd_get_viommu_size(struct device *dev, enum iommu_viommu_type viommu_type)
+static size_t amd_iommufd_get_viommu_size(struct device *dev,
+ enum iommu_viommu_type viommu_type)
{
return VIOMMU_STRUCT_SIZE(struct amd_iommu_viommu, core);
}
-int amd_iommufd_viommu_init(struct iommufd_viommu *viommu, struct iommu_domain *parent,
- const struct iommu_user_data *user_data)
+const struct iommufd_viommu_ops *
+amd_iommufd_get_viommu_ops(struct device *dev, enum iommu_viommu_type viommu_type)
+{
+ return &amd_viommu_ops;
+}
+
+int amd_iommufd_viommu_init(struct iommufd_viommu *viommu,
+ struct device *dev, struct iommu_domain *parent,
+ const struct iommu_user_data *user_data)
{
unsigned long flags;
struct protection_domain *pdom = to_pdomain(parent);
@@ -47,8 +55,6 @@ int amd_iommufd_viommu_init(struct iommufd_viommu *viommu, struct iommu_domain *
xa_init_flags(&aviommu->gdomid_array, XA_FLAGS_ALLOC1);
aviommu->parent = pdom;
- viommu->ops = &amd_viommu_ops;
-
spin_lock_irqsave(&pdom->lock, flags);
list_add(&aviommu->pdom_list, &pdom->viommu_list);
spin_unlock_irqrestore(&pdom->lock, flags);
@@ -73,5 +79,7 @@ static void amd_iommufd_viommu_destroy(struct iommufd_viommu *viommu)
* struct iommufd_viommu_ops - vIOMMU specific operations
*/
static const struct iommufd_viommu_ops amd_viommu_ops = {
+ .get_viommu_size = amd_iommufd_get_viommu_size,
+ .viommu_init = amd_iommufd_viommu_init,
.destroy = amd_iommufd_viommu_destroy,
};
diff --git a/drivers/iommu/amd/iommufd.h b/drivers/iommu/amd/iommufd.h
index 62e9e1bebfbe..0d8e7c0902cf 100644
--- a/drivers/iommu/amd/iommufd.h
+++ b/drivers/iommu/amd/iommufd.h
@@ -8,13 +8,16 @@
#if IS_ENABLED(CONFIG_AMD_IOMMU_IOMMUFD)
void *amd_iommufd_hw_info(struct device *dev, u32 *length, enum iommu_hw_info_type *type);
-size_t amd_iommufd_get_viommu_size(struct device *dev, enum iommu_viommu_type viommu_type);
-int amd_iommufd_viommu_init(struct iommufd_viommu *viommu, struct iommu_domain *parent,
- const struct iommu_user_data *user_data);
+const struct iommufd_viommu_ops *
+amd_iommufd_get_viommu_ops(struct device *dev,
+ enum iommu_viommu_type viommu_type);
+int amd_iommufd_viommu_init(struct iommufd_viommu *viommu,
+ struct device *dev, struct iommu_domain *parent,
+ const struct iommu_user_data *user_data);
#else
#define amd_iommufd_hw_info NULL
#define amd_iommufd_viommu_init NULL
-#define amd_iommufd_get_viommu_size NULL
+#define amd_iommufd_get_viommu_ops NULL
#endif /* CONFIG_AMD_IOMMU_IOMMUFD */
#endif /* AMD_IOMMUFD_H */
diff --git a/drivers/iommu/amd/nested.c b/drivers/iommu/amd/nested.c
index f1c7987fc585..5b07136e7cd9 100644
--- a/drivers/iommu/amd/nested.c
+++ b/drivers/iommu/amd/nested.c
@@ -90,7 +90,7 @@ static void *gdom_info_load_or_alloc_locked(struct xarray *xa,
/*
* This function is assigned to struct iommufd_viommu_ops.alloc_domain_nested()
- * during the call to struct iommu_ops.viommu_init().
+ * when the vIOMMU operations are selected.
*/
struct iommu_domain *
amd_iommu_alloc_domain_nested(struct iommufd_viommu *viommu, u32 flags,
@@ -198,7 +198,7 @@ static void set_dte_nested(struct amd_iommu *iommu, struct iommu_domain *dom,
/*
* The nest parent domain is attached during the call to the
- * struct iommu_ops.viommu_init(), which will be stored as part
+ * struct iommufd_viommu_ops.viommu_init(), which will be stored as part
* of the struct amd_iommu_viommu.parent.
*/
if (WARN_ON(!ndom->viommu || !ndom->viommu->parent))
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c
index 25982bdbcbd9..1dd103196353 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c
@@ -417,20 +417,29 @@ int arm_vsmmu_cache_invalidate(struct iommufd_viommu *viommu,
return ret;
}
+static size_t arm_vsmmu_get_size(struct device *dev,
+ enum iommu_viommu_type viommu_type)
+{
+ return VIOMMU_STRUCT_SIZE(struct arm_vsmmu, core);
+}
+
static const struct iommufd_viommu_ops arm_vsmmu_ops = {
+ .get_viommu_size = arm_vsmmu_get_size,
+ .viommu_init = arm_vsmmu_init,
.alloc_domain_nested = arm_vsmmu_alloc_domain_nested,
.cache_invalidate = arm_vsmmu_cache_invalidate,
.vdevice_init = arm_vsmmu_vdevice_init,
};
-size_t arm_smmu_get_viommu_size(struct device *dev,
- enum iommu_viommu_type viommu_type)
+const struct iommufd_viommu_ops *
+arm_smmu_get_viommu_ops(struct device *dev,
+ enum iommu_viommu_type viommu_type)
{
struct arm_smmu_master *master = dev_iommu_priv_get(dev);
struct arm_smmu_device *smmu = master->smmu;
if (!(smmu->features & ARM_SMMU_FEAT_NESTING))
- return 0;
+ return NULL;
/*
* FORCE_SYNC is not set with FEAT_NESTING. Some study of the exact HW
@@ -438,7 +447,7 @@ size_t arm_smmu_get_viommu_size(struct device *dev,
* any change to remove this.
*/
if (WARN_ON(smmu->options & ARM_SMMU_OPT_CMDQ_FORCE_SYNC))
- return 0;
+ return NULL;
/*
* Must support some way to prevent the VM from bypassing the cache
@@ -450,19 +459,19 @@ size_t arm_smmu_get_viommu_size(struct device *dev,
*/
if (!arm_smmu_master_canwbs(master) &&
!(smmu->features & ARM_SMMU_FEAT_S2FWB))
- return 0;
+ return NULL;
if (viommu_type == IOMMU_VIOMMU_TYPE_ARM_SMMUV3)
- return VIOMMU_STRUCT_SIZE(struct arm_vsmmu, core);
+ return &arm_vsmmu_ops;
- if (!smmu->impl_ops || !smmu->impl_ops->get_viommu_size)
- return 0;
- return smmu->impl_ops->get_viommu_size(viommu_type);
+ if (!smmu->impl_ops || !smmu->impl_ops->get_viommu_ops)
+ return NULL;
+ return smmu->impl_ops->get_viommu_ops(viommu_type);
}
-int arm_vsmmu_init(struct iommufd_viommu *viommu,
- struct iommu_domain *parent_domain,
- const struct iommu_user_data *user_data)
+int arm_vsmmu_init(struct iommufd_viommu *viommu, struct device *dev,
+ struct iommu_domain *parent_domain,
+ const struct iommu_user_data *user_data)
{
struct arm_vsmmu *vsmmu = container_of(viommu, struct arm_vsmmu, core);
struct arm_smmu_device *smmu =
@@ -477,10 +486,8 @@ int arm_vsmmu_init(struct iommufd_viommu *viommu,
/* FIXME Move VMID allocation from the S2 domain allocation to here */
vsmmu->vmid = s2_parent->s2_cfg.vmid;
- if (viommu->type == IOMMU_VIOMMU_TYPE_ARM_SMMUV3) {
- viommu->ops = &arm_vsmmu_ops;
+ if (viommu->type == IOMMU_VIOMMU_TYPE_ARM_SMMUV3)
return 0;
- }
return smmu->impl_ops->vsmmu_init(vsmmu, user_data);
}
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
index 5732f3ba0122..97dfaec6dc58 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
@@ -4389,8 +4389,7 @@ static const struct iommu_ops arm_smmu_ops = {
.get_resv_regions = arm_smmu_get_resv_regions,
.page_response = arm_smmu_page_response,
.def_domain_type = arm_smmu_def_domain_type,
- .get_viommu_size = arm_smmu_get_viommu_size,
- .viommu_init = arm_vsmmu_init,
+ .get_viommu_ops = arm_smmu_get_viommu_ops,
.user_pasid_table = 1,
.owner = THIS_MODULE,
.default_domain_ops = &(const struct iommu_domain_ops) {
@@ -5489,8 +5488,8 @@ static struct arm_smmu_device *arm_smmu_impl_probe(struct arm_smmu_device *smmu)
ops = new_smmu->impl_ops;
if (ops) {
- /* get_viommu_size and vsmmu_init ops must be paired */
- if (WARN_ON(!ops->get_viommu_size != !ops->vsmmu_init)) {
+ /* get_viommu_ops and vsmmu_init ops must be paired */
+ if (WARN_ON(!ops->get_viommu_ops != !ops->vsmmu_init)) {
ret = -EINVAL;
goto err_remove;
}
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
index dd2fee2f560e..79f2adf7a1f5 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
@@ -886,7 +886,8 @@ struct arm_smmu_impl_ops {
*/
void *(*hw_info)(struct arm_smmu_device *smmu, u32 *length,
enum iommu_hw_info_type *type);
- size_t (*get_viommu_size)(enum iommu_viommu_type viommu_type);
+ const struct iommufd_viommu_ops *(*get_viommu_ops)(
+ enum iommu_viommu_type viommu_type);
int (*vsmmu_init)(struct arm_vsmmu *vsmmu,
const struct iommu_user_data *user_data);
};
@@ -1260,11 +1261,12 @@ struct arm_vsmmu {
#if IS_ENABLED(CONFIG_ARM_SMMU_V3_IOMMUFD)
void *arm_smmu_hw_info(struct device *dev, u32 *length,
enum iommu_hw_info_type *type);
-size_t arm_smmu_get_viommu_size(struct device *dev,
- enum iommu_viommu_type viommu_type);
-int arm_vsmmu_init(struct iommufd_viommu *viommu,
- struct iommu_domain *parent_domain,
- const struct iommu_user_data *user_data);
+const struct iommufd_viommu_ops *
+arm_smmu_get_viommu_ops(struct device *dev,
+ enum iommu_viommu_type viommu_type);
+int arm_vsmmu_init(struct iommufd_viommu *viommu, struct device *dev,
+ struct iommu_domain *parent_domain,
+ const struct iommu_user_data *user_data);
int arm_smmu_attach_prepare_vmaster(struct arm_smmu_attach_state *state,
struct arm_smmu_nested_domain *nested_domain);
void arm_smmu_attach_commit_vmaster(struct arm_smmu_attach_state *state);
@@ -1276,7 +1278,7 @@ arm_vsmmu_alloc_domain_nested(struct iommufd_viommu *viommu, u32 flags,
int arm_vsmmu_cache_invalidate(struct iommufd_viommu *viommu,
struct iommu_user_data_array *array);
#else
-#define arm_smmu_get_viommu_size NULL
+#define arm_smmu_get_viommu_ops NULL
#define arm_smmu_hw_info NULL
#define arm_vsmmu_init NULL
#define arm_vsmmu_alloc_domain_nested NULL
diff --git a/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c b/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c
index 6644075c1431..1c8a7939fc1c 100644
--- a/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c
+++ b/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c
@@ -892,11 +892,14 @@ static void *tegra241_cmdqv_hw_info(struct arm_smmu_device *smmu, u32 *length,
return info;
}
-static size_t tegra241_cmdqv_get_vintf_size(enum iommu_viommu_type viommu_type)
+static const struct iommufd_viommu_ops tegra241_cmdqv_viommu_ops;
+
+static const struct iommufd_viommu_ops *
+tegra241_cmdqv_get_viommu_ops(enum iommu_viommu_type viommu_type)
{
if (viommu_type != IOMMU_VIOMMU_TYPE_TEGRA241_CMDQV)
- return 0;
- return VIOMMU_STRUCT_SIZE(struct tegra241_vintf, vsmmu.core);
+ return NULL;
+ return &tegra241_cmdqv_viommu_ops;
}
static struct arm_smmu_impl_ops tegra241_cmdqv_impl_ops = {
@@ -907,7 +910,7 @@ static struct arm_smmu_impl_ops tegra241_cmdqv_impl_ops = {
.device_remove = tegra241_cmdqv_remove,
/* For user-space use */
.hw_info = tegra241_cmdqv_hw_info,
- .get_viommu_size = tegra241_cmdqv_get_vintf_size,
+ .get_viommu_ops = tegra241_cmdqv_get_viommu_ops,
.vsmmu_init = tegra241_cmdqv_init_vintf_user,
};
@@ -1289,7 +1292,15 @@ static int tegra241_vintf_init_vsid(struct iommufd_vdevice *vdev)
return 0;
}
-static struct iommufd_viommu_ops tegra241_cmdqv_viommu_ops = {
+static size_t tegra241_cmdqv_get_vintf_size(struct device *dev,
+ enum iommu_viommu_type viommu_type)
+{
+ return VIOMMU_STRUCT_SIZE(struct tegra241_vintf, vsmmu.core);
+}
+
+static const struct iommufd_viommu_ops tegra241_cmdqv_viommu_ops = {
+ .get_viommu_size = tegra241_cmdqv_get_vintf_size,
+ .viommu_init = arm_vsmmu_init,
.destroy = tegra241_cmdqv_destroy_vintf_user,
.alloc_domain_nested = arm_vsmmu_alloc_domain_nested,
/* Non-accelerated commands will be still handled by the kernel */
@@ -1312,7 +1323,7 @@ tegra241_cmdqv_init_vintf_user(struct arm_vsmmu *vsmmu,
int ret;
/*
- * Unsupported type should be rejected by tegra241_cmdqv_get_vintf_size.
+ * Unsupported type should be rejected by tegra241_cmdqv_get_viommu_ops.
* Seeing one here indicates a kernel bug or some data corruption.
*/
if (WARN_ON(vsmmu->core.type != IOMMU_VIOMMU_TYPE_TEGRA241_CMDQV))
@@ -1364,7 +1375,6 @@ tegra241_cmdqv_init_vintf_user(struct arm_vsmmu *vsmmu,
dev_dbg(cmdqv->dev, "VINTF%u: allocated with vmid (%d)\n", vintf->idx,
vintf->vsmmu.vmid);
- vsmmu->core.ops = &tegra241_cmdqv_viommu_ops;
return 0;
free_mmap:
diff --git a/drivers/iommu/iommufd/selftest.c b/drivers/iommu/iommufd/selftest.c
index e9c825b24356..426ea94467ff 100644
--- a/drivers/iommu/iommufd/selftest.c
+++ b/drivers/iommu/iommufd/selftest.c
@@ -774,7 +774,19 @@ static int mock_hw_queue_init_phys(struct iommufd_hw_queue *hw_queue, u32 index,
return rc;
}
-static struct iommufd_viommu_ops mock_viommu_ops = {
+static int mock_viommu_init(struct iommufd_viommu *viommu, struct device *dev,
+ struct iommu_domain *parent_domain,
+ const struct iommu_user_data *user_data);
+
+static size_t mock_get_viommu_size(struct device *dev,
+ enum iommu_viommu_type viommu_type)
+{
+ return VIOMMU_STRUCT_SIZE(struct mock_viommu, core);
+}
+
+static const struct iommufd_viommu_ops mock_viommu_ops = {
+ .get_viommu_size = mock_get_viommu_size,
+ .viommu_init = mock_viommu_init,
.destroy = mock_viommu_destroy,
.alloc_domain_nested = mock_viommu_alloc_domain_nested,
.cache_invalidate = mock_viommu_cache_invalidate,
@@ -782,17 +794,18 @@ static struct iommufd_viommu_ops mock_viommu_ops = {
.hw_queue_init_phys = mock_hw_queue_init_phys,
};
-static size_t mock_get_viommu_size(struct device *dev,
- enum iommu_viommu_type viommu_type)
+static const struct iommufd_viommu_ops *
+mock_get_viommu_ops(struct device *dev,
+ enum iommu_viommu_type viommu_type)
{
if (viommu_type != IOMMU_VIOMMU_TYPE_SELFTEST)
- return 0;
- return VIOMMU_STRUCT_SIZE(struct mock_viommu, core);
+ return NULL;
+ return &mock_viommu_ops;
}
static int mock_viommu_init(struct iommufd_viommu *viommu,
- struct iommu_domain *parent_domain,
- const struct iommu_user_data *user_data)
+ struct device *dev, struct iommu_domain *parent_domain,
+ const struct iommu_user_data *user_data)
{
struct mock_iommu_device *mock_iommu = container_of(
viommu->iommu_dev, struct mock_iommu_device, iommu_dev);
@@ -834,7 +847,6 @@ static int mock_viommu_init(struct iommufd_viommu *viommu,
mutex_init(&mock_viommu->queue_mutex);
mock_viommu->s2_parent = to_mock_domain(parent_domain);
- viommu->ops = &mock_viommu_ops;
return 0;
err_destroy_mmap:
@@ -861,8 +873,7 @@ static const struct iommu_ops mock_ops = {
.probe_device = mock_probe_device,
.page_response = mock_domain_page_response,
.user_pasid_table = true,
- .get_viommu_size = mock_get_viommu_size,
- .viommu_init = mock_viommu_init,
+ .get_viommu_ops = mock_get_viommu_ops,
};
static void mock_domain_free_nested(struct iommu_domain *domain)
diff --git a/drivers/iommu/iommufd/viommu.c b/drivers/iommu/iommufd/viommu.c
index f3d5b5a7eb4a..e95138ae71d5 100644
--- a/drivers/iommu/iommufd/viommu.c
+++ b/drivers/iommu/iommufd/viommu.c
@@ -32,7 +32,7 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
struct iommufd_viommu *viommu;
struct iommufd_device *idev;
struct iommu_device *iommu_dev;
- const struct iommu_ops *ops;
+ const struct iommufd_viommu_ops *ops;
size_t viommu_size;
int rc;
@@ -48,23 +48,25 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
rc = -EOPNOTSUPP;
goto out_put_idev;
}
- ops = iommu_dev->ops;
- if (!ops->get_viommu_size || !ops->viommu_init) {
+ if (!iommu_dev->ops->get_viommu_ops) {
rc = -EOPNOTSUPP;
goto out_put_idev;
}
-
- viommu_size = ops->get_viommu_size(idev->dev, cmd->type);
- if (!viommu_size) {
+ ops = iommu_dev->ops->get_viommu_ops(idev->dev, cmd->type);
+ if (!ops) {
rc = -EOPNOTSUPP;
goto out_put_idev;
}
-
/*
- * It is a driver bug for providing a viommu_size smaller than the core
- * vIOMMU structure size
+ * It is a driver bug to omit the required operations or provide a size
+ * smaller than the core vIOMMU structure.
*/
- if (WARN_ON_ONCE(viommu_size < sizeof(*viommu))) {
+ if (WARN_ON_ONCE(!ops->get_viommu_size || !ops->viommu_init)) {
+ rc = -EOPNOTSUPP;
+ goto out_put_idev;
+ }
+ viommu_size = ops->get_viommu_size(idev->dev, cmd->type);
+ if (!viommu_size || WARN_ON_ONCE(viommu_size < sizeof(*viommu))) {
rc = -EOPNOTSUPP;
goto out_put_idev;
}
@@ -103,16 +105,12 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
*/
viommu->iommu_dev = iommu_dev;
- rc = ops->viommu_init(viommu, hwpt_paging->common.domain,
+ rc = ops->viommu_init(viommu, idev->dev,
+ hwpt_paging->common.domain,
user_data.len ? &user_data : NULL);
if (rc)
goto out_put_hwpt;
-
- /* It is a driver bug that viommu->ops isn't filled */
- if (WARN_ON_ONCE(!viommu->ops)) {
- rc = -EOPNOTSUPP;
- goto out_put_hwpt;
- }
+ viommu->ops = ops;
cmd->out_viommu_id = viommu->obj.id;
rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
diff --git a/include/linux/iommu.h b/include/linux/iommu.h
index ac43b8b93f14..9931c96e5dd3 100644
--- a/include/linux/iommu.h
+++ b/include/linux/iommu.h
@@ -46,6 +46,7 @@ struct iommu_dma_msi_cookie;
struct iommu_fault_param;
struct iommufd_ctx;
struct iommufd_viommu;
+struct iommufd_viommu_ops;
struct msi_desc;
struct msi_msg;
@@ -669,16 +670,8 @@ __iommu_copy_struct_to_user(const struct iommu_user_data *dst_data,
* - IOMMU_DOMAIN_DMA: must use a dma domain
* - 0: use the default setting
* @default_domain_ops: the default ops for domains
- * @get_viommu_size: Get the size of a driver-level vIOMMU structure for a given
- * @dev corresponding to @viommu_type. Driver should return 0
- * if vIOMMU isn't supported accordingly. It is required for
- * driver to use the VIOMMU_STRUCT_SIZE macro to sanitize the
- * driver-level vIOMMU structure related to the core one
- * @viommu_init: Init the driver-level struct of an iommufd_viommu on a physical
- * IOMMU instance @viommu->iommu_dev, as the set of virtualization
- * resources shared/passed to user space IOMMU instance. Associate
- * it with a nesting @parent_domain. It is required for driver to
- * set @viommu->ops pointing to its own viommu_ops
+ * @get_viommu_ops: Return the vIOMMU operations supported by @dev for a type.
+ * Return NULL if the type is unsupported.
* @owner: Driver module providing these ops
* @identity_domain: An always available, always attachable identity
* translation.
@@ -729,11 +722,8 @@ struct iommu_ops {
int (*def_domain_type)(struct device *dev);
- size_t (*get_viommu_size)(struct device *dev,
- enum iommu_viommu_type viommu_type);
- int (*viommu_init)(struct iommufd_viommu *viommu,
- struct iommu_domain *parent_domain,
- const struct iommu_user_data *user_data);
+ const struct iommufd_viommu_ops *(*get_viommu_ops)(
+ struct device *dev, enum iommu_viommu_type viommu_type);
const struct iommu_domain_ops *default_domain_ops;
struct module *owner;
diff --git a/include/linux/iommufd.h b/include/linux/iommufd.h
index cfcf53b7c9e8..98ac3e5e9c66 100644
--- a/include/linux/iommufd.h
+++ b/include/linux/iommufd.h
@@ -150,6 +150,15 @@ struct iommufd_hw_queue {
/**
* struct iommufd_viommu_ops - vIOMMU specific operations
+ * @get_viommu_size: Get the size of a driver-level vIOMMU structure for a given
+ * @dev corresponding to @viommu_type. Driver should return 0
+ * if vIOMMU isn't supported accordingly. It is required for
+ * driver to use the VIOMMU_STRUCT_SIZE macro to sanitize the
+ * driver-level vIOMMU structure related to the core one
+ * @viommu_init: Init the driver-level struct of an iommufd_viommu on a physical
+ * IOMMU instance @viommu->iommu_dev, as the set of virtualization
+ * resources shared/passed to user space IOMMU instance. Associate
+ * it with a nesting @parent_domain.
* @destroy: Clean up all driver-specific parts of an iommufd_viommu. The memory
* of the vIOMMU will be free-ed by iommufd core after calling this op
* @alloc_domain_nested: Allocate a IOMMU_DOMAIN_NESTED on a vIOMMU that holds a
@@ -191,6 +200,11 @@ struct iommufd_hw_queue {
* does, it should set it to the @hw_queue->destroy pointer
*/
struct iommufd_viommu_ops {
+ size_t (*get_viommu_size)(struct device *dev,
+ enum iommu_viommu_type type);
+ int (*viommu_init)(struct iommufd_viommu *viommu, struct device *dev,
+ struct iommu_domain *parent_domain,
+ const struct iommu_user_data *user_data);
void (*destroy)(struct iommufd_viommu *viommu);
struct iommu_domain *(*alloc_domain_nested)(
struct iommufd_viommu *viommu, u32 flags,
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 14/16] iommufd/viommu: Allow PCI TSM backends to provide vIOMMU operations
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (12 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 13/16] iommufd/viommu: Select vIOMMU operations before allocation Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 15/16] iommufd: Allow vIOMMUs without a parent HWPT Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 16/16] PCI/TSM: wait for vdevice contexts before removing a DSM Aneesh Kumar K.V (Arm)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra
In confidential-computing configurations, a vIOMMU may be implemented by
a TEE Security Manager rather than by the physical IOMMU driver. Such an
implementation also needs the TSM identity when creating and validating
vdevices.
Allow VIOMMU_ALLOC to first query the TSM associated with the target PCI
device for vIOMMU operations. Fall back to the physical IOMMU driver
when the TSM does not provide operations for the requested vIOMMU type.
Resolve the PCI TSM association under pci_tsm_rwsem and pin the module
providing the operations. Store the TSM device in the vIOMMU so that the
backend operations and TSM identity remain valid for its lifetime.
Release the module pin when vIOMMU is destroyed.
This enables the Realm vIOMMU implementation to provide its operations
through the PCI TSM backend and retain the TSM identity needed by its
vdevices.
Cc: jgg@ziepe.ca
Cc: kevin.tian@intel.com
Cc: joro@8bytes.org
Cc: will@kernel.org
Cc: robin.murphy@arm.com
Cc: bhelgaas@google.com
Cc: iommu@lists.linux.dev
Cc: linux-pci@vger.kernel.org
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/iommu/iommufd/viommu.c | 35 +++++++++++++++++++++++++++++-----
drivers/pci/tsm.c | 21 ++++++++++++++++++++
drivers/virt/coco/tsm-core.c | 33 ++++++++++++++++++++++++++++++++
include/linux/iommufd.h | 2 ++
include/linux/pci-tsm.h | 22 +++++++++++++++++++++
include/linux/tsm.h | 23 ++++++++++++++++++++++
6 files changed, 131 insertions(+), 5 deletions(-)
diff --git a/drivers/iommu/iommufd/viommu.c b/drivers/iommu/iommufd/viommu.c
index e95138ae71d5..8628161b37c6 100644
--- a/drivers/iommu/iommufd/viommu.c
+++ b/drivers/iommu/iommufd/viommu.c
@@ -14,6 +14,8 @@ void iommufd_viommu_destroy(struct iommufd_object *obj)
if (viommu->ops && viommu->ops->destroy)
viommu->ops->destroy(viommu);
+ if (viommu->tsm_dev)
+ tsm_put_device(viommu->tsm_dev);
refcount_dec(&viommu->hwpt->common.obj.users);
if (viommu->kvm_file)
fput(viommu->kvm_file);
@@ -29,6 +31,7 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
.len = cmd->data_len,
};
struct iommufd_hwpt_paging *hwpt_paging;
+ struct tsm_dev *tsm_dev = NULL;
struct iommufd_viommu *viommu;
struct iommufd_device *idev;
struct iommu_device *iommu_dev;
@@ -48,15 +51,33 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
rc = -EOPNOTSUPP;
goto out_put_idev;
}
- if (!iommu_dev->ops->get_viommu_ops) {
- rc = -EOPNOTSUPP;
+ tsm_dev = tsm_get_device(idev->dev);
+ if (IS_ERR(tsm_dev)) {
+ rc = PTR_ERR(tsm_dev);
+ tsm_dev = NULL;
goto out_put_idev;
}
- ops = iommu_dev->ops->get_viommu_ops(idev->dev, cmd->type);
- if (!ops) {
- rc = -EOPNOTSUPP;
+ ops = tsm_dev ? tsm_viommu_get_ops(tsm_dev, idev->dev, cmd->type) : NULL;
+ if (IS_ERR(ops)) {
+ rc = PTR_ERR(ops);
goto out_put_idev;
}
+ if (!ops) {
+ if (tsm_dev) {
+ tsm_put_device(tsm_dev);
+ tsm_dev = NULL;
+ }
+ if (!iommu_dev->ops->get_viommu_ops) {
+ rc = -EOPNOTSUPP;
+ goto out_put_idev;
+ }
+ ops = iommu_dev->ops->get_viommu_ops(idev->dev, cmd->type);
+ if (!ops) {
+ rc = -EOPNOTSUPP;
+ goto out_put_idev;
+ }
+ }
+
/*
* It is a driver bug to omit the required operations or provide a size
* smaller than the core vIOMMU structure.
@@ -94,6 +115,8 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
viommu->kvm_file = get_file(idev->kvm_file);
viommu->type = cmd->type;
viommu->ictx = ucmd->ictx;
+ viommu->tsm_dev = tsm_dev;
+ tsm_dev = NULL;
viommu->hwpt = hwpt_paging;
refcount_inc(&viommu->hwpt->common.obj.users);
INIT_LIST_HEAD(&viommu->veventqs);
@@ -118,6 +141,8 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
out_put_hwpt:
iommufd_put_object(ucmd->ictx, &hwpt_paging->common.obj);
out_put_idev:
+ if (tsm_dev)
+ tsm_put_device(tsm_dev);
iommufd_put_object(ucmd->ictx, &idev->obj);
return rc;
}
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index dca9634f8d5e..c8603867e09d 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -9,6 +9,8 @@
#define dev_fmt(fmt) "PCI/TSM: " fmt
#include <linux/bitfield.h>
+#include <linux/iommufd.h>
+#include <linux/module.h>
#include <linux/pci.h>
#include <linux/pci-doe.h>
#include <linux/pci-tsm.h>
@@ -36,6 +38,25 @@ static const struct pci_tsm_ops *to_pci_tsm_ops(struct pci_tsm *tsm)
return tsm->tsm_dev->pci_ops;
}
+struct tsm_dev *pci_tsm_get_device(struct pci_dev *pdev)
+{
+ const struct pci_tsm_ops *ops;
+ struct tsm_dev *tsm_dev;
+
+ guard(rwsem_read)(&pci_tsm_rwsem);
+ if (!pdev->tsm)
+ return NULL;
+
+ tsm_dev = pdev->tsm->tsm_dev;
+ ops = tsm_dev->pci_ops;
+ if (!ops->viommu_get_ops)
+ return NULL;
+ if (!try_module_get(ops->owner))
+ return ERR_PTR(-ENODEV);
+ return tsm_dev;
+}
+EXPORT_SYMBOL_GPL(pci_tsm_get_device);
+
static inline bool is_dsm(struct pci_dev *pdev)
{
return pdev->tsm && pdev->tsm->dsm_dev == pdev;
diff --git a/drivers/virt/coco/tsm-core.c b/drivers/virt/coco/tsm-core.c
index f79135986102..2fce341ff923 100644
--- a/drivers/virt/coco/tsm-core.c
+++ b/drivers/virt/coco/tsm-core.c
@@ -9,6 +9,16 @@
#include <linux/cleanup.h>
#include <linux/pci-tsm.h>
+/* The caller must hold a tsm_get_device() reference for the entire use. */
+const struct iommufd_viommu_ops *tsm_viommu_get_ops(struct tsm_dev *tsm_dev,
+ struct device *dev, enum iommu_viommu_type type)
+{
+ if (!tsm_dev->pci_ops->viommu_get_ops)
+ return NULL;
+ return tsm_dev->pci_ops->viommu_get_ops(dev, type);
+}
+EXPORT_SYMBOL_GPL(tsm_viommu_get_ops);
+
static void tsm_release(struct device *);
static const struct class tsm_class = {
.name = "tsm",
@@ -16,6 +26,29 @@ static const struct class tsm_class = {
};
static DEFINE_IDA(tsm_ida);
+/**
+ * tsm_get_device() - Pin the TSM attached to a device
+ * @dev: device whose TSM is to be pinned
+ *
+ * A successful lookup pins the backend module. Backends providing vIOMMU
+ * operations must not unregister their TSM independently of module unload.
+ * Return: NULL if no vIOMMU-capable TSM is attached, an error if the module
+ * cannot be pinned, or the TSM device associated with the pinned backend.
+ */
+struct tsm_dev *tsm_get_device(struct device *dev)
+{
+ if (!dev_is_pci(dev))
+ return NULL;
+ return pci_tsm_get_device(to_pci_dev(dev));
+}
+EXPORT_SYMBOL_GPL(tsm_get_device);
+
+void tsm_put_device(struct tsm_dev *tsm_dev)
+{
+ module_put(tsm_dev->pci_ops->owner);
+}
+EXPORT_SYMBOL_GPL(tsm_put_device);
+
static int match_id(struct device *dev, const void *data)
{
struct tsm_dev *tsm_dev = container_of(dev, struct tsm_dev, dev);
diff --git a/include/linux/iommufd.h b/include/linux/iommufd.h
index 98ac3e5e9c66..fd4567940243 100644
--- a/include/linux/iommufd.h
+++ b/include/linux/iommufd.h
@@ -23,6 +23,7 @@ struct iommufd_access;
struct iommufd_ctx;
struct iommufd_device;
struct iommufd_viommu_ops;
+struct tsm_dev;
struct page;
struct tsm_guest_req_info;
@@ -105,6 +106,7 @@ struct iommufd_viommu {
struct iommu_device *iommu_dev;
struct iommufd_hwpt_paging *hwpt;
struct file *kvm_file;
+ struct tsm_dev *tsm_dev;
const struct iommufd_viommu_ops *ops;
diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
index eaf0199d01f7..b27f7cf99f22 100644
--- a/include/linux/pci-tsm.h
+++ b/include/linux/pci-tsm.h
@@ -4,6 +4,8 @@
#include <linux/mutex.h>
#include <linux/pci.h>
+struct iommufd_viommu_ops;
+struct module;
struct pci_tsm;
struct pci_tsm_context;
struct tsm_dev;
@@ -16,12 +18,26 @@ struct tsm_dev;
* @devsec_ops: Lock, unlock, and interrogate the security state of the
* function via the platform TSM (typically virtual function
* operations).
+ * @viommu_get_ops: Return operations for a supported TSM-owned vIOMMU type,
+ * or NULL if unsupported
+ * @owner: Module providing the vIOMMU operations, if built as a module
+ *
+ * The vIOMMU owner pins the backend module before querying @viommu_get_ops
+ * and releases it after vIOMMU destruction. A backend providing this callback
+ * must not unregister its TSM while the module is pinned.
*
* This operations are mutually exclusive either a tsm_dev instance
* manages physical link properties or it manages function security
* states like TDISP lock/unlock.
+ *
+ * @viommu_get_ops runs with the backend module pinned. The PCI/TSM
+ * association is checked during acquisition under pci_tsm_rwsem.
*/
struct pci_tsm_ops {
+ struct module *owner;
+ const struct iommufd_viommu_ops *(*viommu_get_ops)(
+ struct device *dev, enum iommu_viommu_type type);
+
/*
* struct pci_tsm_link_ops - Manage physical link and the TSM/DSM session
* @probe: establish context with the TSM (allocate / wrap 'struct
@@ -151,6 +167,7 @@ struct pci_tsm_pf0 *pci_tsm_context_pf0(struct pci_tsm_context *context);
struct pci_dev *pci_tsm_context_dsm_dev(struct pci_tsm_context *context);
bool pci_tsm_context_match_device(struct pci_tsm_context *context,
struct pci_dev *pdev);
+struct tsm_dev *pci_tsm_get_device(struct pci_dev *pdev);
#else
static inline int pci_tsm_register(struct tsm_dev *tsm_dev)
{
@@ -165,6 +182,11 @@ static inline bool pci_tsm_is_configured(struct pci_dev *pdev)
return false;
}
+static inline struct tsm_dev *pci_tsm_get_device(struct pci_dev *pdev)
+{
+ return NULL;
+}
+
static inline struct pci_tsm_context *
pci_tsm_context_get(struct pci_dev *pdev, struct tsm_dev *viommu_tsm_dev)
{
diff --git a/include/linux/tsm.h b/include/linux/tsm.h
index 4bce4c9fa6f0..e5d14f9a23e9 100644
--- a/include/linux/tsm.h
+++ b/include/linux/tsm.h
@@ -110,6 +110,8 @@ struct tsm_report_ops {
};
struct pci_tsm_ops;
+struct iommufd_viommu_ops;
+
struct tsm_dev {
struct device dev;
int id;
@@ -126,6 +128,11 @@ void tsm_unregister(struct tsm_dev *tsm_dev);
struct tsm_dev *find_tsm_dev(int id);
#ifdef CONFIG_TSM
+struct tsm_dev *tsm_get_device(struct device *dev);
+void tsm_put_device(struct tsm_dev *tsm_dev);
+const struct iommufd_viommu_ops *tsm_viommu_get_ops(struct tsm_dev *tsm_dev,
+ struct device *dev, enum iommu_viommu_type type);
+
/**
* struct tsm_guest_req_info - parameters for a guest-initiated TSM request
* @op: operation for the guest-initiated request
@@ -144,6 +151,22 @@ struct tsm_guest_req_info {
u32 resp_len;
};
#else
+static inline struct tsm_dev *tsm_get_device(struct device *dev)
+{
+ return NULL;
+}
+
+static inline void tsm_put_device(struct tsm_dev *tsm_dev)
+{
+}
+
+static inline const struct iommufd_viommu_ops *
+tsm_viommu_get_ops(struct tsm_dev *tsm_dev, struct device *dev,
+ enum iommu_viommu_type type)
+{
+ return NULL;
+}
+
struct tsm_guest_req_info;
#endif
#endif /* __TSM_H */
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 15/16] iommufd: Allow vIOMMUs without a parent HWPT
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (13 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 14/16] iommufd/viommu: Allow PCI TSM backends to provide vIOMMU operations Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
2026-10-08 5:59 ` [PATCH v7 16/16] PCI/TSM: wait for vdevice contexts before removing a DSM Aneesh Kumar K.V (Arm)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, corbet, skhan, rdunlap
VIOMMU_ALLOC currently requires a nesting parent HWPT_PAGING even when
the selected vIOMMU implementation has no use for one. This would force
implementations such as the following Arm Realm vIOMMU to create an
unused parent.
Add IOMMUFD_VIOMMU_NO_HWPT so an implementation can opt out. Require
hwpt_id to be zero in that case and pass a NULL parent domain to its
initialization callback. Keep the existing parent validation and
reference handling for implementations that require a HWPT.
Reject nested domain and hardware queue allocation for a vIOMMU without
a parent, and document the two allocation modes in the UAPI.
Cc: jgg@ziepe.ca
Cc: kevin.tian@intel.com
Cc: corbet@lwn.net
Cc: skhan@linuxfoundation.org
Cc: rdunlap@infradead.org
Cc: joro@8bytes.org
Cc: will@kernel.org
Cc: robin.murphy@arm.com
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
Documentation/userspace-api/iommufd.rst | 44 +++++++++----------
drivers/iommu/iommufd/hw_pagetable.c | 14 +++++--
drivers/iommu/iommufd/viommu.c | 56 +++++++++++++++----------
include/linux/iommufd.h | 13 ++++--
include/uapi/linux/iommufd.h | 3 +-
5 files changed, 77 insertions(+), 53 deletions(-)
diff --git a/Documentation/userspace-api/iommufd.rst b/Documentation/userspace-api/iommufd.rst
index f1c4d21e5c5e..8e2fa3a524f0 100644
--- a/Documentation/userspace-api/iommufd.rst
+++ b/Documentation/userspace-api/iommufd.rst
@@ -82,11 +82,10 @@ Following IOMMUFD objects are exposed to userspace:
* Direct assigned invalidation queues
* Direct assigned interrupts
- Such a vIOMMU object generally has the access to a nesting parent pagetable
- to support some HW-accelerated virtualization features. So, a vIOMMU object
- must be created given a nesting parent HWPT_PAGING object, and then it would
- encapsulate that HWPT_PAGING object. Therefore, a vIOMMU object can be used
- to allocate an HWPT_NESTED object in place of the encapsulated HWPT_PAGING.
+ A vIOMMU object may encapsulate a nesting parent HWPT_PAGING object to
+ support HW-accelerated virtualization features. Whether a parent is required
+ is determined by the selected vIOMMU implementation. Only a parent-backed
+ vIOMMU can be used to allocate an HWPT_NESTED object.
.. note::
@@ -226,9 +225,9 @@ creating the objects and links::
flag is set.
4. IOMMUFD_OBJ_HWPT_NESTED can be only manually created via the IOMMU_HWPT_ALLOC
- uAPI, provided an hwpt_id or a viommu_id of a vIOMMU object encapsulating a
- nesting parent HWPT_PAGING via @pt_id to associate the new HWPT_NESTED object
- to the corresponding HWPT_PAGING object. The associating HWPT_PAGING object
+ uAPI, provided an hwpt_id or a viommu_id of a parent-backed vIOMMU object
+ via @pt_id to associate the new HWPT_NESTED object to the corresponding
+ HWPT_PAGING object. The associating HWPT_PAGING object
must be a nesting parent manually allocated via the same uAPI previously with
an IOMMU_HWPT_ALLOC_NEST_PARENT flag, otherwise the allocation will fail. The
allocation will be further validated by the IOMMU driver to ensure that the
@@ -245,25 +244,28 @@ creating the objects and links::
of the object passed in via the @pt_id field of struct iommufd_hwpt_alloc.
5. IOMMUFD_OBJ_VIOMMU can be only manually created via the IOMMU_VIOMMU_ALLOC
- uAPI, provided a dev_id (for the device's physical IOMMU to back the vIOMMU)
- and an hwpt_id (to associate the vIOMMU to a nesting parent HWPT_PAGING). The
- iommufd core will link the vIOMMU object to the struct iommu_device that the
- struct device is behind. And an IOMMU driver can implement a viommu_alloc op
- to allocate its own vIOMMU data structure embedding the core-level structure
- iommufd_viommu and some driver-specific data. If necessary, the driver can
- also configure its HW virtualization feature for that vIOMMU (and thus for
- the VM). Successful completion of this operation sets up the linkages between
- the vIOMMU object and the HWPT_PAGING, then this vIOMMU object can be used
- as a nesting parent object to allocate an HWPT_NESTED object described above.
+ uAPI, provided a dev_id identifying the device used to select the vIOMMU
+ implementation and, if the selected implementation requires a nesting
+ parent HWPT_PAGING, an hwpt_id. The hwpt_id must be zero for an
+ implementation that does not use a parent. A physical-IOMMU implementation
+ links the vIOMMU object to the struct iommu_device behind the device. Other
+ implementations, such as a TSM associated with the device, can provide their
+ own backing and device association. An implementation can allocate its own
+ vIOMMU data structure embedding the core-level structure iommufd_viommu and
+ some implementation-specific data. If necessary, it can also configure its
+ virtualization resources for that vIOMMU (and thus for the VM). A
+ parent-backed vIOMMU can be used as a nesting parent object to allocate an
+ HWPT_NESTED object described above.
6. IOMMUFD_OBJ_VDEVICE can be only manually created via the IOMMU_VDEVICE_ALLOC
uAPI, provided a viommu_id for an iommufd_viommu object and a dev_id for an
iommufd_device object. The vDEVICE object will be the binding between these
two parent objects. Another @virt_id will be also set via the uAPI providing
the iommufd core an index to store the vDEVICE object to a vDEVICE array per
- vIOMMU. If necessary, the IOMMU driver may choose to implement a vdevce_alloc
- op to init its HW for virtualization feature related to a vDEVICE. Successful
- completion of this operation sets up the linkages between vIOMMU and device.
+ vIOMMU. A physical-IOMMU implementation requires the device to be behind the
+ same iommu_device as the vIOMMU. Other implementations validate the device
+ association while initializing the vDEVICE. Successful initialization sets
+ up the linkage between the vIOMMU and device.
A device can only bind to an iommufd due to DMA ownership claim and attach to at
most one IOAS object (no support of PASID yet).
diff --git a/drivers/iommu/iommufd/hw_pagetable.c b/drivers/iommu/iommufd/hw_pagetable.c
index ef6e119c2a75..bd562bc2db48 100644
--- a/drivers/iommu/iommufd/hw_pagetable.c
+++ b/drivers/iommu/iommufd/hw_pagetable.c
@@ -306,6 +306,10 @@ iommufd_viommu_alloc_hwpt_nested(struct iommufd_viommu *viommu, u32 flags,
return ERR_PTR(-EOPNOTSUPP);
if (!user_data->len)
return ERR_PTR(-EOPNOTSUPP);
+ if (!viommu->hwpt)
+ return ERR_PTR(-EOPNOTSUPP);
+ if (!viommu->iommu_dev)
+ return ERR_PTR(-EOPNOTSUPP);
if (!viommu->ops || !viommu->ops->alloc_domain_nested)
return ERR_PTR(-EOPNOTSUPP);
@@ -404,10 +408,12 @@ int iommufd_hwpt_alloc(struct iommufd_ucmd *ucmd)
struct iommufd_viommu *viommu;
viommu = container_of(pt_obj, struct iommufd_viommu, obj);
- iommu_dev = iommufd_device_get_iommu_dev(idev);
- if (!iommu_dev || viommu->iommu_dev != iommu_dev) {
- rc = -EINVAL;
- goto out_unlock;
+ if (viommu->iommu_dev) {
+ iommu_dev = iommufd_device_get_iommu_dev(idev);
+ if (!iommu_dev || viommu->iommu_dev != iommu_dev) {
+ rc = -EINVAL;
+ goto out_unlock;
+ }
}
hwpt_nested = iommufd_viommu_alloc_hwpt_nested(
viommu, cmd->flags, &user_data);
diff --git a/drivers/iommu/iommufd/viommu.c b/drivers/iommu/iommufd/viommu.c
index 8628161b37c6..9155d0dcb4b4 100644
--- a/drivers/iommu/iommufd/viommu.c
+++ b/drivers/iommu/iommufd/viommu.c
@@ -16,7 +16,8 @@ void iommufd_viommu_destroy(struct iommufd_object *obj)
viommu->ops->destroy(viommu);
if (viommu->tsm_dev)
tsm_put_device(viommu->tsm_dev);
- refcount_dec(&viommu->hwpt->common.obj.users);
+ if (viommu->hwpt)
+ refcount_dec(&viommu->hwpt->common.obj.users);
if (viommu->kvm_file)
fput(viommu->kvm_file);
xa_destroy(&viommu->vdevs);
@@ -30,11 +31,11 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
.uptr = u64_to_user_ptr(cmd->data_uptr),
.len = cmd->data_len,
};
- struct iommufd_hwpt_paging *hwpt_paging;
+ struct iommufd_hwpt_paging *hwpt_paging = NULL;
struct tsm_dev *tsm_dev = NULL;
struct iommufd_viommu *viommu;
struct iommufd_device *idev;
- struct iommu_device *iommu_dev;
+ struct iommu_device *iommu_dev = NULL;
const struct iommufd_viommu_ops *ops;
size_t viommu_size;
int rc;
@@ -46,11 +47,6 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
if (IS_ERR(idev))
return PTR_ERR(idev);
- iommu_dev = iommufd_device_get_iommu_dev(idev);
- if (!iommu_dev) {
- rc = -EOPNOTSUPP;
- goto out_put_idev;
- }
tsm_dev = tsm_get_device(idev->dev);
if (IS_ERR(tsm_dev)) {
rc = PTR_ERR(tsm_dev);
@@ -67,6 +63,11 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
tsm_put_device(tsm_dev);
tsm_dev = NULL;
}
+ iommu_dev = iommufd_device_get_iommu_dev(idev);
+ if (!iommu_dev) {
+ rc = -EOPNOTSUPP;
+ goto out_put_idev;
+ }
if (!iommu_dev->ops->get_viommu_ops) {
rc = -EOPNOTSUPP;
goto out_put_idev;
@@ -92,15 +93,20 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
goto out_put_idev;
}
- hwpt_paging = iommufd_get_hwpt_paging(ucmd, cmd->hwpt_id);
- if (IS_ERR(hwpt_paging)) {
- rc = PTR_ERR(hwpt_paging);
+ if ((ops->flags & IOMMUFD_VIOMMU_NO_HWPT) && cmd->hwpt_id) {
+ rc = -EINVAL;
goto out_put_idev;
}
-
- if (!hwpt_paging->nest_parent) {
- rc = -EINVAL;
- goto out_put_hwpt;
+ if (!(ops->flags & IOMMUFD_VIOMMU_NO_HWPT)) {
+ hwpt_paging = iommufd_get_hwpt_paging(ucmd, cmd->hwpt_id);
+ if (IS_ERR(hwpt_paging)) {
+ rc = PTR_ERR(hwpt_paging);
+ goto out_put_idev;
+ }
+ if (!hwpt_paging->nest_parent) {
+ rc = -EINVAL;
+ goto out_put_hwpt;
+ }
}
viommu = (struct iommufd_viommu *)_iommufd_object_alloc_ucmd(
@@ -118,7 +124,8 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
viommu->tsm_dev = tsm_dev;
tsm_dev = NULL;
viommu->hwpt = hwpt_paging;
- refcount_inc(&viommu->hwpt->common.obj.users);
+ if (viommu->hwpt)
+ refcount_inc(&viommu->hwpt->common.obj.users);
INIT_LIST_HEAD(&viommu->veventqs);
init_rwsem(&viommu->veventqs_rwsem);
/*
@@ -129,7 +136,7 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
viommu->iommu_dev = iommu_dev;
rc = ops->viommu_init(viommu, idev->dev,
- hwpt_paging->common.domain,
+ hwpt_paging ? hwpt_paging->common.domain : NULL,
user_data.len ? &user_data : NULL);
if (rc)
goto out_put_hwpt;
@@ -139,7 +146,8 @@ int iommufd_viommu_alloc_ioctl(struct iommufd_ucmd *ucmd)
rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
out_put_hwpt:
- iommufd_put_object(ucmd->ictx, &hwpt_paging->common.obj);
+ if (hwpt_paging)
+ iommufd_put_object(ucmd->ictx, &hwpt_paging->common.obj);
out_put_idev:
if (tsm_dev)
tsm_put_device(tsm_dev);
@@ -202,10 +210,12 @@ int iommufd_vdevice_alloc_ioctl(struct iommufd_ucmd *ucmd)
goto out_put_viommu;
}
- iommu_dev = iommufd_device_get_iommu_dev(idev);
- if (!iommu_dev || viommu->iommu_dev != iommu_dev) {
- rc = -EINVAL;
- goto out_put_idev;
+ if (viommu->iommu_dev) {
+ iommu_dev = iommufd_device_get_iommu_dev(idev);
+ if (!iommu_dev || viommu->iommu_dev != iommu_dev) {
+ rc = -EINVAL;
+ goto out_put_idev;
+ }
}
mutex_lock(&idev->igroup->lock);
@@ -425,7 +435,7 @@ int iommufd_hw_queue_alloc_ioctl(struct iommufd_ucmd *ucmd)
if (IS_ERR(viommu))
return PTR_ERR(viommu);
- if (!viommu->ops || !viommu->ops->get_hw_queue_size ||
+ if (!viommu->hwpt || !viommu->ops || !viommu->ops->get_hw_queue_size ||
!viommu->ops->hw_queue_init_phys) {
rc = -EOPNOTSUPP;
goto out_put_viommu;
diff --git a/include/linux/iommufd.h b/include/linux/iommufd.h
index fd4567940243..2e67846d6f35 100644
--- a/include/linux/iommufd.h
+++ b/include/linux/iommufd.h
@@ -6,6 +6,7 @@
#ifndef __LINUX_IOMMUFD_H
#define __LINUX_IOMMUFD_H
+#include <linux/bits.h>
#include <linux/err.h>
#include <linux/errno.h>
#include <linux/iommu.h>
@@ -103,6 +104,7 @@ void iommufd_ctx_get(struct iommufd_ctx *ictx);
struct iommufd_viommu {
struct iommufd_object obj;
struct iommufd_ctx *ictx;
+ /* Physical IOMMU backing, or NULL. */
struct iommu_device *iommu_dev;
struct iommufd_hwpt_paging *hwpt;
struct file *kvm_file;
@@ -150,17 +152,19 @@ struct iommufd_hw_queue {
void (*destroy)(struct iommufd_hw_queue *hw_queue);
};
+#define IOMMUFD_VIOMMU_NO_HWPT BIT(0)
+
/**
* struct iommufd_viommu_ops - vIOMMU specific operations
+ * @flags: Properties needed by the core before vIOMMU initialization
* @get_viommu_size: Get the size of a driver-level vIOMMU structure for a given
* @dev corresponding to @viommu_type. Driver should return 0
* if vIOMMU isn't supported accordingly. It is required for
* driver to use the VIOMMU_STRUCT_SIZE macro to sanitize the
* driver-level vIOMMU structure related to the core one
- * @viommu_init: Init the driver-level struct of an iommufd_viommu on a physical
- * IOMMU instance @viommu->iommu_dev, as the set of virtualization
- * resources shared/passed to user space IOMMU instance. Associate
- * it with a nesting @parent_domain.
+ * @viommu_init: Initialize the selected driver-level iommufd_viommu for @dev.
+ * @parent_domain is the nesting parent, or NULL for an
+ * implementation that does not use a parent HWPT.
* @destroy: Clean up all driver-specific parts of an iommufd_viommu. The memory
* of the vIOMMU will be free-ed by iommufd core after calling this op
* @alloc_domain_nested: Allocate a IOMMU_DOMAIN_NESTED on a vIOMMU that holds a
@@ -202,6 +206,7 @@ struct iommufd_hw_queue {
* does, it should set it to the @hw_queue->destroy pointer
*/
struct iommufd_viommu_ops {
+ unsigned long flags;
size_t (*get_viommu_size)(struct device *dev,
enum iommu_viommu_type type);
int (*viommu_init)(struct iommufd_viommu *viommu, struct device *dev,
diff --git a/include/uapi/linux/iommufd.h b/include/uapi/linux/iommufd.h
index 9920138f9eda..3d1ebbdf3948 100644
--- a/include/uapi/linux/iommufd.h
+++ b/include/uapi/linux/iommufd.h
@@ -1130,7 +1130,8 @@ struct iommu_viommu_tegra241_cmdqv {
* @flags: Must be 0
* @type: Type of the virtual IOMMU. Must be defined in enum iommu_viommu_type
* @dev_id: The device's physical IOMMU will be used to back the virtual IOMMU
- * @hwpt_id: ID of a nesting parent HWPT to associate to
+ * @hwpt_id: ID of a nesting parent HWPT to associate to. Must be zero when the
+ * selected vIOMMU implementation does not use a parent HWPT
* @out_viommu_id: Output virtual IOMMU ID for the allocated object
* @data_len: Length of the type specific data
* @__reserved: Must be 0
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v7 16/16] PCI/TSM: wait for vdevice contexts before removing a DSM
2026-10-08 5:59 [PATCH v7 00/16] iommufd: vIOMMUs and TSM guest requests for confidential guests Aneesh Kumar K.V (Arm)
` (14 preceding siblings ...)
2026-10-08 5:59 ` [PATCH v7 15/16] iommufd: Allow vIOMMUs without a parent HWPT Aneesh Kumar K.V (Arm)
@ 2026-10-08 5:59 ` Aneesh Kumar K.V (Arm)
15 siblings, 0 replies; 17+ messages in thread
From: Aneesh Kumar K.V (Arm) @ 2026-10-08 5:59 UTC (permalink / raw)
To: iommu
Cc: Aneesh Kumar K.V (Arm),
Alex Williamson, Alexey Kardashevskiy, Bjorn Helgaas,
Catalin Marinas, Jacob Pan, Jason Gunthorpe, Joerg Roedel,
Jonathan Cameron, Jonathan Hunter, Kevin Tian, Krishna Reddy,
Lukas Wunner, Nicolin Chen, Robin Murphy, Samuel Ortiz,
Shameer Kolothum, Steven Price, Suravee Suthikulpanit,
Suzuki K Poulose, Thierry Reding, Vasant Hegde, Will Deacon,
Xu Yilun, kvm, linux-arm-kernel, linux-coco, linux-kernel,
linux-pci, linux-tegra, Jonathan Cameron
A PF0 DSM can be removed while a sibling function still has a vdevice.
The context pins both the pci dev, but those references do not keep the
PF0 DOE mailbox alive. Ignoring -EBUSY from link disconnect lets PCI
continue to pci_doe_destroy(), leaving the vdevice with a stale mailbox
pointer.
This follows the VFIO PCI removal model: vfio_unregister_group_dev()
prevents new userspace opens and waits for existing users to release the
device before teardown proceeds. Likewise, DSM removal rejects new
contexts and waits for existing vdevice contexts to drain before
destroying the DOE mailbox.
Mark the DSM as removing so no new contexts or subfunctions can attach.
Wait for the last context to be released before disconnecting the link,
VFIO PCI also uses an eventfd to notify userspace that the device should
be released. This patch does not add an equivalent notification for
vdevice contexts; DSM removal waits for userspace to release them. Such
a notification can be added later if required.
Cc: Bjorn Helgaas <bhelgaas@google.com>
Cc: Jonathan Cameron <jonathan.cameron@huawei.com>
Cc: Alexey Kardashevskiy <aik@amd.com>
Cc: Xu Yilun <yilun.xu@linux.intel.com>
Cc: Lukas Wunner <lukas@wunner.de>
Cc: Samuel Ortiz <sameo@rivosinc.com>
Cc: Suzuki K Poulose <suzuki.poulose@arm.com>
Cc: Jason Gunthorpe <jgg@ziepe.ca>
Cc: Kevin Tian <kevin.tian@intel.com>
Signed-off-by: Aneesh Kumar K.V (Arm) <aneesh.kumar@kernel.org>
---
drivers/pci/tsm.c | 63 +++++++++++++++++++++++++++++++++++++++--
include/linux/pci-tsm.h | 5 ++++
2 files changed, 65 insertions(+), 3 deletions(-)
diff --git a/drivers/pci/tsm.c b/drivers/pci/tsm.c
index c8603867e09d..8aa74ba31932 100644
--- a/drivers/pci/tsm.c
+++ b/drivers/pci/tsm.c
@@ -10,10 +10,13 @@
#include <linux/bitfield.h>
#include <linux/iommufd.h>
+#include <linux/jiffies.h>
#include <linux/module.h>
#include <linux/pci.h>
#include <linux/pci-doe.h>
#include <linux/pci-tsm.h>
+#include <linux/pid.h>
+#include <linux/sched.h>
#include <linux/slab.h>
#include <linux/sysfs.h>
#include <linux/tsm.h>
@@ -354,6 +357,12 @@ struct pci_tsm_context *pci_tsm_context_get(struct pci_dev *pdev,
return ERR_PTR(-ENOMEM);
guard(mutex)(&pf0->lock);
+ if (pf0->removing) {
+ kfree(context);
+ return ERR_PTR(-ENODEV);
+ }
+ if (!pf0->context_users)
+ reinit_completion(&pf0->contexts_drained);
pf0->context_users++;
context->pf0 = pf0;
context->pdev = pci_dev_get(pdev);
@@ -368,8 +377,11 @@ void pci_tsm_context_put(struct pci_tsm_context *context)
down_read(&pci_tsm_rwsem);
mutex_lock(&pf0->lock);
- if (!WARN_ON(!pf0->context_users))
- pf0->context_users--;
+ if (WARN_ON(!pf0->context_users))
+ goto out_unlock;
+ if (!--pf0->context_users)
+ complete_all(&pf0->contexts_drained);
+out_unlock:
mutex_unlock(&pf0->lock);
up_read(&pci_tsm_rwsem);
@@ -452,6 +464,9 @@ static ssize_t disconnect_store(struct device *dev,
tsm_dev = pdev->tsm->tsm_dev;
if (!sysfs_streq(buf, dev_name(&tsm_dev->dev)))
return -EINVAL;
+ if (is_link_tsm(tsm_dev) && is_pci_tsm_pf0(pdev) &&
+ to_pci_tsm_pf0(pdev->tsm)->removing)
+ return -ENODEV;
rc = pci_tsm_disconnect(pdev);
if (rc)
@@ -663,6 +678,9 @@ int pci_tsm_pf0_constructor(struct pci_dev *pdev, struct pci_tsm_pf0 *tsm,
struct tsm_dev *tsm_dev)
{
mutex_init(&tsm->lock);
+ init_completion(&tsm->contexts_drained);
+ tsm->context_users = 0;
+ tsm->removing = false;
tsm->doe_mb = pci_find_doe_mailbox(pdev, PCI_VENDOR_ID_PCI_SIG,
PCI_DOE_FEATURE_CMA);
if (!tsm->doe_mb) {
@@ -751,8 +769,44 @@ static void __pci_tsm_destroy(struct pci_dev *pdev, struct tsm_dev *tsm_dev)
void pci_tsm_destroy(struct pci_dev *pdev)
{
- guard(rwsem_write)(&pci_tsm_rwsem);
+ struct pci_tsm_pf0 *pf0 = NULL;
+ struct completion *drained;
+ bool interrupted = false;
+ long rc;
+
+ down_write(&pci_tsm_rwsem);
+ if (pdev->tsm && is_link_tsm(pdev->tsm->tsm_dev) &&
+ is_pci_tsm_pf0(pdev)) {
+ pf0 = to_pci_tsm_pf0(pdev->tsm);
+ drained = &pf0->contexts_drained;
+ mutex_lock(&pf0->lock);
+ pf0->removing = true;
+ mutex_unlock(&pf0->lock);
+
+ /* An unused DSM may never have completed contexts_drained. */
+ rc = pf0->context_users ?
+ try_wait_for_completion(drained) : 1;
+ /* Context release needs the read side of pci_tsm_rwsem. */
+ up_write(&pci_tsm_rwsem);
+ while (rc <= 0) {
+ if (interrupted) {
+ rc = wait_for_completion_timeout(drained, HZ * 10);
+ } else {
+ rc = wait_for_completion_interruptible_timeout(drained,
+ HZ * 10);
+ if (rc < 0) {
+ interrupted = true;
+ pci_warn(pdev, "Task \"%s\" (%d) blocked until vdevices are released\n",
+ current->comm, task_pid_nr(current));
+ }
+ }
+ if (!rc)
+ pci_warn(pdev, "TSM connection is in use, waiting for vdevices\n");
+ }
+ down_write(&pci_tsm_rwsem);
+ }
__pci_tsm_destroy(pdev, NULL);
+ up_write(&pci_tsm_rwsem);
}
void pci_tsm_init(struct pci_dev *pdev)
@@ -779,6 +833,9 @@ void pci_tsm_init(struct pci_dev *pdev)
*/
if (!dsm->tsm)
return;
+ if (is_link_tsm(dsm->tsm->tsm_dev) &&
+ to_pci_tsm_pf0(dsm->tsm)->removing)
+ return;
probe_fn(pdev, dsm);
}
diff --git a/include/linux/pci-tsm.h b/include/linux/pci-tsm.h
index b27f7cf99f22..3adc317d0f9b 100644
--- a/include/linux/pci-tsm.h
+++ b/include/linux/pci-tsm.h
@@ -1,6 +1,7 @@
/* SPDX-License-Identifier: GPL-2.0 */
#ifndef __PCI_TSM_H
#define __PCI_TSM_H
+#include <linux/completion.h>
#include <linux/mutex.h>
#include <linux/pci.h>
@@ -107,12 +108,16 @@ struct pci_tsm {
* @lock: mutual exclustion for pci_tsm_ops invocation
* @context_users: live per-function contexts on this PF0; a nonzero count
* blocks link disconnect
+ * @contexts_drained: completed when the last context is released
+ * @removing: reject new contexts while the DSM is being removed
* @doe_mb: PCIe Data Object Exchange mailbox
*/
struct pci_tsm_pf0 {
struct pci_tsm base_tsm;
struct mutex lock;
unsigned int context_users;
+ struct completion contexts_drained;
+ bool removing;
struct pci_doe_mb *doe_mb;
};
--
2.43.0
^ permalink raw reply [flat|nested] 17+ messages in thread