iommu_group_alloc() eagerly creates two attribute files -
reserved_regions and type - that are read only on the VFIO /
iommufd ioctl path (VFIO_IOMMU_GET_INFO, iommufd type query).
On systems with thousands of SR-IOV VFs each VF typically gets
its own iommu_group, so deferring these two attrs eliminates two
kernfs nodes per group at boot.

iommu_group_ktype is the first non-device_ktype consumer of the
populate mechanism: wire iommu_group_populate_one /
iommu_group_populate_all and call kernfs_set_lazy() on the
group's kernfs node immediately after kobject_init_and_add().
See the in-diff comment at the kernfs_set_lazy() site for the
ordering rationale (must precede kobject_create_and_add("devices")
to interact correctly with kernfs_inc_rev()).

Both callbacks fast-path on iommu_group_sysfs_populated() so
re-entry after the directory is fully populated is cheap and
cannot fire sysfs_warn_dup().

The name attribute is created on demand by iommu_group_set_name()
on a separate API path and is not handled by these callbacks;
groups without a set name return -ENOENT on lookup of name.

No userspace ABI change.  A selftest for this deferral is added
later in the series under tools/testing/selftests/sysfs-lazy/.

Cc: Joerg Roedel <[email protected]>
Cc: Will Deacon <[email protected]>
Cc: Robin Murphy <[email protected]>
Cc: Jason Gunthorpe <[email protected]>
Cc: Kevin Tian <[email protected]>
Cc: Shuah Khan <[email protected]>
Cc: [email protected]
Cc: [email protected]
Cc: [email protected]
Cc: [email protected]
Cc: Greg Kroah-Hartman <[email protected]>
Cc: Danilo Krummrich <[email protected]>
Cc: [email protected]
Assisted-by: Claude:claude-opus-4.7
Signed-off-by: Pavol Sakac <[email protected]>
---
 drivers/iommu/iommu.c | 132 +++++++++++++++++++++++++++++++++++++-----
 1 file changed, 119 insertions(+), 13 deletions(-)

diff --git a/drivers/iommu/iommu.c b/drivers/iommu/iommu.c
index 61c12ba782066..7a8dd43f0a09a 100644
--- a/drivers/iommu/iommu.c
+++ b/drivers/iommu/iommu.c
@@ -21,6 +21,7 @@
 #include <linux/iommufd.h>
 #include <linux/idr.h>
 #include <linux/err.h>
+#include <linux/kernfs.h>
 #include <linux/pci.h>
 #include <linux/pci-ats.h>
 #include <linux/bitops.h>
@@ -71,8 +72,22 @@ struct iommu_group {
        struct list_head entry;
        unsigned int owner_cnt;
        void *owner;
+       /* Embedded lazy-sysfs state (every iommu_group is lazy). */
+       struct sysfs_lazy_state sysfs_lazy;
 };
 
+static inline bool iommu_group_sysfs_populated(const struct iommu_group *group)
+{
+       return READ_ONCE(group->sysfs_lazy.populated);
+}
+
+static inline void
+iommu_group_sysfs_set_populated(struct iommu_group *group)
+{
+       lockdep_assert_held(&group->sysfs_lazy.lock);
+       WRITE_ONCE(group->sysfs_lazy.populated, true);
+}
+
 struct group_device {
        struct list_head list;
        struct device *dev;
@@ -1033,9 +1048,99 @@ static void iommu_group_release(struct kobject *kobj)
        kfree(group);
 }
 
+/* Lazy iommu_group attrs (excludes "name", managed separately). */
+static const struct {
+       const char                      *name;
+       struct iommu_group_attribute    *attr;
+} iommu_group_lazy_attrs[] = {
+       { "reserved_regions", &iommu_group_attr_reserved_regions },
+       { "type",             &iommu_group_attr_type },
+};
+
+static int iommu_group_populate_one(struct kobject *kobj, const char *name)
+{
+       struct iommu_group *group = to_iommu_group(kobj);
+       size_t i;
+       int ret = -ENOENT;
+       bool name_present = false;
+
+       /* Fast path: directory fully populated; kernfs has authoritative 
state. */
+       if (iommu_group_sysfs_populated(group))
+               return -ENOENT;
+
+       mutex_lock(&group->sysfs_lazy.lock);
+       /* Re-check under the lock against a concurrent populate_all. */
+       if (iommu_group_sysfs_populated(group)) {
+               ret = -ENOENT;
+               goto out;
+       }
+
+       /* Materialise the whole table (2 entries); cheaper than per-name 
dispatch. */
+       for (i = 0; i < ARRAY_SIZE(iommu_group_lazy_attrs); i++) {
+               struct iommu_group_attribute *attr =
+                       iommu_group_lazy_attrs[i].attr;
+               int rc;
+
+               if (sysfs_kn_exists(&group->kobj, attr->attr.name))
+                       rc = 0;
+               else
+                       rc = iommu_group_create_file(group, attr);
+
+               if (!strcmp(name, iommu_group_lazy_attrs[i].name)) {
+                       name_present = true;
+                       ret = rc;
+               } else if (rc) {
+                       pr_warn("group %d: lazy-create %s failed: %d\n",
+                               group->id,
+                               iommu_group_lazy_attrs[i].name, rc);
+               }
+       }
+       iommu_group_sysfs_set_populated(group);
+
+out:
+       mutex_unlock(&group->sysfs_lazy.lock);
+       return name_present ? ret : -ENOENT;
+}
+
+static void iommu_group_populate_all(struct kobject *kobj)
+{
+       struct iommu_group *group = to_iommu_group(kobj);
+       size_t i;
+       int ret;
+
+       /* Fast path: directory already fully populated. */
+       if (iommu_group_sysfs_populated(group))
+               return;
+
+       mutex_lock(&group->sysfs_lazy.lock);
+       if (iommu_group_sysfs_populated(group))
+               goto out;
+
+       for (i = 0; i < ARRAY_SIZE(iommu_group_lazy_attrs); i++) {
+               struct iommu_group_attribute *attr =
+                       iommu_group_lazy_attrs[i].attr;
+
+               /* Existence check absorbs the populate_one race. */
+               if (sysfs_kn_exists(&group->kobj, attr->attr.name))
+                       continue;
+
+               ret = iommu_group_create_file(group, attr);
+               if (ret)
+                       pr_warn("group %d: lazy-create %s failed: %d\n",
+                               group->id,
+                               iommu_group_lazy_attrs[i].name, ret);
+       }
+
+       iommu_group_sysfs_set_populated(group);
+out:
+       mutex_unlock(&group->sysfs_lazy.lock);
+}
+
 static const struct kobj_type iommu_group_ktype = {
        .sysfs_ops = &iommu_group_sysfs_ops,
        .release = iommu_group_release,
+       .populate     = iommu_group_populate_one,
+       .populate_all = iommu_group_populate_all,
 };
 
 /**
@@ -1060,6 +1165,7 @@ struct iommu_group *iommu_group_alloc(void)
 
        group->kobj.kset = iommu_group_kset;
        mutex_init(&group->mutex);
+       mutex_init(&group->sysfs_lazy.lock);
        INIT_LIST_HEAD(&group->devices);
        INIT_LIST_HEAD(&group->entry);
        xa_init(&group->pasid_array);
@@ -1078,6 +1184,19 @@ struct iommu_group *iommu_group_alloc(void)
                return ERR_PTR(ret);
        }
 
+       /*
+        * Defer reserved_regions and type - both are read only on
+        * VFIO/iommufd ioctl paths, so let the populate callbacks
+        * materialise them on demand.
+        *
+        * KERNFS_LAZY MUST be set before kobject_create_and_add("devices"),
+        * which calls kernfs_inc_rev() on this kobject's directory and
+        * would otherwise leave a stale negative dentry cached for a
+        * missing reserved_regions/type child.  WARN_ON because
+        * iommu_group_alloc() controls all group kobjects directly.
+        */
+       WARN_ON(kernfs_set_lazy(group->kobj.sd));
+
        group->devices_kobj = kobject_create_and_add("devices", &group->kobj);
        if (!group->devices_kobj) {
                kobject_put(&group->kobj); /* triggers .release & free */
@@ -1091,19 +1210,6 @@ struct iommu_group *iommu_group_alloc(void)
         */
        kobject_put(&group->kobj);
 
-       ret = iommu_group_create_file(group,
-                                     &iommu_group_attr_reserved_regions);
-       if (ret) {
-               kobject_put(group->devices_kobj);
-               return ERR_PTR(ret);
-       }
-
-       ret = iommu_group_create_file(group, &iommu_group_attr_type);
-       if (ret) {
-               kobject_put(group->devices_kobj);
-               return ERR_PTR(ret);
-       }
-
        pr_debug("Allocated group %d\n", group->id);
 
        return group;
-- 
2.47.3




Amazon Web Services Development Center Germany GmbH
Tamara-Danz-Str. 13
10243 Berlin
Geschaeftsfuehrung: Christof Hellmis, Andreas Stieger
Eingetragen am Amtsgericht Charlottenburg unter HRB 257764 B
Sitz: Berlin
Ust-ID: DE 365 538 597


Reply via email to