[PATCH RFC v5 2/6] iommu/arm-smmu-v3: Allocate streams individually
From: Peng Fan (OSS)
Date: Tue Oct 06 2026 - 08:15:56 EST
From: Peng Fan <peng.fan@xxxxxxx>
Change master->streams from an embedded array of struct arm_smmu_stream
to an array of pointers, with each stream individually allocated.
Prepare for shared-SID support where multiple masters will point to the
same stream object. With embedded structs, sharing requires duplicating
stream state and manually keeping fields like ste_installed in sync.
With individually allocated streams, a sharing master can simply point to
the existing stream.
The sort comparator is updated to dereference the pointer indirection.
No functional change.
Suggested-by: Nicolin Chen <nicolinc@xxxxxxxxxx>
Link: https://lore.kernel.org/linux-iommu/arG6hmng3NddGEHm@xxxxxxxxxx/
Assisted-by: LLM
Signed-off-by: Peng Fan <peng.fan@xxxxxxx>
---
.../iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c | 2 +-
drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c | 52 ++++++++++++++--------
drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h | 2 +-
drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c | 2 +-
4 files changed, 37 insertions(+), 21 deletions(-)
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c
index ab1078a97d801..0934a6bbd3e08 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c
@@ -294,7 +294,7 @@ static int arm_vsmmu_vsid_to_sid(struct arm_vsmmu *vsmmu, u32 vsid, u32 *sid)
/* At this moment, iommufd only supports PCI device that has one SID */
if (sid)
- *sid = master->streams[0].id;
+ *sid = master->streams[0]->id;
unlock:
xa_unlock(&vsmmu->core.vdevs);
return ret;
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
index dba6c2942af8f..9d34eac196a65 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
@@ -972,7 +972,7 @@ static void arm_smmu_page_response(struct device *dev, struct iopf_fault *unused
}
arm_smmu_cmdq_issue_cmd(master->smmu,
- arm_smmu_make_cmd_resume(master->streams[0].id,
+ arm_smmu_make_cmd_resume(master->streams[0]->id,
resp->grpid,
resume_resp));
/*
@@ -1504,7 +1504,7 @@ static void arm_smmu_sync_cd(struct arm_smmu_master *master,
for (i = 0; i < master->num_streams; i++)
arm_smmu_cmdq_batch_add_cmd(
smmu, &cmds,
- arm_smmu_make_cmd_cfgi_cd(master->streams[i].id, ssid,
+ arm_smmu_make_cmd_cfgi_cd(master->streams[i]->id, ssid,
leaf));
arm_smmu_cmdq_batch_submit(smmu, &cmds);
@@ -2431,7 +2431,7 @@ static int arm_smmu_atc_inv_master(struct arm_smmu_master *master,
for (i = 0; i < master->num_streams; i++)
arm_smmu_cmdq_batch_add_cmd(
master->smmu, &cmds,
- arm_smmu_make_cmd_atc_inv_all(master->streams[i].id,
+ arm_smmu_make_cmd_atc_inv_all(master->streams[i]->id,
ssid));
return arm_smmu_cmdq_batch_submit(master->smmu, &cmds);
@@ -2978,13 +2978,13 @@ void arm_smmu_install_ste_for_dev(struct arm_smmu_master *master,
STRTAB_STE_1_EATS_TRANS;
for (i = 0; i < master->num_streams; ++i) {
- u32 sid = master->streams[i].id;
+ u32 sid = master->streams[i]->id;
struct arm_smmu_ste *step =
arm_smmu_get_step_for_sid(smmu, sid);
/* Bridged PCI devices may end up with duplicated IDs */
for (j = 0; j < i; j++)
- if (master->streams[j].id == sid)
+ if (master->streams[j]->id == sid)
break;
if (j < i)
continue;
@@ -3276,7 +3276,7 @@ arm_smmu_master_build_invs(struct arm_smmu_master *master, bool ats_enabled,
*/
if (!arm_smmu_master_build_inv(
master, nesting ? INV_TYPE_ATS_FULL : INV_TYPE_ATS,
- master->streams[i].id, ssid, 0))
+ master->streams[i]->id, ssid, 0))
return NULL;
}
@@ -4078,10 +4078,10 @@ static int arm_smmu_init_sid_strtab(struct arm_smmu_device *smmu, u32 sid)
static int arm_smmu_stream_id_cmp(const void *_l, const void *_r)
{
- const typeof_member(struct arm_smmu_stream, id) *l = _l;
- const typeof_member(struct arm_smmu_stream, id) *r = _r;
+ const struct arm_smmu_stream * const *l = _l;
+ const struct arm_smmu_stream * const *r = _r;
- return cmp_int(*l, *r);
+ return cmp_int((*l)->id, (*r)->id);
}
static int arm_smmu_insert_master(struct arm_smmu_device *smmu,
@@ -4111,10 +4111,16 @@ static int arm_smmu_insert_master(struct arm_smmu_device *smmu,
}
for (i = 0; i < fwspec->num_ids; i++) {
- struct arm_smmu_stream *new_stream = &master->streams[i];
+ struct arm_smmu_stream *new_stream;
+ new_stream = kzalloc_obj(*new_stream, GFP_KERNEL);
+ if (!new_stream) {
+ ret = -ENOMEM;
+ goto out_free_streams;
+ }
new_stream->id = fwspec->ids[i];
new_stream->master = master;
+ master->streams[i] = new_stream;
}
/* Put the ids into order for sorted to_merge/to_unref arrays */
@@ -4127,11 +4133,11 @@ static int arm_smmu_insert_master(struct arm_smmu_device *smmu,
* which sort_nonatomic() invalidates by relocating the entries.
*/
for (i = 0; i < fwspec->num_ids; i++)
- RB_CLEAR_NODE(&master->streams[i].node);
+ RB_CLEAR_NODE(&master->streams[i]->node);
mutex_lock(&smmu->streams_mutex);
for (i = 0; i < fwspec->num_ids; i++) {
- struct arm_smmu_stream *new_stream = &master->streams[i];
+ struct arm_smmu_stream *new_stream = master->streams[i];
struct rb_node *existing;
u32 sid = new_stream->id;
@@ -4161,14 +4167,21 @@ static int arm_smmu_insert_master(struct arm_smmu_device *smmu,
if (ret) {
for (i--; i >= 0; i--)
- if (!RB_EMPTY_NODE(&master->streams[i].node))
- rb_erase(&master->streams[i].node,
+ if (!RB_EMPTY_NODE(&master->streams[i]->node))
+ rb_erase(&master->streams[i]->node,
&smmu->streams);
- kfree(master->streams);
- kfree(master->build_invs);
+ mutex_unlock(&smmu->streams_mutex);
+ goto out_free_streams;
}
mutex_unlock(&smmu->streams_mutex);
+ return 0;
+
+out_free_streams:
+ for (i = 0; i < master->num_streams; i++)
+ kfree(master->streams[i]);
+ kfree(master->streams);
+ kfree(master->build_invs);
return ret;
}
@@ -4183,10 +4196,13 @@ static void arm_smmu_remove_master(struct arm_smmu_master *master)
mutex_lock(&smmu->streams_mutex);
for (i = 0; i < fwspec->num_ids; i++)
- if (!RB_EMPTY_NODE(&master->streams[i].node))
- rb_erase(&master->streams[i].node, &smmu->streams);
+ if (!RB_EMPTY_NODE(&master->streams[i]->node))
+ rb_erase(&master->streams[i]->node, &smmu->streams);
mutex_unlock(&smmu->streams_mutex);
+ for (i = 0; i < master->num_streams; i++)
+ kfree(master->streams[i]);
+
kfree(master->streams);
kfree(master->build_invs);
}
diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
index dd2fee2f560e6..ba430078cbce9 100644
--- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
+++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.h
@@ -1000,7 +1000,7 @@ struct arm_smmu_event {
struct arm_smmu_master {
struct arm_smmu_device *smmu;
struct device *dev;
- struct arm_smmu_stream *streams;
+ struct arm_smmu_stream **streams;
/*
* Scratch memory for a to_merge or to_unref array to build a per-domain
* invalidation array. It'll be pre-allocated with enough enries for all
diff --git a/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c b/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c
index 6644075c1431e..bc62a3d5a63f9 100644
--- a/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c
+++ b/drivers/iommu/arm/arm-smmu-v3/tegra241-cmdqv.c
@@ -1257,7 +1257,7 @@ static int tegra241_vintf_init_vsid(struct iommufd_vdevice *vdev)
struct arm_smmu_master *master = dev_iommu_priv_get(dev);
struct tegra241_vintf *vintf = viommu_to_vintf(vdev->viommu);
struct tegra241_vintf_sid *vsid = vdev_to_vsid(vdev);
- struct arm_smmu_stream *stream = &master->streams[0];
+ struct arm_smmu_stream *stream = master->streams[0];
u64 virt_sid = vdev->virt_id;
int sidx;
--
2.34.1