Re: [PATCH RFC v5 3/6] iommu/arm-smmu-v3: Delay stream allocation to inside the mutex

From: Peng Fan

Date: Thu Oct 08 2026 - 09:47:54 EST


On Thu, Oct 08, 2026 at 01:55:36PM +0100, Robin Murphy wrote:
>On 06/10/2026 1:19 pm, Peng Fan (OSS) wrote:
>> From: Peng Fan <peng.fan@xxxxxxx>
>>
>> Move arm_smmu_stream allocation from upfront (before the mutex) into
>> the mutex-protected loop in arm_smmu_insert_master(). Instead of
>> pre-allocating all stream objects and then inserting them into the RB
>> tree, first look up whether the SID already exists in the tree. Only
>> allocate and insert a new stream when no existing entry is found, then
>> avoid unnecessary allocations when bridged PCI devices produce duplicated
>> IDs. Prepare the code for a subsequent patch that will reuse existing
>> streams when stream IDs are shared across masters.
>>
>> The sort is also moved after the mutex section, since streams are now
>> populated inside the loop rather than beforehand.
>>
>> Assisted-by: LLM
>> Signed-off-by: Peng Fan <peng.fan@xxxxxxx>
>> ---
>> drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c | 102 +++++++++++++++-------------
>> 1 file changed, 53 insertions(+), 49 deletions(-)
>>
>> diff --git a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
>> index 9d34eac196a65..69c2c3596b06a 100644
>> --- a/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
>> +++ b/drivers/iommu/arm/arm-smmu-v3/arm-smmu-v3.c
>> @@ -4110,52 +4110,29 @@ static int arm_smmu_insert_master(struct arm_smmu_device *smmu,
>> return -ENOMEM;
>> }
>> - for (i = 0; i < fwspec->num_ids; i++) {
>> - struct arm_smmu_stream *new_stream;
>> -
>> - new_stream = kzalloc_obj(*new_stream, GFP_KERNEL);
>> - if (!new_stream) {
>> - ret = -ENOMEM;
>> - goto out_free_streams;
>> - }
>> - new_stream->id = fwspec->ids[i];
>> - new_stream->master = master;
>> - master->streams[i] = new_stream;
>> - }
>> -
>> - /* Put the ids into order for sorted to_merge/to_unref arrays */
>> - sort(master->streams, master->num_streams,
>> - sizeof(master->streams[0]), arm_smmu_stream_id_cmp,
>> - NULL);
>> -
>> - /*
>> - * Clear after sorting: RB_CLEAR_NODE() records the node's own address,
>> - * which sort_nonatomic() invalidates by relocating the entries.
>> - */
>> - for (i = 0; i < fwspec->num_ids; i++)
>> - RB_CLEAR_NODE(&master->streams[i]->node);
>> -
>> mutex_lock(&smmu->streams_mutex);
>> for (i = 0; i < fwspec->num_ids; i++) {
>> - struct arm_smmu_stream *new_stream = master->streams[i];
>> + struct arm_smmu_stream *stream;
>> struct rb_node *existing;
>> - u32 sid = new_stream->id;
>> + u32 sid = fwspec->ids[i];
>> ret = arm_smmu_init_sid_strtab(smmu, sid);
>> if (ret)
>> break;
>> - /* Insert into SID tree */
>> - existing = rb_find_add(&new_stream->node, &smmu->streams,
>> - arm_smmu_streams_cmp_node);
>> + existing = rb_find(&sid, &smmu->streams,
>> + arm_smmu_streams_cmp_key);
>> if (existing) {
>> struct arm_smmu_master *existing_master =
>> rb_entry(existing, struct arm_smmu_stream, node)
>> ->master;
>> /* Bridged PCI devices may end up with duplicated IDs */
>> - if (existing_master == master)
>> + if (existing_master == master) {
>> + master->streams[i] = rb_entry(existing,
>> + struct arm_smmu_stream, node);
>
>If we're now making the whole stream allocation and tracking business more
>dynamic anyway, could we not just skip inserting duplicate entries entirely,
>and save all the hassle elsewhere?

With duplication, the streams may looks as below:
streams[0] = ptr_to_stream(sid=0x10)
streams[1] = ptr_to_stream(sid=0x10)
streams[2] = ptr_to_stream(sid=0x20)
num_streams = 3

Withou duplicaition:
streams[0] = ptr_to_stream(sid=0x10)
streams[1] = ptr_to_stream(sid=0x20)
num_streams = 2

In next version:
I'll skip duplicate SIDs during insertion and track only unique
entries in master->streams[], with num_streams reflecting the
deduplicated count.

>
>IIRC, the only real reason for not actively deduplicating originally in
>563b5cbe334e ("iommu/arm-smmu-v3: Cope with duplicated Stream IDs") was to
>keep it to the simplest fix that was easier to backport, and at the time it
>was easy to get away with since it only mattered at that one particular
>point. If we have to start copying the double-loop bodge around to multiple
>places, it rather stops looking like the neatest option...

Right, with the old embedded array the duplicate entries were
essentially free (just unused slots with the same SID), so the
"skip and continue" bodge was a reasonable minimal fix. But now
that we individually allocate streams and store shared pointers,
keeping duplicates means dedup logic in every cleanup path.
Agreed it's better to just not insert them in the first place.

As above, I will not keep duplicated SIDs in V6.

Thanks
Peng

>
>Thanks,
>Robin.
>
>> continue;
>> + }
>> dev_warn(master->dev,
>> "Aliasing StreamID 0x%x (from %s) unsupported, expect DMA to be broken\n",
>> @@ -4163,45 +4140,72 @@ static int arm_smmu_insert_master(struct arm_smmu_device *smmu,
>> ret = -ENODEV;
>> break;
>> }
>> +
>> + stream = kzalloc_obj(*stream, GFP_KERNEL);
>> + if (!stream) {
>> + ret = -ENOMEM;
>> + break;
>> + }
>> + stream->id = sid;
>> + stream->master = master;
>> +
>> + rb_find_add(&stream->node, &smmu->streams,
>> + arm_smmu_streams_cmp_node);
>> + master->streams[i] = stream;
>> }
>> if (ret) {
>> - for (i--; i >= 0; i--)
>> - if (!RB_EMPTY_NODE(&master->streams[i]->node))
>> - rb_erase(&master->streams[i]->node,
>> - &smmu->streams);
>> + for (i--; i >= 0; i--) {
>> + int j;
>> +
>> + if (!master->streams[i])
>> + continue;
>> + /* Skip duplicated SID pointers already freed */
>> + for (j = 0; j < i; j++)
>> + if (master->streams[j] == master->streams[i])
>> + break;
>> + if (j < i)
>> + continue;
>> + rb_erase(&master->streams[i]->node, &smmu->streams);
>> + kfree(master->streams[i]);
>> + }
>> mutex_unlock(&smmu->streams_mutex);
>> - goto out_free_streams;
>> + kfree(master->streams);
>> + kfree(master->build_invs);
>> + return ret;
>> }
>> mutex_unlock(&smmu->streams_mutex);
>> - return 0;
>> + /* Put the ids into order for sorted to_merge/to_unref arrays */
>> + sort(master->streams, master->num_streams,
>> + sizeof(master->streams[0]), arm_smmu_stream_id_cmp,
>> + NULL);
>> -out_free_streams:
>> - for (i = 0; i < master->num_streams; i++)
>> - kfree(master->streams[i]);
>> - kfree(master->streams);
>> - kfree(master->build_invs);
>> - return ret;
>> + return 0;
>> }
>> static void arm_smmu_remove_master(struct arm_smmu_master *master)
>> {
>> int i;
>> struct arm_smmu_device *smmu = master->smmu;
>> - struct iommu_fwspec *fwspec = dev_iommu_fwspec_get(master->dev);
>> if (!smmu || !master->streams)
>> return;
>> mutex_lock(&smmu->streams_mutex);
>> - for (i = 0; i < fwspec->num_ids; i++)
>> - if (!RB_EMPTY_NODE(&master->streams[i]->node))
>> - rb_erase(&master->streams[i]->node, &smmu->streams);
>> - mutex_unlock(&smmu->streams_mutex);
>> + for (i = 0; i < master->num_streams; i++) {
>> + int j;
>> - for (i = 0; i < master->num_streams; i++)
>> + /* Skip duplicated SID pointers already freed */
>> + for (j = 0; j < i; j++)
>> + if (master->streams[j] == master->streams[i])
>> + break;
>> + if (j < i)
>> + continue;
>> + rb_erase(&master->streams[i]->node, &smmu->streams);
>> kfree(master->streams[i]);
>> + }
>> + mutex_unlock(&smmu->streams_mutex);
>> kfree(master->streams);
>> kfree(master->build_invs);
>>
>
>