Re: [PATCH] signal: Prevent exec() race
From: Eric W. Biederman
Date: Tue Sep 01 2026 - 13:38:32 EST
Thomas Gleixner <tglx@xxxxxxxxxx> writes:
> On Mon, Aug 31 2026 at 10:26, Eric W. Biederman wrote:
>>
>> This changes partially fixes another bug. Recursive
>> UCOUNT_RLIMIT_SIGPENDING should be decremented when the process exits
>> and not when the process is reaped.
>>
>> Others have noticed possible races flushing the siqueue not
>> holding siglock.
>
> Yes. I doesn't work.
>
>> If I read the history correctly in flush_sigqueue with irqs
>> disabled can trigger the NMI lock-up detector. So flush_sigqueue
>> was moved outside of siglock_irq.
>>
>> Apparently it took KASAN to make kmem_cache_free slow enough
>> to trigger the lock-up detector.
>>
>> The fix to avoid the lock-up detector was not comprehensive and
>> flush_sigqueue is still called in many places with irqs disabled.
>> So if necessary the code can probably just take siglock.
>
> Right, invoke flush_sigqueue() right after setting PF_EXITING.
>
> But we can be smarter than that. See below.
>
>> We can also avoid problems by updating the loops that go:
>> for_each_thread(p, q)
>> flush_sigqueue_mask(p, &flush, &t->pending)
>>
>> To include
>> if (t->flags & PF_EXITING)
>> continue;
>>
>> Or perhaps better tweak flush_sigqueue_mask to take t (and not p) and
>> perform the test of PF_EXITING there. The only current uses I see of
>> the passed in task is to get a reference to signal_struct.
>
> Correct. Though that check would have to be limited to flushing
> tsk::pending not signal::shared_pending.
Acked-by: "Eric W. Biederman" <ebiederm@xxxxxxxxxxxx>
I was just about to suggest removing the entire list under the lock,
and then cleaning up the list entries outside of the lock, then I saw
this email :)
I am not wild about the name sigqueue_splice_pending (what is being
spliced together).
Perhaps call it sigqueue_dequeue_pending? I think that conveys what
is happening a little better.
Eric
>
> Thanks,
>
> tglx
> ---
> --- a/kernel/signal.c
> +++ b/kernel/signal.c
> @@ -457,30 +457,44 @@ static void __sigqueue_free(struct sigqu
> kmem_cache_free(sigqueue_cachep, q);
> }
>
> -void flush_sigqueue(struct sigpending *queue)
> +static void flush_sigqueue_list(struct list_head *head)
> {
> - struct sigqueue *q;
> + struct sigqueue *q, *tmp;
>
> - sigemptyset(&queue->signal);
> - while (!list_empty(&queue->list)) {
> - q = list_entry(queue->list.next, struct sigqueue , list);
> + list_for_each_entry_safe(q, tmp, head, list) {
> list_del_init(&q->list);
> __sigqueue_free(q);
> }
> }
>
> +void flush_sigqueue(struct sigpending *queue)
> +{
> + sigemptyset(&queue->signal);
> + flush_sigqueue_list(&queue->list);
> +}
> +
> +static void sigqueue_splice_pending(struct sigpending *queue, struct list_head *head)
> +{
> + sigemptyset(&queue->signal);
> + list_splice_init(&queue->list, head);
> +}
> +
> /*
> * Flush all pending signals for this kthread.
> */
> void flush_signals(struct task_struct *t)
> {
> - unsigned long flags;
> + LIST_HEAD(pending);
> + LIST_HEAD(shared);
>
> - spin_lock_irqsave(&t->sighand->siglock, flags);
> - clear_tsk_thread_flag(t, TIF_SIGPENDING);
> - flush_sigqueue(&t->pending);
> - flush_sigqueue(&t->signal->shared_pending);
> - spin_unlock_irqrestore(&t->sighand->siglock, flags);
> + scoped_guard(spinlock_irqsave, &t->sighand->siglock) {
> + clear_tsk_thread_flag(t, TIF_SIGPENDING);
> + sigqueue_splice_pending(&t->pending, &pending);
> + sigqueue_splice_pending(&t->signal->shared_pending, &shared);
> + }
> +
> + flush_sigqueue_list(&pending);
> + flush_sigqueue_list(&shared);
> }
> EXPORT_SYMBOL(flush_signals);
>
> @@ -3125,18 +3139,9 @@ static void retarget_shared_pending(stru
> }
> }
>
> -/*
> - * tsk::flags has PF_EXITING set which prevents signals to be queued on
> - * tsk::pending. Nothing else can touch tsk::pending anymore so it can be
> - * flushed lockless.
> - */
> -static inline void flush_pending_unlocked(struct task_struct *tsk)
> -{
> - flush_sigqueue(&tsk->pending);
> -}
> -
> void exit_signals(struct task_struct *tsk)
> {
> + LIST_HEAD(sigq_list);
> int group_stop = 0;
> sigset_t unblocked;
>
> @@ -3147,10 +3152,12 @@ void exit_signals(struct task_struct *ts
> cgroup_threadgroup_change_begin(tsk);
>
> if (thread_group_empty(tsk) || (tsk->signal->flags & SIGNAL_GROUP_EXIT)) {
> - scoped_guard(spinlock_irq, &tsk->sighand->siglock)
> + scoped_guard(spinlock_irq, &tsk->sighand->siglock) {
> tsk->flags |= PF_EXITING;
> + sigqueue_splice_pending(&tsk->pending, &sigq_list);
> + }
> cgroup_threadgroup_change_end(tsk);
> - flush_pending_unlocked(tsk);
> + flush_sigqueue_list(&sigq_list);
> return;
> }
>
> @@ -3160,6 +3167,7 @@ void exit_signals(struct task_struct *ts
> * see wants_signal(), do_signal_stop().
> */
> tsk->flags |= PF_EXITING;
> + sigqueue_splice_pending(&tsk->pending, &sigq_list);
>
> cgroup_threadgroup_change_end(tsk);
>
> @@ -3176,7 +3184,7 @@ void exit_signals(struct task_struct *ts
> out:
> spin_unlock_irq(&tsk->sighand->siglock);
>
> - flush_pending_unlocked(tsk);
> + flush_sigqueue_list(&sigq_list);
>
> /*
> * If group stop has completed, deliver the notification. This