From ac95c32c3e82c6db18b89fe6b7b23a89e2452e8a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Fr=C3=A9d=C3=A9ric=20Desbiens?= Date: Wed, 30 Sep 2026 11:55:52 -0400 Subject: [PATCH] Stopped the Linux port losing a mutex wake-up to a suspend signal The port suspends a thread with a signal whose handler calls sigsuspend and does not return until resumed, and takes its critical section with a bare pthread_mutex_lock. A thread signalled while parked on _tx_linux_mutex never returns to glibc's contended-mutex loop, so the next unlock's wake-up goes to a thread that will not act on it and every other waiter is left parked on a mutex that is free. tx_linux_mutex_lock now calls a helper that waits with pthread_mutex_timedlock and retries every TX_LINUX_MUTEX_RETRY_NSEC, one millisecond, so a lost wake-up costs a retry period instead of the process. The original fix was made by inspection, on a port never observed to deadlock. It has now been observed. A stalled regression test reads the mutex free, owner zero, with the scheduler still on its futex word; it never posts _tx_linux_isr_semaphore, the timer thread never returns from _tx_thread_context_restore, and the simulated clock stops. Bounding a wait cannot save it: one capture held 116 ticks of timeout unchanged across 40 seconds. Measured on netx_tcp_overlapping_packet_test_10, six workers on four pinned CPUs: 5 stalls in 1,266 runs before, 0 in 2,034 after, three held frozen through 60 seconds of untraced /proc sampling. netxduo/default_build_coverage is unmoved at 914/914, 121 skipped, 497 files, 11230/11256 lines and 6912/6925 branches. (cherry picked from commit b567428f2a826fa24e78fc82f6a5dd46a87b079c) Assisted-by: Claude Code (Opus 5) --- ports/linux/gnu/inc/tx_port.h | 13 ++++++- ports/linux/gnu/src/tx_thread_schedule.c | 45 ++++++++++++++++++++++++ 2 files changed, 57 insertions(+), 1 deletion(-) diff --git a/ports/linux/gnu/inc/tx_port.h b/ports/linux/gnu/inc/tx_port.h index dc39b5f9..70a9fb9d 100644 --- a/ports/linux/gnu/inc/tx_port.h +++ b/ports/linux/gnu/inc/tx_port.h @@ -525,7 +525,7 @@ VOID _tx_thread_interrupt_restore(UINT previous_posture); #define TX_RESTORE _tx_linux_debug_entry_insert("RESTORE", __FILE__, __LINE__); \ _tx_thread_interrupt_restore(tx_saved_posture); #endif /* TX_LINUX_DEBUG_ENABLE */ -#define tx_linux_mutex_lock(p) pthread_mutex_lock(&p) +#define tx_linux_mutex_lock(p) _tx_linux_mutex_lock_retry(&p) #define tx_linux_mutex_unlock(p) pthread_mutex_unlock(&p) #define tx_linux_mutex_recursive_unlock(p) {\ int _recursive_count = (int)tx_linux_mutex_recursive_count;\ @@ -566,6 +566,17 @@ extern CHAR _tx_version_id[]; /* Define externals for the Linux port of ThreadX. */ extern pthread_mutex_t _tx_linux_mutex; + +/* Define how long a thread waits on the Linux mutex before retrying. A thread + parked on the mutex can be suspended by the port's signal handler and so never + act on the wake-up the next unlock sends it, which leaves the wake-up lost and + every other waiter parked on a mutex that is free. */ + +#ifndef TX_LINUX_MUTEX_RETRY_NSEC +#define TX_LINUX_MUTEX_RETRY_NSEC 1000000 +#endif + +void _tx_linux_mutex_lock_retry(pthread_mutex_t *mutex); extern sem_t _tx_linux_semaphore; extern sem_t _tx_linux_semaphore_no_idle; extern ULONG _tx_linux_global_int_disabled_flag; diff --git a/ports/linux/gnu/src/tx_thread_schedule.c b/ports/linux/gnu/src/tx_thread_schedule.c index d65104a1..c66aa891 100644 --- a/ports/linux/gnu/src/tx_thread_schedule.c +++ b/ports/linux/gnu/src/tx_thread_schedule.c @@ -8,6 +8,8 @@ * SPDX-License-Identifier: MIT **************************************************************************/ +// Portions of this file were generated with AI assistance. + /**************************************************************************/ /**************************************************************************/ @@ -206,6 +208,49 @@ struct timespec ts; } } +/* Define the ThreadX Linux mutex lock function. The wait is timed and retried + rather than left to pthread_mutex_lock, because a thread can be signalled into + the port's suspend handler while it is parked on this mutex. That handler does + not return until the thread is resumed, so the wake-up the next unlock sends is + delivered to a thread that never retries and is lost. Any other thread parked + on the mutex then waits on a mutex that is free. Retrying on a timeout costs + nothing when the mutex is handed over normally, and turns that lost wake-up into + a delay of at most the retry period. */ + +void _tx_linux_mutex_lock_retry(pthread_mutex_t *mutex) +{ + +INT linux_status; +struct timespec ts; + + + do + { + + /* Set the deadline for this attempt. */ + clock_gettime(CLOCK_REALTIME, &ts); + ts.tv_nsec = ts.tv_nsec + TX_LINUX_MUTEX_RETRY_NSEC; + if (ts.tv_nsec >= 1000000000) + { + + ts.tv_nsec = ts.tv_nsec - 1000000000; + ts.tv_sec++; + } + + linux_status = pthread_mutex_timedlock(mutex, &ts); + + /* Anything but the deadline expiring is a real failure to obtain the + mutex, so stop retrying. */ + if ((linux_status != 0) && (linux_status != ETIMEDOUT)) + { + + break; + } + + } while (linux_status != 0); +} + + void _tx_thread_delete_port_completion(TX_THREAD *thread_ptr, UINT tx_saved_posture) { INT linux_status;