Verified nested IRQ handling on S32Z280 silicon (#612)

#611 exercised the nesting routines on the FVP and established what breaks them:
the interrupt must be acknowledged before nesting starts, or the still-pending
level-asserted timer is retaken the moment IRQ is enabled and recurses until the
stacks are gone. entry.S here carries the same ordering for the same reason, and
irq_dispatch.c splits the same way, board_irq_service taking an
already-acknowledged INTID while board_irq_handler keeps its old shape as the
non-nesting entry point.

What the model could not answer is whether a real GIC-600 agrees, and two things
could have differed.

The first is the number of implemented priority bits. Equal priorities do not
preempt and it is the low bits that vanish, so if the timer and the SGI collapse
to one value after truncation then nesting cannot happen at all -- and the test
would fail without saying why. gicv3_priority_bits discovers the count by writing
0xFF to a priority byte and reading back which bits stick, board_init records the
two effective values, and check P1 requires the SGI to still outrank the timer.
This silicon keeps five bits, the same as the FVP, so 0xA0 and 0x50 stay distinct;
that is now measured and reported rather than assumed.

The second is whether an SGI raised on real hardware is delivered at all. ICC_SGI1R
is a 64-bit AArch32 CP15 register whose encoding does not transcribe from the
AArch64 alias, so check N1 raises one from thread context and requires delivery
before nesting is involved. It arrives.

Verified on S32Z280 silicon:

    priority bits   = 0x00000005
    timer effective = 0x000000A0    sgi effective = 0x00000050
    P1 SGI outranks timer after truncation     PASS
    N1 SGI delivered and dispatched            PASS
    max depth = 0x00000002   nested SGIs = 0x00000032   depth now = 0x00000000
    N2 SGI nested inside another handler       PASS
    N3 nesting unwound to depth zero           PASS
    N4 tick still advancing after nesting      PASS
    N5 lower-priority thread still scheduled   PASS
    N6 no spurious or unexpected interrupts    PASS

Fifty nested SGIs across fifty ticks, one per tick, and depth never exceeded two.

No regression, checked in both configurations on the board. In the nesting build
the ThreadX demo still reports 100 ticks with 20 preemptions and the boot image
still passes its cache and protection checks. In the default build the image links
no nesting symbols at all and the demo is unchanged. Both toolchains compile every
file, GNU with no warnings and Arm Toolchain for Embedded 22.1.0 as well.

Assisted-by: Claude Code (Opus 5) <noreply@anthropic.com>
This commit is contained in:
Frédéric Desbiens
2026-08-14 11:07:06 -04:00
committed by GitHub
parent 559460bed9
commit 5a7f4f8f7c
7 changed files with 554 additions and 4 deletions
@@ -127,3 +127,40 @@ if(TX_R52_ENABLE_VFP)
${EVB_LINK_QUIET_RWX}
)
endif()
# Nested IRQ handling on silicon. Needs the library built with
# TX_R52_ENABLE_IRQ_NESTING, since entry.S only emits the nesting_start and
# nesting_end pairing under that macro, so the image exists in that
# configuration only -- as s32z280_vfp.elf does for the floating-point ABI.
if(TX_R52_ENABLE_IRQ_NESTING)
add_executable(s32z280_nesting.elf EXCLUDE_FROM_ALL
${EVB_DIR}/entry.S
${EVB_DIR}/tx_initialize_low_level.S
${EVB_DIR}/linflexd.c
${EVB_DIR}/timer.c
${EVB_DIR}/mpu.c
${EVB_DIR}/gicv3.c
${EVB_DIR}/irq_dispatch.c
${EVB_DIR}/gic_probe.c
${EVB_DIR}/cache.c
${EVB_DIR}/demo_nesting_s32z280.c
)
target_compile_definitions(s32z280_nesting.elf PRIVATE TX_R52_USE_THREADX_IRQ=1)
target_compile_options(s32z280_nesting.elf PRIVATE -g)
target_link_libraries(s32z280_nesting.elf PRIVATE threadx)
target_include_directories(s32z280_nesting.elf PRIVATE
${EVB_DIR}
${CMAKE_SOURCE_DIR}/common/inc
${CMAKE_SOURCE_DIR}/ports/${THREADX_ARCH}/${THREADX_TOOLCHAIN}/inc
)
target_link_options(s32z280_nesting.elf PRIVATE
-T${EVB_DIR}/link.lds
-nostartfiles
-Wl,-Map=s32z280_nesting.map
${EVB_LINK_QUIET_RWX}
)
endif()
@@ -85,6 +85,13 @@ unsigned int read_sctlr_after_mpu(void);
void board_irq_handler(void);
/* The nesting path splits the handler in three: the acknowledge happens in IRQ
mode before nesting starts, and the end-of-interrupt after it finishes.
board_irq_service does the middle part on an already-acknowledged INTID and
neither acknowledges nor EOIs. */
void board_irq_service(unsigned long intid);
/* Called from _tx_initialize_low_level when the kernel is linked. */
void board_init(void);
@@ -102,4 +109,19 @@ extern volatile unsigned long board_spurious_count;
extern volatile unsigned long board_unexpected_intid;
extern volatile unsigned long board_first_intid;
/* Nesting observability. Inert unless the image was built with
TX_ENABLE_IRQ_NESTING: without it the handler runs in IRQ mode with
interrupts masked and can never be re-entered. */
extern volatile unsigned long board_nest_depth;
extern volatile unsigned long board_nest_max;
extern volatile unsigned long board_sgi_count;
extern volatile unsigned long board_sgi_nested_count;
extern volatile unsigned long board_nest_provoke;
extern volatile unsigned long board_priority_bits;
extern volatile unsigned long board_timer_priority;
extern volatile unsigned long board_sgi_priority;
#define BOARD_NEST_SGI_INTID 8U
#endif /* BOARD_H */
File diff suppressed because it is too large Load Diff
@@ -501,7 +501,39 @@ el1_irq_entry:
.global __tx_irq_processing_return
__tx_irq_processing_return:
#ifdef TX_ENABLE_IRQ_NESTING
/* Nested IRQ handling, the same shape as the FVP example, and the ordering
* matters as much here. The interrupt is acknowledged in IRQ mode with
* interrupts still masked, BEFORE nesting starts: reading ICC_IAR1 raises
* the GIC running priority to this interrupt's own, which masks it and
* everything of equal or lower priority, and only then is re-enabling IRQ
* safe. Start nesting first and the still-pending, still-level-asserted
* timer is retaken the instant IRQ is enabled, recursing until the IRQ and
* System stacks are gone -- a hang with no fault to point at.
*
* The INTID rides in r4, which survives the mode switch because only SP and
* LR are banked, and is pushed on the IRQ stack as well so a nested level
* reusing r4 cannot lose this level's value. End-of-interrupt waits for
* IRQ mode with interrupts masked again, so dropping the running priority
* cannot re-admit this same interrupt.
*/
bl gicv3_acknowledge
mov r4, r0
stmdb sp!, {r4, r5} /* r5 keeps 8-byte alignment */
bl _tx_thread_irq_nesting_start
mov r0, r4
bl board_irq_service
bl _tx_thread_irq_nesting_end
ldmia sp!, {r4, r5}
mov r0, r4
bl gicv3_end_of_interrupt
#else
bl board_irq_handler
#endif
b _tx_thread_context_restore
#else
@@ -195,6 +195,105 @@ void gicv3_enable_ppi(unsigned int intid, unsigned int priority)
/* gicv3_acknowledge */
/**************************************************************************/
/**************************************************************************/
/* gicv3_enable_sgi */
/* */
/* An SGI lives in the same redistributor frame as a PPI, so this is */
/* gicv3_enable_ppi without the ICFGR step: INTIDs 0-15 have no */
/* configurable edge/level, they are always edge-triggered. */
/**************************************************************************/
void gicv3_enable_sgi(unsigned int intid, unsigned int priority)
{
if (intid > 15U)
{
return;
}
REG32(GICR_SGI_BASE + GICR_IGROUPR0) |= (1UL << intid);
REG32(GICR_SGI_BASE + GICR_IPRIORITYR + (intid & ~3U)) &=
~(0xFFUL << ((intid & 3U) * 8U));
REG32(GICR_SGI_BASE + GICR_IPRIORITYR + (intid & ~3U)) |=
((unsigned long) priority & 0xFFUL) << ((intid & 3U) * 8U);
REG32(GICR_SGI_BASE + GICR_ISENABLER0) = (1UL << intid);
}
/**************************************************************************/
/* gicv3_send_sgi */
/* */
/* ICC_SGI1R is 64-bit, so in AArch32 it is an MCRR rather than an MCR. */
/* Fields: INTID in [27:24], TargetList in [15:0], Aff1/2/3 and IRM zero */
/* for this single-core configuration, so TargetList = 1 selects core 0. */
/* */
/* The AArch64 name for this register is S3_0_C12_C11_5, which the */
/* Cortex-A72 example uses, but that encoding does not transcribe to the */
/* AArch32 64-bit CP15 space. The CRm here was confirmed by observing */
/* that the SGI is actually delivered and acknowledged with the expected */
/* INTID rather than by reading it off the A-profile alias. */
/**************************************************************************/
void gicv3_send_sgi(unsigned int intid)
{
unsigned long low = (((unsigned long) intid & 0xFUL) << 24) | 1UL;
unsigned long high = 0UL;
__asm volatile ("mcrr p15, 0, %0, %1, c12" :: "r" (low), "r" (high) : "memory");
instruction_barrier();
}
/**************************************************************************/
/* gicv3_priority_bits */
/* */
/* How many priority bits this GIC actually implements, discovered rather */
/* than assumed: write 0xFF to a priority byte and see which bits stick. */
/* The unimplemented bits are the low ones and they read as zero. */
/* */
/* This matters for nesting. Two interrupts only preempt one another if */
/* their priorities differ AFTER truncation, so a test that picks values */
/* 0x20 apart on a GIC keeping four bits is testing nothing. The FVP */
/* keeps five; real silicon need not agree. */
/* */
/* Called before the SGI is configured, and it leaves the byte at zero, so */
/* the caller must set the priority it wants afterwards. */
/**************************************************************************/
unsigned int gicv3_priority_bits(unsigned int scratch_intid)
{
unsigned long shift;
unsigned long readback;
unsigned int bits = 0U;
if (scratch_intid > 31U)
{
return 0U;
}
shift = (unsigned long) (scratch_intid & 3U) * 8UL;
REG32(GICR_SGI_BASE + GICR_IPRIORITYR + (scratch_intid & ~3U)) |=
(0xFFUL << shift);
readback = (REG32(GICR_SGI_BASE + GICR_IPRIORITYR + (scratch_intid & ~3U))
>> shift) & 0xFFUL;
REG32(GICR_SGI_BASE + GICR_IPRIORITYR + (scratch_intid & ~3U)) &=
~(0xFFUL << shift);
while (readback != 0UL)
{
if ((readback & 1UL) != 0UL)
{
bits++;
}
readback >>= 1;
}
return bits;
}
unsigned long gicv3_acknowledge(void)
{
unsigned long intid;
@@ -49,6 +49,20 @@ void gicv3_init(void);
void gicv3_enable_ppi(unsigned int intid, unsigned int priority);
/* Enable a software-generated interrupt, INTID 0-15. Same registers as a PPI:
the redistributor's SGI/PPI frame covers INTIDs 0-31. */
void gicv3_enable_sgi(unsigned int intid, unsigned int priority);
/* Raise an SGI on this core. */
void gicv3_send_sgi(unsigned int intid);
/* Implemented priority bits, discovered by write and readback. Leaves the
scratch INTID's priority byte at zero. */
unsigned int gicv3_priority_bits(unsigned int scratch_intid);
/* Acknowledge the highest-priority pending Group 1 interrupt, returning its
INTID (possibly GICV3_SPURIOUS_INTID). */
@@ -61,6 +61,15 @@ volatile unsigned long board_spurious_count;
volatile unsigned long board_unexpected_intid;
volatile unsigned long board_first_intid = 0xFFFFFFFFUL;
volatile unsigned long board_nest_depth;
volatile unsigned long board_nest_max;
volatile unsigned long board_sgi_count;
volatile unsigned long board_sgi_nested_count;
volatile unsigned long board_nest_provoke;
volatile unsigned long board_priority_bits;
volatile unsigned long board_timer_priority;
volatile unsigned long board_sgi_priority;
/* The EL1 physical timer PPI. 30 is the architectural assignment and is what
the FVP turned out to use, but it was discovered there by enabling the whole
PPI range and recording whichever INTID arrived in ICC_IAR1 rather than by
@@ -92,6 +101,35 @@ void board_init(void)
gicv3_init();
gicv3_enable_ppi(TIMER_PPI_INTID, TIMER_PPI_PRIORITY);
/* How many priority bits this GIC keeps decides whether the SGI can preempt
the timer at all: equal priorities do not preempt, and it is the low bits
that vanish. Discovered rather than assumed, because the FVP keeps five
and real silicon need not agree. Recorded so the demo can report it, and
a nesting failure caused by the two priorities colliding is visible
instead of mysterious. */
board_priority_bits = gicv3_priority_bits(BOARD_NEST_SGI_INTID);
/* Half the timer's value so the SGI outranks it: numerically lower wins.
With TIMER_PPI_PRIORITY at 0xA0 that is 0x50, which stays distinct even
if only three priority bits survive truncation. */
gicv3_enable_sgi(BOARD_NEST_SGI_INTID, TIMER_PPI_PRIORITY / 2U);
/* The priorities as the GIC will actually compare them. Only the top
board_priority_bits survive; the low ones read back as zero, so two values
that differ only there would collapse together and never preempt. Recorded
here because this is where the constants live. */
{
unsigned long keep = (board_priority_bits >= 8UL)
? 0xFFUL
: ((0xFFUL << (8UL - board_priority_bits)) & 0xFFUL);
board_timer_priority = (unsigned long) TIMER_PPI_PRIORITY & keep;
board_sgi_priority = ((unsigned long) TIMER_PPI_PRIORITY / 2UL) & keep;
}
/* One tick every 10 ms at the measured 8 MHz counter rate. */
timer_start_oneshot_irq(timer_read_cntfrq() / 100U);
@@ -100,10 +138,15 @@ void board_init(void)
#endif
void board_irq_handler(void)
{
unsigned long intid = gicv3_acknowledge();
/**************************************************************************/
/* board_irq_service -- service an already-acknowledged INTID. */
/* */
/* Split out so the nesting path in entry.S can acknowledge in IRQ mode */
/* before nesting starts. Does not acknowledge and does not EOI. */
/**************************************************************************/
void board_irq_service(unsigned long intid)
{
if (intid == GICV3_SPURIOUS_INTID)
{
/* Nothing was actually pending. Must not be acknowledged with an
@@ -113,6 +156,16 @@ void board_irq_handler(void)
return;
}
/* After the spurious check, so a spurious entry does not read as nesting.
Reaches 2 only if a second interrupt is taken while this one is still on
the stack, which cannot happen without TX_ENABLE_IRQ_NESTING. */
board_nest_depth++;
if (board_nest_depth > board_nest_max)
{
board_nest_max = board_nest_depth;
}
if (board_first_intid == 0xFFFFFFFFUL)
{
board_first_intid = intid; /* whatever actually arrived */
@@ -129,6 +182,29 @@ void board_irq_handler(void)
timer_start_oneshot_irq(timer_read_cntfrq() / 100U);
/* Provoke nesting, if this image asked for it. Only useful because
_tx_thread_irq_nesting_start has already re-enabled IRQ; the bounded
spin gives the GIC time to deliver the higher-priority SGI before this
handler returns, so the nesting is deterministic rather than a race we
hope wins. The bound matters: with nesting compiled out the SGI can
never arrive here and this must not become a hang. */
if (board_nest_provoke != 0UL)
{
unsigned long seen = board_sgi_count;
unsigned long guard;
gicv3_send_sgi(BOARD_NEST_SGI_INTID);
for (guard = 0UL; guard < 100000UL; guard++)
{
if (board_sgi_count != seen)
{
break;
}
}
}
#ifdef TX_R52_USE_THREADX_IRQ
/* Drive the kernel's time base. Called after the re-arm so the next
interval is already running while the kernel does its bookkeeping. */
@@ -136,10 +212,40 @@ void board_irq_handler(void)
_tx_timer_interrupt();
#endif
}
else if (intid == (unsigned long) BOARD_NEST_SGI_INTID)
{
board_sgi_count++;
if (board_nest_depth > 1UL)
{
board_sgi_nested_count++;
}
}
else
{
board_unexpected_intid = intid;
}
gicv3_end_of_interrupt(intid);
board_nest_depth--;
}
/**************************************************************************/
/* board_irq_handler -- the non-nesting entry point. */
/* */
/* Acknowledges, services and ends the interrupt in IRQ mode with */
/* interrupts masked, which is what every image built without */
/* TX_ENABLE_IRQ_NESTING did before this split. */
/**************************************************************************/
void board_irq_handler(void)
{
unsigned long intid = gicv3_acknowledge();
board_irq_service(intid);
if (intid != GICV3_SPURIOUS_INTID)
{
gicv3_end_of_interrupt(intid);
}
}