I'm experiencing random deadlocks when running heavily multithreaded workload in qemu-sh4 userspace emulation. This patch fixes them.
The sh4 architecture doesn't support multiprocessing, the processor doesn't have atomic instructions and it uses a technique known as gUSA to provide atomicity guarantees w.r.t. signals or thread scheduling (see the commit 3b894b699c9a for a brief description of gUSA). The function decode_gusa attempts to recognize several well-known gUSA regions and turn them into atomic instructions. When it fails to recognize a known gUSA pattern, it generates a call to helper_exclusive. Qemu will attempt to stop all the other threads and execute a gUSA region exclusively, so that it can't race with anything. The problem is in the function cpu_exec_step_atomic - this function stops all other threads with start_exclusive(), then it executes one instruction and then it releases all other threads with end_exclusive(). This works for all the architectures except sh4. On sh4, executing one instruction atomically is not enough - we must execute the full gUSA region while the other threads are stopped. This patch fixes cpu_exec_step_atomic - it adds two new per-architecture functions: is_uninterruptible and revert_uninterruptible. is_uninterruptible returns true if we are in a gUSA region and we should continue executing code while the other threads are stopped. If we got TB_EXIT_REQUESTED, we must stop executing code - in this case, we call the function revert_uninterruptible that rolls back PC to the beginning of the gUSA region. Cc: [email protected] Signed-off-by: Mikulas Patocka <[email protected]> --- accel/tcg/cpu-exec.c | 15 +++++++++++++++ include/accel/tcg/cpu-ops.h | 20 ++++++++++++++++++++ target/sh4/cpu.c | 25 ++++++++++++++++++++++++- 3 files changed, 59 insertions(+), 1 deletion(-) Index: qemu/accel/tcg/cpu-exec.c =================================================================== --- qemu.orig/accel/tcg/cpu-exec.c 2026-09-20 13:16:20.000000000 +0200 +++ qemu/accel/tcg/cpu-exec.c 2026-09-20 13:16:20.000000000 +0200 @@ -560,6 +560,9 @@ void cpu_exec_step_atomic(CPUState *cpu) g_assert(!cpu->running); cpu->running = true; +#ifdef CONFIG_USER_ONLY +next_instr: +#endif TCGTBCPUState s = cpu->cc->tcg_ops->get_tb_cpu_state(cpu); s.cflags = curr_cflags(cpu); @@ -586,7 +589,19 @@ void cpu_exec_step_atomic(CPUState *cpu) trace_exec_tb(tb, s.pc); cpu_tb_exec(cpu, tb, &tb_exit); cpu_exec_exit(cpu); +#ifdef CONFIG_USER_ONLY + if (cpu->cc->tcg_ops->is_uninterruptible && cpu->cc->tcg_ops->is_uninterruptible(cpu)) { + if ((tb_exit & TB_EXIT_MASK) != TB_EXIT_REQUESTED) + goto next_instr; + if (cpu->cc->tcg_ops->revert_uninterruptible) + cpu->cc->tcg_ops->revert_uninterruptible(cpu); + } +#endif } else { +#ifdef CONFIG_USER_ONLY + if (cpu->cc->tcg_ops->revert_uninterruptible) + cpu->cc->tcg_ops->revert_uninterruptible(cpu); +#endif cpu_exec_longjmp_cleanup(cpu); } Index: qemu/include/accel/tcg/cpu-ops.h =================================================================== --- qemu.orig/include/accel/tcg/cpu-ops.h 2026-09-20 13:16:20.000000000 +0200 +++ qemu/include/accel/tcg/cpu-ops.h 2026-09-20 13:16:20.000000000 +0200 @@ -168,6 +168,26 @@ struct TCGCPUOps { * @addr: tagged guest address */ vaddr (*untagged_addr)(CPUState *cs, vaddr addr); + + /** + * is_uninterruptible: + * @cpu: cpu context + * + * Returns true if we are in the middle of the gUSA region and + * cpu_exec_step_atomic must keep on executing instructions without + * dropping the exclusive lock. + */ + bool (*is_uninterruptible)(CPUState *cs); + + /** + * revert_uninterruptible: + * @cpu: cpu context + * + * This function is called if cpu_exec_step_atomic needs to exit. It + * tests if we are in the gUSA region and rolls back PC to the + * beginning of it. + */ + void (*revert_uninterruptible)(CPUState *cs); #else /** @do_interrupt: Callback for interrupt handling. */ void (*do_interrupt)(CPUState *cpu); Index: qemu/target/sh4/cpu.c =================================================================== --- qemu.orig/target/sh4/cpu.c 2026-09-20 13:16:20.000000000 +0200 +++ qemu/target/sh4/cpu.c 2026-09-20 13:16:20.000000000 +0200 @@ -92,6 +92,26 @@ static void superh_restore_state_to_opc( */ } +#ifdef CONFIG_USER_ONLY +static bool superh_cpu_is_uninterruptible(CPUState *cs) +{ + SuperHCPU *cpu = SUPERH_CPU(cs); + + return cpu->env.gregs[15] >= -128u; +} + +static void superh_cpu_revert_uninterruptible(CPUState *cs) +{ + SuperHCPU *cpu = SUPERH_CPU(cs); + + if (cpu->env.gregs[15] >= -128u && cpu->env.pc < cpu->env.gregs[0]) { + cpu->env.pc = cpu->env.gregs[0] + cpu->env.gregs[15] - 2; + cpu->env.gregs[15] = cpu->env.gregs[1]; + cpu->env.flags &= ~(TB_FLAG_DELAY_SLOT_MASK | TB_FLAG_GUSA_MASK); + } +} +#endif /* CONFIG_USER_ONLY */ + #ifndef CONFIG_USER_ONLY static bool superh_io_recompile_replay_branch(CPUState *cs, const TranslationBlock *tb) @@ -308,7 +328,10 @@ static const TCGCPUOps superh_tcg_ops = .restore_state_to_opc = superh_restore_state_to_opc, .mmu_index = sh4_cpu_mmu_index, -#ifndef CONFIG_USER_ONLY +#ifdef CONFIG_USER_ONLY + .is_uninterruptible = superh_cpu_is_uninterruptible, + .revert_uninterruptible = superh_cpu_revert_uninterruptible, +#else .tlb_fill = superh_cpu_tlb_fill, .pointer_wrap = cpu_pointer_wrap_notreached, .cpu_exec_interrupt = superh_cpu_exec_interrupt,
