From 10d08525f88eefb2f901db999335d72166b64a4e Mon Sep 17 00:00:00 2001 From: Thomas Stokes Date: Sat, 8 Aug 2026 23:29:04 +0800 Subject: [PATCH] wasm: make SMP acquire and release accesses atomic The generic SMP acquire/release helpers implement ordering as a fence around READ_ONCE or WRITE_ONCE. In WebAssembly shared memory those remain non-atomic accesses, so they do not form the atomic reads-from relationship needed to synchronize threads. Override release stores, acquire loads, and conditional loads with __atomic builtins. WebAssembly currently lowers all of these to sequentially consistent atomic instructions. Tell Clang that fields in packed containers retain the natural alignment required by these operations so it emits inline Wasm atomics rather than unsupported library calls. This also replaces many fence-plus-plain-access sequences with one atomic access at the actual synchronization location. In the built kernel, atomic loads increase from 2,088 to 2,850, atomic stores from 420 to 912, and fences fall from 3,134 to 1,900. finish_task_switch(), for example, changes from atomic.fence plus i32.store to i32.atomic.store. Validation: - full Wasm kernel build, including packed objpool release/acquire sites - checkpatch: no errors or warnings - 30/30 four-CPU lifecycle stress runs reached the intended panic with no RuntimeError or timeout - temporary four-CPU lock validation passed with exact counters and no exclusion errors - five-run lifecycle comparison showed no material boot-time regression (5.25 s mean versus 5.32 s for successful baseline runs) Agent-Session: codex:019fe115-244a-76c1-b716-b8af6827da00 --- arch/wasm/include/asm/barrier.h | 50 +++++++++++++++++++++++++++++++++ 1 file changed, 50 insertions(+) diff --git a/arch/wasm/include/asm/barrier.h b/arch/wasm/include/asm/barrier.h index 513e28dd20de1d..667159426df7f0 100644 --- a/arch/wasm/include/asm/barrier.h +++ b/arch/wasm/include/asm/barrier.h @@ -5,6 +5,56 @@ #define __smp_rmb() __atomic_thread_fence(__ATOMIC_ACQ_REL) #define __smp_wmb() __atomic_thread_fence(__ATOMIC_ACQ_REL) +#define __wasm_atomic_ptr(p) \ + ((typeof(p))__builtin_assume_aligned((const void *)(p), sizeof(*(p)))) + +/* + * A fence does not make an adjacent plain Wasm memory access atomic. Use an + * atomic instruction for the access itself so that release/acquire pairs + * synchronize through shared linear memory. + */ +#define __smp_store_release(p, v) \ +do { \ + compiletime_assert_atomic_type(*(p)); \ + __atomic_store_n(__wasm_atomic_ptr(p), (v), __ATOMIC_RELEASE); \ +} while (0) + +#define __smp_load_acquire(p) \ +({ \ + __unqual_scalar_typeof(*(p)) __value; \ + compiletime_assert_atomic_type(*(p)); \ + __value = __atomic_load_n(__wasm_atomic_ptr(p), __ATOMIC_ACQUIRE);\ + (typeof(*(p)))__value; \ +}) + +#define smp_cond_load_relaxed(ptr, cond_expr) \ +({ \ + typeof(ptr) __ptr = (ptr); \ + __unqual_scalar_typeof(*(ptr)) VAL; \ + for (;;) { \ + VAL = __atomic_load_n(__wasm_atomic_ptr(__ptr), \ + __ATOMIC_RELAXED); \ + if (cond_expr) \ + break; \ + cpu_relax(); \ + } \ + (typeof(*(ptr)))VAL; \ +}) + +#define smp_cond_load_acquire(ptr, cond_expr) \ +({ \ + typeof(ptr) __ptr = (ptr); \ + __unqual_scalar_typeof(*(ptr)) VAL; \ + for (;;) { \ + VAL = __atomic_load_n(__wasm_atomic_ptr(__ptr), \ + __ATOMIC_ACQUIRE); \ + if (cond_expr) \ + break; \ + cpu_relax(); \ + } \ + (typeof(*(ptr)))VAL; \ +}) + #include #endif