diff --git a/src/include/port/atomics/generic-msvc.h b/src/include/port/atomics/generic-msvc.h index 6d09cd0e0438e..57fb714bce034 100644 --- a/src/include/port/atomics/generic-msvc.h +++ b/src/include/port/atomics/generic-msvc.h @@ -30,6 +30,34 @@ #define pg_memory_barrier_impl() MemoryBarrier() #endif +/* + * On ARM64 (aarch64) the ordering guarantees PostgreSQL's lock-free code + * assumes are not reliably met by the plain interlocked intrinsics: + * + * - _ReadWriteBarrier() is a compiler barrier only; it emits no instruction. + * + * - A plain volatile load/store is not an acquire/release on this weakly + * ordered architecture (unlike x86/x64 TSO, where it is). + * + * - MSVC may satisfy an interlocked operation with an out-of-line helper + * that, on ARMv8.1 targets, uses an LSE "casal" instruction. "casal" + * provides acquire/release semantics for that single location but is NOT + * a full fence; it is weaker than the "dmb ish" that PostgreSQL's barrier + * macros are expected to provide. Code that assumes full-fence semantics + * (e.g. the standalone pg_memory_barrier()/pg_read_barrier() used around + * lock-free LSN bookkeeping) can then misbehave -- observed as incorrect + * runtime behaviour on Windows ARM64 and reported to Microsoft (MSVC + * Developer Community feedback 10982137). + * + * To be safe we force an explicit full data memory barrier + * (__dmb(_ARM64_BARRIER_ISH)) on the atomic read/write and compare-exchange + * paths on ARM64, rather than relying on the intrinsics' implicit ordering. + * The x86/x64 paths below are unchanged. + */ +#ifdef _M_ARM64 +#define PG_MSVC_ARM64_FENCE 1 +#endif + #define PG_HAVE_ATOMIC_U32_SUPPORT typedef struct pg_atomic_uint32 { @@ -50,12 +78,48 @@ pg_atomic_compare_exchange_u32_impl(volatile pg_atomic_uint32 *ptr, { bool ret; uint32 current; +#ifdef PG_MSVC_ARM64_FENCE + __dmb(_ARM64_BARRIER_ISH); +#endif current = InterlockedCompareExchange(&ptr->value, newval, *expected); +#ifdef PG_MSVC_ARM64_FENCE + __dmb(_ARM64_BARRIER_ISH); +#endif ret = current == *expected; *expected = current; return ret; } +#ifdef PG_MSVC_ARM64_FENCE +/* + * Acquire load / release store for u32. A plain volatile access is not + * ordered on ARM64, so bracket it with a full data memory barrier. + */ +#define PG_HAVE_ATOMIC_READ_U32 +static inline uint32 +pg_atomic_read_u32_impl(volatile pg_atomic_uint32 *ptr) +{ + uint32 v = ptr->value; + + __dmb(_ARM64_BARRIER_ISH); /* acquire */ + return v; +} + +#define PG_HAVE_ATOMIC_WRITE_U32 +static inline void +pg_atomic_write_u32_impl(volatile pg_atomic_uint32 *ptr, uint32 val) +{ + __dmb(_ARM64_BARRIER_ISH); /* release */ + ptr->value = val; +} +#endif /* PG_MSVC_ARM64_FENCE */ + +/* + * On ARM64 leave exchange/fetch_add undefined so generic.h derives them from + * the fenced compare-exchange above. + */ +#ifndef PG_MSVC_ARM64_FENCE + #define PG_HAVE_ATOMIC_EXCHANGE_U32 static inline uint32 pg_atomic_exchange_u32_impl(volatile pg_atomic_uint32 *ptr, uint32 newval) @@ -70,6 +134,8 @@ pg_atomic_fetch_add_u32_impl(volatile pg_atomic_uint32 *ptr, int32 add_) return InterlockedExchangeAdd(&ptr->value, add_); } +#endif /* !PG_MSVC_ARM64_FENCE */ + /* * The non-intrinsics versions are only available in vista upwards, so use the * intrinsic version. Only supported on >486, but we require XP as a minimum @@ -85,15 +151,56 @@ pg_atomic_compare_exchange_u64_impl(volatile pg_atomic_uint64 *ptr, { bool ret; uint64 current; +#ifdef PG_MSVC_ARM64_FENCE + __dmb(_ARM64_BARRIER_ISH); +#endif current = _InterlockedCompareExchange64(&ptr->value, newval, *expected); +#ifdef PG_MSVC_ARM64_FENCE + __dmb(_ARM64_BARRIER_ISH); +#endif ret = current == *expected; *expected = current; return ret; } +#ifdef PG_MSVC_ARM64_FENCE +/* + * Acquire load / release store for u64 (see the u32 versions above for the + * rationale). generic-msvc.h otherwise leaves read/write_u64 to the plain + * volatile implementations in generic.h, which are not ordered on ARM64. + */ +#define PG_HAVE_ATOMIC_READ_U64 +static inline uint64 +pg_atomic_read_u64_impl(volatile pg_atomic_uint64 *ptr) +{ + uint64 v; + + AssertPointerAlignment(ptr, 8); + v = ptr->value; + __dmb(_ARM64_BARRIER_ISH); /* acquire */ + return v; +} + +#define PG_HAVE_ATOMIC_WRITE_U64 +static inline void +pg_atomic_write_u64_impl(volatile pg_atomic_uint64 *ptr, uint64 val) +{ + AssertPointerAlignment(ptr, 8); + __dmb(_ARM64_BARRIER_ISH); /* release */ + ptr->value = val; +} +#endif /* PG_MSVC_ARM64_FENCE */ + /* Only implemented on 64bit builds */ #ifdef _WIN64 +/* + * On ARM64 leave exchange/fetch_add undefined so generic.h derives them from + * the fenced compare-exchange above (see the u32 note). On x86/x64 (TSO) the + * direct intrinsics are fine. + */ +#ifndef PG_MSVC_ARM64_FENCE + #pragma intrinsic(_InterlockedExchange64) #define PG_HAVE_ATOMIC_EXCHANGE_U64 @@ -112,4 +219,6 @@ pg_atomic_fetch_add_u64_impl(volatile pg_atomic_uint64 *ptr, int64 add_) return _InterlockedExchangeAdd64(&ptr->value, add_); } +#endif /* !PG_MSVC_ARM64_FENCE */ + #endif /* _WIN64 */