> rte_atomic_thread_fence() implemented the seq_cst case by calling the
> deprecated rte_smp_mb(). Invert the dependency: the lock add based
> fence moves into rte_atomic_thread_fence() and rte_smp_mb() becomes a
> wrapper around it, so removing the deprecated barriers later is a pure
> deletion. The optimization itself must stay; a plain seq_cst fence is
> an mfence, about twice the cost. No change in generated code.
>
> Drop no longer used rte_smp_mb().
>
> Signed-off-by: Stephen Hemminger <[email protected]>
> ---
> lib/eal/x86/include/rte_atomic.h | 37 ++++++++++++++------------------
> 1 file changed, 16 insertions(+), 21 deletions(-)
>
> diff --git a/lib/eal/x86/include/rte_atomic.h
> b/lib/eal/x86/include/rte_atomic.h
> index e071e4234e..780fdce871 100644
> --- a/lib/eal/x86/include/rte_atomic.h
> +++ b/lib/eal/x86/include/rte_atomic.h
> @@ -60,23 +60,8 @@ extern "C" {
> * Basic idea is to use lock prefixed add with some dummy memory location
> * as the destination. From their experiments 128B(2 cache lines) below
> * current stack pointer looks like a good candidate.
> - * So below we use that technique for rte_smp_mb() implementation.
> */
>
> -static __rte_always_inline void
> -rte_smp_mb(void)
> -{
> -#ifdef RTE_TOOLCHAIN_MSVC
> - _mm_mfence();
> -#else
> -#ifdef RTE_ARCH_I686
> - asm volatile("lock addl $0, -128(%%esp); " ::: "memory");
> -#else
> - asm volatile("lock addl $0, -128(%%rsp); " ::: "memory");
> -#endif
> -#endif
> -}
> -
> #define rte_io_mb() rte_mb()
>
> #define rte_io_wmb() rte_compiler_barrier()
> @@ -86,17 +71,27 @@ rte_smp_mb(void)
> /**
> * Synchronization fence between threads based on the specified memory order.
> *
> - * On x86 the __rte_atomic_thread_fence(rte_memory_order_seq_cst)
> generates full 'mfence'
> - * which is quite expensive. The optimized implementation of rte_smp_mb is
> - * used instead.
> + * On x86 the __rte_atomic_thread_fence(rte_memory_order_seq_cst)
> generates
> + * a full 'mfence' which is quite expensive. The optimized lock add on a
> + * dummy stack location (see above) is used instead.
> */
> static __rte_always_inline void
> rte_atomic_thread_fence(rte_memory_order memorder)
> {
> - if (memorder == rte_memory_order_seq_cst)
> - rte_smp_mb();
> - else
> + if (memorder != rte_memory_order_seq_cst) {
> __rte_atomic_thread_fence(memorder);
> + return;
> + }
> +
> +#ifdef RTE_TOOLCHAIN_MSVC
> + _mm_mfence();
> +#else
> +#ifdef RTE_ARCH_I686
> + asm volatile("lock addl $0, -128(%%esp); " ::: "memory");
> +#else
> + asm volatile("lock addl $0, -128(%%rsp); " ::: "memory");
> +#endif
> +#endif
> }
>
> #ifdef __cplusplus
> --
Acked-by: Konstantin Ananyev <[email protected]>
> 2.53.0