Argon2: fix build with -march=x86-64-v3/v4 and XOP targets (#1906)

* Argon2: fix build with -march=x86-64-v3/v4 and XOP targets

blamka-round-opt.h picked its SSE2, AVX2 or AVX-512 definitions only from
the compiler flags. With global flags such as -march=x86-64-v3 or
-march=x86-64-v4, opt_sse2.c (and with v4 also opt_avx2.c) saw __AVX2__
or __AVX512F__ and got definitions that do not match its code, so the
build failed.

Let opt_sse2.c and opt_avx2.c select the variant they implement, and
fall back to the compiler flags for other includers. This needs no new
compiler options, so it also works with old compilers and with the
plain objects used by NOASM=1 builds.

With XOP targets (-march=bdver*), the header skipped its
_mm_roti_epi64 fallback but did not include the intrinsic; include
<x86intrin.h> for GCC and Clang there.

Hashes are identical through the AVX2, SSE2 and reference code paths,
and match an unmodified build.

Fixes #1867
Refs #1871

* Argon2: update branch comments in blamka-round-opt.h

The #else/#endif comments still named __AVX2__ and __AVX512F__, but the
branches are now selected by ARGON2_BLAMKA_USE_SSE2/AVX2.

Refs #1867
This commit is contained in:
Ville Takio authored and GitHub committed 2026-10-05 09:22:45 +09:00
1 parent 5d4127d9a7
commit 4f76ff7132
3 files changed
+25 -7

No files matched your search

@@ -27,11 +27,25 @@
#endif
#if defined(__XOP__) && (defined(__GNUC__) || defined(__clang__))
//#include <x86intrin.h>
#include <x86intrin.h> /* for _mm_roti_epi64 (-march=bdver*) */
#endif
#if !defined(__AVX512F__)
#if !defined(__AVX2__)
/* The variant is chosen by the file that includes this header
* (ARGON2_BLAMKA_SSE2 in opt_sse2.c, ARGON2_BLAMKA_AVX2 in opt_avx2.c), not
* only by the compiler flags: with global flags such as -march=x86-64-v3 or
* -march=x86-64-v4, every file sees __AVX2__ or __AVX512F__. */
#if defined(ARGON2_BLAMKA_SSE2)
#define ARGON2_BLAMKA_USE_SSE2
#elif defined(ARGON2_BLAMKA_AVX2)
#define ARGON2_BLAMKA_USE_AVX2
#elif !defined(__AVX512F__) && !defined(__AVX2__)
#define ARGON2_BLAMKA_USE_SSE2
#elif !defined(__AVX512F__)
#define ARGON2_BLAMKA_USE_AVX2
#endif
#if defined(ARGON2_BLAMKA_USE_SSE2) || defined(ARGON2_BLAMKA_USE_AVX2)
#if defined(ARGON2_BLAMKA_USE_SSE2)
#if !defined(__XOP__)
#if defined(__SSSE3__)
#define r16 \
@@ -179,7 +193,7 @@ static BLAKE2_INLINE __m128i fBlaMka(__m128i x, __m128i y) {
\
UNDIAGONALIZE(A0, B0, C0, D0, A1, B1, C1, D1); \
} while ((void)0, 0)
#else /* __AVX2__ */
#else /* ARGON2_BLAMKA_USE_AVX2 */
//#include <immintrin.h>
@@ -326,9 +340,9 @@ static BLAKE2_INLINE __m128i fBlaMka(__m128i x, __m128i y) {
UNDIAGONALIZE_2(A0, A1, B0, B1, C0, C1, D0, D1) \
} while((void)0, 0);
#endif /* __AVX2__ */
#endif /* ARGON2_BLAMKA_USE_SSE2 */
#else /* __AVX512F__ */
#else /* AVX-512 */
//#include <immintrin.h>
@@ -468,5 +482,5 @@ static __m512i muladd(__m512i x, __m512i y)
UNSWAP_QUARTERS(D0, D1); \
} while ((void)0, 0)
#endif /* __AVX512F__ */
#endif /* ARGON2_BLAMKA_USE_SSE2 || ARGON2_BLAMKA_USE_AVX2 */
#endif /* BLAKE_ROUND_MKA_OPT_H */
+2
View File
@@ -29,6 +29,8 @@
#include <immintrin.h>
#include "blake2/blake2b.h"
/* 256-bit (AVX2) rounds, also when the compiler flags enable AVX-512. */
#define ARGON2_BLAMKA_AVX2
#include "blake2/blamka-round-opt.h"
/*
+2
View File
@@ -26,6 +26,8 @@
#if CRYPTOPP_BOOL_SSE2_INTRINSICS_AVAILABLE
#include "blake2/blake2b.h"
/* 128-bit (SSE2/SSSE3) rounds, also when the compiler flags enable AVX2. */
#define ARGON2_BLAMKA_SSE2
#include "blake2/blamka-round-opt.h"
/*