From 4f76ff713206c2c134ad97ac7d70cac15699e07f Mon Sep 17 00:00:00 2001 From: Ville Takio <93599126+flatstik@users.noreply.github.com> Date: Mon, 5 Oct 2026 00:22:45 +0000 Subject: [PATCH] Argon2: fix build with -march=x86-64-v3/v4 and XOP targets (#1906) * Argon2: fix build with -march=x86-64-v3/v4 and XOP targets blamka-round-opt.h picked its SSE2, AVX2 or AVX-512 definitions only from the compiler flags. With global flags such as -march=x86-64-v3 or -march=x86-64-v4, opt_sse2.c (and with v4 also opt_avx2.c) saw __AVX2__ or __AVX512F__ and got definitions that do not match its code, so the build failed. Let opt_sse2.c and opt_avx2.c select the variant they implement, and fall back to the compiler flags for other includers. This needs no new compiler options, so it also works with old compilers and with the plain objects used by NOASM=1 builds. With XOP targets (-march=bdver*), the header skipped its _mm_roti_epi64 fallback but did not include the intrinsic; include for GCC and Clang there. Hashes are identical through the AVX2, SSE2 and reference code paths, and match an unmodified build. Fixes #1867 Refs #1871 * Argon2: update branch comments in blamka-round-opt.h The #else/#endif comments still named __AVX2__ and __AVX512F__, but the branches are now selected by ARGON2_BLAMKA_USE_SSE2/AVX2. Refs #1867 --- .../Argon2/src/blake2/blamka-round-opt.h | 28 ++++++++++++++----- src/Crypto/Argon2/src/opt_avx2.c | 2 ++ src/Crypto/Argon2/src/opt_sse2.c | 2 ++ 3 files changed, 25 insertions(+), 7 deletions(-) diff --git a/src/Crypto/Argon2/src/blake2/blamka-round-opt.h b/src/Crypto/Argon2/src/blake2/blamka-round-opt.h index 21fa0349..455a034f 100644 --- a/src/Crypto/Argon2/src/blake2/blamka-round-opt.h +++ b/src/Crypto/Argon2/src/blake2/blamka-round-opt.h @@ -27,11 +27,25 @@ #endif #if defined(__XOP__) && (defined(__GNUC__) || defined(__clang__)) -//#include +#include /* for _mm_roti_epi64 (-march=bdver*) */ #endif -#if !defined(__AVX512F__) -#if !defined(__AVX2__) +/* The variant is chosen by the file that includes this header + * (ARGON2_BLAMKA_SSE2 in opt_sse2.c, ARGON2_BLAMKA_AVX2 in opt_avx2.c), not + * only by the compiler flags: with global flags such as -march=x86-64-v3 or + * -march=x86-64-v4, every file sees __AVX2__ or __AVX512F__. */ +#if defined(ARGON2_BLAMKA_SSE2) +#define ARGON2_BLAMKA_USE_SSE2 +#elif defined(ARGON2_BLAMKA_AVX2) +#define ARGON2_BLAMKA_USE_AVX2 +#elif !defined(__AVX512F__) && !defined(__AVX2__) +#define ARGON2_BLAMKA_USE_SSE2 +#elif !defined(__AVX512F__) +#define ARGON2_BLAMKA_USE_AVX2 +#endif + +#if defined(ARGON2_BLAMKA_USE_SSE2) || defined(ARGON2_BLAMKA_USE_AVX2) +#if defined(ARGON2_BLAMKA_USE_SSE2) #if !defined(__XOP__) #if defined(__SSSE3__) #define r16 \ @@ -179,7 +193,7 @@ static BLAKE2_INLINE __m128i fBlaMka(__m128i x, __m128i y) { \ UNDIAGONALIZE(A0, B0, C0, D0, A1, B1, C1, D1); \ } while ((void)0, 0) -#else /* __AVX2__ */ +#else /* ARGON2_BLAMKA_USE_AVX2 */ //#include @@ -326,9 +340,9 @@ static BLAKE2_INLINE __m128i fBlaMka(__m128i x, __m128i y) { UNDIAGONALIZE_2(A0, A1, B0, B1, C0, C1, D0, D1) \ } while((void)0, 0); -#endif /* __AVX2__ */ +#endif /* ARGON2_BLAMKA_USE_SSE2 */ -#else /* __AVX512F__ */ +#else /* AVX-512 */ //#include @@ -468,5 +482,5 @@ static __m512i muladd(__m512i x, __m512i y) UNSWAP_QUARTERS(D0, D1); \ } while ((void)0, 0) -#endif /* __AVX512F__ */ +#endif /* ARGON2_BLAMKA_USE_SSE2 || ARGON2_BLAMKA_USE_AVX2 */ #endif /* BLAKE_ROUND_MKA_OPT_H */ diff --git a/src/Crypto/Argon2/src/opt_avx2.c b/src/Crypto/Argon2/src/opt_avx2.c index 81d05695..3a2a708f 100644 --- a/src/Crypto/Argon2/src/opt_avx2.c +++ b/src/Crypto/Argon2/src/opt_avx2.c @@ -29,6 +29,8 @@ #include #include "blake2/blake2b.h" +/* 256-bit (AVX2) rounds, also when the compiler flags enable AVX-512. */ +#define ARGON2_BLAMKA_AVX2 #include "blake2/blamka-round-opt.h" /* diff --git a/src/Crypto/Argon2/src/opt_sse2.c b/src/Crypto/Argon2/src/opt_sse2.c index d401954a..c27273b8 100644 --- a/src/Crypto/Argon2/src/opt_sse2.c +++ b/src/Crypto/Argon2/src/opt_sse2.c @@ -26,6 +26,8 @@ #if CRYPTOPP_BOOL_SSE2_INTRINSICS_AVAILABLE #include "blake2/blake2b.h" +/* 128-bit (SSE2/SSSE3) rounds, also when the compiler flags enable AVX2. */ +#define ARGON2_BLAMKA_SSE2 #include "blake2/blamka-round-opt.h" /*