commit 8412665afd9d5d9140642eea48055b43387d1a2b
parent a8f009f774de17a355c28e28e689be946c376ea5
Author: finwo <finwo@pm.me>
Date: Sat, 10 Oct 2026 16:45:09 +0200
Hint unroll factor to compiler
Diffstat:
2 files changed, 20 insertions(+), 1 deletion(-)
diff --git a/README.md b/README.md
@@ -44,7 +44,8 @@ Backends register from constructors; the highest available priority wins
The scalar backend is the donor, pinned to the x86-64 baseline
(`no-avx,no-avx2,no-avx512f,no-bmi,no-bmi2`) so a consumer's `-march=native`
-cannot vectorize it or emit BMI. `scalar+bmi` is the same permutation with
+cannot vectorize it or emit BMI. The round loop is unrolled 4x, which beats
+both no unroll and a full 24x unroll. `scalar+bmi` is the same permutation with
`bmi` enabled, which turns chi's `not`+`and` into `andn`; gcc also gets `rorx`
for the rotates. About 11% faster than the donor.
diff --git a/src/backend/scalar-impl.h b/src/backend/scalar-impl.h
@@ -43,11 +43,29 @@ static const uint64_t RC[24] = {
v = 0; \
REPEAT5(e; v += s;)
+/* Round loop unrolled 4x, after armed-keccak's roundx4: enough straight-line
+ work for the scheduler without the register spills a full 24x unroll causes. */
+#ifndef KF_UNROLL
+#define KF_UNROLL 4
+#endif
+
+#define KF_STR_(x) #x
+#define KF_STR(x) KF_STR_(x)
+
+#if KF_UNROLL > 0 && defined(__clang__)
+#define KF_UNROLL_PRAGMA _Pragma(KF_STR(clang loop unroll_count(KF_UNROLL)))
+#elif KF_UNROLL > 0 && defined(__GNUC__)
+#define KF_UNROLL_PRAGMA _Pragma(KF_STR(GCC unroll KF_UNROLL))
+#else
+#define KF_UNROLL_PRAGMA
+#endif
+
#define KF_ROUNDS(FIRST) \
uint64_t *a = (uint64_t *)state; \
uint64_t b[5] = {0}; \
uint64_t t = 0; \
uint8_t x, y; \
+ KF_UNROLL_PRAGMA \
for (int i = (FIRST); i < 24; i++) { \
/* theta */ \
FOR5(x, 1, b[x] = 0; FOR5(y, 5, b[x] ^= a[x + y];)) \