keccak-fast.c

Minimal SIMD keccak implementation
git clone git://git.finwo.net/lib/keccak-fast.c
Log | Files | Refs | README | LICENSE

commit 8412665afd9d5d9140642eea48055b43387d1a2b
parent a8f009f774de17a355c28e28e689be946c376ea5
Author: finwo <finwo@pm.me>
Date:   Sat, 10 Oct 2026 16:45:09 +0200

Hint unroll factor to compiler

Diffstat:
MREADME.md | 3++-
Msrc/backend/scalar-impl.h | 18++++++++++++++++++
2 files changed, 20 insertions(+), 1 deletion(-)

diff --git a/README.md b/README.md @@ -44,7 +44,8 @@ Backends register from constructors; the highest available priority wins The scalar backend is the donor, pinned to the x86-64 baseline (`no-avx,no-avx2,no-avx512f,no-bmi,no-bmi2`) so a consumer's `-march=native` -cannot vectorize it or emit BMI. `scalar+bmi` is the same permutation with +cannot vectorize it or emit BMI. The round loop is unrolled 4x, which beats +both no unroll and a full 24x unroll. `scalar+bmi` is the same permutation with `bmi` enabled, which turns chi's `not`+`and` into `andn`; gcc also gets `rorx` for the rotates. About 11% faster than the donor. diff --git a/src/backend/scalar-impl.h b/src/backend/scalar-impl.h @@ -43,11 +43,29 @@ static const uint64_t RC[24] = { v = 0; \ REPEAT5(e; v += s;) +/* Round loop unrolled 4x, after armed-keccak's roundx4: enough straight-line + work for the scheduler without the register spills a full 24x unroll causes. */ +#ifndef KF_UNROLL +#define KF_UNROLL 4 +#endif + +#define KF_STR_(x) #x +#define KF_STR(x) KF_STR_(x) + +#if KF_UNROLL > 0 && defined(__clang__) +#define KF_UNROLL_PRAGMA _Pragma(KF_STR(clang loop unroll_count(KF_UNROLL))) +#elif KF_UNROLL > 0 && defined(__GNUC__) +#define KF_UNROLL_PRAGMA _Pragma(KF_STR(GCC unroll KF_UNROLL)) +#else +#define KF_UNROLL_PRAGMA +#endif + #define KF_ROUNDS(FIRST) \ uint64_t *a = (uint64_t *)state; \ uint64_t b[5] = {0}; \ uint64_t t = 0; \ uint8_t x, y; \ + KF_UNROLL_PRAGMA \ for (int i = (FIRST); i < 24; i++) { \ /* theta */ \ FOR5(x, 1, b[x] = 0; FOR5(y, 5, b[x] ^= a[x + y];)) \