keccak-fast.c

Minimal SIMD keccak implementation
git clone git://git.finwo.net/lib/keccak-fast.c
Log | Files | Refs | README | LICENSE

commit a8f009f774de17a355c28e28e689be946c376ea5
parent d5efc1d3ff93f16f9ee6f5c2862ffd33c21eb1f8
Author: finwo <finwo@pm.me>
Date:   Sat, 10 Oct 2026 16:32:59 +0200

Add turboshake

Diffstat:
MREADME.md | 10++++++----
Msrc/backend/scalar-impl.h | 113+++++++++++++++++++++++++++++++++++++++++++++++--------------------------------
Msrc/backend/scalar.c | 2++
Msrc/backend/scalar_bmi.c | 2++
Msrc/keccak-fast-internal.h | 4++++
Msrc/keccak-fast.c | 2++
Msrc/keccak-fast.h | 2++
Mtest/kat.c | 18+++++++++++++++++-
8 files changed, 103 insertions(+), 50 deletions(-)

diff --git a/README.md b/README.md @@ -1,8 +1,8 @@ # keccak-fast -Drop-in-speed replacement for `coruus/keccak-tiny`. Six SHA-3/SHAKE one-shots -plus batch entry points. Plain C11, libc only, no build flags, -runtime-dispatched, thread-safe. +Drop-in-speed replacement for `coruus/keccak-tiny`. SHA-3/SHAKE one-shots, +TurboSHAKE128/256, and batch entry points. Plain C11, libc only, no build +flags, runtime-dispatched, thread-safe. ## Use @@ -16,7 +16,9 @@ kf_shake256_batch(count, in, inlen, out, outlen); /* equal-length, packed */ The one-shots (`kf_shake128`, `kf_shake256`, `kf_sha3_224`, `kf_sha3_256`, `kf_sha3_384`, `kf_sha3_512`) are `kf_hash_fn` table entries: a call is one indirect jump. Batch variants are `kf_<algo>_batch`, taking `count` -equal-length messages packed back to back. +equal-length messages packed back to back. `kf_turboshake128` and +`kf_turboshake256` implement TurboSHAKE (same sponge over Keccak-p[1600,12], +domain `0x1F`), about 1.9x faster than SHAKE. ## Build diff --git a/src/backend/scalar-impl.h b/src/backend/scalar-impl.h @@ -43,31 +43,36 @@ static const uint64_t RC[24] = { v = 0; \ REPEAT5(e; v += s;) -static inline void keccakf(void *state) { - uint64_t *a = (uint64_t *)state; - uint64_t b[5] = {0}; - uint64_t t = 0; - uint8_t x, y; - - for (int i = 0; i < 24; i++) { - /* theta */ - FOR5(x, 1, b[x] = 0; FOR5(y, 5, b[x] ^= a[x + y];)) - FOR5(x, 1, FOR5(y, 5, a[y + x] ^= b[(x + 4) % 5] ^ rol(b[(x + 1) % 5], 1);)) - /* rho and pi */ - t = a[1]; - x = 0; - REPEAT24(b[0] = a[pi[x]]; - a[pi[x]] = rol(t, rho[x]); - t = b[0]; - x++;) - /* chi */ - FOR5(y, - 5, - FOR5(x, 1, b[x] = a[y + x];) - FOR5(x, 1, a[y + x] = b[x] ^ ((~b[(x + 1) % 5]) & b[(x + 2) % 5]);)) - /* iota */ - a[0] ^= RC[i]; +#define KF_ROUNDS(FIRST) \ + uint64_t *a = (uint64_t *)state; \ + uint64_t b[5] = {0}; \ + uint64_t t = 0; \ + uint8_t x, y; \ + for (int i = (FIRST); i < 24; i++) { \ + /* theta */ \ + FOR5(x, 1, b[x] = 0; FOR5(y, 5, b[x] ^= a[x + y];)) \ + FOR5(x, 1, FOR5(y, 5, a[y + x] ^= b[(x + 4) % 5] ^ rol(b[(x + 1) % 5], 1);)) \ + /* rho and pi */ \ + t = a[1]; \ + x = 0; \ + REPEAT24(b[0] = a[pi[x]]; \ + a[pi[x]] = rol(t, rho[x]); \ + t = b[0]; \ + x++;) \ + /* chi */ \ + FOR5(y, 5, \ + FOR5(x, 1, b[x] = a[y + x];) \ + FOR5(x, 1, a[y + x] = b[x] ^ ((~b[(x + 1) % 5]) & b[(x + 2) % 5]);)) \ + /* iota */ \ + a[0] ^= RC[i]; \ } + +static inline void keccakf(void *state) { + KF_ROUNDS(0) +} + +static inline void keccak12(void *state) { + KF_ROUNDS(12) } #define _(S) do { S } while (0) @@ -87,36 +92,46 @@ mkapply_sd(setout, dst[i] = src[i]) #define Plen 200 -#define foldP(I, L, F) \ - while (L >= rate) { \ - F(a, I, rate); \ - keccakf(a); \ - I += rate; \ - L -= rate; \ +#define foldP(I, L, F, PERM) \ + while (L >= rate) { \ + F(a, I, rate); \ + PERM(a); \ + I += rate; \ + L -= rate; \ } -static inline int kf_hash_impl(uint8_t *out, size_t outlen, - const uint8_t *in, size_t inlen, - size_t rate, uint8_t delim) { +#define KF_HASH_BODY(PERM) \ + uint8_t a[Plen] = {0}; \ + foldP(in, inlen, xorin, PERM); \ + a[inlen] ^= delim; \ + a[rate - 1] ^= 0x80; \ + xorin(a, in, inlen); \ + PERM(a); \ + foldP(out, outlen, setout, PERM); \ + setout(a, out, outlen); \ + memset_s(a, 200, 0, 200); \ + return 0 + +static inline int kf_hash24(uint8_t *out, size_t outlen, const uint8_t *in, + size_t inlen, size_t rate, uint8_t delim) { if ((out == NULL) || ((in == NULL) && inlen != 0) || (rate >= Plen)) { return -1; } - uint8_t a[Plen] = {0}; - foldP(in, inlen, xorin); - a[inlen] ^= delim; - a[rate - 1] ^= 0x80; - xorin(a, in, inlen); - keccakf(a); - foldP(out, outlen, setout); - setout(a, out, outlen); - memset_s(a, 200, 0, 200); - return 0; + KF_HASH_BODY(keccakf); +} + +static inline int kf_hash12(uint8_t *out, size_t outlen, const uint8_t *in, + size_t inlen, size_t rate, uint8_t delim) { + if ((out == NULL) || ((in == NULL) && inlen != 0) || (rate >= Plen)) { + return -1; + } + KF_HASH_BODY(keccak12); } #define defshake(bits) \ KF_LINKAGE int KF_FN(shake##bits)(uint8_t *out, size_t outlen, \ const uint8_t *in, size_t inlen) { \ - return kf_hash_impl(out, outlen, in, inlen, 200 - (bits / 4), 0x1f); \ + return kf_hash24(out, outlen, in, inlen, 200 - (bits / 4), 0x1f); \ } #define defsha3(bits) \ KF_LINKAGE int KF_FN(sha3_##bits)(uint8_t *out, size_t outlen, \ @@ -124,7 +139,12 @@ static inline int kf_hash_impl(uint8_t *out, size_t outlen, if (outlen > (bits / 8)) { \ return -1; \ } \ - return kf_hash_impl(out, outlen, in, inlen, 200 - (bits / 4), 0x06); \ + return kf_hash24(out, outlen, in, inlen, 200 - (bits / 4), 0x06); \ + } +#define defturboshake(bits) \ + KF_LINKAGE int KF_FN(turboshake##bits)(uint8_t *out, size_t outlen, \ + const uint8_t *in, size_t inlen) { \ + return kf_hash12(out, outlen, in, inlen, 200 - (bits / 4), 0x1f); \ } defshake(128) @@ -133,8 +153,11 @@ defsha3(224) defsha3(256) defsha3(384) defsha3(512) +defturboshake(128) +defturboshake(256) #undef defshake #undef defsha3 +#undef defturboshake #endif /* FINWO_SCALAR_IMPL_H */ diff --git a/src/backend/scalar.c b/src/backend/scalar.c @@ -43,6 +43,8 @@ __attribute__((constructor)) static void kf_scalar_register(void) { kf_sha3_256 = kf_scalar_sha3_256; kf_sha3_384 = kf_scalar_sha3_384; kf_sha3_512 = kf_scalar_sha3_512; + kf_turboshake128 = kf_scalar_turboshake128; + kf_turboshake256 = kf_scalar_turboshake256; kf_backend_priority = KF_PRIO_SCALAR; kf_backend_label = "scalar"; } diff --git a/src/backend/scalar_bmi.c b/src/backend/scalar_bmi.c @@ -43,6 +43,8 @@ __attribute__((constructor)) static void kf_bmi_register(void) { kf_sha3_256 = kf_bmi_sha3_256; kf_sha3_384 = kf_bmi_sha3_384; kf_sha3_512 = kf_bmi_sha3_512; + kf_turboshake128 = kf_bmi_turboshake128; + kf_turboshake256 = kf_bmi_turboshake256; kf_backend_priority = KF_PRIO_BMI; kf_backend_label = "scalar+bmi"; } diff --git a/src/keccak-fast-internal.h b/src/keccak-fast-internal.h @@ -35,6 +35,10 @@ int kf_scalar_sha3_384(uint8_t *out, size_t outlen, const uint8_t *in, size_t inlen); int kf_scalar_sha3_512(uint8_t *out, size_t outlen, const uint8_t *in, size_t inlen); +int kf_scalar_turboshake128(uint8_t *out, size_t outlen, const uint8_t *in, + size_t inlen); +int kf_scalar_turboshake256(uint8_t *out, size_t outlen, const uint8_t *in, + size_t inlen); void kf_scalar_permute(uint64_t state[25]); void kf_scalar_batch(uint64_t states[][25], size_t count); diff --git a/src/keccak-fast.c b/src/keccak-fast.c @@ -14,6 +14,8 @@ kf_hash_fn kf_sha3_224 = kf_scalar_sha3_224; kf_hash_fn kf_sha3_256 = kf_scalar_sha3_256; kf_hash_fn kf_sha3_384 = kf_scalar_sha3_384; kf_hash_fn kf_sha3_512 = kf_scalar_sha3_512; +kf_hash_fn kf_turboshake128 = kf_scalar_turboshake128; +kf_hash_fn kf_turboshake256 = kf_scalar_turboshake256; int kf_backend_priority = KF_PRIO_SCALAR; const char *kf_backend_label = "scalar"; diff --git a/src/keccak-fast.h b/src/keccak-fast.h @@ -13,6 +13,8 @@ extern kf_hash_fn kf_sha3_224; extern kf_hash_fn kf_sha3_256; extern kf_hash_fn kf_sha3_384; extern kf_hash_fn kf_sha3_512; +extern kf_hash_fn kf_turboshake128; +extern kf_hash_fn kf_turboshake256; int kf_shake128_batch(size_t count, const uint8_t *in, size_t inlen, uint8_t *out, size_t outlen); diff --git a/test/kat.c b/test/kat.c @@ -39,7 +39,7 @@ int main(void) { const uint8_t *abc = (const uint8_t *)"abc"; const uint8_t *nil = NULL; - printf("1..16\n"); + printf("1..22\n"); check("shake128 empty", kf_shake128, nil, 0, "7f9c2ba4e88f827d616045507605853ed73b8093f6efbc88eb1a6eacfa66ef26"); @@ -75,6 +75,22 @@ int main(void) { "b751850b1a57168a5693cd924b6b096e08f621827444f70d884f5d0240d2712e" "10e116e9192af3c91a7ec57647e3934057340b4cf408d5a56592f8274eec53f0"); + check("turboshake128 empty", kf_turboshake128, nil, 0, + "1e415f1c5983aff2169217277d17bb538cd945a397ddec541f1ce41af2c1b74c"); + check("turboshake128 abc", kf_turboshake128, abc, 3, + "dcf1646dfe993a8eb6b782d1faaca6d82416a5dcf1de98ee3c6dbc5e1dc63018"); + check("turboshake128 200", kf_turboshake128, qbf, 200, + "5b96f9aac16a9047cd03b83c950fb6b074a31ea05f0aaa3aa3a7f98deaab289c"); + check("turboshake256 empty", kf_turboshake256, nil, 0, + "367a329dafea871c7802ec67f905ae13c57695dc2c6663c61035f59a18f8e7db" + "11edc0e12e91ea60eb6b32df06dd7f002fbafabb6e13ec1cc20d995547600db0"); + check("turboshake256 abc", kf_turboshake256, abc, 3, + "63824b1431a7372e85edc022c9d7afdd027472fcfa33c887d6f5aaf8dc5d4db6" + "8afbcb5714b49b7ffd8dd115dd5bd5436f837236845a230d6969a4083a113617"); + check("turboshake256 200", kf_turboshake256, qbf, 200, + "c27b5422da9177efaf2a335b6ea79cb57961a8cc3ca8f870f17ebbbfd662086a" + "fe62a10085fb7c413fcc0f586aa938d152254a77aa0d682eb846c82199428a71"); + { uint8_t out[64]; int bad1 = kf_sha3_256(out, 33, abc, 3);