commit a8f009f774de17a355c28e28e689be946c376ea5
parent d5efc1d3ff93f16f9ee6f5c2862ffd33c21eb1f8
Author: finwo <finwo@pm.me>
Date: Sat, 10 Oct 2026 16:32:59 +0200
Add turboshake
Diffstat:
8 files changed, 103 insertions(+), 50 deletions(-)
diff --git a/README.md b/README.md
@@ -1,8 +1,8 @@
# keccak-fast
-Drop-in-speed replacement for `coruus/keccak-tiny`. Six SHA-3/SHAKE one-shots
-plus batch entry points. Plain C11, libc only, no build flags,
-runtime-dispatched, thread-safe.
+Drop-in-speed replacement for `coruus/keccak-tiny`. SHA-3/SHAKE one-shots,
+TurboSHAKE128/256, and batch entry points. Plain C11, libc only, no build
+flags, runtime-dispatched, thread-safe.
## Use
@@ -16,7 +16,9 @@ kf_shake256_batch(count, in, inlen, out, outlen); /* equal-length, packed */
The one-shots (`kf_shake128`, `kf_shake256`, `kf_sha3_224`, `kf_sha3_256`,
`kf_sha3_384`, `kf_sha3_512`) are `kf_hash_fn` table entries: a call is one
indirect jump. Batch variants are `kf_<algo>_batch`, taking `count`
-equal-length messages packed back to back.
+equal-length messages packed back to back. `kf_turboshake128` and
+`kf_turboshake256` implement TurboSHAKE (same sponge over Keccak-p[1600,12],
+domain `0x1F`), about 1.9x faster than SHAKE.
## Build
diff --git a/src/backend/scalar-impl.h b/src/backend/scalar-impl.h
@@ -43,31 +43,36 @@ static const uint64_t RC[24] = {
v = 0; \
REPEAT5(e; v += s;)
-static inline void keccakf(void *state) {
- uint64_t *a = (uint64_t *)state;
- uint64_t b[5] = {0};
- uint64_t t = 0;
- uint8_t x, y;
-
- for (int i = 0; i < 24; i++) {
- /* theta */
- FOR5(x, 1, b[x] = 0; FOR5(y, 5, b[x] ^= a[x + y];))
- FOR5(x, 1, FOR5(y, 5, a[y + x] ^= b[(x + 4) % 5] ^ rol(b[(x + 1) % 5], 1);))
- /* rho and pi */
- t = a[1];
- x = 0;
- REPEAT24(b[0] = a[pi[x]];
- a[pi[x]] = rol(t, rho[x]);
- t = b[0];
- x++;)
- /* chi */
- FOR5(y,
- 5,
- FOR5(x, 1, b[x] = a[y + x];)
- FOR5(x, 1, a[y + x] = b[x] ^ ((~b[(x + 1) % 5]) & b[(x + 2) % 5]);))
- /* iota */
- a[0] ^= RC[i];
+#define KF_ROUNDS(FIRST) \
+ uint64_t *a = (uint64_t *)state; \
+ uint64_t b[5] = {0}; \
+ uint64_t t = 0; \
+ uint8_t x, y; \
+ for (int i = (FIRST); i < 24; i++) { \
+ /* theta */ \
+ FOR5(x, 1, b[x] = 0; FOR5(y, 5, b[x] ^= a[x + y];)) \
+ FOR5(x, 1, FOR5(y, 5, a[y + x] ^= b[(x + 4) % 5] ^ rol(b[(x + 1) % 5], 1);)) \
+ /* rho and pi */ \
+ t = a[1]; \
+ x = 0; \
+ REPEAT24(b[0] = a[pi[x]]; \
+ a[pi[x]] = rol(t, rho[x]); \
+ t = b[0]; \
+ x++;) \
+ /* chi */ \
+ FOR5(y, 5, \
+ FOR5(x, 1, b[x] = a[y + x];) \
+ FOR5(x, 1, a[y + x] = b[x] ^ ((~b[(x + 1) % 5]) & b[(x + 2) % 5]);)) \
+ /* iota */ \
+ a[0] ^= RC[i]; \
}
+
+static inline void keccakf(void *state) {
+ KF_ROUNDS(0)
+}
+
+static inline void keccak12(void *state) {
+ KF_ROUNDS(12)
}
#define _(S) do { S } while (0)
@@ -87,36 +92,46 @@ mkapply_sd(setout, dst[i] = src[i])
#define Plen 200
-#define foldP(I, L, F) \
- while (L >= rate) { \
- F(a, I, rate); \
- keccakf(a); \
- I += rate; \
- L -= rate; \
+#define foldP(I, L, F, PERM) \
+ while (L >= rate) { \
+ F(a, I, rate); \
+ PERM(a); \
+ I += rate; \
+ L -= rate; \
}
-static inline int kf_hash_impl(uint8_t *out, size_t outlen,
- const uint8_t *in, size_t inlen,
- size_t rate, uint8_t delim) {
+#define KF_HASH_BODY(PERM) \
+ uint8_t a[Plen] = {0}; \
+ foldP(in, inlen, xorin, PERM); \
+ a[inlen] ^= delim; \
+ a[rate - 1] ^= 0x80; \
+ xorin(a, in, inlen); \
+ PERM(a); \
+ foldP(out, outlen, setout, PERM); \
+ setout(a, out, outlen); \
+ memset_s(a, 200, 0, 200); \
+ return 0
+
+static inline int kf_hash24(uint8_t *out, size_t outlen, const uint8_t *in,
+ size_t inlen, size_t rate, uint8_t delim) {
if ((out == NULL) || ((in == NULL) && inlen != 0) || (rate >= Plen)) {
return -1;
}
- uint8_t a[Plen] = {0};
- foldP(in, inlen, xorin);
- a[inlen] ^= delim;
- a[rate - 1] ^= 0x80;
- xorin(a, in, inlen);
- keccakf(a);
- foldP(out, outlen, setout);
- setout(a, out, outlen);
- memset_s(a, 200, 0, 200);
- return 0;
+ KF_HASH_BODY(keccakf);
+}
+
+static inline int kf_hash12(uint8_t *out, size_t outlen, const uint8_t *in,
+ size_t inlen, size_t rate, uint8_t delim) {
+ if ((out == NULL) || ((in == NULL) && inlen != 0) || (rate >= Plen)) {
+ return -1;
+ }
+ KF_HASH_BODY(keccak12);
}
#define defshake(bits) \
KF_LINKAGE int KF_FN(shake##bits)(uint8_t *out, size_t outlen, \
const uint8_t *in, size_t inlen) { \
- return kf_hash_impl(out, outlen, in, inlen, 200 - (bits / 4), 0x1f); \
+ return kf_hash24(out, outlen, in, inlen, 200 - (bits / 4), 0x1f); \
}
#define defsha3(bits) \
KF_LINKAGE int KF_FN(sha3_##bits)(uint8_t *out, size_t outlen, \
@@ -124,7 +139,12 @@ static inline int kf_hash_impl(uint8_t *out, size_t outlen,
if (outlen > (bits / 8)) { \
return -1; \
} \
- return kf_hash_impl(out, outlen, in, inlen, 200 - (bits / 4), 0x06); \
+ return kf_hash24(out, outlen, in, inlen, 200 - (bits / 4), 0x06); \
+ }
+#define defturboshake(bits) \
+ KF_LINKAGE int KF_FN(turboshake##bits)(uint8_t *out, size_t outlen, \
+ const uint8_t *in, size_t inlen) { \
+ return kf_hash12(out, outlen, in, inlen, 200 - (bits / 4), 0x1f); \
}
defshake(128)
@@ -133,8 +153,11 @@ defsha3(224)
defsha3(256)
defsha3(384)
defsha3(512)
+defturboshake(128)
+defturboshake(256)
#undef defshake
#undef defsha3
+#undef defturboshake
#endif /* FINWO_SCALAR_IMPL_H */
diff --git a/src/backend/scalar.c b/src/backend/scalar.c
@@ -43,6 +43,8 @@ __attribute__((constructor)) static void kf_scalar_register(void) {
kf_sha3_256 = kf_scalar_sha3_256;
kf_sha3_384 = kf_scalar_sha3_384;
kf_sha3_512 = kf_scalar_sha3_512;
+ kf_turboshake128 = kf_scalar_turboshake128;
+ kf_turboshake256 = kf_scalar_turboshake256;
kf_backend_priority = KF_PRIO_SCALAR;
kf_backend_label = "scalar";
}
diff --git a/src/backend/scalar_bmi.c b/src/backend/scalar_bmi.c
@@ -43,6 +43,8 @@ __attribute__((constructor)) static void kf_bmi_register(void) {
kf_sha3_256 = kf_bmi_sha3_256;
kf_sha3_384 = kf_bmi_sha3_384;
kf_sha3_512 = kf_bmi_sha3_512;
+ kf_turboshake128 = kf_bmi_turboshake128;
+ kf_turboshake256 = kf_bmi_turboshake256;
kf_backend_priority = KF_PRIO_BMI;
kf_backend_label = "scalar+bmi";
}
diff --git a/src/keccak-fast-internal.h b/src/keccak-fast-internal.h
@@ -35,6 +35,10 @@ int kf_scalar_sha3_384(uint8_t *out, size_t outlen, const uint8_t *in,
size_t inlen);
int kf_scalar_sha3_512(uint8_t *out, size_t outlen, const uint8_t *in,
size_t inlen);
+int kf_scalar_turboshake128(uint8_t *out, size_t outlen, const uint8_t *in,
+ size_t inlen);
+int kf_scalar_turboshake256(uint8_t *out, size_t outlen, const uint8_t *in,
+ size_t inlen);
void kf_scalar_permute(uint64_t state[25]);
void kf_scalar_batch(uint64_t states[][25], size_t count);
diff --git a/src/keccak-fast.c b/src/keccak-fast.c
@@ -14,6 +14,8 @@ kf_hash_fn kf_sha3_224 = kf_scalar_sha3_224;
kf_hash_fn kf_sha3_256 = kf_scalar_sha3_256;
kf_hash_fn kf_sha3_384 = kf_scalar_sha3_384;
kf_hash_fn kf_sha3_512 = kf_scalar_sha3_512;
+kf_hash_fn kf_turboshake128 = kf_scalar_turboshake128;
+kf_hash_fn kf_turboshake256 = kf_scalar_turboshake256;
int kf_backend_priority = KF_PRIO_SCALAR;
const char *kf_backend_label = "scalar";
diff --git a/src/keccak-fast.h b/src/keccak-fast.h
@@ -13,6 +13,8 @@ extern kf_hash_fn kf_sha3_224;
extern kf_hash_fn kf_sha3_256;
extern kf_hash_fn kf_sha3_384;
extern kf_hash_fn kf_sha3_512;
+extern kf_hash_fn kf_turboshake128;
+extern kf_hash_fn kf_turboshake256;
int kf_shake128_batch(size_t count, const uint8_t *in, size_t inlen,
uint8_t *out, size_t outlen);
diff --git a/test/kat.c b/test/kat.c
@@ -39,7 +39,7 @@ int main(void) {
const uint8_t *abc = (const uint8_t *)"abc";
const uint8_t *nil = NULL;
- printf("1..16\n");
+ printf("1..22\n");
check("shake128 empty", kf_shake128, nil, 0,
"7f9c2ba4e88f827d616045507605853ed73b8093f6efbc88eb1a6eacfa66ef26");
@@ -75,6 +75,22 @@ int main(void) {
"b751850b1a57168a5693cd924b6b096e08f621827444f70d884f5d0240d2712e"
"10e116e9192af3c91a7ec57647e3934057340b4cf408d5a56592f8274eec53f0");
+ check("turboshake128 empty", kf_turboshake128, nil, 0,
+ "1e415f1c5983aff2169217277d17bb538cd945a397ddec541f1ce41af2c1b74c");
+ check("turboshake128 abc", kf_turboshake128, abc, 3,
+ "dcf1646dfe993a8eb6b782d1faaca6d82416a5dcf1de98ee3c6dbc5e1dc63018");
+ check("turboshake128 200", kf_turboshake128, qbf, 200,
+ "5b96f9aac16a9047cd03b83c950fb6b074a31ea05f0aaa3aa3a7f98deaab289c");
+ check("turboshake256 empty", kf_turboshake256, nil, 0,
+ "367a329dafea871c7802ec67f905ae13c57695dc2c6663c61035f59a18f8e7db"
+ "11edc0e12e91ea60eb6b32df06dd7f002fbafabb6e13ec1cc20d995547600db0");
+ check("turboshake256 abc", kf_turboshake256, abc, 3,
+ "63824b1431a7372e85edc022c9d7afdd027472fcfa33c887d6f5aaf8dc5d4db6"
+ "8afbcb5714b49b7ffd8dd115dd5bd5436f837236845a230d6969a4083a113617");
+ check("turboshake256 200", kf_turboshake256, qbf, 200,
+ "c27b5422da9177efaf2a335b6ea79cb57961a8cc3ca8f870f17ebbbfd662086a"
+ "fe62a10085fb7c413fcc0f586aa938d152254a77aa0d682eb846c82199428a71");
+
{
uint8_t out[64];
int bad1 = kf_sha3_256(out, 33, abc, 3);