scalar_bmi.c (1991B)
1 /* Runtime BMI1/BMI2 variant of the scalar backend. Same source as scalar.c, 2 * but the target enables `andn` (replaces `not`+`and` in chi) and `rorx`. */ 3 4 #include <stddef.h> 5 #include <stdint.h> 6 #include <string.h> 7 8 #include "../keccak-fast-internal.h" 9 10 #if (defined(__x86_64__) || defined(__i386__)) && !defined(KECCAK_FAST_NO_BMI) 11 12 /* clang schedules `rorx` worse than `rol`; gcc is faster with it. */ 13 #if defined(__clang__) 14 #pragma clang attribute push( \ 15 __attribute__((target("no-avx,no-avx2,no-avx512f,bmi,no-bmi2"))), \ 16 apply_to=function) 17 #elif defined(__GNUC__) 18 #pragma GCC target("no-avx,no-avx2,no-avx512f,bmi,bmi2") 19 #endif 20 21 #define KF_PREFIX kf_bmi_ 22 #define KF_LINKAGE static 23 #include "scalar-impl.h" 24 25 void kf_bmi_batch(uint64_t states[][25], size_t count) { 26 for (size_t i = 0; i < count; i++) { 27 keccakf(states[i]); 28 } 29 } 30 31 void kf_bmi_batch12(uint64_t states[][25], size_t count) { 32 for (size_t i = 0; i < count; i++) { 33 keccak12(states[i]); 34 } 35 } 36 37 __attribute__((constructor)) static void kf_bmi_register(void) { 38 #if defined(__clang__) 39 if (!__builtin_cpu_supports("bmi")) { 40 #else 41 if (!__builtin_cpu_supports("bmi") || !__builtin_cpu_supports("bmi2")) { 42 #endif 43 return; 44 } 45 if (kf_backend_priority <= KF_PRIO_BMI) { 46 kf_shake128 = kf_bmi_shake128; 47 kf_shake256 = kf_bmi_shake256; 48 kf_sha3_224 = kf_bmi_sha3_224; 49 kf_sha3_256 = kf_bmi_sha3_256; 50 kf_sha3_384 = kf_bmi_sha3_384; 51 kf_sha3_512 = kf_bmi_sha3_512; 52 kf_turboshake128 = kf_bmi_turboshake128; 53 kf_turboshake256 = kf_bmi_turboshake256; 54 kf_backend_priority = KF_PRIO_BMI; 55 kf_backend_label = "scalar+bmi"; 56 } 57 if (kf_batch_priority <= KF_PRIO_BMI) { 58 kf_batch_perm = kf_bmi_batch; 59 kf_batch_perm12 = kf_bmi_batch12; 60 kf_batch_priority = KF_PRIO_BMI; 61 kf_batch_lanes = 1; 62 kf_batch_label = "scalar+bmi"; 63 } 64 } 65 66 #if defined(__clang__) 67 #pragma clang attribute pop 68 #endif 69 70 #endif