From e6abd33bb42e3b09631b423929e218994cd1df1d Mon Sep 17 00:00:00 2001 From: Daniele Lacamera Date: Fri, 2 Oct 2026 21:16:53 +0200 Subject: [PATCH 1/4] Falcon: keep fpr_add and fpr_floor selects branchless --- wolfcrypt/src/falcon.c | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/wolfcrypt/src/falcon.c b/wolfcrypt/src/falcon.c index 6ae564110d3..0b0d846414c 100644 --- a/wolfcrypt/src/falcon.c +++ b/wolfcrypt/src/falcon.c @@ -692,6 +692,19 @@ static int falcon_sign_core(falcon_sampler_ctx* spc, const fpr* expanded, /* operand-dependent timing on platforms whose shift is data dependent. */ /* ------------------------------------------------------------------------- */ +/* Return x unchanged but opaque to the optimizer, so that masks derived from + * it are not turned into conditional branches. */ +static WC_MAYBE_UNUSED WC_INLINE word32 fpr_ct_opaque32(word32 x) +{ +#if defined(__GNUC__) && !defined(WOLFSSL_NO_ASM) + __asm__ __volatile__("" : "+r"(x)); +#else + volatile word32 v = x; + x = v; +#endif + return x; +} + /* Right-shift a 64-bit unsigned value by n (0..63), constant-time. */ static WC_MAYBE_UNUSED WC_INLINE fpr fpr_ursh(word64 x, int n) { @@ -885,7 +898,8 @@ sword64 fpr_floor(fpr x) /* If the true shift count was 64 or more, replace xi with 0 (nonnegative) * or -1 (negative). This also fixes the bogus implicit-bit assumption for * a zero input. */ - xi ^= (xi ^ -(sword64)t) & -(sword64)((word32)(63 - cc) >> 31); + xi ^= (xi ^ -(sword64)t) + & -(sword64)fpr_ct_opaque32((word32)(63 - cc) >> 31); return xi; } @@ -929,6 +943,7 @@ fpr fpr_add(fpr x, fpr y) za = (x & m) - (y & m); cs = (word32)(za >> 63) | ((1U - (word32)(((word64)0 - za) >> 63)) & (word32)(x >> 63)); + cs = fpr_ct_opaque32(cs); m = (x ^ y) & ((word64)0 - (word64)cs); x ^= m; y ^= m; From f9dad2356e40a28daa363f0821dd18ad7c6085cb Mon Sep 17 00:00:00 2001 From: Daniele Lacamera Date: Fri, 2 Oct 2026 21:20:59 +0200 Subject: [PATCH 2/4] Falcon: branchless fpr shift helpers on 32-bit targets --- wolfcrypt/src/falcon.c | 49 +++++++++++++++++++++++++++++++++++++++--- 1 file changed, 46 insertions(+), 3 deletions(-) diff --git a/wolfcrypt/src/falcon.c b/wolfcrypt/src/falcon.c index 0b0d846414c..c4209282176 100644 --- a/wolfcrypt/src/falcon.c +++ b/wolfcrypt/src/falcon.c @@ -708,22 +708,65 @@ static WC_MAYBE_UNUSED WC_INLINE word32 fpr_ct_opaque32(word32 x) /* Right-shift a 64-bit unsigned value by n (0..63), constant-time. */ static WC_MAYBE_UNUSED WC_INLINE fpr fpr_ursh(word64 x, int n) { - x ^= (x ^ (x >> 32)) & ((word64)0 - (word64)(n >> 5)); +#ifdef WC_64BIT_CPU + x ^= (x ^ (x >> 32)) + & ((word64)0 - (word64)fpr_ct_opaque32((word32)n >> 5)); return x >> (n & 31); +#else + word32 lo = (word32)x; + word32 hi = (word32)(x >> 32); + word32 m = 0U - fpr_ct_opaque32((word32)n >> 5); + word32 s = (word32)n & 31; + + lo ^= (lo ^ hi) & m; + hi &= ~m; + lo = (lo >> s) | ((hi << (31 - s)) << 1); + hi >>= s; + return ((word64)hi << 32) | lo; +#endif } /* Right-shift a 64-bit signed value by n (0..63), constant-time. */ static WC_MAYBE_UNUSED WC_INLINE sword64 fpr_irsh(sword64 x, int n) { - x ^= (x ^ (x >> 32)) & ((sword64)0 - (sword64)(n >> 5)); +#ifdef WC_64BIT_CPU + x ^= (x ^ (x >> 32)) + & ((sword64)0 - (sword64)fpr_ct_opaque32((word32)n >> 5)); return x >> (n & 31); +#else + word32 lo = (word32)x; + word32 hi = (word32)((word64)x >> 32); + word32 sg = (word32)((sword32)hi >> 31); + word32 m = 0U - fpr_ct_opaque32((word32)n >> 5); + word32 s = (word32)n & 31; + + lo ^= (lo ^ hi) & m; + hi ^= (hi ^ sg) & m; + lo = (lo >> s) | ((hi << (31 - s)) << 1); + hi = (word32)((sword32)hi >> s); + return (sword64)(((word64)hi << 32) | lo); +#endif } /* Left-shift a 64-bit unsigned value by n (0..63), constant-time. */ static WC_MAYBE_UNUSED WC_INLINE word64 fpr_ulsh(word64 x, int n) { - x ^= (x ^ (x << 32)) & ((word64)0 - (word64)(n >> 5)); +#ifdef WC_64BIT_CPU + x ^= (x ^ (x << 32)) + & ((word64)0 - (word64)fpr_ct_opaque32((word32)n >> 5)); return x << (n & 31); +#else + word32 lo = (word32)x; + word32 hi = (word32)(x >> 32); + word32 m = 0U - fpr_ct_opaque32((word32)n >> 5); + word32 s = (word32)n & 31; + + hi ^= (hi ^ lo) & m; + lo &= ~m; + hi = (hi << s) | ((lo >> (31 - s)) >> 1); + lo <<= s; + return ((word64)hi << 32) | lo; +#endif } /* Pack a sign s (0/1), unbiased exponent e and mantissa m (2^54 <= m < 2^55, From ca554c4248ac8653e84412393dbc8f2fac05489c Mon Sep 17 00:00:00 2001 From: Daniele Lacamera Date: Fri, 2 Oct 2026 21:20:59 +0200 Subject: [PATCH 3/4] Falcon: keep fpr_mul selects branchless --- wolfcrypt/src/falcon.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/wolfcrypt/src/falcon.c b/wolfcrypt/src/falcon.c index c4209282176..b2e06976c32 100644 --- a/wolfcrypt/src/falcon.c +++ b/wolfcrypt/src/falcon.c @@ -1103,7 +1103,7 @@ fpr fpr_mul(fpr x, fpr y) /* Normalize zu to 2^54..2^55-1; it may be one bit too large. The * conditional right-shift preserves the sticky bit. */ zv = (zu >> 1) | (zu & 1); - w = zu >> 55; + w = fpr_ct_opaque32((word32)(zu >> 55)); zu ^= (zu ^ zv) & ((word64)0 - w); /* Aggregate scaling factor: sum the exponents, remove 2*(1023+52), then @@ -1116,7 +1116,7 @@ fpr fpr_mul(fpr x, fpr y) s = (int)((x ^ y) >> 63); /* Corrective action: if either operand is zero, clamp the mantissa. */ - d = ((ex + 0x7FF) & (ey + 0x7FF)) >> 11; + d = (int)fpr_ct_opaque32((word32)(((ex + 0x7FF) & (ey + 0x7FF)) >> 11)); zu &= (word64)0 - (word64)d; return FPR(s, e, zu); From e4013538d5d7f03976cb4208f1634df719595e36 Mon Sep 17 00:00:00 2001 From: Daniele Lacamera Date: Fri, 2 Oct 2026 21:59:28 +0200 Subject: [PATCH 4/4] Falcon: white-box vectors for the fpr shift helpers --- tests/unit-mcdc/test_falcon_whitebox.c | 56 ++++++++++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/tests/unit-mcdc/test_falcon_whitebox.c b/tests/unit-mcdc/test_falcon_whitebox.c index ac963ca8220..a12e17ebcbe 100644 --- a/tests/unit-mcdc/test_falcon_whitebox.c +++ b/tests/unit-mcdc/test_falcon_whitebox.c @@ -66,6 +66,61 @@ static int wb_fail = 0; #if defined(HAVE_FALCON) +/* ------------------------------------------------------------------ * + * fpr_ursh / fpr_irsh / fpr_ulsh: compared against plain 64-bit shifts + * at the 32-bit word boundaries (0, 1, 31, 32, 33, 63) on edge values, + * then at every count 0..63 on pseudo-random values. Covers both the + * WC_64BIT_CPU and the 32-bit halves implementation. + * ------------------------------------------------------------------ */ +static int wb_fpr_shift_check(word64 x, int n) +{ + sword64 xs = (sword64)x; + + return (fpr_ursh(x, n) == (x >> n)) && (fpr_ulsh(x, n) == (x << n)) && + (fpr_irsh(xs, n) == (xs >> n)); +} + +static void wb_fpr_shifts(void) +{ + static const word64 vals[] = { + 0x0000000000000000ULL, 0x0000000000000001ULL, + 0xFFFFFFFFFFFFFFFFULL, 0x8000000000000000ULL, + 0x7FFFFFFFFFFFFFFFULL, 0x0000000080000000ULL, + 0x00000000FFFFFFFFULL, 0xFFFFFFFF00000000ULL, + 0x8000000080000000ULL, 0x7FFFFFFF7FFFFFFFULL, + 0x80000000FFFFFFFFULL, 0xC000000000000001ULL, + 0x0123456789ABCDEFULL, 0xFEDCBA9876543210ULL + }; + static const int counts[] = { 0, 1, 31, 32, 33, 63 }; + word64 st = 0x9E3779B97F4A7C15ULL; + size_t i, j; + int n, bad = 0; + + for (i = 0; i < sizeof(vals) / sizeof(vals[0]); i++) { + for (j = 0; j < sizeof(counts) / sizeof(counts[0]); j++) { + if (!wb_fpr_shift_check(vals[i], counts[j])) { + bad = 1; + } + } + } + for (i = 0; i < 4096; i++) { + st ^= st << 13; + st ^= st >> 7; + st ^= st << 17; + for (n = 0; n < 64; n++) { + if (!wb_fpr_shift_check(st, n)) { + bad = 1; + } + } + } + if (bad) { + WB_FAIL("fpr shift helpers differ from 64-bit shifts"); + } + else { + WB_OK("fpr_ursh/fpr_irsh/fpr_ulsh match 64-bit shifts for 0..63"); + } +} + /* ------------------------------------------------------------------ * * falcon_comp_encode: for-loop range guard x[u] < -2047 || x[u] > 2047 * both FALSE (in range), left TRUE (x<-2047), right TRUE with left FALSE. @@ -2111,6 +2166,7 @@ int main(void) WB_FAIL("wc_InitRng failed; RNG-dependent paths skipped"); } + wb_fpr_shifts(); wb_comp_encode(); wb_trim_i8(); wb_privkey();