From 829dfe66a7ed3f94a70f037c04653b223e9a4b45 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:43 +0200 Subject: [PATCH 01/11] Reduce ML-DSA small memory heap footprint Small memory signing kept vector w1 as k full polynomials although it is only ever consumed in its encoded form, once by the commit hash and once by MakeHint, which only tests each coefficient for zero. Keep the encoded w1 and build it a polynomial at a time from a single scratch polynomial of w. This also removes the separate w1 encode buffer that was allocated and freed on every rejection attempt. Add WOLFSSL_MLDSA_SIGN_SMALLEST_MEM. It generates matrix A a column at a time so that one polynomial of y is held instead of the whole vector and is transformed once rather than once per row, decomposes w into w0 in place, and regenerates y for the z calculation. Peak signing heap drops by about half against the small memory path and signing is quicker as ML-DSA-87 does 7 forward transforms of y per attempt instead of 56. It cannot be combined with the PRECALC options, and WOLFSSL_MLDSA_SMALL_MEM_ POLY64 has no effect on signing in this mode. Small memory key generation held t as a full vector for one final vector encode. Encode t a polynomial at a time and overlay t and the single decoded s2 polynomial on the s2 vector, which is dead once s2 has been encoded into the private key. Peak signing heap, in bytes: before after smallest ML-DSA-44 16201 13897 9801 ML-DSA-65 21321 16969 11849 ML-DSA-87 27465 21321 14153 Peak key generation heap falls from 14153, 19273 and 25417 bytes to 10057, 13129 and 17225 bytes. Signatures are unchanged. Problems found while making these changes are fixed here too: - mldsa_vec_expand_mask dispatched to AVX2 generators that only handle dimensions 4, 5 and 7, and silently left y untouched for any other dimension. Fall through to the C implementation instead. - WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A greater than k indexed A, w0 and w1 past their allocations, as the loops used the raw macro rather than min(macro, k). - WOLFSSL_MLDSA_SIGN_CHECK_W0 passed two arguments to the three argument mldsa_vec_check_low, so the option never compiled with small memory signing. - WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM did not build alongside small memory signing, as mldsa_vec_ntt_full and mldsa_vec_check_low were unused. - WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM sized z at l polynomials while only ever using one. --- ChangeLog.md | 6 + wolfcrypt/src/wc_mldsa.c | 642 ++++++++++++++++++++++++++++++---- wolfssl/wolfcrypt/dilithium.h | 8 + 3 files changed, 584 insertions(+), 72 deletions(-) diff --git a/ChangeLog.md b/ChangeLog.md index 6bb746f5d5e..b6696c1ec3b 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -1,3 +1,9 @@ +# wolfSSL Release (unreleased) + +## Post-Quantum Cryptography (PQC) + +* Reduced the ML-DSA small memory heap footprint: signing now keeps w1 only in its encoded form and key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` generates matrix A a column at a time so that only one polynomial of y is held and w0 replaces w in place, roughly halving the signing peak. by @Frauschi + # wolfSSL Release 5.9.4 (Sep 25, 2026) Release 5.9.4 has been developed according to wolfSSL's development and QA diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 8fade7cfb5f..047294fe3e6 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -70,6 +70,21 @@ * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A Default: OFF * Compiles signature implementation that uses smaller amounts of memory but * is slower. Allocates matrix A and calculates it upfront. + * WOLFSSL_MLDSA_SIGN_SMALLEST_MEM Default: OFF + * Compiles the smallest memory signature implementation. Implies + * WOLFSSL_MLDSA_SIGN_SMALL_MEM. Matrix A is generated a column at a time so + * that only one polynomial of vector y is ever held, and w0 replaces w in + * place. Vector y is regenerated for the z calculation. + * Cannot be used with WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC or + * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A. + * WOLFSSL_MLDSA_SMALL_MEM_POLY64 has no effect on signing in this mode as a + * 64-bit accumulator would be needed for every row of w. + * Which of the two is quicker depends on the target. Where the C code runs + * this is the quicker of the two, as each polynomial of y is transformed + * once rather than once per row of matrix A. Where the assembly runs it is + * the slower of the two, as generating y a polynomial at a time cannot use + * the generators that produce the whole vector in one pass. Prefer this on + * targets without a vector unit and WOLFSSL_MLDSA_SIGN_SMALL_MEM elsewhere. * WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM Default: OFF * Compiles key generation implementation that uses smaller amounts of memory * but is slower. @@ -202,6 +217,12 @@ #error "PRECALC and PRECALC_A are equivalent to non small mem" #endif #endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM + #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + #error "SMALLEST_MEM keeps nothing pre-calculated" + #endif +#endif #if defined(USE_INTEL_SPEEDUP) static cpuid_flags_t cpuid_flags = WC_CPUID_INITIALIZER; @@ -1214,7 +1235,9 @@ static void mldsa_vec_encode_eta_bits(const sword32* s, byte d, byte eta, } #endif /* !WOLFSSL_MLDSA_NO_MAKE_KEY */ -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || defined(WOLFSSL_MLDSA_CHECK_KEY) +#if !defined(WOLFSSL_MLDSA_NO_SIGN) || defined(WOLFSSL_MLDSA_CHECK_KEY) || \ + (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) #if !defined(WOLFSSL_NO_ML_DSA_44) || !defined(WOLFSSL_NO_ML_DSA_87) /* Decode polynomial with range -2..2. @@ -1350,8 +1373,11 @@ static void mldsa_decode_eta_4_bits(const byte* p, sword32* s) #endif #if defined(WOLFSSL_MLDSA_CHECK_KEY) || \ + (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (defined(WC_MLDSA_CACHE_PRIV_VECTORS) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) || \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM))) /* Decode vector of polynomials with range -ETA..ETA. * @@ -2636,9 +2662,56 @@ WOLFSSL_TEST_VIS void wc_mldsa_encode_w1_32(const sword32* w1, byte* w1e) #endif #endif -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || \ - (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) +/* Decode one polynomial of w1 from its encoded form. + * + * The small memory signing implementations keep w1 only in the encoded form + * that the commit hash needs. MakeHint wants the polynomial back. + * + * @param [in] w1e Encoded polynomial. + * @param [in] gamma2 Low-order rounding range, GAMMA2. + * @param [out] w1 Decoded polynomial. + */ +static void mldsa_decode_w1(const byte* w1e, sword32 gamma2, sword32* w1) +{ + unsigned int j; + +#ifndef WOLFSSL_NO_ML_DSA_44 + if (gamma2 == MLDSA_Q_LOW_88) { + /* 6 bits per number. 4 numbers in 3 bytes. */ + for (j = 0; j < MLDSA_N; j += 4) { + w1[j + 0] = (sword32)( w1e[0] & 0x3f); + w1[j + 1] = (sword32)(((w1e[0] >> 6) | (w1e[1] << 2)) & 0x3f); + w1[j + 2] = (sword32)(((w1e[1] >> 4) | (w1e[2] << 4)) & 0x3f); + w1[j + 3] = (sword32)( (w1e[2] >> 2) & 0x3f); + /* Move to next place to decode from. */ + w1e += 3; + } + } + else +#endif +#if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) + if (gamma2 == MLDSA_Q_LOW_32) { + /* 4 bits per number. 2 numbers in 1 byte. */ + for (j = 0; j < MLDSA_N; j += 2) { + w1[j + 0] = (sword32)( w1e[0] & 0xf); + w1[j + 1] = (sword32)((w1e[0] >> 4) & 0xf); + /* Move to next place to decode from. */ + w1e++; + } + } + else +#endif + { + } +} +#endif + +#if (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A))) || \ + (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) /* Encode w1 with range of 0..((q-1)/(2*GAMMA2)-1). * * FIPS 204 Section 7.2, Algorithm 28 w1Encode(w1) @@ -5187,22 +5260,35 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, #endif if (IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { + /* Each generator handles one dimension only, so anything else, such + * as the single polynomial the smallest memory signing asks for, + * falls through to the C implementation. */ + byte done = 0; + #ifndef WOLFSSL_NO_ML_DSA_44 if (l == 4) { ret = wc_mldsa_gen_y_4_avx2(y, seed, kappa); + done = 1; } #endif #ifndef WOLFSSL_NO_ML_DSA_65 if (l == 5) { ret = wc_mldsa_gen_y_5_avx2(y, seed, kappa, shake256); + done = 1; } #endif #ifndef WOLFSSL_NO_ML_DSA_87 if (l == 7) { ret = wc_mldsa_gen_y_7_avx2(y, seed, kappa); + done = 1; } #endif RESTORE_VECTOR_REGISTERS(); + + if (!done) { + ret = mldsa_vec_expand_mask_c(shake256, seed, kappa, gamma1_bits, + y, l); + } } else #endif @@ -5706,9 +5792,12 @@ static int mldsa_check_low(const sword32* a, sword32 hi) return (int)(in >> 31); } -#if !defined(WOLFSSL_MLDSA_NO_VERIFY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))) /* Check that the values of the vector are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -5733,7 +5822,6 @@ static int mldsa_vec_check_low_c(const sword32* a, byte l, sword32 hi) return ret; } -#endif /* Check that the values of the vector are in range. * @@ -5765,6 +5853,7 @@ static int mldsa_vec_check_low(const sword32* a, byte l, sword32 hi) return ret; } #endif +#endif /****************************************************************************** * Hint operations @@ -6952,10 +7041,13 @@ static void mldsa_vec_ntt(sword32* r, byte l) #endif #endif -#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC)))) || \ +#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + (!defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) || \ + defined(WC_MLDSA_CACHE_PUB_VECTORS))) || \ + (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A))) || \ (defined(WOLFSSL_MLDSA_SMALL) && \ (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) @@ -8935,8 +9027,10 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) /* Allocate memory for large intermediates. */ if (ret == 0) { - /* s1-l, s2-k, t-k, a-1 */ - allocSz = (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz + + /* s1-l, s2-k, a-1. t is encoded a polynomial at a time, so it and the + * one decoded s2 polynomial share the s2 vector, which is dead once + * s2 has been encoded into the private key. */ + allocSz = (unsigned int)params->s1Sz + params->s2Sz + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE + (unsigned int)MLDSA_POLY_SIZE; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 @@ -8949,8 +9043,8 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) } else { s2 = s1 + params->s1Sz / sizeof(*s1); - t = s2 + params->s2Sz / sizeof(*s2); - h = (byte*)(t + params->s2Sz / sizeof(*t)); + t = s2; + h = (byte*)(s2 + params->s2Sz / sizeof(*s2)); a = (sword32*)(h + MLDSA_REJ_NTT_POLY_H_SIZE); #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 t64 = (sword64*)(a + MLDSA_N); @@ -8997,8 +9091,11 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) byte* t0 = s2p + params->s2EncSz; byte* t1 = key->p + MLDSA_PUB_SEED_SZ; byte aseed[MLDSA_GEN_A_SEED_SZ]; - sword32* s2t = s2; + /* One decoded s2 polynomial, held after t in the dead s2 vector. */ + sword32* s2t = t + MLDSA_N; sword32* tt = t; + const byte* s2pt = s2p; + word32 s2Stride = (word32)params->s2EncSz / params->k; /* Step 9: Move k down to after public seed. */ XMEMCPY(k, k + MLDSA_PRIV_SEED_SZ, MLDSA_K_SZ); @@ -9110,18 +9207,22 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) } #endif mldsa_invntt_full(tt); + /* s2 was overwritten by t, so recover this polynomial from the + * copy already encoded into the private key. */ + mldsa_vec_decode_eta_bits(s2pt, params->eta, s2t, 1); mldsa_add(tt, s2t); /* Make positive for decomposing. */ mldsa_make_pos(tt); - tt += MLDSA_N; - s2t += MLDSA_N; - } + /* Step 6, Step 7, Step 9. Alg 22 Steps 2-4, Alg 24 Steps 8-10. + * Decompose t in t0 and t1 and encode into public and private + * key. */ + mldsa_vec_encode_t0_t1(tt, 1, t0, t1); - /* Step 6, Step 7, Step 9. Alg 22 Steps 2-4, Alg 24 Steps 8-10. - * Decompose t in t0 and t1 and encode into public and private key. - */ - mldsa_vec_encode_t0_t1(t, params->k, t0, t1); + s2pt += s2Stride; + t0 += MLDSA_D * MLDSA_N / 8; + t1 += MLDSA_U * MLDSA_N / 8; + } /* Step 8. Alg 24, Step 1: Hash public key into private key. */ ret = mldsa_shake256(&key->shake, key->p, params->pkSz, tr, MLDSA_TR_SZ); @@ -9654,7 +9755,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } XFREE(y, key->heap, DYNAMIC_TYPE_MLDSA); return ret; -#else +#elif !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) int ret = 0; const wc_MlDsaParams* params = key->params; const byte* pub_seed = key->k; @@ -9673,7 +9774,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sword32* y = NULL; sword32* y_ntt = NULL; sword32* w0 = NULL; - sword32* w1 = NULL; + sword32* w = NULL; sword32* c = NULL; sword32* z = NULL; sword32* ct0 = NULL; @@ -9681,9 +9782,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sword64* t64 = NULL; #endif byte* blocks = NULL; + byte* w1e = NULL; byte priv_rand_seed[MLDSA_Y_SEED_SZ]; byte* h = sig + params->lambda / 4 + params->zEncSz; unsigned int allocSz = 0; + /* Bytes of encoded w1 per polynomial. */ + unsigned int w1Stride = (unsigned int)params->w1EncSz / params->k; #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A byte maxK = (byte)min(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A, params->k); @@ -9711,16 +9815,23 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Allocate memory for large intermediates. */ if (ret == 0) { - /* y-l, w0-k, w1-k, blocks, c-1, z-1, A-1 */ - allocSz = (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz + - (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE + + /* y-l, w0-k, w-1, c-1, z-1, A-1, w1e, blocks. + * w1 is only ever consumed in its encoded form, so just w1e is kept + * and a single polynomial of w is enough to build it row by row. */ + allocSz = (unsigned int)params->s1Sz + params->s2Sz + (unsigned int)MLDSA_POLY_SIZE + (unsigned int)MLDSA_POLY_SIZE + - (unsigned int)MLDSA_POLY_SIZE; + (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)params->w1EncSz + + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE; #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC allocSz += (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz; #elif defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) - allocSz += (unsigned int)maxK * params->l * + /* The pre-calculated rows of A are multiplied as a vector, so w must + * hold that many polynomials. */ + allocSz += (unsigned int)(maxK - 1) * (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)maxK * params->l * (unsigned int)MLDSA_POLY_SIZE; #endif #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 @@ -9735,9 +9846,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, y_check = y; #endif w0 = y + params->s1Sz / sizeof(*y_ntt); - w1 = w0 + params->s2Sz / sizeof(*w0); - blocks = (byte*)(w1 + params->s2Sz / sizeof(*w1)); - c = (sword32*)(blocks + MLDSA_REJ_NTT_POLY_H_SIZE); + w = w0 + params->s2Sz / sizeof(*w0); + #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + c = w + (unsigned int)maxK * MLDSA_N; + #else + c = w + MLDSA_N; + #endif z = c + MLDSA_N; a = z + MLDSA_N; ct0 = z; @@ -9746,24 +9860,36 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, s1 = z; s2 = z; t0 = z; + w1e = (byte*)(a + (1 + maxK * params->l) * MLDSA_N); #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(a + (1 + maxK * params->l) * MLDSA_N); + t64 = (sword64*)(w1e + params->w1EncSz); + blocks = (byte*)(t64 + MLDSA_N); + #else + blocks = w1e + params->w1EncSz; #endif #elif defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) y_ntt = z; s1 = a + MLDSA_N; s2 = s1 + params->s1Sz / sizeof(*s1); t0 = s2 + params->s2Sz / sizeof(*s2); + w1e = (byte*)(t0 + params->s2Sz / sizeof(*t0)); #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(t0 + params->s2Sz / sizeof(*t0)); + t64 = (sword64*)(w1e + params->w1EncSz); + blocks = (byte*)(t64 + MLDSA_N); + #else + blocks = w1e + params->w1EncSz; #endif #else y_ntt = z; s1 = z; s2 = z; t0 = z; + w1e = (byte*)(a + MLDSA_N); #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(a + MLDSA_N); + t64 = (sword64*)(w1e + params->w1EncSz); + blocks = (byte*)(t64 + MLDSA_N); + #else + blocks = w1e + params->w1EncSz; #endif #endif } @@ -9794,23 +9920,19 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Step 11: Start rejection sampling loop */ do { byte aseed[MLDSA_GEN_A_SEED_SZ]; - WC_DECLARE_VAR(w1e, byte, MLDSA_MAX_W1_ENC_SZ, 0); - sword32* w = w1; byte* commit = sig; byte r; byte s; sword32 hi; sword32* wt = w; sword32* w0t = w0; - sword32* w1t = w1; + byte* w1et = w1e; sword32* at = a; #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - w0t += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * MLDSA_N; - w1t += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * MLDSA_N; - wt += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * MLDSA_N; - at += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * params->l * - MLDSA_N; + w0t += (unsigned int)maxK * MLDSA_N; + w1et += (unsigned int)maxK * w1Stride; + at += (unsigned int)maxK * params->l * MLDSA_N; #endif valid = 1; @@ -9828,18 +9950,20 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_ntt_full(y_ntt, params->l); mldsa_matrix_mul(w, a, y_ntt, maxK, params->l); #ifdef WOLFSSL_MLDSA_SMALL - mldsa_vec_red(w, params->k); + mldsa_vec_red(w, maxK); #endif mldsa_vec_invntt_full(w, maxK); /* Step 14, Step 22: Make values positive and decompose. */ mldsa_vec_make_pos(w, maxK); - mldsa_vec_decompose(w, maxK, params->gamma2, w0, w1); + mldsa_vec_decompose(w, maxK, params->gamma2, w0, w); + /* Step 15: Encode the w1 polynomials just computed. */ + mldsa_vec_encode_w1(w, maxK, params->gamma2, w1e); #endif /* Step 5: Create the matrix A from the public seed. */ /* Copy the seed into a buffer that has space for s and r. */ XMEMCPY(aseed, pub_seed, MLDSA_PUB_SEED_SZ); #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - r = WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A; + r = maxK; #else r = 0; #endif @@ -9995,27 +10119,30 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, if (params->gamma2 == MLDSA_Q_LOW_88) { /* For each value of polynomial. */ for (e = 0; e < MLDSA_N; e++) { - /* Decompose value into two vectors. */ - mldsa_decompose_q88(wt[e], &w0t[e], &w1t[e]); + /* Decompose value into two vectors, w1 in place. */ + mldsa_decompose_q88(wt[e], &w0t[e], &wt[e]); } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_88(wt, w1et); } #endif #if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) if (params->gamma2 == MLDSA_Q_LOW_32) { /* For each value of polynomial. */ for (e = 0; e < MLDSA_N; e++) { - /* Decompose value into two vectors. */ - mldsa_decompose_q32(wt[e], &w0t[e], &w1t[e]); + /* Decompose value into two vectors, w1 in place. */ + mldsa_decompose_q32(wt[e], &w0t[e], &wt[e]); } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_32(wt, w1et); } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_vec_check_low(w0t, + valid = mldsa_vec_check_low(w0t, 1, params->gamma2 - params->beta); #endif - wt += MLDSA_N; w0t += MLDSA_N; - w1t += MLDSA_N; + w1et += w1Stride; } if ((ret == 0) && valid) { sword32* yt = y; @@ -10024,18 +10151,10 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #endif byte* ze = sig + params->lambda / 4; - /* Step 15: Encode w1. */ - WC_ALLOC_VAR_EX(w1e, byte, MLDSA_MAX_W1_ENC_SZ, - key->heap, DYNAMIC_TYPE_MLDSA, ret=MEMORY_E); - if (WC_VAR_OK(w1e)) { - mldsa_vec_encode_w1(w1, params->k, params->gamma2, - w1e); - /* Step 15: Hash mu and encoded w1. - * Step 32: Hash is stored in signature. */ - ret = mldsa_hash256(&key->shake, mu, MLDSA_MU_SZ, - w1e, params->w1EncSz, commit, params->lambda / 4); - } - WC_FREE_VAR_EX(w1e, key->heap, DYNAMIC_TYPE_MLDSA); + /* Step 15: Hash mu and encoded w1. + * Step 32: Hash is stored in signature. */ + ret = mldsa_hash256(&key->shake, mu, MLDSA_MU_SZ, + w1e, params->w1EncSz, commit, params->lambda / 4); if (ret == 0) { /* Step 17: Compute c from first 256 bits of commit. */ ret = mldsa_sample_in_ball_ex(params->level, @@ -10110,7 +10229,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sword32* cs2 = ct0; byte idx = 0; w0t = w0; - w1t = w1; + w1et = w1e; for (r = 0; valid && (r < params->k); r++) { #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC @@ -10163,13 +10282,17 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_add(ct0, w0t); mldsa_poly_red(ct0); + /* w1 is only kept encoded, so recover the polynomial + * the hint needs. */ + mldsa_decode_w1(w1et, params->gamma2, w); + /* Step 26, 27: Make hint from ct0 and w1 and check * number of hints is valid. * Step 32: h is encoded into signature. */ #ifndef WOLFSSL_NO_ML_DSA_44 if (params->gamma2 == MLDSA_Q_LOW_88) { - valid = (mldsa_make_hint_88(ct0, w1t, h, + valid = (mldsa_make_hint_88(ct0, w, h, &idx) == 0); /* Alg 14, Step 10: Store count of hints for * polynomial at end of list. */ @@ -10179,7 +10302,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #if !defined(WOLFSSL_NO_ML_DSA_65) || \ !defined(WOLFSSL_NO_ML_DSA_87) if (params->gamma2 == MLDSA_Q_LOW_32) { - valid = (mldsa_make_hint_32(ct0, w1t, + valid = (mldsa_make_hint_32(ct0, w, params->omega, h, &idx) == 0); /* Alg 14, Step 10: Store count of hints for * polynomial at end of list. */ @@ -10192,7 +10315,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, t0pt += MLDSA_D * MLDSA_N / 8; #endif w0t += MLDSA_N; - w1t += MLDSA_N; + w1et += w1Stride; } /* Set remaining hints to zero. */ XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); @@ -10222,6 +10345,373 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } XFREE(y, key->heap, DYNAMIC_TYPE_MLDSA); return ret; +#else + int ret = 0; + const wc_MlDsaParams* params = key->params; + const byte* pub_seed = key->k; + const byte* k = pub_seed + MLDSA_PUB_SEED_SZ; + const byte* tr = k + MLDSA_K_SZ; + const byte* s1p = tr + MLDSA_TR_SZ; + const byte* s2p = s1p + params->s1EncSz; + const byte* t0p = s2p + params->s2EncSz; + const byte* mu = seedMu + MLDSA_RND_SZ; + sword32* w = NULL; + sword32* y = NULL; + sword32* a = NULL; + sword32* c = NULL; + sword32* z = NULL; + byte* w1e = NULL; + byte* blocks = NULL; + byte priv_rand_seed[MLDSA_Y_SEED_SZ]; + byte* h = sig + params->lambda / 4 + params->zEncSz; + unsigned int allocSz = 0; + /* Bytes of encoded w1 per polynomial. */ + unsigned int w1Stride = (unsigned int)params->w1EncSz / params->k; + /* Bytes of encoded s1 or s2 per polynomial. */ + unsigned int sStride = (unsigned int)params->s2EncSz / params->k; +#ifdef WC_MLDSA_FAULT_HARDEN + sword32* w_check; +#endif + + /* priv_rand_seed will hold the secret signing seed (rho'') derived below; + * baseline-zero and register it up front (single-exit function) so any + * later exit before the ForceZero is covered. */ +#ifdef WOLFSSL_CHECK_MEM_ZERO + XMEMSET(priv_rand_seed, 0, sizeof(priv_rand_seed)); + wc_MemZero_Add("mldsa sign priv_rand_seed", priv_rand_seed, + sizeof(priv_rand_seed)); +#endif + /* Check the signature buffer isn't too small. */ + if (*sigLen < params->sigSz) { + ret = BUFFER_E; + } + if (ret == 0) { + /* Return the size of the signature. */ + *sigLen = params->sigSz; + } + + /* Allocate memory for large intermediates. */ + if (ret == 0) { + /* w-k, y-1, a-1, c-1, z-1, w1e, blocks. + * w0 replaces w in place and w1 is only kept encoded, so w is the + * only vector held. */ + allocSz = (unsigned int)params->s2Sz + + 4U * (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)params->w1EncSz + + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE; + w = (sword32*)XMALLOC(allocSz, key->heap, DYNAMIC_TYPE_MLDSA); + if (w == NULL) { + ret = MEMORY_E; + } + else { + #ifdef WC_MLDSA_FAULT_HARDEN + w_check = w; + #endif + y = w + params->s2Sz / sizeof(*w); + a = y + MLDSA_N; + c = a + MLDSA_N; + z = c + MLDSA_N; + w1e = (byte*)(z + MLDSA_N); + blocks = w1e + params->w1EncSz; + } + } + + if (ret == 0) { + /* Step 9: Compute private random using hash. */ + ret = mldsa_hash256(&key->shake, k, MLDSA_K_SZ, seedMu, + MLDSA_RND_SZ + MLDSA_MU_SZ, priv_rand_seed, + MLDSA_PRIV_RAND_SEED_SZ); + } + if (ret == 0) { + word16 kappa = 0; + int valid; + + /* Step 11: Start rejection sampling loop */ + do { + byte aseed[MLDSA_GEN_A_SEED_SZ]; + byte* commit = sig; + byte* w1et = w1e; + const byte* sp; + sword32* wt = w; + unsigned int e; + sword32 hi; + byte r; + byte s; + + valid = 1; + /* Copy the seed into a buffer that has space for s and r. */ + XMEMCPY(aseed, pub_seed, MLDSA_PUB_SEED_SZ); + + /* Alg 26. Step 2: Loop over second dimension of matrix. Working + * down the columns keeps one polynomial of y and transforms it + * once rather than once per row. */ + for (s = 0; (ret == 0) && valid && (s < params->l); s++) { + #ifdef WC_MLDSA_FAULT_HARDEN + if (w_check != w) { + valid = 0; + ret = BAD_COND_E; + break; + } + #endif + /* Step 12: Compute polynomial of y from seed and kappa. */ + ret = mldsa_vec_expand_mask(&key->shake, priv_rand_seed, + (word16)(kappa + s), params->gamma1_bits, y, 1, key->heap); + if (ret != 0) { + break; + } + #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y + valid = mldsa_vec_check_low(y, 1, + ((sword32)1 << params->gamma1_bits) - params->beta); + if (!valid) { + break; + } + #endif + /* Step 13: NTT(y) */ + mldsa_ntt_full(y); + + /* Put s into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 0] = s; + /* Alg 26. Step 1: Loop over first dimension of matrix. */ + for (r = 0; r < params->k; r++) { + /* Put r/i into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 1] = r; + /* Alg 26. Step 3: Create polynomial from hashing seed. */ + ret = mldsa_rej_ntt_poly_ex(&key->shake, aseed, a, blocks); + if (ret != 0) { + break; + } + wt = w + (unsigned int)r * MLDSA_N; + /* Step 13: A o NTT(y), accumulated down the column. */ + if (s == 0) { + #ifdef WOLFSSL_MLDSA_SMALL + for (e = 0; e < MLDSA_N; e++) { + wt[e] = mldsa_mont_red((sword64)a[e] * y[e]); + } + #else + for (e = 0; e < MLDSA_N; e += 8) { + wt[e+0] = mldsa_mont_red((sword64)a[e+0] * y[e+0]); + wt[e+1] = mldsa_mont_red((sword64)a[e+1] * y[e+1]); + wt[e+2] = mldsa_mont_red((sword64)a[e+2] * y[e+2]); + wt[e+3] = mldsa_mont_red((sword64)a[e+3] * y[e+3]); + wt[e+4] = mldsa_mont_red((sword64)a[e+4] * y[e+4]); + wt[e+5] = mldsa_mont_red((sword64)a[e+5] * y[e+5]); + wt[e+6] = mldsa_mont_red((sword64)a[e+6] * y[e+6]); + wt[e+7] = mldsa_mont_red((sword64)a[e+7] * y[e+7]); + } + #endif + } + else { + #ifdef WOLFSSL_MLDSA_SMALL + for (e = 0; e < MLDSA_N; e++) { + wt[e] += mldsa_mont_red((sword64)a[e] * y[e]); + } + #else + for (e = 0; e < MLDSA_N; e += 8) { + wt[e+0] += mldsa_mont_red((sword64)a[e+0] * y[e+0]); + wt[e+1] += mldsa_mont_red((sword64)a[e+1] * y[e+1]); + wt[e+2] += mldsa_mont_red((sword64)a[e+2] * y[e+2]); + wt[e+3] += mldsa_mont_red((sword64)a[e+3] * y[e+3]); + wt[e+4] += mldsa_mont_red((sword64)a[e+4] * y[e+4]); + wt[e+5] += mldsa_mont_red((sword64)a[e+5] * y[e+5]); + wt[e+6] += mldsa_mont_red((sword64)a[e+6] * y[e+6]); + wt[e+7] += mldsa_mont_red((sword64)a[e+7] * y[e+7]); + } + #endif + } + } + } + + wt = w; + for (r = 0; (ret == 0) && valid && (r < params->k); r++) { + /* Step 13: w = NTT-1(A o NTT(y)) */ + mldsa_invntt_full(wt); + /* Step 14, Step 22: Make values positive and decompose. */ + mldsa_make_pos(wt); + #ifndef WOLFSSL_NO_ML_DSA_44 + if (params->gamma2 == MLDSA_Q_LOW_88) { + /* For each value of polynomial. */ + for (e = 0; e < MLDSA_N; e++) { + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q88(wt[e], &wt[e], &a[e]); + } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_88(a, w1et); + } + #endif + #if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) + if (params->gamma2 == MLDSA_Q_LOW_32) { + /* For each value of polynomial. */ + for (e = 0; e < MLDSA_N; e++) { + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q32(wt[e], &wt[e], &a[e]); + } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_32(a, w1et); + } + #endif + #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 + valid = mldsa_vec_check_low(wt, 1, + params->gamma2 - params->beta); + #endif + wt += MLDSA_N; + w1et += w1Stride; + } + + if ((ret == 0) && valid) { + /* Step 15: Hash mu and encoded w1. + * Step 32: Hash is stored in signature. */ + ret = mldsa_hash256(&key->shake, mu, MLDSA_MU_SZ, w1e, + params->w1EncSz, commit, params->lambda / 4); + } + if ((ret == 0) && valid) { + /* Step 17: Compute c from first 256 bits of commit. */ + ret = mldsa_sample_in_ball_ex(params->level, &key->shake, + commit, params->lambda / 4, params->tau, c, blocks); + } + if ((ret == 0) && valid) { + /* Step 18: NTT(c). */ + mldsa_ntt_small(c); + } + + if ((ret == 0) && valid) { + byte* ze = sig + params->lambda / 4; + + sp = s1p; + hi = ((sword32)1 << params->gamma1_bits) - params->beta; + for (s = 0; (ret == 0) && valid && (s < params->l); s++) { + /* Vector y is not kept, so regenerate this polynomial. */ + ret = mldsa_vec_expand_mask(&key->shake, priv_rand_seed, + (word16)(kappa + s), params->gamma1_bits, y, 1, + key->heap); + if (ret != 0) { + break; + } + mldsa_vec_decode_eta_bits(sp, params->eta, a, 1); + mldsa_ntt_small(a); + /* Step 19: cs1 = NTT-1(c o s1) */ + mldsa_mul(z, c, a); + mldsa_invntt(z); + /* Step 21: z = y + cs1 */ + mldsa_add(z, y); + mldsa_poly_red(z); + /* Step 23: Check z has low enough values. */ + valid = mldsa_check_low(z, hi); + if (valid) { + /* Step 32: Encode z into signature. + * Commit (c) and h already encoded into signature. */ + #if !defined(WOLFSSL_NO_ML_DSA_44) + if (params->gamma1_bits == MLDSA_GAMMA1_BITS_17) { + mldsa_encode_gamma1_17_bits(z, ze); + /* Move to next place to encode to. */ + ze += MLDSA_GAMMA1_17_ENC_BITS / 2 * MLDSA_N / 4; + } + #endif + #if !defined(WOLFSSL_NO_ML_DSA_65) || \ + !defined(WOLFSSL_NO_ML_DSA_87) + if (params->gamma1_bits == MLDSA_GAMMA1_BITS_19) { + mldsa_encode_gamma1_19_bits(z, ze); + /* Move to next place to encode to. */ + ze += MLDSA_GAMMA1_19_ENC_BITS / 2 * MLDSA_N / 4; + } + #endif + } + sp += sStride; + } + } + if ((ret == 0) && valid) { + const byte* t0pt = t0p; + byte idx = 0; + + sp = s2p; + wt = w; + w1et = w1e; + for (r = 0; (ret == 0) && valid && (r < params->k); r++) { + mldsa_vec_decode_eta_bits(sp, params->eta, a, 1); + mldsa_ntt_small(a); + /* Step 20: cs2 = NTT-1(c o s2) */ + mldsa_mul(z, c, a); + mldsa_invntt(z); + /* Step 22: w0 - cs2 */ + mldsa_sub(wt, z); + mldsa_poly_red(wt); + /* Step 23: Check w0 - cs2 has low enough values. */ + valid = mldsa_check_low(wt, + params->gamma2 - params->beta); + if (valid) { + mldsa_decode_t0(t0pt, a); + mldsa_ntt(a); + /* Step 25: ct0 = NTT-1(c o t0) */ + mldsa_mul(z, c, a); + mldsa_invntt(z); + /* Step 27: Check ct0 has low enough values. */ + valid = mldsa_check_low(z, params->gamma2); + } + if (valid) { + /* Step 26: ct0 = ct0 + w0 */ + mldsa_add(z, wt); + mldsa_poly_red(z); + + /* w1 is only kept encoded, so recover the polynomial + * the hint needs. */ + mldsa_decode_w1(w1et, params->gamma2, a); + + /* Step 26, 27: Make hint from ct0 and w1 and check + * number of hints is valid. + * Step 32: h is encoded into signature. + */ + #ifndef WOLFSSL_NO_ML_DSA_44 + if (params->gamma2 == MLDSA_Q_LOW_88) { + valid = (mldsa_make_hint_88(z, a, h, &idx) == 0); + /* Alg 14, Step 10: Store count of hints for + * polynomial at end of list. */ + h[PARAMS_ML_DSA_44_OMEGA + r] = idx; + } + #endif + #if !defined(WOLFSSL_NO_ML_DSA_65) || \ + !defined(WOLFSSL_NO_ML_DSA_87) + if (params->gamma2 == MLDSA_Q_LOW_32) { + valid = (mldsa_make_hint_32(z, a, params->omega, + h, &idx) == 0); + /* Alg 14, Step 10: Store count of hints for + * polynomial at end of list. */ + h[params->omega + r] = idx; + } + #endif + } + + sp += sStride; + t0pt += MLDSA_D * MLDSA_N / 8; + wt += MLDSA_N; + w1et += w1Stride; + } + /* Set remaining hints to zero. */ + XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); + } + + if (!valid) { + /* Too many attempts - something wrong with implementation. */ + if ((kappa > (word16)(kappa + params->l))) { + ret = BAD_COND_E; + } + + /* Step 30: increment value to append to seed to unique value. + */ + kappa = (word16)(kappa + params->l); + } + } + /* Step 11: Check we have a valid signature. */ + while ((ret == 0) && (!valid)); + } + + ForceZero(priv_rand_seed, sizeof(priv_rand_seed)); +#ifdef WOLFSSL_CHECK_MEM_ZERO + wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); +#endif + if (w != NULL) { + ForceZero(w, allocSz); + } + XFREE(w, key->heap, DYNAMIC_TYPE_MLDSA); + return ret; #endif } @@ -10936,6 +11426,14 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, #ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM /* Bytes of encoded z per polynomial - z is streamed one poly at a time. */ word32 zStride = (word32)(MLDSA_N / 8) * (word32)(params->gamma1_bits + 1); +#endif +#ifndef WOLFSSL_MLDSA_VERIFY_NO_MALLOC +#ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM + /* Only one polynomial of z is held at a time in this mode. */ + unsigned int zSz = (unsigned int)MLDSA_POLY_SIZE; +#else + unsigned int zSz = (unsigned int)params->s1Sz; +#endif #endif /* Ensure the signature is the right size for the parameters. */ @@ -10953,7 +11451,7 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, /* z, c, w, t1, w1e. */ unsigned int allocSz; - allocSz = (unsigned int)params->s1Sz + params->w1EncSz + + allocSz = zSz + params->w1EncSz + 3U * (unsigned int)MLDSA_POLY_SIZE + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 @@ -10965,7 +11463,7 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, } else { XMEMSET(z, 0, allocSz); - c = z + params->s1Sz / sizeof(*t1); + c = z + zSz / sizeof(*t1); w = c + MLDSA_N; t1 = w + MLDSA_N; block = (byte*)(t1 + MLDSA_N); diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index 7d39cf378f9..a6ce87f3945 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -176,6 +176,14 @@ #define WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC #endif #endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM + /* Smallest signing RAM: on top of the small-mem path, generate matrix A a + * column at a time so only one polynomial of y is held, decompose w into + * w0 in place and keep w1 encoded. Vector y is regenerated for z. */ + #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + #endif +#endif #ifdef WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC_A #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A #define WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A \ From 410dfbf6cea200a0ac66f6470bdc0d3fd6af709b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:43 +0200 Subject: [PATCH 02/11] Fix ML-DSA cached vector build and test failures Two combinations of the ML-DSA cache options did not work. WC_MLDSA_CACHE_PUB_VECTORS declares a cached t1 vector in wc_MlDsaKey and WOLFSSL_MLDSA_VERIFY_NO_MALLOC declares a pinned t1 scratch polynomial in the same structure, so enabling both failed to compile with a duplicate member. Rename the pinned scratch to vt1. The matrix A cache regression test asserts that signing populates key->a. That holds for the full signing implementation, which is what the test was written for, but the small memory implementations stream matrix A rather than caching it, so they leave key->a as NULL and the test failed. Run the test only when small memory signing is off. WC_MLDSA_CACHE_PRIV_VECTORS enables WC_MLDSA_CACHE_MATRIX_A, so it saw the same failure. --- ChangeLog.md | 1 + wolfcrypt/src/wc_mldsa.c | 2 +- wolfcrypt/test/test.c | 10 ++++++++-- wolfssl/wolfcrypt/wc_mldsa.h | 4 +++- 4 files changed, 13 insertions(+), 4 deletions(-) diff --git a/ChangeLog.md b/ChangeLog.md index b6696c1ec3b..6e89d93e071 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -2,6 +2,7 @@ ## Post-Quantum Cryptography (PQC) +* Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, and stopped the matrix A cache regression test running against the small memory signing implementations, which stream matrix A rather than caching it. by @Frauschi * Reduced the ML-DSA small memory heap footprint: signing now keeps w1 only in its encoded form and key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` generates matrix A a column at a time so that only one polynomial of y is held and w0 replaces w in place, roughly halving the signing peak. by @Frauschi # wolfSSL Release 5.9.4 (Sep 25, 2026) diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 047294fe3e6..6c52ba263ef 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -11479,7 +11479,7 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, z = key->z; c = key->c; w = key->w; - t1 = key->t1; + t1 = key->vt1; w1e = key->w1e; aBuf = t1; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 diff --git a/wolfcrypt/test/test.c b/wolfcrypt/test/test.c index 09cd3d50e30..52888cb58f6 100644 --- a/wolfcrypt/test/test.c +++ b/wolfcrypt/test/test.c @@ -66475,6 +66475,7 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_NO_VERIFY) @@ -66490,6 +66491,10 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) * key after make_key, then signing. The post-condition asserts that key->a * was populated (proving the allocation made it into the key, not the local) * and that signing produces a verifiable signature. + * + * The small memory signing implementations stream matrix A rather than + * caching it, so they never populate key->a and the post-condition does not + * apply to them. */ static wc_test_ret_t mldsa_sign_cache_alloc_test(int param, WC_RNG* rng) { @@ -66570,8 +66575,8 @@ static wc_test_ret_t mldsa_sign_cache_alloc_test(int param, WC_RNG* rng) return ret; } #endif /* WC_MLDSA_CACHE_MATRIX_A && !WC_MLDSA_FIXED_ARRAY && - * !WOLFSSL_MLDSA_NO_MAKE_KEY && !WOLFSSL_MLDSA_NO_SIGN && - * !WOLFSSL_MLDSA_NO_VERIFY */ + * !WOLFSSL_MLDSA_SIGN_SMALL_MEM && !WOLFSSL_MLDSA_NO_MAKE_KEY && + * !WOLFSSL_MLDSA_NO_SIGN && !WOLFSSL_MLDSA_NO_VERIFY */ #if (defined(WOLFSSL_MLDSA_PRIVATE_KEY) && \ @@ -67562,6 +67567,7 @@ WOLFSSL_TEST_SUBROUTINE wc_test_ret_t mldsa_test(void) #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_NO_VERIFY) diff --git a/wolfssl/wolfcrypt/wc_mldsa.h b/wolfssl/wolfcrypt/wc_mldsa.h index 34ba33efc26..89a50fbd760 100644 --- a/wolfssl/wolfcrypt/wc_mldsa.h +++ b/wolfssl/wolfcrypt/wc_mldsa.h @@ -684,7 +684,9 @@ struct wc_MlDsaKey { #endif sword32 c[MLDSA_N]; sword32 w[MLDSA_N]; - sword32 t1[MLDSA_N]; + /* One t1 polynomial, also used for a polynomial of A. Not named t1 as + * WC_MLDSA_CACHE_PUB_VECTORS already has a member of that name. */ + sword32 vt1[MLDSA_N]; byte w1e[MLDSA_MAX_W1_ENC_SZ]; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 sword64 t64[MLDSA_N]; From b191e7b7ded415051bdd33f55d80f08672d1d0a1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:43 +0200 Subject: [PATCH 03/11] Separate ML-DSA smallest memory verify from pinned buffers WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM defined WOLFSSL_MLDSA_VERIFY_NO_MALLOC, so streaming vector z a polynomial at a time was only available with the verify buffers pinned against the key for the life of the key. The two are independent, so let them be selected independently. The pinned buffers are still available by defining WOLFSSL_MLDSA_VERIFY_NO_MALLOC as well. Peak verify heap with allocated buffers, in bytes: small mem smallest mem ML-DSA-44 8777 5705 ML-DSA-65 9801 5705 ML-DSA-87 12105 5961 Structure sizes when the buffers are pinned instead are 20176 bytes for small memory verify and 14032 bytes for smallest memory verify. Also document WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM in the option list, which it was missing from. --- ChangeLog.md | 1 + wolfcrypt/src/wc_mldsa.c | 5 +++++ wolfssl/wolfcrypt/dilithium.h | 11 +++++------ 3 files changed, 11 insertions(+), 6 deletions(-) diff --git a/ChangeLog.md b/ChangeLog.md index 6e89d93e071..c8cf2833ccc 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -2,6 +2,7 @@ ## Post-Quantum Cryptography (PQC) +* `WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM` no longer forces `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, so streaming vector z can be used with allocated buffers instead of buffers pinned against the key. by @Frauschi * Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, and stopped the matrix A cache regression test running against the small memory signing implementations, which stream matrix A rather than caching it. by @Frauschi * Reduced the ML-DSA small memory heap footprint: signing now keeps w1 only in its encoded form and key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` generates matrix A a column at a time so that only one polynomial of y is held and w0 replaces w in place, roughly halving the signing peak. by @Frauschi diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 6c52ba263ef..17331540e4f 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -47,6 +47,11 @@ * Compiles in only the verification and public key operations. * WOLFSSL_MLDSA_VERIFY_SMALL_MEM Default: OFF * Compiles verification implementation that uses smaller amounts of memory. + * WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM Default: OFF + * Implies WOLFSSL_MLDSA_VERIFY_SMALL_MEM. Decodes and transforms vector z a + * polynomial at a time instead of holding the whole vector, k times rather + * than once, so unlike the signing option it trades time for memory. + * Add WOLFSSL_MLDSA_VERIFY_NO_MALLOC to pin the buffers against the key. * WOLFSSL_MLDSA_VERIFY_NO_MALLOC Default: OFF * Only works with WOLFSSL_MLDSA_VERIFY_SMALL_MEM. * Don't allocate memory with XMALLOC. Memory is pinned against key. diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index a6ce87f3945..1a6c95712a8 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -151,15 +151,14 @@ #endif #endif #ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM - /* Smallest verify RAM: on top of the small-mem, no-malloc path, stream the - * signature's z vector one polynomial at a time instead of pinning the whole - * l-vector (~6 KB for ML-DSA-87) at the cost of a per-row z decode+NTT. */ + /* Smallest verify RAM: on top of the small-mem path, stream the + * signature's z vector one polynomial at a time instead of holding the + * whole l-vector (~6 KB for ML-DSA-87) at the cost of a per-row z + * decode+NTT. Combine with WOLFSSL_MLDSA_VERIFY_NO_MALLOC to pin the + * buffers against the key; on its own the buffers are still allocated. */ #ifndef WOLFSSL_MLDSA_VERIFY_SMALL_MEM #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM #endif - #ifndef WOLFSSL_MLDSA_VERIFY_NO_MALLOC - #define WOLFSSL_MLDSA_VERIFY_NO_MALLOC - #endif #endif #ifdef WOLFSSL_DILITHIUM_MAKE_KEY_SMALL_MEM #ifndef WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM From 178e2f8de3d01b9d300e5d2d17d1c2badba7cf57 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:45 +0200 Subject: [PATCH 04/11] Transform each polynomial of y once when signing with small memory Small memory signing walked matrix A a row at a time and copied and transformed the polynomial of y it needed for every element of the row. Each polynomial of y was therefore transformed k times with an identical result: 156 transforms per ML-DSA-65 signature where 31 are needed. Walk the matrix a column at a time instead. Every row of the matrix uses the same column of y, so the transform is done once and the products are accumulated down the column into w0, which is already a vector of k polynomials. Inverting, decomposing and encoding then run as a second pass over the rows, with w0 replaced in place and the scratch polynomial holding w1. Vector y stays resident, so unlike the smallest memory implementation this does not have to regenerate it, and it keeps generating y as a whole vector so the assembly generators are still used. Memory is unchanged: w0 was already held for the hints and the row accumulator becomes the scratch the transform runs in. WOLFSSL_MLDSA_SMALL_MEM_POLY64 no longer applies to signing as a 64-bit accumulator would now be needed for every row of w, which also returns the 2KB it was allocating. Signatures are unchanged. Signing is 12 to 29 percent quicker with the C code and 6 to 12 percent quicker with AVX-512, which makes it faster than the smallest memory implementation on every target rather than only where the assembly runs. --- ChangeLog.md | 1 + wolfcrypt/src/wc_mldsa.c | 167 ++++++++++++--------------------------- 2 files changed, 53 insertions(+), 115 deletions(-) diff --git a/ChangeLog.md b/ChangeLog.md index c8cf2833ccc..8270402c84b 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -5,6 +5,7 @@ * `WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM` no longer forces `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, so streaming vector z can be used with allocated buffers instead of buffers pinned against the key. by @Frauschi * Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, and stopped the matrix A cache regression test running against the small memory signing implementations, which stream matrix A rather than caching it. by @Frauschi * Reduced the ML-DSA small memory heap footprint: signing now keeps w1 only in its encoded form and key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` generates matrix A a column at a time so that only one polynomial of y is held and w0 replaces w in place, roughly halving the signing peak. by @Frauschi +* Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of vector y is transformed once rather than once per row of A. Signatures are unchanged, memory is unchanged, and signing is 12 to 29 percent quicker with the C code and 6 to 12 percent quicker with AVX-512, making `WOLFSSL_MLDSA_SIGN_SMALL_MEM` faster than `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` on every target. by @Frauschi # wolfSSL Release 5.9.4 (Sep 25, 2026) diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 17331540e4f..cea0448635d 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -67,7 +67,9 @@ * Cannot be used with WOLFSSL_MLDSA_ASSIGN_KEY. * WOLFSSL_MLDSA_SIGN_SMALL_MEM Default: OFF * Compiles signature implementation that uses smaller amounts of memory but - * is considerably slower. + * is considerably slower. Matrix A is streamed a polynomial at a time rather + * than held, and is walked a column at a time so that each polynomial of + * vector y is transformed once rather than once per row of A. * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC Default: OFF * Compiles signature implementation that uses smaller amounts of memory but * is considerably slower. Allocates vectors and decodes private key data @@ -84,18 +86,19 @@ * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A. * WOLFSSL_MLDSA_SMALL_MEM_POLY64 has no effect on signing in this mode as a * 64-bit accumulator would be needed for every row of w. - * Which of the two is quicker depends on the target. Where the C code runs - * this is the quicker of the two, as each polynomial of y is transformed - * once rather than once per row of matrix A. Where the assembly runs it is - * the slower of the two, as generating y a polynomial at a time cannot use - * the generators that produce the whole vector in one pass. Prefer this on - * targets without a vector unit and WOLFSSL_MLDSA_SIGN_SMALL_MEM elsewhere. + * This is the slower of the two on every target as vector y has to be + * generated twice, and generating it a polynomial at a time cannot use the + * generators that produce the whole vector in one pass. Turn it on only + * when the polynomials of y that WOLFSSL_MLDSA_SIGN_SMALL_MEM keeps are + * more memory than the target has to spare. * WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM Default: OFF * Compiles key generation implementation that uses smaller amounts of memory * but is slower. * WOLFSSL_MLDSA_SMALL_MEM_POLY64 Default: OFF * Compiles the small memory implementations to use a 64-bit polynomial. * Uses 2KB of memory but is slightly quicker (2.75-7%). + * Has no effect on signing, which accumulates a column at a time and would + * need a 64-bit accumulator for every row of w. * * WOLFSSL_MLDSA_ALIGNMENT Default: 8 * Use to indicate whether loading and storing of words needs to be aligned. @@ -9783,9 +9786,6 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sword32* c = NULL; sword32* z = NULL; sword32* ct0 = NULL; -#ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - sword64* t64 = NULL; -#endif byte* blocks = NULL; byte* w1e = NULL; byte priv_rand_seed[MLDSA_Y_SEED_SZ]; @@ -9821,8 +9821,8 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Allocate memory for large intermediates. */ if (ret == 0) { /* y-l, w0-k, w-1, c-1, z-1, A-1, w1e, blocks. - * w1 is only ever consumed in its encoded form, so just w1e is kept - * and a single polynomial of w is enough to build it row by row. */ + * Walking A a column at a time makes w0 the vector of accumulators, + * w1 is only kept encoded, and w is the scratch for the hints. */ allocSz = (unsigned int)params->s1Sz + params->s2Sz + (unsigned int)MLDSA_POLY_SIZE + (unsigned int)MLDSA_POLY_SIZE + @@ -9838,9 +9838,6 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, allocSz += (unsigned int)(maxK - 1) * (unsigned int)MLDSA_POLY_SIZE + (unsigned int)maxK * params->l * (unsigned int)MLDSA_POLY_SIZE; - #endif - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - allocSz += (unsigned int)MLDSA_POLY_SIZE * 2U; #endif y = (sword32*)XMALLOC(allocSz, key->heap, DYNAMIC_TYPE_MLDSA); if (y == NULL) { @@ -9866,36 +9863,21 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, s2 = z; t0 = z; w1e = (byte*)(a + (1 + maxK * params->l) * MLDSA_N); - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(w1e + params->w1EncSz); - blocks = (byte*)(t64 + MLDSA_N); - #else blocks = w1e + params->w1EncSz; - #endif #elif defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) y_ntt = z; s1 = a + MLDSA_N; s2 = s1 + params->s1Sz / sizeof(*s1); t0 = s2 + params->s2Sz / sizeof(*s2); w1e = (byte*)(t0 + params->s2Sz / sizeof(*t0)); - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(w1e + params->w1EncSz); - blocks = (byte*)(t64 + MLDSA_N); - #else blocks = w1e + params->w1EncSz; - #endif #else y_ntt = z; s1 = z; s2 = z; t0 = z; w1e = (byte*)(a + MLDSA_N); - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(w1e + params->w1EncSz); - blocks = (byte*)(t64 + MLDSA_N); - #else blocks = w1e + params->w1EncSz; - #endif #endif } } @@ -9928,11 +9910,13 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, byte* commit = sig; byte r; byte s; + byte rStart; sword32 hi; sword32* wt = w; sword32* w0t = w0; byte* w1et = w1e; sword32* at = a; + sword32* y_ntt_t; #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A w0t += (unsigned int)maxK * MLDSA_N; @@ -9968,22 +9952,18 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Copy the seed into a buffer that has space for s and r. */ XMEMCPY(aseed, pub_seed, MLDSA_PUB_SEED_SZ); #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - r = maxK; + rStart = (byte)maxK; + y_ntt_t = z; #else - r = 0; + rStart = 0; + y_ntt_t = y_ntt; #endif - /* Alg 26. Step 1: Loop over first dimension of matrix. */ - for (; (ret == 0) && valid && (r < params->k); r++) { + /* Alg 26. Step 2: Loop over second dimension of matrix. + * The matrix is walked a column at a time so that each polynomial + * of y is transformed once rather than once per row. */ + for (s = 0; (ret == 0) && valid && (s < params->l); s++) { unsigned int e; - sword32* yt = y; - #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - sword32* y_ntt_t = z; - #else - sword32* y_ntt_t = y_ntt; - #endif - #ifdef WC_MLDSA_FAULT_HARDEN - sword32* yt_check = yt; - #endif + #ifdef WC_MLDSA_FAULT_HARDEN if (y_check != y) { valid = 0; @@ -9991,29 +9971,25 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, break; } #endif + /* Step 13: NTT(y) for this column of the matrix. */ + XMEMCPY(y_ntt_t, y + (unsigned int)s * MLDSA_N, + MLDSA_POLY_SIZE); + mldsa_ntt_full(y_ntt_t); - /* Put r/i into buffer to be hashed. */ - aseed[MLDSA_PUB_SEED_SZ + 1] = r; - /* Alg 26. Step 2: Loop over second dimension of matrix. */ - for (s = 0; s < params->l; s++) { - /* Put s into buffer to be hashed. */ - aseed[MLDSA_PUB_SEED_SZ + 0] = s; + /* Put s into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 0] = s; + wt = w0t; + /* Alg 26. Step 1: Loop over first dimension of matrix. */ + for (r = rStart; r < params->k; r++) { + /* Put r/i into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 1] = r; /* Alg 26. Step 3: Create polynomial from hashing seed. */ ret = mldsa_rej_ntt_poly_ex(&key->shake, aseed, at, blocks); if (ret != 0) { break; } - XMEMCPY(y_ntt_t, yt, MLDSA_POLY_SIZE); - #ifdef WC_MLDSA_FAULT_HARDEN - if (yt_check + s * MLDSA_N != yt) { - ret = BAD_COND_E; - break; - } - #endif - mldsa_ntt_full(y_ntt_t); - /* Matrix multiply. */ - #ifndef WOLFSSL_MLDSA_SMALL_MEM_POLY64 + /* Step 13: A o NTT(y), accumulated down the column. */ if (s == 0) { #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { @@ -10068,78 +10044,39 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif } - #else - if (s == 0) { - #ifdef WOLFSSL_MLDSA_SMALL - for (e = 0; e < MLDSA_N; e++) { - t64[e] = (sword64)at[e] * y_ntt_t[e]; - } - #else - for (e = 0; e < MLDSA_N; e += 8) { - t64[e+0] = (sword64)at[e+0] * y_ntt_t[e+0]; - t64[e+1] = (sword64)at[e+1] * y_ntt_t[e+1]; - t64[e+2] = (sword64)at[e+2] * y_ntt_t[e+2]; - t64[e+3] = (sword64)at[e+3] * y_ntt_t[e+3]; - t64[e+4] = (sword64)at[e+4] * y_ntt_t[e+4]; - t64[e+5] = (sword64)at[e+5] * y_ntt_t[e+5]; - t64[e+6] = (sword64)at[e+6] * y_ntt_t[e+6]; - t64[e+7] = (sword64)at[e+7] * y_ntt_t[e+7]; - } - #endif - } - else { - #ifdef WOLFSSL_MLDSA_SMALL - for (e = 0; e < MLDSA_N; e++) { - t64[e] += (sword64)at[e] * y_ntt_t[e]; - } - #else - for (e = 0; e < MLDSA_N; e += 8) { - t64[e+0] += (sword64)at[e+0] * y_ntt_t[e+0]; - t64[e+1] += (sword64)at[e+1] * y_ntt_t[e+1]; - t64[e+2] += (sword64)at[e+2] * y_ntt_t[e+2]; - t64[e+3] += (sword64)at[e+3] * y_ntt_t[e+3]; - t64[e+4] += (sword64)at[e+4] * y_ntt_t[e+4]; - t64[e+5] += (sword64)at[e+5] * y_ntt_t[e+5]; - t64[e+6] += (sword64)at[e+6] * y_ntt_t[e+6]; - t64[e+7] += (sword64)at[e+7] * y_ntt_t[e+7]; - } - #endif - } - #endif - /* Next polynomial. */ - yt += MLDSA_N; - } - if (ret != 0) { - break; - } - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - for (e = 0; e < MLDSA_N; e++) { - wt[e] = mldsa_mont_red(t64[e]); + wt += MLDSA_N; } - #endif - mldsa_invntt_full(wt); + } + + /* Steps 13-15: Invert transform, decompose and encode each row of + * w now that every column has been accumulated into it. */ + for (r = rStart; (ret == 0) && valid && (r < params->k); r++) { + unsigned int e; + + /* Step 13: w = NTT-1(A o NTT(y)) */ + mldsa_invntt_full(w0t); /* Step 14, Step 22: Make values positive and decompose. */ - mldsa_make_pos(wt); + mldsa_make_pos(w0t); #ifndef WOLFSSL_NO_ML_DSA_44 if (params->gamma2 == MLDSA_Q_LOW_88) { /* For each value of polynomial. */ for (e = 0; e < MLDSA_N; e++) { - /* Decompose value into two vectors, w1 in place. */ - mldsa_decompose_q88(wt[e], &w0t[e], &wt[e]); + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q88(w0t[e], &w0t[e], &at[e]); } /* Step 15: Encode this polynomial of w1. */ - mldsa_encode_w1_88(wt, w1et); + mldsa_encode_w1_88(at, w1et); } #endif #if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) if (params->gamma2 == MLDSA_Q_LOW_32) { /* For each value of polynomial. */ for (e = 0; e < MLDSA_N; e++) { - /* Decompose value into two vectors, w1 in place. */ - mldsa_decompose_q32(wt[e], &w0t[e], &wt[e]); + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q32(w0t[e], &w0t[e], &at[e]); } /* Step 15: Encode this polynomial of w1. */ - mldsa_encode_w1_32(wt, w1et); + mldsa_encode_w1_32(at, w1et); } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 From 008e483ccc4ab9b9959f830e2bce7c7316aa712f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:46 +0200 Subject: [PATCH 05/11] Record where the ML-DSA 64-bit polynomial option still applies Signing no longer uses a 64-bit accumulator, so the option now does nothing unless key generation or verification is also built for small memory. Say so where the option is documented and in the ChangeLog, as a build that sets it alongside only WOLFSSL_MLDSA_SIGN_SMALL_MEM used to get something for it and now does not. --- ChangeLog.md | 2 +- wolfcrypt/src/wc_mldsa.c | 6 ++++-- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/ChangeLog.md b/ChangeLog.md index 8270402c84b..a3151442649 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -5,7 +5,7 @@ * `WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM` no longer forces `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, so streaming vector z can be used with allocated buffers instead of buffers pinned against the key. by @Frauschi * Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, and stopped the matrix A cache regression test running against the small memory signing implementations, which stream matrix A rather than caching it. by @Frauschi * Reduced the ML-DSA small memory heap footprint: signing now keeps w1 only in its encoded form and key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` generates matrix A a column at a time so that only one polynomial of y is held and w0 replaces w in place, roughly halving the signing peak. by @Frauschi -* Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of vector y is transformed once rather than once per row of A. Signatures are unchanged, memory is unchanged, and signing is 12 to 29 percent quicker with the C code and 6 to 12 percent quicker with AVX-512, making `WOLFSSL_MLDSA_SIGN_SMALL_MEM` faster than `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` on every target. by @Frauschi +* Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of vector y is transformed once rather than once per row of A. Signatures are unchanged, memory is unchanged, and signing is 12 to 29 percent quicker with the C code and 6 to 12 percent quicker with AVX-512, making `WOLFSSL_MLDSA_SIGN_SMALL_MEM` faster than `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` on every target. `WOLFSSL_MLDSA_SMALL_MEM_POLY64` no longer applies to signing, which would need a 64-bit accumulator per row of w, so it now affects only `WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM` and `WOLFSSL_MLDSA_VERIFY_SMALL_MEM` and returns the 2KB it was allocating when signing. by @Frauschi # wolfSSL Release 5.9.4 (Sep 25, 2026) diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index cea0448635d..d763ba68b45 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -97,8 +97,10 @@ * WOLFSSL_MLDSA_SMALL_MEM_POLY64 Default: OFF * Compiles the small memory implementations to use a 64-bit polynomial. * Uses 2KB of memory but is slightly quicker (2.75-7%). - * Has no effect on signing, which accumulates a column at a time and would - * need a 64-bit accumulator for every row of w. + * Only WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM and WOLFSSL_MLDSA_VERIFY_SMALL_MEM + * are affected. Signing accumulates a column at a time and would need a + * 64-bit accumulator for every row of w, so it does not use one, and this + * does nothing at all without one of those two options. * * WOLFSSL_MLDSA_ALIGNMENT Default: 8 * Use to indicate whether loading and storing of words needs to be aligned. From 95e92e7e32343a2b6f6453302e4a4523b22b52d9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:46 +0200 Subject: [PATCH 06/11] Fix ML-DSA PRECALC_A buffer sizing Signing with WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A sized w for maxK polynomials, but two helpers ignore their dimension for ML-DSA-44: the q88 decompose kernels take no dimension argument at all, and mldsa_vec_encode_w1() bounds its q88 loops by PARAMS_ML_DSA_44_K. With PRECALC_A set to 1, decomposing w ran past it into the first polynomial of matrix A, which has to survive every rejection restart, so signing returned success with a signature the verifier rejects. Size w on params->k and honour the caller's dimension in mldsa_vec_encode_w1(). Align the preprocessor guards of the vector helpers with their call sites. The PRECALC_A path drives seven whole-vector helpers whose guards had drifted, so introduce MLDSA_SIGN_VEC_HELPERS and use it consistently, move the PRECALC and PRECALC_A implications into dilithium.h so every translation unit agrees, and match mldsa_rej_ntt_poly() and mldsa_expand_a() to their callers. Reject a PRECALC_A value below one, which underflowed the allocation size. Return NOT_COMPILED_IN from wc_CheckPrivateKey() for ML-DSA when WOLFSSL_MLDSA_CHECK_KEY is not defined, matching what the RSA arm already does. The call to wc_MlDsaKey_CheckKey() was unguarded while the function itself is compiled out, so WOLFSSL_MLDSA_NO_CHECK_KEY builds failed to compile. Reduce w before the inverse transform on both column accumulating paths now that the 64-bit accumulator is gone, and bind the two expansions of y under WC_MLDSA_FAULT_HARDEN. Add functional and -Wconversion CI rows for the two configurations this branch enables but nothing built: SIGN_SMALLEST_MEM, and VERIFY_SMALLEST_MEM with malloc. --- .github/configs/pq-all.json | 14 +- .github/workflows/wolfCrypt-Wconversion.yml | 24 +++ .wolfssl_known_macro_extras | 1 + ChangeLog.md | 15 +- tests/unit-mcdc/test_wc_mldsa_whitebox.c | 7 + wolfcrypt/src/asn.c | 6 + wolfcrypt/src/wc_mldsa.c | 167 +++++++++++++------- wolfcrypt/test/test.c | 16 +- wolfssl/wolfcrypt/dilithium.h | 13 ++ 9 files changed, 192 insertions(+), 71 deletions(-) diff --git a/.github/configs/pq-all.json b/.github/configs/pq-all.json index a250cadbb64..fb4bc7488b2 100644 --- a/.github/configs/pq-all.json +++ b/.github/configs/pq-all.json @@ -228,5 +228,17 @@ {"name": "pkcs7-mldsa-only", "minutes": 0.5, "comment": "PKCS#7 SignedData with ML-DSA as the only signature algorithm (no RSA, no ECC); guards the ML-DSA-only PKCS7 build path", "configure": ["--enable-cryptonly", "--enable-mldsa", - "--enable-pkcs7", "--disable-rsa", "--disable-ecc"]} + "--enable-pkcs7", "--disable-rsa", "--disable-ecc"]}, +{"name": "mldsa-smallest-mem", "minutes": 2, + "comment": "Smallest memory ML-DSA signing and verification; the only functional coverage of WOLFSSL_MLDSA_SIGN_SMALLEST_MEM and of VERIFY_SMALLEST_MEM with malloc", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM"]}, +{"name": "mldsa-smallest-mem-checks", "minutes": 2, + "comment": "Smallest memory ML-DSA signing with the optional y and w0 rejection checks and the small code arms, which are otherwise never compiled", + "configure": ["--enable-cryptonly", "--enable-mldsa=yes,small", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]}, +{"name": "mldsa-44-precalc-a-intelasm", "minutes": 2, + "comment": "ML-DSA-44 with one pre-calculated row of matrix A on the Intel assembly paths; the q88 decompose and w1 encode kernels take no dimension, so this pins that w is sized for a full vector", + "configure": ["--enable-cryptonly", "--enable-mldsa", "--enable-intelasm", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]} ] diff --git a/.github/workflows/wolfCrypt-Wconversion.yml b/.github/workflows/wolfCrypt-Wconversion.yml index 0002f3809c2..832c033c689 100644 --- a/.github/workflows/wolfCrypt-Wconversion.yml +++ b/.github/workflows/wolfCrypt-Wconversion.yml @@ -90,6 +90,30 @@ jobs: "--enable-xmss", "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM -DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC -DWOLFSSL_WC_LMS_SERIALIZE_STATE -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual"], "check": false}, + {"name": "smallest-mem", "minutes": 1, + "configure": ["--enable-cryptonly", "--enable-all-crypto", + "--disable-examples", "--disable-benchmark", + "--disable-crypttests", "--enable-mlkem", + "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", + "--enable-xmss", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual -Wdeclaration-after-statement"], + "check": false}, + {"name": "smallest-mem-no-verify", "minutes": 1, + "configure": ["--enable-cryptonly", "--enable-all-crypto", + "--disable-examples", "--disable-benchmark", + "--disable-crypttests", "--enable-mlkem", + "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", + "--enable-xmss", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_NO_VERIFY -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual"], + "check": false}, + {"name": "precalc-a-intelasm", "minutes": 1, + "configure": ["--enable-cryptonly", "--enable-all-crypto", + "--enable-intelasm", "--disable-examples", "--disable-benchmark", + "--disable-crypttests", "--enable-mlkem", + "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", + "--enable-xmss", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1 -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual"], + "check": false}, {"name": "precalc-a-no-int128", "minutes": 1, "configure": ["--enable-cryptonly", "--enable-all-crypto", "--disable-examples", "--disable-benchmark", diff --git a/.wolfssl_known_macro_extras b/.wolfssl_known_macro_extras index 4f53192b6d2..a2b953f9465 100644 --- a/.wolfssl_known_macro_extras +++ b/.wolfssl_known_macro_extras @@ -998,6 +998,7 @@ WOLFSSL_MANUALLY_SELECT_DEVICE_CONFIG WOLFSSL_MCDC_ALLOC_SWEEP WOLFSSL_MDK5 WOLFSSL_MICROCHIP_AESGCM +WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC WOLFSSL_MLDSA_VERIFY_PRECOMP_A WOLFSSL_MLKEM_ASM_TEST WOLFSSL_MLKEM_INVNTT_UNROLL diff --git a/ChangeLog.md b/ChangeLog.md index a3151442649..218f5777408 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -1,11 +1,18 @@ # wolfSSL Release (unreleased) +## Behavioral Changes + +* **Behavioral change (`WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM` allocates)**: the option no longer forces `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, so the smallest memory verify now allocates its scratch buffers rather than pinning them in `wc_MlDsaKey`, and the key structure is smaller. Define `WOLFSSL_MLDSA_VERIFY_NO_MALLOC` as well to keep the heapless verify. by @Frauschi + +* **Behavioral change (`WOLFSSL_NO_MALLOC` pins the ML-DSA verify buffers)**: when ML-DSA verification is compiled in, `WOLFSSL_NO_MALLOC` now selects the small memory verify along with `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, including in builds that use the canonical option names. The verify scratch buffers then live in `wc_MlDsaKey`, which makes each key about 12 kB larger. Previously the no-malloc option selected nothing on its own, so without a working allocator verification failed with `MEMORY_E`. A build that defines `WOLFSSL_NO_MALLOC` but still has an allocator, through `WOLFSSL_STATIC_MEMORY` or `wolfSSL_SetAllocators()`, can keep the smaller key and the allocating verify by defining `WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC`. by @Frauschi + ## Post-Quantum Cryptography (PQC) -* `WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM` no longer forces `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, so streaming vector z can be used with allocated buffers instead of buffers pinned against the key. by @Frauschi -* Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, and stopped the matrix A cache regression test running against the small memory signing implementations, which stream matrix A rather than caching it. by @Frauschi -* Reduced the ML-DSA small memory heap footprint: signing now keeps w1 only in its encoded form and key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` generates matrix A a column at a time so that only one polynomial of y is held and w0 replaces w in place, roughly halving the signing peak. by @Frauschi -* Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of vector y is transformed once rather than once per row of A. Signatures are unchanged, memory is unchanged, and signing is 12 to 29 percent quicker with the C code and 6 to 12 percent quicker with AVX-512, making `WOLFSSL_MLDSA_SIGN_SMALL_MEM` faster than `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` on every target. `WOLFSSL_MLDSA_SMALL_MEM_POLY64` no longer applies to signing, which would need a 64-bit accumulator per row of w, so it now affects only `WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM` and `WOLFSSL_MLDSA_VERIFY_SMALL_MEM` and returns the 2KB it was allocating when signing. by @Frauschi +* Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, renaming the verify scratch member `t1` to `vt1`. by @Frauschi +* Reduced the ML-DSA small memory heap footprint: signing keeps w1 only in encoded form, key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` holds one polynomial of y, roughly halving the signing peak. by @Frauschi +* Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of y is transformed once rather than once per row. `WOLFSSL_MLDSA_SMALL_MEM_POLY64` no longer applies to signing. by @Frauschi +* Fixed `--enable-mldsa=` naming only a parameter set, which left key generation, signing and verification all disabled. by @Frauschi +* `wc_CheckPrivateKey()` now reports `NOT_COMPILED_IN` for an ML-DSA certificate and key when the key pair check is compiled out with `WOLFSSL_MLDSA_NO_CHECK_KEY`, rather than failing to build. Such a build cannot confirm the pair matches, so loading the two together fails. by @Frauschi # wolfSSL Release 5.9.4 (Sep 25, 2026) diff --git a/tests/unit-mcdc/test_wc_mldsa_whitebox.c b/tests/unit-mcdc/test_wc_mldsa_whitebox.c index 41409587cce..aa76dd38431 100644 --- a/tests/unit-mcdc/test_wc_mldsa_whitebox.c +++ b/tests/unit-mcdc/test_wc_mldsa_whitebox.c @@ -251,6 +251,12 @@ static void wb_check_low(void) WB_NOTE("mldsa_check_low(>=hi) expected 0"); } +#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM)) || \ + (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))) /* Vector level: two polynomials, both in range -> (ret==1)&&(i early ret 0. */ for (j = 0; j < 2 * MLDSA_N; j++) { @@ -265,6 +271,7 @@ static void wb_check_low(void) if (ret != 0) { WB_NOTE("mldsa_vec_check_low_c(out) expected 0"); } +#endif WB_OK("mldsa_check_low / vec_check_low_c operand pairs exercised"); } #endif diff --git a/wolfcrypt/src/asn.c b/wolfcrypt/src/asn.c index 3c1a2eda0d2..2da88874799 100644 --- a/wolfcrypt/src/asn.c +++ b/wolfcrypt/src/asn.c @@ -9950,9 +9950,15 @@ int wc_CheckPrivateKey(const byte* privKey, word32 privKeySz, keyIdx = 0; if ((ret = wc_MlDsaKey_ImportPubRaw(key_pair, pubKey, pubKeySz)) == 0) { + #ifdef WOLFSSL_MLDSA_CHECK_KEY /* Public and private extracted successfully. Sanity check. */ if ((ret = wc_MlDsaKey_CheckKey(key_pair)) == 0) ret = 1; + #else + /* Without the key check the pair cannot be confirmed to + * match, so do not claim that it does. */ + ret = NOT_COMPILED_IN; + #endif } } wc_MlDsaKey_Free(key_pair); diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index d763ba68b45..0640c116db9 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -78,19 +78,13 @@ * Compiles signature implementation that uses smaller amounts of memory but * is slower. Allocates matrix A and calculates it upfront. * WOLFSSL_MLDSA_SIGN_SMALLEST_MEM Default: OFF - * Compiles the smallest memory signature implementation. Implies - * WOLFSSL_MLDSA_SIGN_SMALL_MEM. Matrix A is generated a column at a time so - * that only one polynomial of vector y is ever held, and w0 replaces w in - * place. Vector y is regenerated for the z calculation. - * Cannot be used with WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC or - * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A. - * WOLFSSL_MLDSA_SMALL_MEM_POLY64 has no effect on signing in this mode as a - * 64-bit accumulator would be needed for every row of w. - * This is the slower of the two on every target as vector y has to be - * generated twice, and generating it a polynomial at a time cannot use the - * generators that produce the whole vector in one pass. Turn it on only - * when the polynomials of y that WOLFSSL_MLDSA_SIGN_SMALL_MEM keeps are - * more memory than the target has to spare. + * Compiles the smallest memory signature implementation, slower again than + * WOLFSSL_MLDSA_SIGN_SMALL_MEM, which it implies. Matrix A is generated a + * column at a time so only one polynomial of vector y is held, and y is + * regenerated for z. + * Cannot be used with WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC or _PRECALC_A. + * WOLFSSL_MLDSA_SMALL_MEM_POLY64, WC_MLDSA_CACHE_PRIV_VECTORS and + * WC_MLDSA_CACHE_MATRIX_A do nothing in this mode. * WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM Default: OFF * Compiles key generation implementation that uses smaller amounts of memory * but is slower. @@ -217,15 +211,12 @@ #endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + #error "PRECALC and PRECALC_A are equivalent to non small mem" #endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC - #error "PRECALC and PRECALC_A are equivalent to non small mem" - #endif + (WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A < 1) + #error "PRECALC_A must pre-calculate at least one row of matrix A" #endif #ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ @@ -234,6 +225,15 @@ #endif #endif +/* Signing drives the whole-vector helpers when it holds a full vector: the + * default implementation, and PRECALC_A which multiplies the pre-calculated + * rows of matrix A as a vector. */ +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A)) + #define MLDSA_SIGN_VEC_HELPERS +#endif + #if defined(USE_INTEL_SPEEDUP) static cpuid_flags_t cpuid_flags = WC_CPUID_INITIALIZER; @@ -1388,6 +1388,7 @@ static void mldsa_decode_eta_4_bits(const byte* p, sword32* s) (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (defined(WC_MLDSA_CACHE_PRIV_VECTORS) || \ defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM))) /* Decode vector of polynomials with range -ETA..ETA. * @@ -1756,6 +1757,7 @@ static void mldsa_decode_t0(const byte* t0, sword32* t) #if defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (defined(WC_MLDSA_CACHE_PRIV_VECTORS) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM))) /* Decode bottom D bits of t as t0. * @@ -1899,7 +1901,8 @@ static void mldsa_decode_t1(const byte* t1, sword32* t) #endif #if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ + (!defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM) || \ + defined(WC_MLDSA_CACHE_PUB_VECTORS))) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) /* Decode top bits of t as t1. * @@ -2713,6 +2716,7 @@ static void mldsa_decode_w1(const byte* w1e, sword32 gamma2, sword32* w1) else #endif { + XMEMSET(w1, 0, MLDSA_POLY_SIZE); } } #endif @@ -2742,8 +2746,6 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, { unsigned int i; - (void)k; - #ifndef WOLFSSL_NO_ML_DSA_44 if (gamma2 == MLDSA_Q_LOW_88) { unsigned int e = MLDSA_Q_HI_88_ENC_BITS * 2 * MLDSA_N / 16; @@ -2754,7 +2756,7 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, * encoder needs no polynomial pairing and does the entire vector. */ if (IS_INTEL_AVX512_VBMI(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { - for (; i < PARAMS_ML_DSA_44_K; i++) { + for (; i < k; i++) { wc_mldsa_encode_w1_88_avx512_vbmi(w1, w1e); w1 += MLDSA_N; w1e += e; @@ -2766,7 +2768,7 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, /* Two polynomials per pass; an odd trailing one falls through. */ if ((i == 0) && USE_INTEL_AVX512(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { - for (; i + 1 < PARAMS_ML_DSA_44_K; i += 2) { + for (; i + 1 < k; i += 2) { wc_mldsa_encode_w1_88_x2_avx512(w1, w1 + MLDSA_N, w1e, w1e + e); w1 += 2 * MLDSA_N; @@ -2776,7 +2778,7 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, } #endif /* Step 2. For each polynomial of vector. */ - for (; i < PARAMS_ML_DSA_44_K; i++) { + for (; i < k; i++) { mldsa_encode_w1_88(w1, w1e); /* Next polynomial. */ w1 += MLDSA_N; @@ -3065,10 +3067,10 @@ static int mldsa_rej_ntt_poly_ex(wc_Shake* shake128, byte* seed, sword32* a, #if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) || \ + defined(WC_MLDSA_CACHE_MATRIX_A) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) + !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ + defined(MLDSA_SIGN_VEC_HELPERS) /* Generate a random polynomial by rejection. * * @param [in, out] shake128 SHAKE-128 object. @@ -3113,11 +3115,10 @@ static int mldsa_rej_ntt_poly(wc_Shake* shake128, byte* seed, sword32* a, #if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ + defined(WC_MLDSA_CACHE_MATRIX_A) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WC_MLDSA_CACHE_MATRIX_A))) + defined(MLDSA_SIGN_VEC_HELPERS) #if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) #define SHA3_128_BYTES (WC_SHA3_128_COUNT * 8) @@ -5260,9 +5261,10 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, #if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) #ifdef WOLFSSL_MLDSA_HAVE_INTEL_AVX512 - /* Whole vector in one eight-way run, whatever the dimension. */ - if (USE_INTEL_AVX512(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && - (SAVE_VECTOR_REGISTERS2() == 0)) { + /* Whole vector in one eight-way run. Only worth its scratch when enough + * of the eight lanes are used, so short vectors fall through. */ + if ((l >= 4) && USE_INTEL_AVX512(cpuid_flags) && + IS_INTEL_BMI2(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { ret = wc_mldsa_gen_y_avx512(y, seed, kappa, gamma1_bits, l, heap); RESTORE_VECTOR_REGISTERS(); } @@ -6985,9 +6987,7 @@ static void mldsa_ntt(sword32* r) #endif #if !defined(WOLFSSL_MLDSA_NO_VERIFY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC))) || \ + !defined(WOLFSSL_MLDSA_NO_SIGN) || \ (defined(WOLFSSL_MLDSA_SMALL) && \ (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) @@ -8047,8 +8047,7 @@ static void mldsa_invntt_full(sword32* r) defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + defined(MLDSA_SIGN_VEC_HELPERS) /* Inverse Number-Theoretic Transform. * * @param [in, out] r Vector of polynomials to transform. @@ -8080,8 +8079,7 @@ static void mldsa_vec_invntt_full(sword32* r, byte l) defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + defined(MLDSA_SIGN_VEC_HELPERS) /* Matrix multiplication. * * @param [out] r Vector of polynomials that is result. @@ -8526,13 +8524,37 @@ static void mldsa_poly_red(sword32* a) } } +#if defined(WC_MLDSA_FAULT_HARDEN) && \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) +/* Checksum of a polynomial. + * + * Binds two computations of a value that is not kept between them, so a fault + * injected into the second is detectable. + * + * @param [in] a Polynomial to checksum. + * @return Checksum of the polynomial. + */ +static sword32 mldsa_poly_checksum(const sword32* a) +{ + unsigned int i; + word32 chk = 0; + + for (i = 0; i < MLDSA_N; i++) { + chk = (chk * 3) ^ (word32)a[i]; + } + + return (sword32)chk; +} +#endif + #if (defined(WOLFSSL_MLDSA_SMALL) && \ (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) || \ + (defined(MLDSA_SIGN_VEC_HELPERS) && defined(WOLFSSL_MLDSA_SMALL)) /* Modulo reduce values in polynomials of vector. Range (-2^31)..(2^31-1). * * @param [in, out] a Vector of polynomials. @@ -8761,8 +8783,7 @@ static void mldsa_make_pos(sword32* a) #if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + defined(MLDSA_SIGN_VEC_HELPERS) /* Make values in polynomials of vector be in positive range. * * @param [in, out] a Vector of polynomials. @@ -9226,16 +9247,19 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) /* Step 6, Step 7, Step 9. Alg 22 Steps 2-4, Alg 24 Steps 8-10. * Decompose t in t0 and t1 and encode into public and private - * key. */ + * key. One polynomial at a time trades the AVX-512 paired encoder + * for the s2Sz saving this implementation exists to make. */ mldsa_vec_encode_t0_t1(tt, 1, t0, t1); s2pt += s2Stride; t0 += MLDSA_D * MLDSA_N / 8; t1 += MLDSA_U * MLDSA_N / 8; } - /* Step 8. Alg 24, Step 1: Hash public key into private key. */ - ret = mldsa_shake256(&key->shake, key->p, params->pkSz, tr, - MLDSA_TR_SZ); + if (ret == 0) { + /* Step 8. Alg 24, Step 1: Hash public key into private key. */ + ret = mldsa_shake256(&key->shake, key->p, params->pkSz, tr, + MLDSA_TR_SZ); + } } if (ret == 0) { /* Public key and private key are available. */ @@ -9835,9 +9859,10 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC allocSz += (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz; #elif defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) - /* The pre-calculated rows of A are multiplied as a vector, so w must - * hold that many polynomials. */ - allocSz += (unsigned int)(maxK - 1) * (unsigned int)MLDSA_POLY_SIZE + + /* w is decomposed and encoded as a full k vector regardless of how + * many rows of A are pre-calculated, so it must hold k polynomials. */ + allocSz += (unsigned int)(params->k - 1) * + (unsigned int)MLDSA_POLY_SIZE + (unsigned int)maxK * params->l * (unsigned int)MLDSA_POLY_SIZE; #endif @@ -9852,7 +9877,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, w0 = y + params->s1Sz / sizeof(*y_ntt); w = w0 + params->s2Sz / sizeof(*w0); #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) - c = w + (unsigned int)maxK * MLDSA_N; + c = w + (unsigned int)params->k * MLDSA_N; #else c = w + MLDSA_N; #endif @@ -9914,7 +9939,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, byte s; byte rStart; sword32 hi; - sword32* wt = w; + sword32* wt; sword32* w0t = w0; byte* w1et = w1e; sword32* at = a; @@ -9977,12 +10002,26 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, XMEMCPY(y_ntt_t, y + (unsigned int)s * MLDSA_N, MLDSA_POLY_SIZE); mldsa_ntt_full(y_ntt_t); + #ifdef WC_MLDSA_FAULT_HARDEN + if (s >= params->l) { + valid = 0; + ret = BAD_COND_E; + break; + } + #endif /* Put s into buffer to be hashed. */ aseed[MLDSA_PUB_SEED_SZ + 0] = s; wt = w0t; /* Alg 26. Step 1: Loop over first dimension of matrix. */ for (r = rStart; r < params->k; r++) { + #ifdef WC_MLDSA_FAULT_HARDEN + if (wt != w0t + (unsigned int)(r - rStart) * MLDSA_N) { + valid = 0; + ret = BAD_COND_E; + break; + } + #endif /* Put r/i into buffer to be hashed. */ aseed[MLDSA_PUB_SEED_SZ + 1] = r; /* Alg 26. Step 3: Create polynomial from hashing seed. */ @@ -10056,6 +10095,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, unsigned int e; /* Step 13: w = NTT-1(A o NTT(y)) */ + mldsa_poly_red(w0t); mldsa_invntt_full(w0t); /* Step 14, Step 22: Make values positive and decompose. */ mldsa_make_pos(w0t); @@ -10315,6 +10355,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, unsigned int sStride = (unsigned int)params->s2EncSz / params->k; #ifdef WC_MLDSA_FAULT_HARDEN sword32* w_check; + sword32 yChk[MLDSA_MAX_L_VECTOR_COUNT / MLDSA_N]; #endif /* priv_rand_seed will hold the secret signing seed (rho'') derived below; @@ -10376,7 +10417,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, byte* commit = sig; byte* w1et = w1e; const byte* sp; - sword32* wt = w; + sword32* wt; unsigned int e; sword32 hi; byte r; @@ -10403,8 +10444,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, if (ret != 0) { break; } + #ifdef WC_MLDSA_FAULT_HARDEN + /* Bind this polynomial of y to the one regenerated for z. */ + yChk[s] = mldsa_poly_checksum(y); + #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_vec_check_low(y, 1, + valid = mldsa_check_low(y, ((sword32)1 << params->gamma1_bits) - params->beta); if (!valid) { break; @@ -10468,6 +10513,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, wt = w; for (r = 0; (ret == 0) && valid && (r < params->k); r++) { /* Step 13: w = NTT-1(A o NTT(y)) */ + mldsa_poly_red(wt); mldsa_invntt_full(wt); /* Step 14, Step 22: Make values positive and decompose. */ mldsa_make_pos(wt); @@ -10494,7 +10540,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_vec_check_low(wt, 1, + valid = mldsa_check_low(wt, params->gamma2 - params->beta); #endif wt += MLDSA_N; @@ -10530,6 +10576,13 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, if (ret != 0) { break; } + #ifdef WC_MLDSA_FAULT_HARDEN + if (yChk[s] != mldsa_poly_checksum(y)) { + valid = 0; + ret = BAD_COND_E; + break; + } + #endif mldsa_vec_decode_eta_bits(sp, params->eta, a, 1); mldsa_ntt_small(a); /* Step 19: cs1 = NTT-1(c o s1) */ diff --git a/wolfcrypt/test/test.c b/wolfcrypt/test/test.c index 52888cb58f6..ecb1901c791 100644 --- a/wolfcrypt/test/test.c +++ b/wolfcrypt/test/test.c @@ -66475,7 +66475,6 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_NO_VERIFY) @@ -66491,10 +66490,6 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) * key after make_key, then signing. The post-condition asserts that key->a * was populated (proving the allocation made it into the key, not the local) * and that signing produces a verifiable signature. - * - * The small memory signing implementations stream matrix A rather than - * caching it, so they never populate key->a and the post-condition does not - * apply to them. */ static wc_test_ret_t mldsa_sign_cache_alloc_test(int param, WC_RNG* rng) { @@ -66552,13 +66547,17 @@ static wc_test_ret_t mldsa_sign_cache_alloc_test(int param, WC_RNG* rng) if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); +#ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM /* With the fix, signing must populate key->a (allocated buffer is owned * by the key, not leaked to a local). Without the fix, key->a remains - * NULL because the XMALLOC result was assigned to a local variable. */ + * NULL because the XMALLOC result was assigned to a local variable. + * The small memory implementations stream matrix A and never populate + * key->a, so only the round trip below applies to them. */ if (key->a == NULL) ERROR_OUT(WC_TEST_RET_ENC_NC, out); if (key->aSet != 1) ERROR_OUT(WC_TEST_RET_ENC_NC, out); +#endif ret = wc_MlDsaKey_VerifyCtx(key, sig, sigLen, NULL, 0, msg, (word32)sizeof(msg), &res); @@ -66575,8 +66574,8 @@ static wc_test_ret_t mldsa_sign_cache_alloc_test(int param, WC_RNG* rng) return ret; } #endif /* WC_MLDSA_CACHE_MATRIX_A && !WC_MLDSA_FIXED_ARRAY && - * !WOLFSSL_MLDSA_SIGN_SMALL_MEM && !WOLFSSL_MLDSA_NO_MAKE_KEY && - * !WOLFSSL_MLDSA_NO_SIGN && !WOLFSSL_MLDSA_NO_VERIFY */ + * !WOLFSSL_MLDSA_NO_MAKE_KEY && !WOLFSSL_MLDSA_NO_SIGN && + * !WOLFSSL_MLDSA_NO_VERIFY */ #if (defined(WOLFSSL_MLDSA_PRIVATE_KEY) && \ @@ -67567,7 +67566,6 @@ WOLFSSL_TEST_SUBROUTINE wc_test_ret_t mldsa_test(void) #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_NO_VERIFY) diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index 1a6c95712a8..93cfeb1163c 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -183,12 +183,22 @@ #define WOLFSSL_MLDSA_SIGN_SMALL_MEM #endif #endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC + #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + #endif +#endif #ifdef WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC_A #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A #define WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A \ WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC_A #endif #endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A + #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + #endif +#endif #ifdef WOLFSSL_DILITHIUM_SIGN_CHECK_W0 #ifndef WOLFSSL_MLDSA_SIGN_CHECK_W0 #define WOLFSSL_MLDSA_SIGN_CHECK_W0 @@ -435,6 +445,9 @@ #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) && !defined(WOLFSSL_DILITHIUM_SIGN_SMALL_MEM) #define WOLFSSL_DILITHIUM_SIGN_SMALL_MEM #endif +#if defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) && !defined(WOLFSSL_DILITHIUM_VERIFY_SMALLEST_MEM) + #define WOLFSSL_DILITHIUM_VERIFY_SMALLEST_MEM +#endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) && !defined(WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC) #define WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC #endif From 0101a858467ff53f0e823f6fee2141fa0062a79e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:47 +0200 Subject: [PATCH 07/11] Harden the ML-DSA signing range checks and fix small memory build gates Signing retries until a candidate passes four infinity norm checks. mldsa_check_low() stopped at the first out of range coefficient, so its running time revealed which coefficient failed. Replace it with a branchless mldsa_check_low_ct() at every signing check site. This costs nothing: the accepting path already scanned all 256 coefficients, so only a failing check changes, and ML-DSA-44 signing measures no slower than before. The rejecting index itself carries no key information for z and w0-cs2. Both are masked by the uniform vector y, and the count of y values that push a coefficient out of range is 2*beta+1 whatever the secret contributes, so every coefficient rejects with the same probability. The reference implementation relies on the same argument. Those loops therefore keep their early exit. ct0 is different: it is c*t0 with no mask, so the index of a failing polynomial does depend on t0. Check every polynomial of ct0 rather than stopping at the first failure. This is free in practice because that check rarely rejects. The smallest and small memory signers compute ct0 in the same row loop as w0-cs2, so they accumulate the ct0 result in a flag of its own: the loop still stops at the first rejecting w0-cs2 row, but a failing ct0 row no longer ends it, and the hint is made for every row the loop reaches whatever ct0 decided. Move the canonical option implications out of the legacy name gate region of dilithium.h. WOLFSSL_NO_DILITHIUM_LEGACY_GATES suppresses legacy to canonical name translation, but it was also suppressing SIGN_SMALLEST_MEM, PRECALC and PRECALC_A implying WOLFSSL_MLDSA_SIGN_SMALL_MEM, so those builds silently got the full memory signer. Derive WC_ENABLE_ASYM_KEY_IMPORT and WC_ENABLE_ASYM_KEY_EXPORT from WOLFSSL_HAVE_MLDSA rather than the legacy HAVE_DILITHIUM, so an ML-DSA build that opts out of the legacy gates still gets the RFC 5958 helpers it calls. settings.h maps HAVE_DILITHIUM onto WOLFSSL_HAVE_MLDSA unconditionally before either use, so the canonical name alone covers both spellings. Make WOLFSSL_NO_MALLOC select the small memory verify along with WOLFSSL_MLDSA_VERIFY_NO_MALLOC. The pinned verify buffers only exist under the small memory verify, so the no-malloc option alone selected nothing and every verification failed with MEMORY_E. Set only the canonical names there and gate on WOLFSSL_HAVE_MLDSA; dilithium.h mirrors them onto the legacy names for code that still reads those. Zeroize the fault hardening checksums of the signing mask, and narrow the mldsa_vec_check_low() guard away from the smallest memory signer that does not call it. Wipe the key's SHAKE object when the smallest memory signer returns. Its last use of the object regenerates a polynomial of y, so signing left rho'' in the key: the whole of it could be read back from the object on the generic path, and on the Intel path the Keccak state it leaves is a permutation of it. The other signers finish by hashing the public commitment. Every later use sets the object up afresh, so only the device id needs restoring. Add a signing KAT to testwolfcrypt. ML-DSA signing is deterministic given the key seed and the signing seed, so every signer must produce the same bytes; the test compares a SHAKE-256 digest of the signature with the default signer's. tests/api already holds full signing KATs, but unit.test is not built in the --enable-cryptonly configurations that every smallest memory and PRECALC_A CI row uses, so without this nothing checked that those signers produce the FIPS 204 signature rather than merely one that verifies. Add CI rows for the configurations none of this was covered by: the legacy gate opt out, and smallest memory signing with the y and w0 checks. --- .github/configs/pq-all.json | 10 +- tests/api/test_ossl_x509_crypto.c | 1 + tests/unit-mcdc/test_wc_mldsa_whitebox.c | 5 +- wolfcrypt/src/asn.c | 1 + wolfcrypt/src/wc_mldsa.c | 143 +++++++++++++++++----- wolfcrypt/test/test.c | 148 +++++++++++++++++++++++ wolfssl/wolfcrypt/dilithium.h | 61 +++++----- wolfssl/wolfcrypt/settings.h | 19 ++- 8 files changed, 321 insertions(+), 67 deletions(-) diff --git a/.github/configs/pq-all.json b/.github/configs/pq-all.json index fb4bc7488b2..3d2ff936a6e 100644 --- a/.github/configs/pq-all.json +++ b/.github/configs/pq-all.json @@ -240,5 +240,13 @@ {"name": "mldsa-44-precalc-a-intelasm", "minutes": 2, "comment": "ML-DSA-44 with one pre-calculated row of matrix A on the Intel assembly paths; the q88 decompose and w1 encode kernels take no dimension, so this pins that w is sized for a full vector", "configure": ["--enable-cryptonly", "--enable-mldsa", "--enable-intelasm", - "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]} + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]}, +{"name": "mldsa-no-legacy-gates-smallest", "minutes": 2, + "comment": "Canonical option names only; pins that SIGN_SMALLEST_MEM still implies SIGN_SMALL_MEM when the legacy name gates are opted out", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_NO_DILITHIUM_LEGACY_GATES -DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM"]}, +{"name": "mldsa-smallest-checks-no-verify", "minutes": 2, + "comment": "Smallest memory signing with the y and w0 checks and no verify, the pairing that pulls in the vector range check helper it does not use", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0 -DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM"]} ] diff --git a/tests/api/test_ossl_x509_crypto.c b/tests/api/test_ossl_x509_crypto.c index 71650077b32..056abd3467c 100644 --- a/tests/api/test_ossl_x509_crypto.c +++ b/tests/api/test_ossl_x509_crypto.c @@ -83,6 +83,7 @@ int test_wolfSSL_X509_check_private_key_mldsa(void) defined(HAVE_DILITHIUM) && !defined(WOLFSSL_DILITHIUM_NO_SIGN) && \ !defined(WOLFSSL_DILITHIUM_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_NO_ASN1) && \ + defined(WOLFSSL_MLDSA_CHECK_KEY) && \ (defined(OPENSSL_ALL) || defined(WOLFSSL_WPAS_SMALL)) && \ (!defined(WOLFSSL_NO_ML_DSA_44) || !defined(WOLFSSL_NO_ML_DSA_65) || \ !defined(WOLFSSL_NO_ML_DSA_87)) diff --git a/tests/unit-mcdc/test_wc_mldsa_whitebox.c b/tests/unit-mcdc/test_wc_mldsa_whitebox.c index aa76dd38431..a7fe9d9eaeb 100644 --- a/tests/unit-mcdc/test_wc_mldsa_whitebox.c +++ b/tests/unit-mcdc/test_wc_mldsa_whitebox.c @@ -255,8 +255,9 @@ static void wb_check_low(void) !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))) + (!defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) && \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))))) /* Vector level: two polynomials, both in range -> (ret==1)&&(i early ret 0. */ for (j = 0; j < 2 * MLDSA_N; j++) { diff --git a/wolfcrypt/src/asn.c b/wolfcrypt/src/asn.c index 2da88874799..0181fe8cf72 100644 --- a/wolfcrypt/src/asn.c +++ b/wolfcrypt/src/asn.c @@ -9958,6 +9958,7 @@ int wc_CheckPrivateKey(const byte* privKey, word32 privKeySz, /* Without the key check the pair cannot be confirmed to * match, so do not claim that it does. */ ret = NOT_COMPILED_IN; + WOLFSSL_ERROR_VERBOSE(ret); #endif } } diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 0640c116db9..57bc9b8ce3f 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -5764,7 +5764,72 @@ static void mldsa_vec_decompose(const sword32* r, byte k, sword32 gamma2, * Range check operation ******************************************************************************/ -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || !defined(WOLFSSL_MLDSA_NO_VERIFY) +#ifndef WOLFSSL_MLDSA_NO_SIGN +/* Check that the values of the polynomial are in range, in constant time. + * + * Signing range checks run on values derived from the private key, so the + * index of the first failing coefficient must not be observable. Every + * coefficient is examined and no branch depends on a value. + * + * @param [in] a Polynomial. + * @param [in] hi Largest value in range. + * @return 1 when all values are in range, 0 when any is not. + */ +static int mldsa_check_low_ct(const sword32* a, sword32 hi) +{ + unsigned int j; + /* Calculate lowest range value. */ + sword32 nhi = -hi; + word32 good = 1; + + /* For each value of polynomial. */ + for (j = 0; j < MLDSA_N; j++) { + /* Sign bit of the difference is set when the comparison holds. */ + word32 lt = (word32)((word64)((sword64)a[j] - (sword64)hi) >> 63); + word32 gt = (word32)((word64)((sword64)nhi - (sword64)a[j]) >> 63); + + /* Range is -(hi-1)..(hi-1). */ + good &= lt & gt; + } + + return (int)good; +} + +/* The smallest memory signer holds one polynomial at a time and uses the + * scalar form for these checks. */ +#if (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_W0)) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) +/* Check that the values of the vector are in range, in constant time. + * + * @param [in] a Vector of polynomials. + * @param [in] l Dimension of vector. + * @param [in] hi Largest value in range. + * @return 1 when all values are in range, 0 when any is not. + */ +static int mldsa_vec_check_low_ct(const sword32* a, byte l, sword32 hi) +{ + unsigned int i; + int good = 1; + + /* For each polynomial of vector. */ + for (i = 0; i < l; i++) { + good &= mldsa_check_low_ct(a, hi); + /* Next polynomial. */ + a += MLDSA_N; + } + + return good; +} +#endif /* (CHECK_Y || CHECK_W0) && !SIGN_SMALLEST_MEM */ +#endif /* !WOLFSSL_MLDSA_NO_SIGN */ + +#if !defined(WOLFSSL_MLDSA_NO_VERIFY) || \ + (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) && \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))))) /* Check that the values of the polynomial are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -5808,8 +5873,9 @@ static int mldsa_check_low(const sword32* a, sword32 hi) !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))) + (!defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) && \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ + defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))))) /* Check that the values of the vector are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -9654,7 +9720,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_expand_mask(&key->shake, priv_rand_seed, kappa, params->gamma1_bits, y, params->l, key->heap); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_vec_check_low(y, params->l, + valid = mldsa_vec_check_low_ct(y, params->l, ((sword32)1 << params->gamma1_bits) - params->beta); if (valid) #endif @@ -9679,7 +9745,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_make_pos(w, params->k); mldsa_vec_decompose(w, params->k, params->gamma2, w0, w1); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_vec_check_low(w0, params->k, + valid = mldsa_vec_check_low_ct(w0, params->k, params->gamma2 - params->beta); } if (valid) { @@ -9715,7 +9781,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Step 22: w0 - cs2 */ mldsa_sub(w0 + i * MLDSA_N, cs2 + i * MLDSA_N); /* Step 23: Check w0 - cs2 has low enough values. */ - valid = mldsa_vec_check_low(w0 + i * MLDSA_N, 1, hi); + valid = mldsa_check_low_ct(w0 + i * MLDSA_N, hi); } hi = ((sword32)1 << params->gamma1_bits) - params->beta; for (i = 0; valid && i < params->l; i++) { @@ -9726,15 +9792,21 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_add(z + i * MLDSA_N, y + i * MLDSA_N); mldsa_poly_red(z + i * MLDSA_N); /* Step 23: Check z has low enough values. */ - valid = mldsa_vec_check_low(z + i * MLDSA_N, 1, hi); + valid = mldsa_check_low_ct(z + i * MLDSA_N, hi); } - hi = params->gamma2; - for (i = 0; valid && i < params->k; i++) { - /* Step 25: ct0 = NTT-1(c o t0) */ - mldsa_mul_invntt(ct0 + i * MLDSA_N, c, - t0 + i * MLDSA_N); - /* Step 27: Check ct0 has low enough values. */ - valid = mldsa_vec_check_low(ct0 + i * MLDSA_N, 1, hi); + if (valid) { + hi = params->gamma2; + /* Unlike z and w0-cs2, ct0 carries no uniform mask, so + * the index of a failing polynomial depends on t0. + * Check them all. */ + for (i = 0; i < params->k; i++) { + /* Step 25: ct0 = NTT-1(c o t0) */ + mldsa_mul_invntt(ct0 + i * MLDSA_N, c, + t0 + i * MLDSA_N); + /* Step 27: Check ct0 has low enough values. */ + valid &= mldsa_check_low_ct(ct0 + i * MLDSA_N, + hi); + } } if (valid) { /* Step 26: ct0 = ct0 + w0 */ @@ -9956,7 +10028,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_expand_mask(&key->shake, priv_rand_seed, kappa, params->gamma1_bits, y, params->l, key->heap); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_vec_check_low(y, params->l, + valid = mldsa_vec_check_low_ct(y, params->l, ((sword32)1 << params->gamma1_bits) - params->beta); #endif @@ -10122,7 +10194,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_vec_check_low(w0t, 1, + valid = mldsa_check_low_ct(w0t, params->gamma2 - params->beta); #endif w0t += MLDSA_N; @@ -10179,7 +10251,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_poly_red(z); /* Step 23: Check z has low enough values. */ hi = ((sword32)1 << params->gamma1_bits) - params->beta; - valid = mldsa_check_low(z, hi); + valid = mldsa_check_low_ct(z, hi); if (valid) { /* Step 32: Encode z into signature. * Commit (c) and h already encoded into signature. */ @@ -10212,6 +10284,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #endif sword32* cs2 = ct0; byte idx = 0; + int ct0Valid = 1; w0t = w0; w1et = w1e; @@ -10245,7 +10318,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_poly_red(w0t); /* Step 23: Check w0 - cs2 has low enough values. */ hi = params->gamma2 - params->beta; - valid = mldsa_check_low(w0t, hi); + valid = mldsa_check_low_ct(w0t, hi); if (valid) { #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC mldsa_decode_t0(t0pt, t0); @@ -10258,10 +10331,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_mul(ct0, c, t0 + r * MLDSA_N); #endif mldsa_invntt(ct0); - /* Step 27: Check ct0 has low enough values. */ - valid = mldsa_check_low(ct0, params->gamma2); - } - if (valid) { + /* Step 27: Check ct0 has low enough values. Every + * row: ct0 has no mask, so a failing index leaks t0. */ + ct0Valid &= mldsa_check_low_ct(ct0, params->gamma2); /* Step 26: ct0 = ct0 + w0 */ mldsa_add(ct0, w0t); mldsa_poly_red(ct0); @@ -10301,6 +10373,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, w0t += MLDSA_N; w1et += w1Stride; } + valid &= ct0Valid; /* Set remaining hints to zero. */ XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); } @@ -10449,7 +10522,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, yChk[s] = mldsa_poly_checksum(y); #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_check_low(y, + valid = mldsa_check_low_ct(y, ((sword32)1 << params->gamma1_bits) - params->beta); if (!valid) { break; @@ -10540,7 +10613,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_check_low(wt, + valid = mldsa_check_low_ct(wt, params->gamma2 - params->beta); #endif wt += MLDSA_N; @@ -10592,7 +10665,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_add(z, y); mldsa_poly_red(z); /* Step 23: Check z has low enough values. */ - valid = mldsa_check_low(z, hi); + valid = mldsa_check_low_ct(z, hi); if (valid) { /* Step 32: Encode z into signature. * Commit (c) and h already encoded into signature. */ @@ -10618,6 +10691,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, if ((ret == 0) && valid) { const byte* t0pt = t0p; byte idx = 0; + int ct0Valid = 1; sp = s2p; wt = w; @@ -10632,7 +10706,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_sub(wt, z); mldsa_poly_red(wt); /* Step 23: Check w0 - cs2 has low enough values. */ - valid = mldsa_check_low(wt, + valid = mldsa_check_low_ct(wt, params->gamma2 - params->beta); if (valid) { mldsa_decode_t0(t0pt, a); @@ -10640,10 +10714,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Step 25: ct0 = NTT-1(c o t0) */ mldsa_mul(z, c, a); mldsa_invntt(z); - /* Step 27: Check ct0 has low enough values. */ - valid = mldsa_check_low(z, params->gamma2); - } - if (valid) { + /* Step 27: Check ct0 has low enough values. Every + * row: ct0 has no mask, so a failing index leaks t0. */ + ct0Valid &= mldsa_check_low_ct(z, params->gamma2); /* Step 26: ct0 = ct0 + w0 */ mldsa_add(z, wt); mldsa_poly_red(z); @@ -10681,6 +10754,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, wt += MLDSA_N; w1et += w1Stride; } + valid &= ct0Valid; /* Set remaining hints to zero. */ XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); } @@ -10701,8 +10775,17 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } ForceZero(priv_rand_seed, sizeof(priv_rand_seed)); + /* The last expansion of y leaves rho'' recoverable from key->shake. */ + ForceZero(&key->shake, sizeof(key->shake)); +#ifdef WOLF_CRYPTO_CB + key->shake.devId = INVALID_DEVID; +#endif #ifdef WOLFSSL_CHECK_MEM_ZERO wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); +#endif +#ifdef WC_MLDSA_FAULT_HARDEN + /* Checksums are derived from the secret mask y. */ + ForceZero(yChk, sizeof(yChk)); #endif if (w != NULL) { ForceZero(w, allocSz); diff --git a/wolfcrypt/test/test.c b/wolfcrypt/test/test.c index ecb1901c791..b373d20c613 100644 --- a/wolfcrypt/test/test.c +++ b/wolfcrypt/test/test.c @@ -66473,6 +66473,132 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) } #endif +/* Cross-implementation signature KAT. ML-DSA signing is deterministic given + * the key and signing seeds, so every signer here must produce the same bytes. + * Expected values are SHAKE-256 digests of the default signer's signature; + * regenerate them from a default-signer build if the seeds or message change. + * Skipped for the signers that are deliberately not FIPS 204 conformant: the + * draft domain separation and the CHECK_Y/CHECK_W0 early rejects. */ +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_FIPS204_DRAFT) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) + +/* Seed the key pair is generated from. */ +static const byte mldsa_kat_key_seed[MLDSA_SEED_SZ] = { + 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, + 0x08, 0x09, 0x0a, 0x0b, 0x0c, 0x0d, 0x0e, 0x0f, + 0x10, 0x11, 0x12, 0x13, 0x14, 0x15, 0x16, 0x17, + 0x18, 0x19, 0x1a, 0x1b, 0x1c, 0x1d, 0x1e, 0x1f +}; +/* Seed used in place of the per-signature randomness. */ +static const byte mldsa_kat_sig_seed[MLDSA_RND_SZ] = { + 0x20, 0x21, 0x22, 0x23, 0x24, 0x25, 0x26, 0x27, + 0x28, 0x29, 0x2a, 0x2b, 0x2c, 0x2d, 0x2e, 0x2f, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x3a, 0x3b, 0x3c, 0x3d, 0x3e, 0x3f +}; +static const byte mldsa_kat_msg[] = { + 0x77, 0x6f, 0x6c, 0x66, 0x53, 0x53, 0x4c, 0x20, /* "wolfSSL " */ + 0x4d, 0x4c, 0x2d, 0x44, 0x53, 0x41, 0x20, 0x4b, /* "ML-DSA K" */ + 0x41, 0x54 /* "AT" */ +}; + +/* SHAKE-256 digests of the signature the default signer produces. */ +#ifndef WOLFSSL_NO_ML_DSA_44 +static const byte mldsa_kat_digest_44[32] = { + 0x17, 0xc1, 0xa1, 0x07, 0x90, 0xe6, 0xce, 0xc3, + 0x38, 0x17, 0x18, 0x02, 0x41, 0xaf, 0x0a, 0x3f, + 0xbd, 0x2c, 0xb9, 0x0d, 0xbc, 0x3f, 0x5d, 0x8b, + 0x07, 0x98, 0xc6, 0xe3, 0x75, 0x66, 0x8b, 0x3c +}; +#endif +#ifndef WOLFSSL_NO_ML_DSA_65 +static const byte mldsa_kat_digest_65[32] = { + 0x10, 0xb5, 0x77, 0xb1, 0x8f, 0xf0, 0x21, 0x0c, + 0x17, 0x31, 0x54, 0xe9, 0x3e, 0x79, 0xc8, 0x05, + 0x22, 0xdf, 0x27, 0x03, 0xfb, 0x99, 0xc0, 0x8b, + 0xf8, 0x25, 0x4c, 0xda, 0x36, 0xf7, 0x6f, 0xb1 +}; +#endif +#ifndef WOLFSSL_NO_ML_DSA_87 +static const byte mldsa_kat_digest_87[32] = { + 0xbe, 0xf0, 0xb7, 0xe5, 0x5f, 0x86, 0x4a, 0xdb, + 0x48, 0xfc, 0x56, 0x80, 0x93, 0x13, 0xdd, 0x96, + 0x08, 0x2d, 0x0f, 0x86, 0x1b, 0xf1, 0x89, 0x52, + 0x9f, 0x97, 0xb9, 0xca, 0xd3, 0x8f, 0xc3, 0xbf +}; +#endif + +static wc_test_ret_t mldsa_sign_kat_test(int param, const byte* expDigest) +{ + wc_test_ret_t ret; + wc_MlDsaKey* key = NULL; + byte* sig = NULL; + word32 sigLen; + int sigSz = 0; + byte digest[32]; + wc_Shake shake; + int keyInit = 0; + int shakeInit = 0; + + key = (wc_MlDsaKey*)XMALLOC(sizeof(wc_MlDsaKey), HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + sig = (byte*)XMALLOC(MLDSA_MAX_SIG_SIZE, HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + if ((key == NULL) || (sig == NULL)) + ERROR_OUT(WC_TEST_RET_ENC_NC, out); + + ret = wc_MlDsaKey_Init(key, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + keyInit = 1; + ret = wc_MlDsaKey_SetParams(key, param); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + + /* Deterministic key and deterministic signature. */ + ret = wc_MlDsaKey_MakeKeyFromSeed(key, mldsa_kat_key_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + + ret = wc_MlDsaKey_GetSigLen(key, &sigSz); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + sigLen = (word32)sigSz; + ret = wc_MlDsaKey_SignCtxWithSeed(key, NULL, 0, sig, &sigLen, + mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), mldsa_kat_sig_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + + ret = wc_InitShake256(&shake, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + shakeInit = 1; + ret = wc_Shake256_Update(&shake, sig, sigLen); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_Shake256_Final(&shake, digest, (word32)sizeof(digest)); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + + if (XMEMCMP(digest, expDigest, sizeof(digest)) != 0) + ERROR_OUT(WC_TEST_RET_ENC_NC, out); + + ret = 0; +out: + if (shakeInit) + wc_Shake256_Free(&shake); + if (keyInit) + wc_MlDsaKey_Free(key); + XFREE(sig, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + XFREE(key, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + return ret; +} + +#endif /* !NO_SIGN && !NO_MAKE_KEY && !FIPS204_DRAFT */ + #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ @@ -67564,6 +67690,28 @@ WOLFSSL_TEST_SUBROUTINE wc_test_ret_t mldsa_test(void) #endif #endif +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_FIPS204_DRAFT) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) +#ifndef WOLFSSL_NO_ML_DSA_44 + ret = mldsa_sign_kat_test(WC_ML_DSA_44, mldsa_kat_digest_44); + if (ret != 0) + ERROR_OUT(ret, out); +#endif +#ifndef WOLFSSL_NO_ML_DSA_65 + ret = mldsa_sign_kat_test(WC_ML_DSA_65, mldsa_kat_digest_65); + if (ret != 0) + ERROR_OUT(ret, out); +#endif +#ifndef WOLFSSL_NO_ML_DSA_87 + ret = mldsa_sign_kat_test(WC_ML_DSA_87, mldsa_kat_digest_87); + if (ret != 0) + ERROR_OUT(ret, out); +#endif +#endif + #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index 93cfeb1163c..a661ce6cb9e 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -150,16 +150,6 @@ #define WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM #endif #endif -#ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM - /* Smallest verify RAM: on top of the small-mem path, stream the - * signature's z vector one polynomial at a time instead of holding the - * whole l-vector (~6 KB for ML-DSA-87) at the cost of a per-row z - * decode+NTT. Combine with WOLFSSL_MLDSA_VERIFY_NO_MALLOC to pin the - * buffers against the key; on its own the buffers are still allocated. */ - #ifndef WOLFSSL_MLDSA_VERIFY_SMALL_MEM - #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM - #endif -#endif #ifdef WOLFSSL_DILITHIUM_MAKE_KEY_SMALL_MEM #ifndef WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM #define WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM @@ -175,30 +165,12 @@ #define WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC #endif #endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM - /* Smallest signing RAM: on top of the small-mem path, generate matrix A a - * column at a time so only one polynomial of y is held, decompose w into - * w0 in place and keep w1 encoded. Vector y is regenerated for z. */ - #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #endif -#endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC - #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #endif -#endif #ifdef WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC_A #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A #define WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A \ WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC_A #endif #endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #endif -#endif #ifdef WOLFSSL_DILITHIUM_SIGN_CHECK_W0 #ifndef WOLFSSL_MLDSA_SIGN_CHECK_W0 #define WOLFSSL_MLDSA_SIGN_CHECK_W0 @@ -348,6 +320,37 @@ /* === Derived canonical gates ========================================== */ +/* Canonical option implications. These derive one canonical option from + * another and must apply whether or not the legacy name gates are enabled. */ +#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM + /* Smallest signing RAM: on top of the small-mem path, generate matrix A a + * column at a time so only one polynomial of y is held, decompose w into + * w0 in place and keep w1 encoded. Vector y is regenerated for z. */ + #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + #endif +#endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC + #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + #endif +#endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A + #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + #endif +#endif +#ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM + /* Smallest verify RAM: on top of the small-mem path, stream the + * signature's z vector one polynomial at a time instead of holding the + * whole l-vector (~6 KB for ML-DSA-87) at the cost of a per-row z + * decode+NTT. Combine with WOLFSSL_MLDSA_VERIFY_NO_MALLOC to pin the + * buffers against the key; on its own the buffers are still allocated. */ + #ifndef WOLFSSL_MLDSA_VERIFY_SMALL_MEM + #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM + #endif +#endif + /* Derive secondary canonical gates from the primary NO_* gates. Lives in * this file (rather than in wc_mldsa.h alongside the struct definition) * so the reverse arm at the bottom of this file sees the derived set @@ -378,6 +381,8 @@ !defined(WOLFSSL_MLDSA_NO_SIGN) #define WOLFSSL_MLDSA_PRIVATE_KEY #endif +/* wc_CheckPrivateKey() needs this to match an ML-DSA certificate to its + * private key. */ #if defined(WOLFSSL_MLDSA_PUBLIC_KEY) && \ defined(WOLFSSL_MLDSA_PRIVATE_KEY) && \ !defined(WOLFSSL_MLDSA_NO_CHECK_KEY) && \ diff --git a/wolfssl/wolfcrypt/settings.h b/wolfssl/wolfcrypt/settings.h index 80151de2464..49344fc3d25 100644 --- a/wolfssl/wolfcrypt/settings.h +++ b/wolfssl/wolfcrypt/settings.h @@ -3760,7 +3760,7 @@ (defined(HAVE_CURVE25519) && defined(HAVE_CURVE25519_KEY_EXPORT)) || \ (defined(HAVE_ED448) && defined(HAVE_ED448_KEY_EXPORT)) || \ (defined(HAVE_CURVE448) && defined(HAVE_CURVE448_KEY_EXPORT)) || \ - defined(HAVE_FALCON) || defined(HAVE_DILITHIUM) || \ + defined(HAVE_FALCON) || defined(WOLFSSL_HAVE_MLDSA) || \ defined(WOLFSSL_HAVE_FRODOKEM) || \ defined(WOLFSSL_HAVE_SLHDSA) || \ (defined(WOLFSSL_HAVE_LMS) && !defined(WOLFSSL_LMS_VERIFY_ONLY)) || \ @@ -3773,7 +3773,7 @@ (defined(HAVE_CURVE25519) && defined(HAVE_CURVE25519_KEY_IMPORT)) || \ (defined(HAVE_ED448) && defined(HAVE_ED448_KEY_IMPORT)) || \ (defined(HAVE_CURVE448) && defined(HAVE_CURVE448_KEY_IMPORT)) || \ - defined(HAVE_FALCON) || defined(HAVE_DILITHIUM) || \ + defined(HAVE_FALCON) || defined(WOLFSSL_HAVE_MLDSA) || \ defined(WOLFSSL_HAVE_FRODOKEM) || \ defined(WOLFSSL_HAVE_SLHDSA) || \ (defined(WOLFSSL_HAVE_LMS) && !defined(WOLFSSL_LMS_VERIFY_ONLY)) || \ @@ -5348,10 +5348,17 @@ blinding by defining WC_BLINDING_NO_RNG_ACKNOWLEDGE_WEAKNESS." #error Experimental settings without WOLFSSL_EXPERIMENTAL_SETTINGS #endif -/* If no malloc then make sure the valid Dilithium settings are used */ -#if defined(HAVE_DILITHIUM) && defined(WOLFSSL_NO_MALLOC) - #undef WOLFSSL_DILITHIUM_VERIFY_NO_MALLOC - #define WOLFSSL_DILITHIUM_VERIFY_NO_MALLOC +/* If no malloc then make sure the valid ML-DSA settings are used */ +#if defined(WOLFSSL_HAVE_MLDSA) && defined(WOLFSSL_NO_MALLOC) + #undef WOLFSSL_MLDSA_VERIFY_NO_MALLOC + #define WOLFSSL_MLDSA_VERIFY_NO_MALLOC + /* The pinned buffers only exist under the small memory verify; without it + * every verification fails with MEMORY_E. */ + #if !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC) + #undef WOLFSSL_MLDSA_VERIFY_SMALL_MEM + #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM + #endif #endif #if defined(WOLFSSL_HAVE_MLKEM) && \ From c988e4f3cc611d56ce37fea3750ac8e5183f9453 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:47 +0200 Subject: [PATCH 08/11] Fix the guard and CI regressions from the range check rework Narrowing the range check gates left three configurations unbuildable. mldsa_vec_check_low() and mldsa_vec_check_low_c() lost their last caller when signing moved to the constant time helpers, but their gate still carried the sign clause, so --enable-mldsa=make,sign and VERIFY_SMALLEST_MEM paired with the default signer compiled a static function with no references. The hardened CFLAGS put -Wunused-function after the -Wno-unused in AM_CFLAGS, so that is a build failure rather than a warning. Reduce both gates to the verify condition, and gate the constant time vector form on the two arms that actually call it. The MC/DC whitebox test had its inner guard updated to match the library but not its outer guard or its call site, so SIGN_SMALLEST_MEM with NO_VERIFY referenced a function that is no longer compiled. Mirror the library condition in both places. The pq-all row named mldsa-smallest-checks-no-verify set VERIFY_SMALLEST_MEM rather than NO_VERIFY, so it neither matched its name nor reached the case it was added for. Bind the two derivations of the signing mask unconditionally. The smallest memory signer is the only path that derives y twice per rejection round, and the check that the second derivation matches the first was gated on WC_MLDSA_FAULT_HARDEN. A glitch in the second derivation yields z computed from one mask against a commitment computed from another, which is the shape a fault attack on the private vector needs. Compile the check always and accumulate the checksum with FNV-1a so a single altered coefficient changes the whole result. Signing in that mode is 2.6 percent slower; no other path derives y twice. Scrub the caller's signature buffer when signing fails. The commit hash and each accepted polynomial of z are written into it as they are produced, but the round is only accepted after later checks, so an error return could leave part of a signature behind. The fault hardening check added to the small memory column walk tested the loop variable against its own bound, which can never hold, and sat after the copy it was meant to guard. Validate the derived pointer before the access instead. Range check WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A so that a definition with no value produces the intended error rather than a preprocessor syntax error, move the canonical option implications inside the ML-DSA guard, default the ML-DSA operations when --enable-mldsa names only a parameter set, and add CI rows for sign only, smallest memory verify with the default signer, and small memory signing with only the w0 check. --- .github/configs/pq-all.json | 49 ++++++++++++- .github/workflows/wolfCrypt-Wconversion.yml | 4 +- configure.ac | 13 ++++ tests/unit-mcdc/test_wc_mldsa_whitebox.c | 13 ++-- wolfcrypt/src/wc_mldsa.c | 80 ++++++++++++--------- wolfssl/wolfcrypt/dilithium.h | 16 ++--- 6 files changed, 119 insertions(+), 56 deletions(-) diff --git a/.github/configs/pq-all.json b/.github/configs/pq-all.json index 3d2ff936a6e..9659ab8e756 100644 --- a/.github/configs/pq-all.json +++ b/.github/configs/pq-all.json @@ -239,14 +239,57 @@ "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]}, {"name": "mldsa-44-precalc-a-intelasm", "minutes": 2, "comment": "ML-DSA-44 with one pre-calculated row of matrix A on the Intel assembly paths; the q88 decompose and w1 encode kernels take no dimension, so this pins that w is sized for a full vector", - "configure": ["--enable-cryptonly", "--enable-mldsa", "--enable-intelasm", + "configure": ["--enable-cryptonly", "--enable-mldsa=44", "--enable-intelasm", "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]}, +{"name": "mldsa-precalc-a-saturated", "minutes": 2, + "comment": "Pre-calculated matrix A with more rows than any parameter set has, so maxK saturates at k and the streaming loops do not run at all", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=8"]}, +{"name": "mldsa-precalc-a-split", "minutes": 2, + "comment": "Pre-calculated matrix A split part way, which is neither the single row nor the saturated case", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=2"]}, +{"name": "mldsa-checks-default-signer", "minutes": 2, + "comment": "The y and w0 rejection checks on the default signer, the only combination that compiles the vector range check helper and its two call sites", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]}, +{"name": "mldsa-cache-pub-vectors-verify-no-malloc", "minutes": 2, + "comment": "Cached public vectors alongside the pinned verify buffers, the combination whose key structure member clash stopped it building; the clashing member only exists when the small memory verify is selected too", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_PUB_VECTORS -DWOLFSSL_MLDSA_VERIFY_NO_MALLOC -DWOLFSSL_MLDSA_VERIFY_SMALL_MEM"]}, +{"name": "mldsa-cache-matrix-precalc-a", "minutes": 2, + "comment": "Cached matrix A with a pre-calculated row, the combination the signing cache allocation test gates on: only the full vector signer assigns the cache", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_MATRIX_A -DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]}, +{"name": "mldsa-no-check-key-small-mem", "minutes": 2, + "comment": "Key pair checking compiled out alongside the small memory key generation and the smallest memory signer, which leaves the whole vector helpers with no caller", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_NO_CHECK_KEY -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM -DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM"]}, +{"name": "mldsa-verify-only-no-malloc", "minutes": 2, + "comment": "Heapless verify-only ML-DSA named only by the canonical options; the no-malloc gate must select the pinned small memory verify or every verification fails with MEMORY_E", + "configure": ["--enable-cryptonly", "--enable-mldsa=verify-only", + "CPPFLAGS=-DWOLFSSL_NO_DILITHIUM_LEGACY_GATES -DWOLFSSL_NO_MALLOC"]}, +{"name": "mldsa-no-check-key", "minutes": 3, + "comment": "Key pair checking compiled out, which is the only build that reaches the NOT_COMPILED_IN arm of wc_CheckPrivateKey; needs the OpenSSL compatibility layer so the X509_check_private_key callers and their tests are compiled too", + "configure": ["--enable-opensslall", "--enable-mldsa", "--enable-certgen", + "--enable-certreq", "CPPFLAGS=-DWOLFSSL_MLDSA_NO_CHECK_KEY"]}, {"name": "mldsa-no-legacy-gates-smallest", "minutes": 2, "comment": "Canonical option names only; pins that SIGN_SMALLEST_MEM still implies SIGN_SMALL_MEM when the legacy name gates are opted out", "configure": ["--enable-cryptonly", "--enable-mldsa", "CPPFLAGS=-DWOLFSSL_NO_DILITHIUM_LEGACY_GATES -DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM"]}, {"name": "mldsa-smallest-checks-no-verify", "minutes": 2, - "comment": "Smallest memory signing with the y and w0 checks and no verify, the pairing that pulls in the vector range check helper it does not use", + "comment": "Smallest memory signing with the y and w0 checks and verify compiled out, which leaves neither vector range check helper with a caller", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0 -DWOLFSSL_MLDSA_NO_VERIFY"]}, +{"name": "mldsa-sign-only", "minutes": 2, + "comment": "Default signer with verify compiled out; the non constant time vector range check has no caller here", + "configure": ["--enable-cryptonly", "--enable-mldsa=make,sign"]}, +{"name": "mldsa-verify-smallest-default-sign", "minutes": 2, + "comment": "Smallest memory verify paired with the default signer, a combination every other row avoids", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM"]}, +{"name": "mldsa-small-check-w0-only", "minutes": 2, + "comment": "Small memory signing with the w0 check but not the y check, which decides whether the vector range check helper is reachable", "configure": ["--enable-cryptonly", "--enable-mldsa", - "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0 -DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM"]} + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]} ] diff --git a/.github/workflows/wolfCrypt-Wconversion.yml b/.github/workflows/wolfCrypt-Wconversion.yml index 832c033c689..789886ee0f7 100644 --- a/.github/workflows/wolfCrypt-Wconversion.yml +++ b/.github/workflows/wolfCrypt-Wconversion.yml @@ -104,7 +104,7 @@ jobs: "--disable-crypttests", "--enable-mlkem", "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", "--enable-xmss", - "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_NO_VERIFY -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual"], + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_NO_VERIFY -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual -Wdeclaration-after-statement"], "check": false}, {"name": "precalc-a-intelasm", "minutes": 1, "configure": ["--enable-cryptonly", "--enable-all-crypto", @@ -112,7 +112,7 @@ jobs: "--disable-crypttests", "--enable-mlkem", "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", "--enable-xmss", - "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1 -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual"], + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1 -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual -Wdeclaration-after-statement"], "check": false}, {"name": "precalc-a-no-int128", "minutes": 1, "configure": ["--enable-cryptonly", "--enable-all-crypto", diff --git a/configure.ac b/configure.ac index cbbcb77a948..4ed06c38632 100644 --- a/configure.ac +++ b/configure.ac @@ -2423,6 +2423,19 @@ then ENABLED_MLDSA87=yes fi +# Likewise, naming only a parameter set, for example --enable-mldsa=44, +# selects no operation, which would disable everything. verify-only names an +# operation, so it is unaffected. +if test "$ENABLED_MLDSA" != "no" && \ + test "$ENABLED_MLDSA_MAKE_KEY" = "no" && \ + test "$ENABLED_MLDSA_SIGN" = "no" && \ + test "$ENABLED_MLDSA_VERIFY" = "no" +then + ENABLED_MLDSA_MAKE_KEY=yes + ENABLED_MLDSA_SIGN=yes + ENABLED_MLDSA_VERIFY=yes +fi + # XMSS AC_ARG_ENABLE([xmss], [AS_HELP_STRING([--enable-xmss],[Enable stateful XMSS/XMSS^MT signatures (default: disabled)])], diff --git a/tests/unit-mcdc/test_wc_mldsa_whitebox.c b/tests/unit-mcdc/test_wc_mldsa_whitebox.c index a7fe9d9eaeb..a696fb5e56d 100644 --- a/tests/unit-mcdc/test_wc_mldsa_whitebox.c +++ b/tests/unit-mcdc/test_wc_mldsa_whitebox.c @@ -219,7 +219,7 @@ static void wb_get_params(void) * Drive independence pairs: both F (in range), left T (a<=nhi), * right T with left F (a>=hi). Plus the vector-level (ret==1)&&(i=hi) expected 0"); } -#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) && \ - (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))))) +#if !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) /* Vector level: two polynomials, both in range -> (ret==1)&&(i early ret 0. */ for (j = 0; j < 2 * MLDSA_N; j++) { @@ -1428,7 +1423,7 @@ int main(void) return 0; #else wb_get_params(); -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || !defined(WOLFSSL_MLDSA_NO_VERIFY) +#ifndef WOLFSSL_MLDSA_NO_VERIFY wb_check_low(); #endif #ifndef WOLFSSL_MLDSA_NO_SIGN diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 57bc9b8ce3f..e78ce292b56 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -215,7 +215,7 @@ #error "PRECALC and PRECALC_A are equivalent to non small mem" #endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) && \ - (WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A < 1) + ((WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A + 0) < 1) #error "PRECALC_A must pre-calculate at least one row of matrix A" #endif #ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM @@ -5797,9 +5797,10 @@ static int mldsa_check_low_ct(const sword32* a, sword32 hi) /* The smallest memory signer holds one polynomial at a time and uses the * scalar form for these checks. */ -#if (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_W0)) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) +#if (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM)) || \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) /* Check that the values of the vector are in range, in constant time. * * @param [in] a Vector of polynomials. @@ -5821,15 +5822,11 @@ static int mldsa_vec_check_low_ct(const sword32* a, byte l, sword32 hi) return good; } -#endif /* (CHECK_Y || CHECK_W0) && !SIGN_SMALLEST_MEM */ +#endif /* (CHECK_Y && !SIGN_SMALLEST_MEM) || + * (CHECK_W0 && !SIGN_SMALL_MEM) */ #endif /* !WOLFSSL_MLDSA_NO_SIGN */ -#if !defined(WOLFSSL_MLDSA_NO_VERIFY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) && \ - (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))))) +#ifndef WOLFSSL_MLDSA_NO_VERIFY /* Check that the values of the polynomial are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -5869,13 +5866,8 @@ static int mldsa_check_low(const sword32* a, sword32 hi) return (int)(in >> 31); } -#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) && \ - (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) || \ - defined(WOLFSSL_MLDSA_SIGN_CHECK_W0))))) +#if !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) /* Check that the values of the vector are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -8590,12 +8582,13 @@ static void mldsa_poly_red(sword32* a) } } -#if defined(WC_MLDSA_FAULT_HARDEN) && \ +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) /* Checksum of a polynomial. * * Binds two computations of a value that is not kept between them, so a fault - * injected into the second is detectable. + * or glitch in the second is detectable. FNV-1a is used so that a single + * altered coefficient changes the whole result. * * @param [in] a Polynomial to checksum. * @return Checksum of the polynomial. @@ -8603,10 +8596,10 @@ static void mldsa_poly_red(sword32* a) static sword32 mldsa_poly_checksum(const sword32* a) { unsigned int i; - word32 chk = 0; + word32 chk = 2166136261U; for (i = 0; i < MLDSA_N; i++) { - chk = (chk * 3) ^ (word32)a[i]; + chk = (chk ^ (word32)a[i]) * 16777619U; } return (sword32)chk; @@ -9849,6 +9842,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #ifdef WOLFSSL_CHECK_MEM_ZERO wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); #endif + /* Parts of the commit and z are written into the caller's buffer as they + * are produced. Do not leave them there on failure. The length is only + * set once the buffer was known to be big enough. */ + if ((ret != 0) && (*sigLen == params->sigSz)) { + ForceZero(sig, params->sigSz); + } if (y != NULL) { word32 zeroSz = allocSz; #ifndef WC_MLDSA_CACHE_MATRIX_A @@ -10062,6 +10061,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, * of y is transformed once rather than once per row. */ for (s = 0; (ret == 0) && valid && (s < params->l); s++) { unsigned int e; + #ifdef WC_MLDSA_FAULT_HARDEN + const sword32* yc = y + (unsigned int)s * MLDSA_N; + #endif #ifdef WC_MLDSA_FAULT_HARDEN if (y_check != y) { @@ -10070,17 +10072,19 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, break; } #endif - /* Step 13: NTT(y) for this column of the matrix. */ - XMEMCPY(y_ntt_t, y + (unsigned int)s * MLDSA_N, - MLDSA_POLY_SIZE); - mldsa_ntt_full(y_ntt_t); #ifdef WC_MLDSA_FAULT_HARDEN - if (s >= params->l) { + /* The polynomial of y this column reads must lie inside y. */ + if ((yc < y) || + (yc > y + (unsigned int)(params->l - 1) * MLDSA_N)) { valid = 0; ret = BAD_COND_E; break; } #endif + /* Step 13: NTT(y) for this column of the matrix. */ + XMEMCPY(y_ntt_t, y + (unsigned int)s * MLDSA_N, + MLDSA_POLY_SIZE); + mldsa_ntt_full(y_ntt_t); /* Put s into buffer to be hashed. */ aseed[MLDSA_PUB_SEED_SZ + 0] = s; @@ -10397,6 +10401,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #ifdef WOLFSSL_CHECK_MEM_ZERO wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); #endif + /* Parts of the commit and z are written into the caller's buffer as they + * are produced. Do not leave them there on failure. The length is only + * set once the buffer was known to be big enough. */ + if ((ret != 0) && (*sigLen == params->sigSz)) { + ForceZero(sig, params->sigSz); + } if (y != NULL) { ForceZero(y, allocSz); } @@ -10426,9 +10436,10 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, unsigned int w1Stride = (unsigned int)params->w1EncSz / params->k; /* Bytes of encoded s1 or s2 per polynomial. */ unsigned int sStride = (unsigned int)params->s2EncSz / params->k; + /* Checksum of each polynomial of y, to bind the two derivations. */ + sword32 yChk[MLDSA_MAX_L_VECTOR_COUNT / MLDSA_N]; #ifdef WC_MLDSA_FAULT_HARDEN sword32* w_check; - sword32 yChk[MLDSA_MAX_L_VECTOR_COUNT / MLDSA_N]; #endif /* priv_rand_seed will hold the secret signing seed (rho'') derived below; @@ -10517,10 +10528,8 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, if (ret != 0) { break; } - #ifdef WC_MLDSA_FAULT_HARDEN /* Bind this polynomial of y to the one regenerated for z. */ yChk[s] = mldsa_poly_checksum(y); - #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y valid = mldsa_check_low_ct(y, ((sword32)1 << params->gamma1_bits) - params->beta); @@ -10649,13 +10658,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, if (ret != 0) { break; } - #ifdef WC_MLDSA_FAULT_HARDEN + /* y must match the derivation w was built from. */ if (yChk[s] != mldsa_poly_checksum(y)) { valid = 0; ret = BAD_COND_E; break; } - #endif mldsa_vec_decode_eta_bits(sp, params->eta, a, 1); mldsa_ntt_small(a); /* Step 19: cs1 = NTT-1(c o s1) */ @@ -10783,10 +10791,14 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #ifdef WOLFSSL_CHECK_MEM_ZERO wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); #endif -#ifdef WC_MLDSA_FAULT_HARDEN + /* Parts of the commit and z are written into the caller's buffer as they + * are produced. Do not leave them there on failure. The length is only + * set once the buffer was known to be big enough. */ + if ((ret != 0) && (*sigLen == params->sigSz)) { + ForceZero(sig, params->sigSz); + } /* Checksums are derived from the secret mask y. */ ForceZero(yChk, sizeof(yChk)); -#endif if (w != NULL) { ForceZero(w, allocSz); } diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index a661ce6cb9e..29d27ba1eb3 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -320,6 +320,14 @@ /* === Derived canonical gates ========================================== */ +/* Derive secondary canonical gates from the primary NO_* gates. Lives in + * this file (rather than in wc_mldsa.h alongside the struct definition) + * so the reverse arm at the bottom of this file sees the derived set + * fully populated without needing wc_mldsa.h to finish parsing first. + * wc_mldsa.h includes this file at its top, so by the time control + * returns from that include the gates are already set and wc_mldsa.h's + * struct definition / conditional declarations read them directly. */ +#if defined(WOLFSSL_HAVE_MLDSA) /* Canonical option implications. These derive one canonical option from * another and must apply whether or not the legacy name gates are enabled. */ #ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM @@ -351,14 +359,6 @@ #endif #endif -/* Derive secondary canonical gates from the primary NO_* gates. Lives in - * this file (rather than in wc_mldsa.h alongside the struct definition) - * so the reverse arm at the bottom of this file sees the derived set - * fully populated without needing wc_mldsa.h to finish parsing first. - * wc_mldsa.h includes this file at its top, so by the time control - * returns from that include the gates are already set and wc_mldsa.h's - * struct definition / conditional declarations read them directly. */ -#if defined(WOLFSSL_HAVE_MLDSA) #if defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ From 3f18f924318484d90a4e6ce35679f0497c236ee6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:47 +0200 Subject: [PATCH 09/11] Keep the ML-DSA option derivation include-order independent Moving the canonical option implications inside the WOLFSSL_HAVE_MLDSA guard made the wc_MlDsaKey layout depend on include order. dilithium.h has a single include and it sits after that guard, so a translation unit that reaches dilithium.h before any settings-bearing header evaluates the guard with WOLFSSL_HAVE_MLDSA still undefined and never derives WOLFSSL_MLDSA_VERIFY_SMALL_MEM. The verify scratch tail of the key structure is gated on that macro together with WOLFSSL_MLDSA_VERIFY_NO_MALLOC, so the library would be built with the full structure while an application that includes dilithium.h first gets one without it, and verification then writes past the end of the caller's key. Derive the implications unconditionally again, and do it in wc_mldsa.h straight after its include of dilithium.h rather than in the legacy compatibility shim: every route to the key structure passes through that point, and the implications have to outlive the shim. Checksum the signing mask with rotation and exclusive-or rather than a multiplicative hash. The coefficients are secret, and a multiply chain over them is not constant time on cores with an operand dependent multiplier, which is what the smallest memory option targets. It also hands power analysis a clean per-coefficient hypothesis. Rotating keeps a changed coefficient's position significant, which is all the fault check needs. Clear the rows of w that PRECALC_A does not multiply into. The ML-DSA-44 decompose kernels take no dimension and touch all k rows, so the rows above the pre-calculated ones were read before being written. Cover the constant time range checks in the ML-DSA whitebox test: both boundaries of the accepted range, and a bad coefficient in the last position of a polynomial and in the last polynomial of a vector, which only fail because neither helper exits early. --- tests/unit-mcdc/test_wc_mldsa_whitebox.c | 151 ++++++++++++++++++++++- wolfcrypt/src/wc_mldsa.c | 137 +++++++++++--------- wolfssl/wolfcrypt/dilithium.h | 36 +----- wolfssl/wolfcrypt/wc_mldsa.h | 13 ++ 4 files changed, 248 insertions(+), 89 deletions(-) diff --git a/tests/unit-mcdc/test_wc_mldsa_whitebox.c b/tests/unit-mcdc/test_wc_mldsa_whitebox.c index a696fb5e56d..44576919ce9 100644 --- a/tests/unit-mcdc/test_wc_mldsa_whitebox.c +++ b/tests/unit-mcdc/test_wc_mldsa_whitebox.c @@ -251,8 +251,7 @@ static void wb_check_low(void) WB_NOTE("mldsa_check_low(>=hi) expected 0"); } -#if !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) +#ifndef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM /* Vector level: two polynomials, both in range -> (ret==1)&&(i early ret 0. */ for (j = 0; j < 2 * MLDSA_N; j++) { @@ -272,6 +271,148 @@ static void wb_check_low(void) } #endif +#ifndef WOLFSSL_MLDSA_NO_SIGN +/* ------------------------------------------------------------------ * + * mldsa_check_low_ct / mldsa_vec_check_low_ct: the constant time forms + * gate every signing range check, so drive the four boundary values and + * confirm no early exit hides a later failure. + * ------------------------------------------------------------------ */ +static void wb_check_low_ct(void) +{ + sword32 a[2 * MLDSA_N]; + sword32 hi = 1 << 17; + unsigned int j; + int ret = 0; + + for (j = 0; j < 2 * MLDSA_N; j++) { + a[j] = 0; + } + + /* Boundaries: hi-1 and -hi+1 are in range, hi and -hi are not. */ + a[0] = hi - 1; + if (mldsa_check_low_ct(a, hi) != 1) { + WB_NOTE("mldsa_check_low_ct(hi-1) expected 1"); + } + a[0] = -hi + 1; + if (mldsa_check_low_ct(a, hi) != 1) { + WB_NOTE("mldsa_check_low_ct(-hi+1) expected 1"); + } + a[0] = hi; + if (mldsa_check_low_ct(a, hi) != 0) { + WB_NOTE("mldsa_check_low_ct(hi) expected 0"); + } + a[0] = -hi; + if (mldsa_check_low_ct(a, hi) != 0) { + WB_NOTE("mldsa_check_low_ct(-hi) expected 0"); + } + + /* The last coefficient must count: an early exit would miss it. */ + a[0] = 0; + a[MLDSA_N - 1] = hi; + if (mldsa_check_low_ct(a, hi) != 0) { + WB_NOTE("mldsa_check_low_ct(last coeff) expected 0"); + } + a[MLDSA_N - 1] = 0; + + /* Agreement with the branching form wherever both are compiled. */ +#if !defined(WOLFSSL_MLDSA_NO_VERIFY) + for (j = 0; j < MLDSA_N; j++) { + a[j] = (j & 1) ? (hi - 1) : (-hi + 1); + } + ret = mldsa_check_low(a, hi); + if (mldsa_check_low_ct(a, hi) != ret) { + WB_NOTE("check_low_ct disagrees with check_low (in range)"); + } + a[MLDSA_N / 2] = hi; + ret = mldsa_check_low(a, hi); + if (mldsa_check_low_ct(a, hi) != ret) { + WB_NOTE("check_low_ct disagrees with check_low (out of range)"); + } +#else + (void)ret; +#endif + +#if (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM)) || \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + /* Vector form must fail on a bad coefficient in the LAST polynomial, + * which only holds because it does not exit early. */ + for (j = 0; j < 2 * MLDSA_N; j++) { + a[j] = 0; + } + if (mldsa_vec_check_low_ct(a, 2, hi) != 1) { + WB_NOTE("mldsa_vec_check_low_ct(in-range,l=2) expected 1"); + } + a[2 * MLDSA_N - 1] = hi; + if (mldsa_vec_check_low_ct(a, 2, hi) != 0) { + WB_NOTE("mldsa_vec_check_low_ct(last poly) expected 0"); + } +#endif + + WB_OK("mldsa_check_low_ct / vec_check_low_ct boundaries exercised"); +} +#endif + +#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM +/* ------------------------------------------------------------------ * + * mldsa_poly_checksum: the smallest memory signer binds the polynomial of y + * it regenerates for z to the one it used for w, raising BAD_COND_E when they + * differ. That branch needs a fault to reach, so the property is tested + * directly: any single changed coefficient must change the checksum. + * ------------------------------------------------------------------ */ +static void wb_poly_checksum(void) +{ + sword32 a[MLDSA_N]; + sword32 base; + unsigned int j; + unsigned int pos[4]; + unsigned int p; + + for (j = 0; j < MLDSA_N; j++) { + a[j] = (sword32)(j * 7); + } + base = mldsa_poly_checksum(a); + + /* Same input, same checksum: the comparison in the signer only fires + * on a real difference. */ + if (mldsa_poly_checksum(a) != base) { + WB_NOTE("mldsa_poly_checksum is not deterministic"); + } + + /* First, last and two interior coefficients. The rotate in the + * checksum is what makes position matter. */ + pos[0] = 0; + pos[1] = 1; + pos[2] = MLDSA_N / 2; + pos[3] = MLDSA_N - 1; + for (p = 0; p < 4; p++) { + sword32 keep = a[pos[p]]; + + a[pos[p]] = keep ^ 1; + if (mldsa_poly_checksum(a) == base) { + WB_NOTE("mldsa_poly_checksum missed a changed coefficient"); + } + a[pos[p]] = keep; + } + + /* Two coefficients swapped: same multiset, different polynomial. */ + if (MLDSA_N >= 2) { + sword32 k0 = a[0]; + + a[0] = a[1]; + a[1] = k0; + if (mldsa_poly_checksum(a) == base) { + WB_NOTE("mldsa_poly_checksum missed a reordering"); + } + a[1] = a[0]; + a[0] = k0; + } + + WB_OK("mldsa_poly_checksum change detection exercised"); +} +#endif /* WOLFSSL_MLDSA_SIGN_SMALLEST_MEM */ + /* ------------------------------------------------------------------ * * mldsa_check_hint: two inner loop decisions that the 3-outcome test * above never reaches because their FALSE/TRUE pair only shows up with @@ -1426,6 +1567,12 @@ int main(void) #ifndef WOLFSSL_MLDSA_NO_VERIFY wb_check_low(); #endif +#ifndef WOLFSSL_MLDSA_NO_SIGN + wb_check_low_ct(); +#endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM + wb_poly_checksum(); +#endif #ifndef WOLFSSL_MLDSA_NO_SIGN #ifndef WOLFSSL_NO_ML_DSA_44 wb_make_hint_88(); diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index e78ce292b56..0bcee8cb60a 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -55,6 +55,10 @@ * WOLFSSL_MLDSA_VERIFY_NO_MALLOC Default: OFF * Only works with WOLFSSL_MLDSA_VERIFY_SMALL_MEM. * Don't allocate memory with XMALLOC. Memory is pinned against key. + * WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC Default: OFF + * Declines the small memory verify and pinned buffers that WOLFSSL_NO_MALLOC + * selects automatically, keeping the default verify and a key about 12kB + * smaller. Only for a build whose XMALLOC does work. * WOLFSSL_MLDSA_ASSIGN_KEY Default: OFF * Key data is assigned into ML-DSA key rather than copied. * Life of key data passed in is tightly coupled to life of ML-DSA key. @@ -5270,39 +5274,34 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, } else #endif - if (IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + /* Each generator handles one dimension only, so the dimension is tested + * before the vector registers are saved: anything else, such as the + * single polynomial the smallest memory signing asks for, goes straight + * to the C implementation rather than save and restore around no work. */ +#ifndef WOLFSSL_NO_ML_DSA_44 + if ((l == 4) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { - /* Each generator handles one dimension only, so anything else, such - * as the single polynomial the smallest memory signing asks for, - * falls through to the C implementation. */ - byte done = 0; - - #ifndef WOLFSSL_NO_ML_DSA_44 - if (l == 4) { - ret = wc_mldsa_gen_y_4_avx2(y, seed, kappa); - done = 1; - } - #endif - #ifndef WOLFSSL_NO_ML_DSA_65 - if (l == 5) { - ret = wc_mldsa_gen_y_5_avx2(y, seed, kappa, shake256); - done = 1; - } - #endif - #ifndef WOLFSSL_NO_ML_DSA_87 - if (l == 7) { - ret = wc_mldsa_gen_y_7_avx2(y, seed, kappa); - done = 1; - } - #endif + ret = wc_mldsa_gen_y_4_avx2(y, seed, kappa); + RESTORE_VECTOR_REGISTERS(); + } + else +#endif +#ifndef WOLFSSL_NO_ML_DSA_65 + if ((l == 5) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + (SAVE_VECTOR_REGISTERS2() == 0)) { + ret = wc_mldsa_gen_y_5_avx2(y, seed, kappa, shake256); RESTORE_VECTOR_REGISTERS(); - - if (!done) { - ret = mldsa_vec_expand_mask_c(shake256, seed, kappa, gamma1_bits, - y, l); - } } else +#endif +#ifndef WOLFSSL_NO_ML_DSA_87 + if ((l == 7) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + (SAVE_VECTOR_REGISTERS2() == 0)) { + ret = wc_mldsa_gen_y_7_avx2(y, seed, kappa); + RESTORE_VECTOR_REGISTERS(); + } + else +#endif #endif { ret = mldsa_vec_expand_mask_c(shake256, seed, kappa, gamma1_bits, y, @@ -5822,7 +5821,7 @@ static int mldsa_vec_check_low_ct(const sword32* a, byte l, sword32 hi) return good; } -#endif /* (CHECK_Y && !SIGN_SMALLEST_MEM) || +#endif /* (CHECK_Y && !SIGN_SMALLEST_MEM) || * (CHECK_W0 && !SIGN_SMALL_MEM) */ #endif /* !WOLFSSL_MLDSA_NO_SIGN */ @@ -8587,8 +8586,9 @@ static void mldsa_poly_red(sword32* a) /* Checksum of a polynomial. * * Binds two computations of a value that is not kept between them, so a fault - * or glitch in the second is detectable. FNV-1a is used so that a single - * altered coefficient changes the whole result. + * in the second is detectable. Rotates and exclusive-ors rather than + * multiplying: the coefficients are secret, and a multiply chain over them is + * not constant time on cores with an operand dependent multiplier. * * @param [in] a Polynomial to checksum. * @return Checksum of the polynomial. @@ -8596,10 +8596,11 @@ static void mldsa_poly_red(sword32* a) static sword32 mldsa_poly_checksum(const sword32* a) { unsigned int i; - word32 chk = 2166136261U; + word32 chk = 0; for (i = 0; i < MLDSA_N; i++) { - chk = (chk ^ (word32)a[i]) * 16777619U; + /* Rotate so that the position of a changed coefficient matters. */ + chk = ((chk << 1) | (chk >> 31)) ^ (word32)a[i]; } return (sword32)chk; @@ -8762,7 +8763,8 @@ static void mldsa_add(sword32* r, const sword32* a) } } -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) @@ -8840,7 +8842,8 @@ static void mldsa_make_pos(sword32* a) } } -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ defined(MLDSA_SIGN_VEC_HELPERS) /* Make values in polynomials of vector be in positive range. @@ -9948,6 +9951,14 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, w0 = y + params->s1Sz / sizeof(*y_ntt); w = w0 + params->s2Sz / sizeof(*w0); #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + /* Only maxK rows of w are multiplied into, but the ML-DSA-44 + * decompose kernels take no dimension and touch all k. One + * zeroing covers every rejection round: decompose of zero is + * (0, 0), so the rows from maxK up keep the value they hold. */ + if (maxK < params->k) { + XMEMSET(w + (unsigned int)maxK * MLDSA_N, 0, + (size_t)(params->k - maxK) * MLDSA_POLY_SIZE); + } c = w + (unsigned int)params->k * MLDSA_N; #else c = w + MLDSA_N; @@ -10015,6 +10026,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, byte* w1et = w1e; sword32* at = a; sword32* y_ntt_t; + #ifdef WC_MLDSA_FAULT_HARDEN + const sword32* yc = y; + #endif #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A w0t += (unsigned int)maxK * MLDSA_N; @@ -10057,13 +10071,11 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, y_ntt_t = y_ntt; #endif /* Alg 26. Step 2: Loop over second dimension of matrix. - * The matrix is walked a column at a time so that each polynomial - * of y is transformed once rather than once per row. */ - for (s = 0; (ret == 0) && valid && (s < params->l); s++) { + * A column at a time transforms each polynomial of y once. With + * every row pre-calculated there is nothing left to stream. */ + for (s = 0; (ret == 0) && valid && (rStart < params->k) && + (s < params->l); s++) { unsigned int e; - #ifdef WC_MLDSA_FAULT_HARDEN - const sword32* yc = y + (unsigned int)s * MLDSA_N; - #endif #ifdef WC_MLDSA_FAULT_HARDEN if (y_check != y) { @@ -10071,11 +10083,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, ret = BAD_COND_E; break; } - #endif - #ifdef WC_MLDSA_FAULT_HARDEN - /* The polynomial of y this column reads must lie inside y. */ - if ((yc < y) || - (yc > y + (unsigned int)(params->l - 1) * MLDSA_N)) { + /* yc walks y independently of s: a fault in either the index + * or the pointer breaks the agreement. */ + if (yc != y + (unsigned int)s * MLDSA_N) { valid = 0; ret = BAD_COND_E; break; @@ -10163,6 +10173,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } wt += MLDSA_N; } + #ifdef WC_MLDSA_FAULT_HARDEN + yc += MLDSA_N; + #endif } /* Steps 13-15: Invert transform, decompose and encode each row of @@ -10442,13 +10455,16 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sword32* w_check; #endif - /* priv_rand_seed will hold the secret signing seed (rho'') derived below; - * baseline-zero and register it up front (single-exit function) so any + /* priv_rand_seed will hold the secret signing seed (rho'') derived below + * and yChk a checksum over each polynomial of the secret mask y; + * baseline-zero and register both up front (single-exit function) so any * later exit before the ForceZero is covered. */ #ifdef WOLFSSL_CHECK_MEM_ZERO XMEMSET(priv_rand_seed, 0, sizeof(priv_rand_seed)); wc_MemZero_Add("mldsa sign priv_rand_seed", priv_rand_seed, sizeof(priv_rand_seed)); + XMEMSET(yChk, 0, sizeof(yChk)); + wc_MemZero_Add("mldsa sign yChk", yChk, sizeof(yChk)); #endif /* Check the signature buffer isn't too small. */ if (*sigLen < params->sigSz) { @@ -10514,7 +10530,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Alg 26. Step 2: Loop over second dimension of matrix. Working * down the columns keeps one polynomial of y and transforms it * once rather than once per row. */ - for (s = 0; (ret == 0) && valid && (s < params->l); s++) { + for (s = 0; (ret == 0) && (s < params->l); s++) { #ifdef WC_MLDSA_FAULT_HARDEN if (w_check != w) { valid = 0; @@ -10531,11 +10547,11 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Bind this polynomial of y to the one regenerated for z. */ yChk[s] = mldsa_poly_checksum(y); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_check_low_ct(y, + /* Accumulate rather than leave the loop: an early exit would + * make the number of polynomials of matrix A generated depend + * on the secret mask y. */ + valid &= mldsa_check_low_ct(y, ((sword32)1 << params->gamma1_bits) - params->beta); - if (!valid) { - break; - } #endif /* Step 13: NTT(y) */ mldsa_ntt_full(y); @@ -10650,6 +10666,8 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sp = s1p; hi = ((sword32)1 << params->gamma1_bits) - params->beta; + /* z and w0 - cs2 stop at the first rejecting polynomial of a + * discarded round; y masks which one it was. */ for (s = 0; (ret == 0) && valid && (s < params->l); s++) { /* Vector y is not kept, so regenerate this polynomial. */ ret = mldsa_vec_expand_mask(&key->shake, priv_rand_seed, @@ -10763,8 +10781,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, w1et += w1Stride; } valid &= ct0Valid; - /* Set remaining hints to zero. */ - XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); + /* Set remaining hints to zero. Only for a valid attempt: a + * rejected one leaves a partial count that is never + * published, and the next attempt rebuilds from index 0. */ + if (valid) { + XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); + } } if (!valid) { @@ -10799,6 +10821,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } /* Checksums are derived from the secret mask y. */ ForceZero(yChk, sizeof(yChk)); +#ifdef WOLFSSL_CHECK_MEM_ZERO + wc_MemZero_Check(yChk, sizeof(yChk)); +#endif if (w != NULL) { ForceZero(w, allocSz); } diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index 29d27ba1eb3..a8d249e1bae 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -83,6 +83,11 @@ #ifndef WOLF_CRYPT_DILITHIUM_H #define WOLF_CRYPT_DILITHIUM_H +/* Read the build configuration before deriving any option: an application + * that reaches this header before settings.h would otherwise translate an + * empty option set and get a different wc_MlDsaKey layout than the library. */ +#include + /* === Sub-config build-gate translations =============================== */ /* The two sub-gates that (auto-generated, no @@ -328,37 +333,6 @@ * returns from that include the gates are already set and wc_mldsa.h's * struct definition / conditional declarations read them directly. */ #if defined(WOLFSSL_HAVE_MLDSA) -/* Canonical option implications. These derive one canonical option from - * another and must apply whether or not the legacy name gates are enabled. */ -#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM - /* Smallest signing RAM: on top of the small-mem path, generate matrix A a - * column at a time so only one polynomial of y is held, decompose w into - * w0 in place and keep w1 encoded. Vector y is regenerated for z. */ - #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #endif -#endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC - #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #endif -#endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #endif -#endif -#ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM - /* Smallest verify RAM: on top of the small-mem path, stream the - * signature's z vector one polynomial at a time instead of holding the - * whole l-vector (~6 KB for ML-DSA-87) at the cost of a per-row z - * decode+NTT. Combine with WOLFSSL_MLDSA_VERIFY_NO_MALLOC to pin the - * buffers against the key; on its own the buffers are still allocated. */ - #ifndef WOLFSSL_MLDSA_VERIFY_SMALL_MEM - #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM - #endif -#endif - #if defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ diff --git a/wolfssl/wolfcrypt/wc_mldsa.h b/wolfssl/wolfcrypt/wc_mldsa.h index 89a50fbd760..31ecc255680 100644 --- a/wolfssl/wolfcrypt/wc_mldsa.h +++ b/wolfssl/wolfcrypt/wc_mldsa.h @@ -82,6 +82,19 @@ * legacy compatibility shim is dropped. */ #include +/* Unconditional so the key layout does not depend on include order. */ +#if (defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A)) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM +#endif +/* Allocates its buffers unless WOLFSSL_MLDSA_VERIFY_NO_MALLOC is also set. */ +#if defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM) + #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM +#endif + #if defined(WOLFSSL_HAVE_MLDSA) #include From a870148eceec07b6e88c23ed25bc49f986fc2c20 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 14:16:47 +0200 Subject: [PATCH 10/11] Finish the ML-DSA small memory guard rework and its test coverage Three whole-vector helpers are only called by the full-vector key generation, but their guards still admitted WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM, and mldsa_vec_ntt_full still carried a WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC clause with no call site behind it. Two configurations therefore compiled a static function with no references: MAKE_KEY_SMALL_MEM with NO_CHECK_KEY and the smallest memory sign and verify (mldsa_vec_invntt_full, mldsa_matrix_mul, and mldsa_vec_red once WOLFSSL_MLDSA_SMALL is added), and NO_VERIFY or VERIFY_SMALLEST_MEM with SIGN_SMALL_MEM_PRECALC (mldsa_vec_ntt_full). Neither built at all before the guard rework, so this finishes that work rather than fixing a regression. Give matrix_mul, invntt_full and vec_red the MAKE_KEY_SMALL_MEM exclusion their siblings already have, and reduce the ntt_full signing clause to MLDSA_SIGN_VEC_HELPERS, which already spells the set of call sites that exist. Accumulate the WOLFSSL_MLDSA_SIGN_CHECK_W0 result in both small memory signers instead of leaving the loop on the first rejecting row. w0 is LowBits(A o y) and so is derived from the secret mask, exactly as the y check immediately above it is, and that check already runs every column for this reason. Read the cached public vector in the small memory verify. Widening the mldsa_vec_decode_t1 guard let WC_MLDSA_CACHE_PUB_VECTORS build alongside the small memory verify, but nothing consumed the cache: key->pubVecSet was only tested by the full-vector verifier, so every public key import paid an allocation of s2Sz and a full decode and transform that was then discarded and re-done a polynomial at a time. The cached vector is already in the form the row loop wants. Mix the coefficient index into mldsa_poly_checksum. Rotation alone has period 32 over 256 coefficients, so a fault applying the same delta to coefficients 32 apart cancelled. Single coefficient faults were always caught, which is what the white-box test drives. Match the white-box guard for wb_poly_checksum to the definition guard of the function it calls, as the three neighbouring guards in that file already do. Compile test_wolfSSL_X509_check_private_key_mldsa in a WOLFSSL_MLDSA_NO_CHECK_KEY build and assert that the pair is rejected there, rather than gating the test off and leaving the NOT_COMPILED_IN arm of wc_CheckPrivateKey with no coverage. Drop the key's caches when the small memory key generation replaces a key. The default key generation clears aSet, privVecsSet and pubVecSet on success, but the small memory arm never did, so a key generated into an object that had imported or used another key kept that key's matrix A and vectors, and its signatures failed to verify. The full-vector verifier has the same exposure on master; reading the cache in the small memory verify widened it. Rework the signing KAT to generate the key into a fresh object and into one that already signed and verified with a different key, and require both to produce the same signature and to verify it. A sign and verify on the reused object alone does not catch this, because the signer and the verifier read the same stale caches and agree. The comparison needs no known answer, so it also runs for the draft and CHECK_Y/CHECK_W0 signers, which only skip the digest check. Add a pq-all row for cached matrix A on the default signer: the existing row pairs the cache with PRECALC_A, which implies the small memory signer and preprocesses away the key->a and aSet assertions the test exists for. Correct that row's comment to say what it does cover. Add another row that builds every cache with the small memory key generation, the only CI configuration where the comparison reaches the fix. --- .github/configs/pq-all.json | 10 +- ChangeLog.md | 3 +- tests/api/test_ossl_x509_crypto.c | 8 +- tests/unit-mcdc/test_wc_mldsa_whitebox.c | 6 +- wolfcrypt/src/wc_mldsa.c | 57 +++++++--- wolfcrypt/test/test.c | 131 ++++++++++++++++------- 6 files changed, 156 insertions(+), 59 deletions(-) diff --git a/.github/configs/pq-all.json b/.github/configs/pq-all.json index 9659ab8e756..ead05645b98 100644 --- a/.github/configs/pq-all.json +++ b/.github/configs/pq-all.json @@ -257,10 +257,18 @@ "comment": "Cached public vectors alongside the pinned verify buffers, the combination whose key structure member clash stopped it building; the clashing member only exists when the small memory verify is selected too", "configure": ["--enable-cryptonly", "--enable-mldsa", "CPPFLAGS=-DWC_MLDSA_CACHE_PUB_VECTORS -DWOLFSSL_MLDSA_VERIFY_NO_MALLOC -DWOLFSSL_MLDSA_VERIFY_SMALL_MEM"]}, +{"name": "mldsa-cache-matrix", "minutes": 2, + "comment": "Cached matrix A on the default full vector signer, the only build where the signing cache allocation test's key->a and aSet assertions are compiled", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_MATRIX_A"]}, {"name": "mldsa-cache-matrix-precalc-a", "minutes": 2, - "comment": "Cached matrix A with a pre-calculated row, the combination the signing cache allocation test gates on: only the full vector signer assigns the cache", + "comment": "Cached matrix A with a pre-calculated row; the small memory signer streams A, so this exercises the sign and verify round trip rather than the cache assertions", "configure": ["--enable-cryptonly", "--enable-mldsa", "CPPFLAGS=-DWC_MLDSA_CACHE_MATRIX_A -DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]}, +{"name": "mldsa-cache-small-keygen", "minutes": 2, + "comment": "Every key cache alongside the small memory key generation, which must drop the caches of the key it replaces", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_PRIV_VECTORS -DWC_MLDSA_CACHE_PUB_VECTORS -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM"]}, {"name": "mldsa-no-check-key-small-mem", "minutes": 2, "comment": "Key pair checking compiled out alongside the small memory key generation and the smallest memory signer, which leaves the whole vector helpers with no caller", "configure": ["--enable-cryptonly", "--enable-mldsa", diff --git a/ChangeLog.md b/ChangeLog.md index 218f5777408..9f6b427850e 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -12,7 +12,8 @@ * Reduced the ML-DSA small memory heap footprint: signing keeps w1 only in encoded form, key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` holds one polynomial of y, roughly halving the signing peak. by @Frauschi * Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of y is transformed once rather than once per row. `WOLFSSL_MLDSA_SMALL_MEM_POLY64` no longer applies to signing. by @Frauschi * Fixed `--enable-mldsa=` naming only a parameter set, which left key generation, signing and verification all disabled. by @Frauschi -* `wc_CheckPrivateKey()` now reports `NOT_COMPILED_IN` for an ML-DSA certificate and key when the key pair check is compiled out with `WOLFSSL_MLDSA_NO_CHECK_KEY`, rather than failing to build. Such a build cannot confirm the pair matches, so loading the two together fails. by @Frauschi +* `wc_CheckPrivateKey()` now reports `NOT_COMPILED_IN` for an ML-DSA certificate and key when the key pair check is compiled out with `WOLFSSL_MLDSA_NO_CHECK_KEY`, rather than failing to build. Such a build cannot confirm the pair matches, so `wolfSSL_CTX_check_private_key()`, `wolfSSL_check_private_key()` and `wolfSSL_X509_check_private_key()` report a mismatch for it; loading the certificate and key is unaffected. by @Frauschi +* Fixed ML-DSA key generation with `WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM` keeping the matrix and vector caches (`WC_MLDSA_CACHE_*`) of the key it replaced. A key generated into an object that already held another key was signed and verified against the old key's cached values, so its signatures failed to verify. by @Frauschi # wolfSSL Release 5.9.4 (Sep 25, 2026) diff --git a/tests/api/test_ossl_x509_crypto.c b/tests/api/test_ossl_x509_crypto.c index 056abd3467c..230cfaf3938 100644 --- a/tests/api/test_ossl_x509_crypto.c +++ b/tests/api/test_ossl_x509_crypto.c @@ -83,7 +83,6 @@ int test_wolfSSL_X509_check_private_key_mldsa(void) defined(HAVE_DILITHIUM) && !defined(WOLFSSL_DILITHIUM_NO_SIGN) && \ !defined(WOLFSSL_DILITHIUM_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_NO_ASN1) && \ - defined(WOLFSSL_MLDSA_CHECK_KEY) && \ (defined(OPENSSL_ALL) || defined(WOLFSSL_WPAS_SMALL)) && \ (!defined(WOLFSSL_NO_ML_DSA_44) || !defined(WOLFSSL_NO_ML_DSA_65) || \ !defined(WOLFSSL_NO_ML_DSA_87)) @@ -156,7 +155,14 @@ int test_wolfSSL_X509_check_private_key_mldsa(void) ExpectNotNull(x509 = X509_load_certificate_file( cases[i].certPath, SSL_FILETYPE_ASN1)); + #ifdef WOLFSSL_MLDSA_CHECK_KEY ExpectIntEQ(X509_check_private_key(x509, pkey), 1); + #else + /* Without the key pair check the match cannot be confirmed, so + * wc_CheckPrivateKey() reports NOT_COMPILED_IN and the pair is + * rejected. */ + ExpectIntEQ(X509_check_private_key(x509, pkey), 0); + #endif if (cases[i].mismatchCertPath != NULL) { ExpectNotNull(mismatchX509 = X509_load_certificate_file( diff --git a/tests/unit-mcdc/test_wc_mldsa_whitebox.c b/tests/unit-mcdc/test_wc_mldsa_whitebox.c index 44576919ce9..cc79ad9286c 100644 --- a/tests/unit-mcdc/test_wc_mldsa_whitebox.c +++ b/tests/unit-mcdc/test_wc_mldsa_whitebox.c @@ -354,7 +354,8 @@ static void wb_check_low_ct(void) } #endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) /* ------------------------------------------------------------------ * * mldsa_poly_checksum: the smallest memory signer binds the polynomial of y * it regenerates for z to the one it used for w, raising BAD_COND_E when they @@ -1570,7 +1571,8 @@ int main(void) #ifndef WOLFSSL_MLDSA_NO_SIGN wb_check_low_ct(); #endif -#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) wb_poly_checksum(); #endif #ifndef WOLFSSL_MLDSA_NO_SIGN diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 0bcee8cb60a..e97411f6e43 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -7111,10 +7111,7 @@ static void mldsa_vec_ntt(sword32* r, byte l) #if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ (!defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) || \ defined(WC_MLDSA_CACHE_PUB_VECTORS))) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ - defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A))) || \ + defined(MLDSA_SIGN_VEC_HELPERS) || \ (defined(WOLFSSL_MLDSA_SMALL) && \ (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) @@ -8100,7 +8097,8 @@ static void mldsa_invntt_full(sword32* r) } } -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ @@ -8132,7 +8130,8 @@ static void mldsa_vec_invntt_full(sword32* r, byte l) } #endif -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ @@ -8599,8 +8598,10 @@ static sword32 mldsa_poly_checksum(const sword32* a) word32 chk = 0; for (i = 0; i < MLDSA_N; i++) { - /* Rotate so that the position of a changed coefficient matters. */ - chk = ((chk << 1) | (chk >> 31)) ^ (word32)a[i]; + /* Rotate so the position of a changed coefficient matters, and mix in + * the index: rotation alone repeats every 32 of the 256 coefficients, + * so a fault applying the same delta 32 apart would cancel. */ + chk = ((chk << 1) | (chk >> 31)) ^ ((word32)a[i] + i); } return (sword32)chk; @@ -8608,7 +8609,8 @@ static sword32 mldsa_poly_checksum(const sword32* a) #endif #if (defined(WOLFSSL_MLDSA_SMALL) && \ - (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ + ((!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) || \ @@ -9327,6 +9329,16 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) /* Public key and private key are available. */ key->prvKeySet = 1; key->pubKeySet = 1; + /* Any cached matrix or vectors belong to the key this replaced. */ +#ifdef WC_MLDSA_CACHE_MATRIX_A + key->aSet = 0; +#endif +#ifdef WC_MLDSA_CACHE_PRIV_VECTORS + key->privVecsSet = 0; +#endif +#ifdef WC_MLDSA_CACHE_PUB_VECTORS + key->pubVecSet = 0; +#endif } /* Zeroize the whole buffer before freeing. It holds the private vectors @@ -10180,7 +10192,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Steps 13-15: Invert transform, decompose and encode each row of * w now that every column has been accumulated into it. */ - for (r = rStart; (ret == 0) && valid && (r < params->k); r++) { + for (r = rStart; (ret == 0) && (r < params->k); r++) { unsigned int e; /* Step 13: w = NTT-1(A o NTT(y)) */ @@ -10211,7 +10223,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_check_low_ct(w0t, + valid &= mldsa_check_low_ct(w0t, params->gamma2 - params->beta); #endif w0t += MLDSA_N; @@ -10609,7 +10621,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } wt = w; - for (r = 0; (ret == 0) && valid && (r < params->k); r++) { + for (r = 0; (ret == 0) && (r < params->k); r++) { /* Step 13: w = NTT-1(A o NTT(y)) */ mldsa_poly_red(wt); mldsa_invntt_full(wt); @@ -10638,7 +10650,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_check_low_ct(wt, + /* Accumulate: w0 comes from the secret mask, so an early exit + * would leak which row rejected. */ + valid &= mldsa_check_low_ct(wt, params->gamma2 - params->beta); #endif wt += MLDSA_N; @@ -11657,13 +11671,22 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, unsigned int e; const sword32* zt = z; - /* Step 1: Decode and NTT vector t1. */ - mldsa_decode_t1(t1p, w); +#ifdef WC_MLDSA_CACHE_PUB_VECTORS + if (key->pubVecSet) { + /* Cached vector is already decoded and transformed. */ + XMEMCPY(w, key->t1 + (size_t)r * MLDSA_N, + sizeof(sword32) * MLDSA_N); + } + else +#endif + { + /* Step 1: Decode and NTT vector t1. */ + mldsa_decode_t1(t1p, w); + mldsa_ntt_full(w); + } /* Next polynomial. */ t1p += MLDSA_U * MLDSA_N / 8; - /* Step 10: - NTT(c) o NTT(t1)) */ - mldsa_ntt_full(w); #ifndef WOLFSSL_MLDSA_SMALL_MEM_POLY64 #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { diff --git a/wolfcrypt/test/test.c b/wolfcrypt/test/test.c index b373d20c613..b51c394c2d0 100644 --- a/wolfcrypt/test/test.c +++ b/wolfcrypt/test/test.c @@ -66473,18 +66473,7 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) } #endif -/* Cross-implementation signature KAT. ML-DSA signing is deterministic given - * the key and signing seeds, so every signer here must produce the same bytes. - * Expected values are SHAKE-256 digests of the default signer's signature; - * regenerate them from a default-signer build if the seeds or message change. - * Skipped for the signers that are deliberately not FIPS 204 conformant: the - * draft domain separation and the CHECK_Y/CHECK_W0 early rejects. */ -#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ - !defined(WOLFSSL_MLDSA_FIPS204_DRAFT) && \ - !defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ - !defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) - +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) /* Seed the key pair is generated from. */ static const byte mldsa_kat_key_seed[MLDSA_SEED_SZ] = { 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, @@ -66505,6 +66494,11 @@ static const byte mldsa_kat_msg[] = { 0x41, 0x54 /* "AT" */ }; +/* Signers deliberately not FIPS 204 conformant have no known answer. */ +#if !defined(WOLFSSL_MLDSA_FIPS204_DRAFT) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) +#define MLDSA_KAT_DIGEST(d) (d) /* SHAKE-256 digests of the signature the default signer produces. */ #ifndef WOLFSSL_NO_ML_DSA_44 static const byte mldsa_kat_digest_44[32] = { @@ -66530,40 +66524,87 @@ static const byte mldsa_kat_digest_87[32] = { 0x9f, 0x97, 0xb9, 0xca, 0xd3, 0x8f, 0xc3, 0xbf }; #endif +#else +#define MLDSA_KAT_DIGEST(d) NULL +#endif -static wc_test_ret_t mldsa_sign_kat_test(int param, const byte* expDigest) +/* A key generated into an object that already held a key must sign and verify + * exactly like one generated into a fresh object, whatever the key caches. */ +static wc_test_ret_t mldsa_make_key_reuse_test(int param, const byte* expDigest) { wc_test_ret_t ret; wc_MlDsaKey* key = NULL; + wc_MlDsaKey* freshKey = NULL; byte* sig = NULL; + byte* freshSig = NULL; word32 sigLen; + word32 freshSigLen; int sigSz = 0; byte digest[32]; wc_Shake shake; int keyInit = 0; + int freshKeyInit = 0; int shakeInit = 0; +#ifndef WOLFSSL_MLDSA_NO_VERIFY + int res = 0; +#endif key = (wc_MlDsaKey*)XMALLOC(sizeof(wc_MlDsaKey), HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + freshKey = (wc_MlDsaKey*)XMALLOC(sizeof(wc_MlDsaKey), HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); sig = (byte*)XMALLOC(MLDSA_MAX_SIG_SIZE, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); - if ((key == NULL) || (sig == NULL)) + freshSig = (byte*)XMALLOC(MLDSA_MAX_SIG_SIZE, HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + if ((key == NULL) || (freshKey == NULL) || (sig == NULL) || + (freshSig == NULL)) ERROR_OUT(WC_TEST_RET_ENC_NC, out); - ret = wc_MlDsaKey_Init(key, HEAP_HINT, INVALID_DEVID); + ret = wc_MlDsaKey_Init(freshKey, HEAP_HINT, INVALID_DEVID); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); - keyInit = 1; - ret = wc_MlDsaKey_SetParams(key, param); + freshKeyInit = 1; + ret = wc_MlDsaKey_SetParams(freshKey, param); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_MlDsaKey_GetSigLen(freshKey, &sigSz); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_MlDsaKey_MakeKeyFromSeed(freshKey, mldsa_kat_key_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + freshSigLen = (word32)sigSz; + ret = wc_MlDsaKey_SignCtxWithSeed(freshKey, NULL, 0, freshSig, + &freshSigLen, mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), + mldsa_kat_sig_seed); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (expDigest != NULL) { + ret = wc_InitShake256(&shake, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + shakeInit = 1; + ret = wc_Shake256_Update(&shake, freshSig, freshSigLen); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_Shake256_Final(&shake, digest, (word32)sizeof(digest)); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (XMEMCMP(digest, expDigest, sizeof(digest)) != 0) + ERROR_OUT(WC_TEST_RET_ENC_NC, out); + } - /* Deterministic key and deterministic signature. */ - ret = wc_MlDsaKey_MakeKeyFromSeed(key, mldsa_kat_key_seed); + ret = wc_MlDsaKey_Init(key, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + keyInit = 1; + ret = wc_MlDsaKey_SetParams(key, param); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); - ret = wc_MlDsaKey_GetSigLen(key, &sigSz); + /* Fill the caches from another key before regenerating. */ + ret = wc_MlDsaKey_MakeKeyFromSeed(key, mldsa_kat_sig_seed); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); sigLen = (word32)sigSz; @@ -66571,20 +66612,34 @@ static wc_test_ret_t mldsa_sign_kat_test(int param, const byte* expDigest) mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), mldsa_kat_sig_seed); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); - - ret = wc_InitShake256(&shake, HEAP_HINT, INVALID_DEVID); +#ifndef WOLFSSL_MLDSA_NO_VERIFY + ret = wc_MlDsaKey_VerifyCtx(key, sig, sigLen, NULL, 0, mldsa_kat_msg, + (word32)sizeof(mldsa_kat_msg), &res); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); - shakeInit = 1; - ret = wc_Shake256_Update(&shake, sig, sigLen); + if (res != 1) + ERROR_OUT(WC_TEST_RET_ENC_I(res), out); +#endif + + ret = wc_MlDsaKey_MakeKeyFromSeed(key, mldsa_kat_key_seed); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); - ret = wc_Shake256_Final(&shake, digest, (word32)sizeof(digest)); + sigLen = (word32)sigSz; + ret = wc_MlDsaKey_SignCtxWithSeed(key, NULL, 0, sig, &sigLen, + mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), mldsa_kat_sig_seed); if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); - - if (XMEMCMP(digest, expDigest, sizeof(digest)) != 0) + if ((sigLen != freshSigLen) || (XMEMCMP(sig, freshSig, sigLen) != 0)) ERROR_OUT(WC_TEST_RET_ENC_NC, out); +#ifndef WOLFSSL_MLDSA_NO_VERIFY + res = 0; + ret = wc_MlDsaKey_VerifyCtx(key, freshSig, freshSigLen, NULL, 0, + mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), &res); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (res != 1) + ERROR_OUT(WC_TEST_RET_ENC_I(res), out); +#endif ret = 0; out: @@ -66592,12 +66647,15 @@ static wc_test_ret_t mldsa_sign_kat_test(int param, const byte* expDigest) wc_Shake256_Free(&shake); if (keyInit) wc_MlDsaKey_Free(key); + if (freshKeyInit) + wc_MlDsaKey_Free(freshKey); + XFREE(freshSig, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); XFREE(sig, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + XFREE(freshKey, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); XFREE(key, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); return ret; } - -#endif /* !NO_SIGN && !NO_MAKE_KEY && !FIPS204_DRAFT */ +#endif /* !WOLFSSL_MLDSA_NO_SIGN && !WOLFSSL_MLDSA_NO_MAKE_KEY */ #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ @@ -67690,23 +67748,22 @@ WOLFSSL_TEST_SUBROUTINE wc_test_ret_t mldsa_test(void) #endif #endif -#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ - !defined(WOLFSSL_MLDSA_FIPS204_DRAFT) && \ - !defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ - !defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) #ifndef WOLFSSL_NO_ML_DSA_44 - ret = mldsa_sign_kat_test(WC_ML_DSA_44, mldsa_kat_digest_44); + ret = mldsa_make_key_reuse_test(WC_ML_DSA_44, + MLDSA_KAT_DIGEST(mldsa_kat_digest_44)); if (ret != 0) ERROR_OUT(ret, out); #endif #ifndef WOLFSSL_NO_ML_DSA_65 - ret = mldsa_sign_kat_test(WC_ML_DSA_65, mldsa_kat_digest_65); + ret = mldsa_make_key_reuse_test(WC_ML_DSA_65, + MLDSA_KAT_DIGEST(mldsa_kat_digest_65)); if (ret != 0) ERROR_OUT(ret, out); #endif #ifndef WOLFSSL_NO_ML_DSA_87 - ret = mldsa_sign_kat_test(WC_ML_DSA_87, mldsa_kat_digest_87); + ret = mldsa_make_key_reuse_test(WC_ML_DSA_87, + MLDSA_KAT_DIGEST(mldsa_kat_digest_87)); if (ret != 0) ERROR_OUT(ret, out); #endif From c1376d8a6138fef87809b497d56e3abb31c0f78c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tobias=20Frauenschl=C3=A4ger?= Date: Sun, 6 Sep 2026 18:23:54 +0200 Subject: [PATCH 11/11] Stop the smallest memory signer allocating a mask buffer per polynomial mldsa_vec_expand_mask_c() owns a 680 byte scratch that it allocates, zeroizes and frees on every call, which is an XMALLOC/XFREE pair under WOLFSSL_SMALL_STACK. Expanding the whole vector in one call amortised that over l polynomials, but the smallest memory signer expands one polynomial at a time and calls it twice per polynomial - once to build w and once to regenerate y for z - so the cost is paid 2*l times per rejection attempt. The l == 1 argument also bypasses every assembly dispatch arm, so that path was always going to land in the C implementation anyway. Split the per-polynomial work into mldsa_expand_mask_poly(), which takes the scratch from the caller, and let the signer pass the z polynomial: z is not live at either call site, being written by the multiply that follows, and it is larger than MLDSA_MAX_V. ML-DSA-44 signing drops from 42 allocations per signature to 7. Throughput is unchanged, since the SHAKE work dominates. Read the cached public vector in place when verifying. The small memory verify copied one already decoded and transformed polynomial out of the cache into w, then immediately overwrote w with the result of the pointwise multiply. Point the multiply at the cache instead. No functional change, one buffer less touched per polynomial. Compile mldsa_vec_expand_mask(), mldsa_vec_expand_mask_c() and the AVX2 and AVX-512 mask generators only when the smallest memory signer is not selected. That signer no longer calls them, and SIGN_SMALLEST_MEM implies SIGN_SMALL_MEM, whose signer is the only other caller, so every smallest memory build warned about an unused static function and an in-tree -Werror build failed. Move the ExpandMask algorithm comment back onto the vector function it documents. The AVX-512 generator's dimension gate only ever turned away that single polynomial, so drop it. --- wolfcrypt/src/wc_mldsa.c | 124 ++++++++++++++++++++++++--------------- 1 file changed, 77 insertions(+), 47 deletions(-) diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index e97411f6e43..a4ba5227d5e 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -342,6 +342,8 @@ void print_data(const char* name, const byte* d, int len) #define MLDSA_MAX_V_BLOCKS 5 /* Maximum number of bytes to generate into v to make y. */ #define MLDSA_MAX_V (MLDSA_MAX_V_BLOCKS * 8 * 17) +/* The smallest memory signer expands the mask into a polynomial buffer. */ +wc_static_assert(sizeof(sword32) * MLDSA_N >= MLDSA_MAX_V); /* 2 blocks, each block 136 bytes = 272 bytes. @@ -4836,7 +4838,8 @@ static int mldsa_expand_s(wc_Shake* shake256, byte* priv_seed, byte eta, #endif /* !WOLFSSL_MLDSA_NO_MAKE_KEY */ #ifndef WOLFSSL_MLDSA_NO_SIGN -#if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) +#if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) #define SHA3_256_BYTES (WC_SHA3_256_COUNT * 8) #ifndef WOLFSSL_NO_ML_DSA_44 @@ -5187,6 +5190,38 @@ static int wc_mldsa_gen_y_avx512(sword32* y, byte* seed, word16 kappa, #endif /* WOLFSSL_MLDSA_HAVE_INTEL_AVX512 */ #endif +/* Expand the private random seed into one polynomial of vector y. + * + * @param [in, out] shake256 SHAKE-256 object. + * @param [in, out] seed Buffer containing seed to expand. + * Has space for two bytes to be appended. + * @param [in] n Value to append to seed. + * @param [in] gamma1_bits Number of bits per value. + * @param [out] y Polynomial. + * @param [out] v Scratch of MLDSA_MAX_V bytes. Holds the + * secret mask, so the caller must zeroize it. + * @return 0 on success. + * @return Negative on hash error. + */ +static int mldsa_expand_mask_poly(wc_Shake* shake256, byte* seed, word16 n, + byte gamma1_bits, sword32* y, byte* v) +{ + int ret; + + /* Step 4: Append to seed and squeeze out data. */ + seed[MLDSA_PRIV_RAND_SEED_SZ + 0] = (byte)n; + seed[MLDSA_PRIV_RAND_SEED_SZ + 1] = (byte)(n >> 8); + ret = mldsa_squeeze256(shake256, seed, MLDSA_Y_SEED_SZ, v, + MLDSA_MAX_V_BLOCKS); + if (ret == 0) { + /* Decode v into polynomial. */ + mldsa_decode_gamma1(v, gamma1_bits, y); + } + + return ret; +} + +#ifndef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM /* Expand the private random seed into vector y. * * FIPS 204 Section 7.3, Algorithm 34 ExpandMask(rho, mu) @@ -5222,19 +5257,10 @@ static int mldsa_vec_expand_mask_c(wc_Shake* shake256, byte* seed, /* Step 2: For each polynomial of vector. */ for (r = 0; (ret == 0) && (r < l); r++) { /* Step 3: Calculate value to append to seed. */ - word16 n = (word16)(kappa + r); - - /* Step 4: Append to seed and squeeze out data. */ - seed[MLDSA_PRIV_RAND_SEED_SZ + 0] = (byte)n; - seed[MLDSA_PRIV_RAND_SEED_SZ + 1] = (byte)(n >> 8); - ret = mldsa_squeeze256(shake256, seed, MLDSA_Y_SEED_SZ, v, - MLDSA_MAX_V_BLOCKS); - if (ret == 0) { - /* Decode v into polynomial. */ - mldsa_decode_gamma1(v, gamma1_bits, y); - /* Next polynomial. */ - y += MLDSA_N; - } + ret = mldsa_expand_mask_poly(shake256, seed, (word16)(kappa + r), + gamma1_bits, y, v); + /* Next polynomial. */ + y += MLDSA_N; } /* v holds the secret mask y. */ @@ -5265,19 +5291,16 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, #if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) #ifdef WOLFSSL_MLDSA_HAVE_INTEL_AVX512 - /* Whole vector in one eight-way run. Only worth its scratch when enough - * of the eight lanes are used, so short vectors fall through. */ - if ((l >= 4) && USE_INTEL_AVX512(cpuid_flags) && - IS_INTEL_BMI2(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { + /* Whole vector in one eight-way run, whatever the dimension. */ + if (USE_INTEL_AVX512(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + (SAVE_VECTOR_REGISTERS2() == 0)) { ret = wc_mldsa_gen_y_avx512(y, seed, kappa, gamma1_bits, l, heap); RESTORE_VECTOR_REGISTERS(); } else #endif - /* Each generator handles one dimension only, so the dimension is tested - * before the vector registers are saved: anything else, such as the - * single polynomial the smallest memory signing asks for, goes straight - * to the C implementation rather than save and restore around no work. */ + /* Each generator handles one dimension only, so test it before saving + * the vector registers. */ #ifndef WOLFSSL_NO_ML_DSA_44 if ((l == 4) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { @@ -5310,6 +5333,7 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, return ret; } +#endif /* !WOLFSSL_MLDSA_SIGN_SMALLEST_MEM */ #endif #if !defined(WOLFSSL_MLDSA_NO_SIGN) || !defined(WOLFSSL_MLDSA_NO_VERIFY) @@ -10551,8 +10575,10 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif /* Step 12: Compute polynomial of y from seed and kappa. */ - ret = mldsa_vec_expand_mask(&key->shake, priv_rand_seed, - (word16)(kappa + s), params->gamma1_bits, y, 1, key->heap); + /* z is not live yet, so it doubles as the expansion scratch + * rather than allocating one per polynomial. */ + ret = mldsa_expand_mask_poly(&key->shake, priv_rand_seed, + (word16)(kappa + s), params->gamma1_bits, y, (byte*)z); if (ret != 0) { break; } @@ -10684,9 +10710,10 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, * discarded round; y masks which one it was. */ for (s = 0; (ret == 0) && valid && (s < params->l); s++) { /* Vector y is not kept, so regenerate this polynomial. */ - ret = mldsa_vec_expand_mask(&key->shake, priv_rand_seed, - (word16)(kappa + s), params->gamma1_bits, y, 1, - key->heap); + /* z is overwritten by the multiply below, so it doubles + * as the expansion scratch. */ + ret = mldsa_expand_mask_poly(&key->shake, priv_rand_seed, + (word16)(kappa + s), params->gamma1_bits, y, (byte*)z); if (ret != 0) { break; } @@ -11670,12 +11697,15 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, unsigned int s; unsigned int e; const sword32* zt = z; + /* Source of this polynomial of t1 for the multiply below. The + * result goes to w either way, so a cached polynomial is read in + * place rather than copied. */ + const sword32* t1v = w; #ifdef WC_MLDSA_CACHE_PUB_VECTORS if (key->pubVecSet) { /* Cached vector is already decoded and transformed. */ - XMEMCPY(w, key->t1 + (size_t)r * MLDSA_N, - sizeof(sword32) * MLDSA_N); + t1v = key->t1 + (size_t)r * MLDSA_N; } else #endif @@ -11690,35 +11720,35 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, #ifndef WOLFSSL_MLDSA_SMALL_MEM_POLY64 #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { - w[e] = -mldsa_mont_red((sword64)c[e] * w[e]); + w[e] = -mldsa_mont_red((sword64)c[e] * t1v[e]); } #else for (e = 0; e < MLDSA_N; e += 8) { - w[e+0] = -mldsa_mont_red((sword64)c[e+0] * w[e+0]); - w[e+1] = -mldsa_mont_red((sword64)c[e+1] * w[e+1]); - w[e+2] = -mldsa_mont_red((sword64)c[e+2] * w[e+2]); - w[e+3] = -mldsa_mont_red((sword64)c[e+3] * w[e+3]); - w[e+4] = -mldsa_mont_red((sword64)c[e+4] * w[e+4]); - w[e+5] = -mldsa_mont_red((sword64)c[e+5] * w[e+5]); - w[e+6] = -mldsa_mont_red((sword64)c[e+6] * w[e+6]); - w[e+7] = -mldsa_mont_red((sword64)c[e+7] * w[e+7]); + w[e+0] = -mldsa_mont_red((sword64)c[e+0] * t1v[e+0]); + w[e+1] = -mldsa_mont_red((sword64)c[e+1] * t1v[e+1]); + w[e+2] = -mldsa_mont_red((sword64)c[e+2] * t1v[e+2]); + w[e+3] = -mldsa_mont_red((sword64)c[e+3] * t1v[e+3]); + w[e+4] = -mldsa_mont_red((sword64)c[e+4] * t1v[e+4]); + w[e+5] = -mldsa_mont_red((sword64)c[e+5] * t1v[e+5]); + w[e+6] = -mldsa_mont_red((sword64)c[e+6] * t1v[e+6]); + w[e+7] = -mldsa_mont_red((sword64)c[e+7] * t1v[e+7]); } #endif #else #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { - t64[e] = -(sword64)c[e] * w[e]; + t64[e] = -(sword64)c[e] * t1v[e]; } #else for (e = 0; e < MLDSA_N; e += 8) { - t64[e+0] = -(sword64)c[e+0] * w[e+0]; - t64[e+1] = -(sword64)c[e+1] * w[e+1]; - t64[e+2] = -(sword64)c[e+2] * w[e+2]; - t64[e+3] = -(sword64)c[e+3] * w[e+3]; - t64[e+4] = -(sword64)c[e+4] * w[e+4]; - t64[e+5] = -(sword64)c[e+5] * w[e+5]; - t64[e+6] = -(sword64)c[e+6] * w[e+6]; - t64[e+7] = -(sword64)c[e+7] * w[e+7]; + t64[e+0] = -(sword64)c[e+0] * t1v[e+0]; + t64[e+1] = -(sword64)c[e+1] * t1v[e+1]; + t64[e+2] = -(sword64)c[e+2] * t1v[e+2]; + t64[e+3] = -(sword64)c[e+3] * t1v[e+3]; + t64[e+4] = -(sword64)c[e+4] * t1v[e+4]; + t64[e+5] = -(sword64)c[e+5] * t1v[e+5]; + t64[e+6] = -(sword64)c[e+6] * t1v[e+6]; + t64[e+7] = -(sword64)c[e+7] * t1v[e+7]; } #endif #endif