diff --git a/.github/configs/pq-all.json b/.github/configs/pq-all.json index a250cadbb64..ead05645b98 100644 --- a/.github/configs/pq-all.json +++ b/.github/configs/pq-all.json @@ -228,5 +228,76 @@ {"name": "pkcs7-mldsa-only", "minutes": 0.5, "comment": "PKCS#7 SignedData with ML-DSA as the only signature algorithm (no RSA, no ECC); guards the ML-DSA-only PKCS7 build path", "configure": ["--enable-cryptonly", "--enable-mldsa", - "--enable-pkcs7", "--disable-rsa", "--disable-ecc"]} + "--enable-pkcs7", "--disable-rsa", "--disable-ecc"]}, +{"name": "mldsa-smallest-mem", "minutes": 2, + "comment": "Smallest memory ML-DSA signing and verification; the only functional coverage of WOLFSSL_MLDSA_SIGN_SMALLEST_MEM and of VERIFY_SMALLEST_MEM with malloc", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM"]}, +{"name": "mldsa-smallest-mem-checks", "minutes": 2, + "comment": "Smallest memory ML-DSA signing with the optional y and w0 rejection checks and the small code arms, which are otherwise never compiled", + "configure": ["--enable-cryptonly", "--enable-mldsa=yes,small", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]}, +{"name": "mldsa-44-precalc-a-intelasm", "minutes": 2, + "comment": "ML-DSA-44 with one pre-calculated row of matrix A on the Intel assembly paths; the q88 decompose and w1 encode kernels take no dimension, so this pins that w is sized for a full vector", + "configure": ["--enable-cryptonly", "--enable-mldsa=44", "--enable-intelasm", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]}, +{"name": "mldsa-precalc-a-saturated", "minutes": 2, + "comment": "Pre-calculated matrix A with more rows than any parameter set has, so maxK saturates at k and the streaming loops do not run at all", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=8"]}, +{"name": "mldsa-precalc-a-split", "minutes": 2, + "comment": "Pre-calculated matrix A split part way, which is neither the single row nor the saturated case", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=2"]}, +{"name": "mldsa-checks-default-signer", "minutes": 2, + "comment": "The y and w0 rejection checks on the default signer, the only combination that compiles the vector range check helper and its two call sites", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]}, +{"name": "mldsa-cache-pub-vectors-verify-no-malloc", "minutes": 2, + "comment": "Cached public vectors alongside the pinned verify buffers, the combination whose key structure member clash stopped it building; the clashing member only exists when the small memory verify is selected too", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_PUB_VECTORS -DWOLFSSL_MLDSA_VERIFY_NO_MALLOC -DWOLFSSL_MLDSA_VERIFY_SMALL_MEM"]}, +{"name": "mldsa-cache-matrix", "minutes": 2, + "comment": "Cached matrix A on the default full vector signer, the only build where the signing cache allocation test's key->a and aSet assertions are compiled", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_MATRIX_A"]}, +{"name": "mldsa-cache-matrix-precalc-a", "minutes": 2, + "comment": "Cached matrix A with a pre-calculated row; the small memory signer streams A, so this exercises the sign and verify round trip rather than the cache assertions", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_MATRIX_A -DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1"]}, +{"name": "mldsa-cache-small-keygen", "minutes": 2, + "comment": "Every key cache alongside the small memory key generation, which must drop the caches of the key it replaces", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWC_MLDSA_CACHE_PRIV_VECTORS -DWC_MLDSA_CACHE_PUB_VECTORS -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM"]}, +{"name": "mldsa-no-check-key-small-mem", "minutes": 2, + "comment": "Key pair checking compiled out alongside the small memory key generation and the smallest memory signer, which leaves the whole vector helpers with no caller", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_NO_CHECK_KEY -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM -DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM"]}, +{"name": "mldsa-verify-only-no-malloc", "minutes": 2, + "comment": "Heapless verify-only ML-DSA named only by the canonical options; the no-malloc gate must select the pinned small memory verify or every verification fails with MEMORY_E", + "configure": ["--enable-cryptonly", "--enable-mldsa=verify-only", + "CPPFLAGS=-DWOLFSSL_NO_DILITHIUM_LEGACY_GATES -DWOLFSSL_NO_MALLOC"]}, +{"name": "mldsa-no-check-key", "minutes": 3, + "comment": "Key pair checking compiled out, which is the only build that reaches the NOT_COMPILED_IN arm of wc_CheckPrivateKey; needs the OpenSSL compatibility layer so the X509_check_private_key callers and their tests are compiled too", + "configure": ["--enable-opensslall", "--enable-mldsa", "--enable-certgen", + "--enable-certreq", "CPPFLAGS=-DWOLFSSL_MLDSA_NO_CHECK_KEY"]}, +{"name": "mldsa-no-legacy-gates-smallest", "minutes": 2, + "comment": "Canonical option names only; pins that SIGN_SMALLEST_MEM still implies SIGN_SMALL_MEM when the legacy name gates are opted out", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_NO_DILITHIUM_LEGACY_GATES -DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM"]}, +{"name": "mldsa-smallest-checks-no-verify", "minutes": 2, + "comment": "Smallest memory signing with the y and w0 checks and verify compiled out, which leaves neither vector range check helper with a caller", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_Y -DWOLFSSL_MLDSA_SIGN_CHECK_W0 -DWOLFSSL_MLDSA_NO_VERIFY"]}, +{"name": "mldsa-sign-only", "minutes": 2, + "comment": "Default signer with verify compiled out; the non constant time vector range check has no caller here", + "configure": ["--enable-cryptonly", "--enable-mldsa=make,sign"]}, +{"name": "mldsa-verify-smallest-default-sign", "minutes": 2, + "comment": "Smallest memory verify paired with the default signer, a combination every other row avoids", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM"]}, +{"name": "mldsa-small-check-w0-only", "minutes": 2, + "comment": "Small memory signing with the w0 check but not the y check, which decides whether the vector range check helper is reachable", + "configure": ["--enable-cryptonly", "--enable-mldsa", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM -DWOLFSSL_MLDSA_SIGN_CHECK_W0"]} ] diff --git a/.github/workflows/wolfCrypt-Wconversion.yml b/.github/workflows/wolfCrypt-Wconversion.yml index 0002f3809c2..789886ee0f7 100644 --- a/.github/workflows/wolfCrypt-Wconversion.yml +++ b/.github/workflows/wolfCrypt-Wconversion.yml @@ -90,6 +90,30 @@ jobs: "--enable-xmss", "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM -DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC -DWOLFSSL_WC_LMS_SERIALIZE_STATE -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual"], "check": false}, + {"name": "smallest-mem", "minutes": 1, + "configure": ["--enable-cryptonly", "--enable-all-crypto", + "--disable-examples", "--disable-benchmark", + "--disable-crypttests", "--enable-mlkem", + "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", + "--enable-xmss", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_VERIFY_SMALLEST_MEM -DWOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual -Wdeclaration-after-statement"], + "check": false}, + {"name": "smallest-mem-no-verify", "minutes": 1, + "configure": ["--enable-cryptonly", "--enable-all-crypto", + "--disable-examples", "--disable-benchmark", + "--disable-crypttests", "--enable-mlkem", + "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", + "--enable-xmss", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALLEST_MEM -DWOLFSSL_MLDSA_NO_VERIFY -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual -Wdeclaration-after-statement"], + "check": false}, + {"name": "precalc-a-intelasm", "minutes": 1, + "configure": ["--enable-cryptonly", "--enable-all-crypto", + "--enable-intelasm", "--disable-examples", "--disable-benchmark", + "--disable-crypttests", "--enable-mlkem", + "--enable-slhdsa=yes,sha2", "--enable-mldsa", "--enable-lms", + "--enable-xmss", + "CPPFLAGS=-DWOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A=1 -Wconversion -Warith-conversion -Wenum-conversion -Wfloat-conversion -Wsign-conversion -Wcast-qual -Wdeclaration-after-statement"], + "check": false}, {"name": "precalc-a-no-int128", "minutes": 1, "configure": ["--enable-cryptonly", "--enable-all-crypto", "--disable-examples", "--disable-benchmark", diff --git a/.wolfssl_known_macro_extras b/.wolfssl_known_macro_extras index 4f53192b6d2..a2b953f9465 100644 --- a/.wolfssl_known_macro_extras +++ b/.wolfssl_known_macro_extras @@ -998,6 +998,7 @@ WOLFSSL_MANUALLY_SELECT_DEVICE_CONFIG WOLFSSL_MCDC_ALLOC_SWEEP WOLFSSL_MDK5 WOLFSSL_MICROCHIP_AESGCM +WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC WOLFSSL_MLDSA_VERIFY_PRECOMP_A WOLFSSL_MLKEM_ASM_TEST WOLFSSL_MLKEM_INVNTT_UNROLL diff --git a/ChangeLog.md b/ChangeLog.md index 6bb746f5d5e..9f6b427850e 100644 --- a/ChangeLog.md +++ b/ChangeLog.md @@ -1,3 +1,20 @@ +# wolfSSL Release (unreleased) + +## Behavioral Changes + +* **Behavioral change (`WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM` allocates)**: the option no longer forces `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, so the smallest memory verify now allocates its scratch buffers rather than pinning them in `wc_MlDsaKey`, and the key structure is smaller. Define `WOLFSSL_MLDSA_VERIFY_NO_MALLOC` as well to keep the heapless verify. by @Frauschi + +* **Behavioral change (`WOLFSSL_NO_MALLOC` pins the ML-DSA verify buffers)**: when ML-DSA verification is compiled in, `WOLFSSL_NO_MALLOC` now selects the small memory verify along with `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, including in builds that use the canonical option names. The verify scratch buffers then live in `wc_MlDsaKey`, which makes each key about 12 kB larger. Previously the no-malloc option selected nothing on its own, so without a working allocator verification failed with `MEMORY_E`. A build that defines `WOLFSSL_NO_MALLOC` but still has an allocator, through `WOLFSSL_STATIC_MEMORY` or `wolfSSL_SetAllocators()`, can keep the smaller key and the allocating verify by defining `WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC`. by @Frauschi + +## Post-Quantum Cryptography (PQC) + +* Fixed the ML-DSA key structure member clash that stopped `WC_MLDSA_CACHE_PUB_VECTORS` building alongside `WOLFSSL_MLDSA_VERIFY_NO_MALLOC`, renaming the verify scratch member `t1` to `vt1`. by @Frauschi +* Reduced the ML-DSA small memory heap footprint: signing keeps w1 only in encoded form, key generation encodes t a polynomial at a time, and the new `WOLFSSL_MLDSA_SIGN_SMALLEST_MEM` holds one polynomial of y, roughly halving the signing peak. by @Frauschi +* Sped up ML-DSA small memory signing by walking matrix A a column at a time so each polynomial of y is transformed once rather than once per row. `WOLFSSL_MLDSA_SMALL_MEM_POLY64` no longer applies to signing. by @Frauschi +* Fixed `--enable-mldsa=` naming only a parameter set, which left key generation, signing and verification all disabled. by @Frauschi +* `wc_CheckPrivateKey()` now reports `NOT_COMPILED_IN` for an ML-DSA certificate and key when the key pair check is compiled out with `WOLFSSL_MLDSA_NO_CHECK_KEY`, rather than failing to build. Such a build cannot confirm the pair matches, so `wolfSSL_CTX_check_private_key()`, `wolfSSL_check_private_key()` and `wolfSSL_X509_check_private_key()` report a mismatch for it; loading the certificate and key is unaffected. by @Frauschi +* Fixed ML-DSA key generation with `WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM` keeping the matrix and vector caches (`WC_MLDSA_CACHE_*`) of the key it replaced. A key generated into an object that already held another key was signed and verified against the old key's cached values, so its signatures failed to verify. by @Frauschi + # wolfSSL Release 5.9.4 (Sep 25, 2026) Release 5.9.4 has been developed according to wolfSSL's development and QA diff --git a/configure.ac b/configure.ac index cbbcb77a948..4ed06c38632 100644 --- a/configure.ac +++ b/configure.ac @@ -2423,6 +2423,19 @@ then ENABLED_MLDSA87=yes fi +# Likewise, naming only a parameter set, for example --enable-mldsa=44, +# selects no operation, which would disable everything. verify-only names an +# operation, so it is unaffected. +if test "$ENABLED_MLDSA" != "no" && \ + test "$ENABLED_MLDSA_MAKE_KEY" = "no" && \ + test "$ENABLED_MLDSA_SIGN" = "no" && \ + test "$ENABLED_MLDSA_VERIFY" = "no" +then + ENABLED_MLDSA_MAKE_KEY=yes + ENABLED_MLDSA_SIGN=yes + ENABLED_MLDSA_VERIFY=yes +fi + # XMSS AC_ARG_ENABLE([xmss], [AS_HELP_STRING([--enable-xmss],[Enable stateful XMSS/XMSS^MT signatures (default: disabled)])], diff --git a/tests/api/test_ossl_x509_crypto.c b/tests/api/test_ossl_x509_crypto.c index 71650077b32..230cfaf3938 100644 --- a/tests/api/test_ossl_x509_crypto.c +++ b/tests/api/test_ossl_x509_crypto.c @@ -155,7 +155,14 @@ int test_wolfSSL_X509_check_private_key_mldsa(void) ExpectNotNull(x509 = X509_load_certificate_file( cases[i].certPath, SSL_FILETYPE_ASN1)); + #ifdef WOLFSSL_MLDSA_CHECK_KEY ExpectIntEQ(X509_check_private_key(x509, pkey), 1); + #else + /* Without the key pair check the match cannot be confirmed, so + * wc_CheckPrivateKey() reports NOT_COMPILED_IN and the pair is + * rejected. */ + ExpectIntEQ(X509_check_private_key(x509, pkey), 0); + #endif if (cases[i].mismatchCertPath != NULL) { ExpectNotNull(mismatchX509 = X509_load_certificate_file( diff --git a/tests/unit-mcdc/test_wc_mldsa_whitebox.c b/tests/unit-mcdc/test_wc_mldsa_whitebox.c index 41409587cce..cc79ad9286c 100644 --- a/tests/unit-mcdc/test_wc_mldsa_whitebox.c +++ b/tests/unit-mcdc/test_wc_mldsa_whitebox.c @@ -219,7 +219,7 @@ static void wb_get_params(void) * Drive independence pairs: both F (in range), left T (a<=nhi), * right T with left F (a>=hi). Plus the vector-level (ret==1)&&(i=hi) expected 0"); } +#ifndef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM /* Vector level: two polynomials, both in range -> (ret==1)&&(i early ret 0. */ for (j = 0; j < 2 * MLDSA_N; j++) { @@ -265,10 +266,154 @@ static void wb_check_low(void) if (ret != 0) { WB_NOTE("mldsa_vec_check_low_c(out) expected 0"); } +#endif WB_OK("mldsa_check_low / vec_check_low_c operand pairs exercised"); } #endif +#ifndef WOLFSSL_MLDSA_NO_SIGN +/* ------------------------------------------------------------------ * + * mldsa_check_low_ct / mldsa_vec_check_low_ct: the constant time forms + * gate every signing range check, so drive the four boundary values and + * confirm no early exit hides a later failure. + * ------------------------------------------------------------------ */ +static void wb_check_low_ct(void) +{ + sword32 a[2 * MLDSA_N]; + sword32 hi = 1 << 17; + unsigned int j; + int ret = 0; + + for (j = 0; j < 2 * MLDSA_N; j++) { + a[j] = 0; + } + + /* Boundaries: hi-1 and -hi+1 are in range, hi and -hi are not. */ + a[0] = hi - 1; + if (mldsa_check_low_ct(a, hi) != 1) { + WB_NOTE("mldsa_check_low_ct(hi-1) expected 1"); + } + a[0] = -hi + 1; + if (mldsa_check_low_ct(a, hi) != 1) { + WB_NOTE("mldsa_check_low_ct(-hi+1) expected 1"); + } + a[0] = hi; + if (mldsa_check_low_ct(a, hi) != 0) { + WB_NOTE("mldsa_check_low_ct(hi) expected 0"); + } + a[0] = -hi; + if (mldsa_check_low_ct(a, hi) != 0) { + WB_NOTE("mldsa_check_low_ct(-hi) expected 0"); + } + + /* The last coefficient must count: an early exit would miss it. */ + a[0] = 0; + a[MLDSA_N - 1] = hi; + if (mldsa_check_low_ct(a, hi) != 0) { + WB_NOTE("mldsa_check_low_ct(last coeff) expected 0"); + } + a[MLDSA_N - 1] = 0; + + /* Agreement with the branching form wherever both are compiled. */ +#if !defined(WOLFSSL_MLDSA_NO_VERIFY) + for (j = 0; j < MLDSA_N; j++) { + a[j] = (j & 1) ? (hi - 1) : (-hi + 1); + } + ret = mldsa_check_low(a, hi); + if (mldsa_check_low_ct(a, hi) != ret) { + WB_NOTE("check_low_ct disagrees with check_low (in range)"); + } + a[MLDSA_N / 2] = hi; + ret = mldsa_check_low(a, hi); + if (mldsa_check_low_ct(a, hi) != ret) { + WB_NOTE("check_low_ct disagrees with check_low (out of range)"); + } +#else + (void)ret; +#endif + +#if (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM)) || \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + /* Vector form must fail on a bad coefficient in the LAST polynomial, + * which only holds because it does not exit early. */ + for (j = 0; j < 2 * MLDSA_N; j++) { + a[j] = 0; + } + if (mldsa_vec_check_low_ct(a, 2, hi) != 1) { + WB_NOTE("mldsa_vec_check_low_ct(in-range,l=2) expected 1"); + } + a[2 * MLDSA_N - 1] = hi; + if (mldsa_vec_check_low_ct(a, 2, hi) != 0) { + WB_NOTE("mldsa_vec_check_low_ct(last poly) expected 0"); + } +#endif + + WB_OK("mldsa_check_low_ct / vec_check_low_ct boundaries exercised"); +} +#endif + +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) +/* ------------------------------------------------------------------ * + * mldsa_poly_checksum: the smallest memory signer binds the polynomial of y + * it regenerates for z to the one it used for w, raising BAD_COND_E when they + * differ. That branch needs a fault to reach, so the property is tested + * directly: any single changed coefficient must change the checksum. + * ------------------------------------------------------------------ */ +static void wb_poly_checksum(void) +{ + sword32 a[MLDSA_N]; + sword32 base; + unsigned int j; + unsigned int pos[4]; + unsigned int p; + + for (j = 0; j < MLDSA_N; j++) { + a[j] = (sword32)(j * 7); + } + base = mldsa_poly_checksum(a); + + /* Same input, same checksum: the comparison in the signer only fires + * on a real difference. */ + if (mldsa_poly_checksum(a) != base) { + WB_NOTE("mldsa_poly_checksum is not deterministic"); + } + + /* First, last and two interior coefficients. The rotate in the + * checksum is what makes position matter. */ + pos[0] = 0; + pos[1] = 1; + pos[2] = MLDSA_N / 2; + pos[3] = MLDSA_N - 1; + for (p = 0; p < 4; p++) { + sword32 keep = a[pos[p]]; + + a[pos[p]] = keep ^ 1; + if (mldsa_poly_checksum(a) == base) { + WB_NOTE("mldsa_poly_checksum missed a changed coefficient"); + } + a[pos[p]] = keep; + } + + /* Two coefficients swapped: same multiset, different polynomial. */ + if (MLDSA_N >= 2) { + sword32 k0 = a[0]; + + a[0] = a[1]; + a[1] = k0; + if (mldsa_poly_checksum(a) == base) { + WB_NOTE("mldsa_poly_checksum missed a reordering"); + } + a[1] = a[0]; + a[0] = k0; + } + + WB_OK("mldsa_poly_checksum change detection exercised"); +} +#endif /* WOLFSSL_MLDSA_SIGN_SMALLEST_MEM */ + /* ------------------------------------------------------------------ * * mldsa_check_hint: two inner loop decisions that the 3-outcome test * above never reaches because their FALSE/TRUE pair only shows up with @@ -1420,9 +1565,16 @@ int main(void) return 0; #else wb_get_params(); -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || !defined(WOLFSSL_MLDSA_NO_VERIFY) +#ifndef WOLFSSL_MLDSA_NO_VERIFY wb_check_low(); #endif +#ifndef WOLFSSL_MLDSA_NO_SIGN + wb_check_low_ct(); +#endif +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) + wb_poly_checksum(); +#endif #ifndef WOLFSSL_MLDSA_NO_SIGN #ifndef WOLFSSL_NO_ML_DSA_44 wb_make_hint_88(); diff --git a/wolfcrypt/src/asn.c b/wolfcrypt/src/asn.c index 3c1a2eda0d2..0181fe8cf72 100644 --- a/wolfcrypt/src/asn.c +++ b/wolfcrypt/src/asn.c @@ -9950,9 +9950,16 @@ int wc_CheckPrivateKey(const byte* privKey, word32 privKeySz, keyIdx = 0; if ((ret = wc_MlDsaKey_ImportPubRaw(key_pair, pubKey, pubKeySz)) == 0) { + #ifdef WOLFSSL_MLDSA_CHECK_KEY /* Public and private extracted successfully. Sanity check. */ if ((ret = wc_MlDsaKey_CheckKey(key_pair)) == 0) ret = 1; + #else + /* Without the key check the pair cannot be confirmed to + * match, so do not claim that it does. */ + ret = NOT_COMPILED_IN; + WOLFSSL_ERROR_VERBOSE(ret); + #endif } } wc_MlDsaKey_Free(key_pair); diff --git a/wolfcrypt/src/wc_mldsa.c b/wolfcrypt/src/wc_mldsa.c index 8fade7cfb5f..a4ba5227d5e 100644 --- a/wolfcrypt/src/wc_mldsa.c +++ b/wolfcrypt/src/wc_mldsa.c @@ -47,9 +47,18 @@ * Compiles in only the verification and public key operations. * WOLFSSL_MLDSA_VERIFY_SMALL_MEM Default: OFF * Compiles verification implementation that uses smaller amounts of memory. + * WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM Default: OFF + * Implies WOLFSSL_MLDSA_VERIFY_SMALL_MEM. Decodes and transforms vector z a + * polynomial at a time instead of holding the whole vector, k times rather + * than once, so unlike the signing option it trades time for memory. + * Add WOLFSSL_MLDSA_VERIFY_NO_MALLOC to pin the buffers against the key. * WOLFSSL_MLDSA_VERIFY_NO_MALLOC Default: OFF * Only works with WOLFSSL_MLDSA_VERIFY_SMALL_MEM. * Don't allocate memory with XMALLOC. Memory is pinned against key. + * WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC Default: OFF + * Declines the small memory verify and pinned buffers that WOLFSSL_NO_MALLOC + * selects automatically, keeping the default verify and a key about 12kB + * smaller. Only for a build whose XMALLOC does work. * WOLFSSL_MLDSA_ASSIGN_KEY Default: OFF * Key data is assigned into ML-DSA key rather than copied. * Life of key data passed in is tightly coupled to life of ML-DSA key. @@ -62,7 +71,9 @@ * Cannot be used with WOLFSSL_MLDSA_ASSIGN_KEY. * WOLFSSL_MLDSA_SIGN_SMALL_MEM Default: OFF * Compiles signature implementation that uses smaller amounts of memory but - * is considerably slower. + * is considerably slower. Matrix A is streamed a polynomial at a time rather + * than held, and is walked a column at a time so that each polynomial of + * vector y is transformed once rather than once per row of A. * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC Default: OFF * Compiles signature implementation that uses smaller amounts of memory but * is considerably slower. Allocates vectors and decodes private key data @@ -70,12 +81,24 @@ * WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A Default: OFF * Compiles signature implementation that uses smaller amounts of memory but * is slower. Allocates matrix A and calculates it upfront. + * WOLFSSL_MLDSA_SIGN_SMALLEST_MEM Default: OFF + * Compiles the smallest memory signature implementation, slower again than + * WOLFSSL_MLDSA_SIGN_SMALL_MEM, which it implies. Matrix A is generated a + * column at a time so only one polynomial of vector y is held, and y is + * regenerated for z. + * Cannot be used with WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC or _PRECALC_A. + * WOLFSSL_MLDSA_SMALL_MEM_POLY64, WC_MLDSA_CACHE_PRIV_VECTORS and + * WC_MLDSA_CACHE_MATRIX_A do nothing in this mode. * WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM Default: OFF * Compiles key generation implementation that uses smaller amounts of memory * but is slower. * WOLFSSL_MLDSA_SMALL_MEM_POLY64 Default: OFF * Compiles the small memory implementations to use a 64-bit polynomial. * Uses 2KB of memory but is slightly quicker (2.75-7%). + * Only WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM and WOLFSSL_MLDSA_VERIFY_SMALL_MEM + * are affected. Signing accumulates a column at a time and would need a + * 64-bit accumulator for every row of w, so it does not use one, and this + * does nothing at all without one of those two options. * * WOLFSSL_MLDSA_ALIGNMENT Default: 8 * Use to indicate whether loading and storing of words needs to be aligned. @@ -192,17 +215,29 @@ #endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + #error "PRECALC and PRECALC_A are equivalent to non small mem" #endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) - #define WOLFSSL_MLDSA_SIGN_SMALL_MEM - #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC - #error "PRECALC and PRECALC_A are equivalent to non small mem" + ((WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A + 0) < 1) + #error "PRECALC_A must pre-calculate at least one row of matrix A" +#endif +#ifdef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM + #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + #error "SMALLEST_MEM keeps nothing pre-calculated" #endif #endif +/* Signing drives the whole-vector helpers when it holds a full vector: the + * default implementation, and PRECALC_A which multiplies the pre-calculated + * rows of matrix A as a vector. */ +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A)) + #define MLDSA_SIGN_VEC_HELPERS +#endif + #if defined(USE_INTEL_SPEEDUP) static cpuid_flags_t cpuid_flags = WC_CPUID_INITIALIZER; @@ -307,6 +342,8 @@ void print_data(const char* name, const byte* d, int len) #define MLDSA_MAX_V_BLOCKS 5 /* Maximum number of bytes to generate into v to make y. */ #define MLDSA_MAX_V (MLDSA_MAX_V_BLOCKS * 8 * 17) +/* The smallest memory signer expands the mask into a polynomial buffer. */ +wc_static_assert(sizeof(sword32) * MLDSA_N >= MLDSA_MAX_V); /* 2 blocks, each block 136 bytes = 272 bytes. @@ -1214,7 +1251,9 @@ static void mldsa_vec_encode_eta_bits(const sword32* s, byte d, byte eta, } #endif /* !WOLFSSL_MLDSA_NO_MAKE_KEY */ -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || defined(WOLFSSL_MLDSA_CHECK_KEY) +#if !defined(WOLFSSL_MLDSA_NO_SIGN) || defined(WOLFSSL_MLDSA_CHECK_KEY) || \ + (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) #if !defined(WOLFSSL_NO_ML_DSA_44) || !defined(WOLFSSL_NO_ML_DSA_87) /* Decode polynomial with range -2..2. @@ -1350,8 +1389,12 @@ static void mldsa_decode_eta_4_bits(const byte* p, sword32* s) #endif #if defined(WOLFSSL_MLDSA_CHECK_KEY) || \ + (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (defined(WC_MLDSA_CACHE_PRIV_VECTORS) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM))) /* Decode vector of polynomials with range -ETA..ETA. * @@ -1720,6 +1763,7 @@ static void mldsa_decode_t0(const byte* t0, sword32* t) #if defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ (defined(WC_MLDSA_CACHE_PRIV_VECTORS) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM))) /* Decode bottom D bits of t as t0. * @@ -1863,7 +1907,8 @@ static void mldsa_decode_t1(const byte* t1, sword32* t) #endif #if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ + (!defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM) || \ + defined(WC_MLDSA_CACHE_PUB_VECTORS))) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) /* Decode top bits of t as t1. * @@ -2636,9 +2681,57 @@ WOLFSSL_TEST_VIS void wc_mldsa_encode_w1_32(const sword32* w1, byte* w1e) #endif #endif -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || \ - (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) +/* Decode one polynomial of w1 from its encoded form. + * + * The small memory signing implementations keep w1 only in the encoded form + * that the commit hash needs. MakeHint wants the polynomial back. + * + * @param [in] w1e Encoded polynomial. + * @param [in] gamma2 Low-order rounding range, GAMMA2. + * @param [out] w1 Decoded polynomial. + */ +static void mldsa_decode_w1(const byte* w1e, sword32 gamma2, sword32* w1) +{ + unsigned int j; + +#ifndef WOLFSSL_NO_ML_DSA_44 + if (gamma2 == MLDSA_Q_LOW_88) { + /* 6 bits per number. 4 numbers in 3 bytes. */ + for (j = 0; j < MLDSA_N; j += 4) { + w1[j + 0] = (sword32)( w1e[0] & 0x3f); + w1[j + 1] = (sword32)(((w1e[0] >> 6) | (w1e[1] << 2)) & 0x3f); + w1[j + 2] = (sword32)(((w1e[1] >> 4) | (w1e[2] << 4)) & 0x3f); + w1[j + 3] = (sword32)( (w1e[2] >> 2) & 0x3f); + /* Move to next place to decode from. */ + w1e += 3; + } + } + else +#endif +#if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) + if (gamma2 == MLDSA_Q_LOW_32) { + /* 4 bits per number. 2 numbers in 1 byte. */ + for (j = 0; j < MLDSA_N; j += 2) { + w1[j + 0] = (sword32)( w1e[0] & 0xf); + w1[j + 1] = (sword32)((w1e[0] >> 4) & 0xf); + /* Move to next place to decode from. */ + w1e++; + } + } + else +#endif + { + XMEMSET(w1, 0, MLDSA_POLY_SIZE); + } +} +#endif + +#if (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ + (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A))) || \ + (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) /* Encode w1 with range of 0..((q-1)/(2*GAMMA2)-1). * * FIPS 204 Section 7.2, Algorithm 28 w1Encode(w1) @@ -2659,8 +2752,6 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, { unsigned int i; - (void)k; - #ifndef WOLFSSL_NO_ML_DSA_44 if (gamma2 == MLDSA_Q_LOW_88) { unsigned int e = MLDSA_Q_HI_88_ENC_BITS * 2 * MLDSA_N / 16; @@ -2671,7 +2762,7 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, * encoder needs no polynomial pairing and does the entire vector. */ if (IS_INTEL_AVX512_VBMI(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { - for (; i < PARAMS_ML_DSA_44_K; i++) { + for (; i < k; i++) { wc_mldsa_encode_w1_88_avx512_vbmi(w1, w1e); w1 += MLDSA_N; w1e += e; @@ -2683,7 +2774,7 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, /* Two polynomials per pass; an odd trailing one falls through. */ if ((i == 0) && USE_INTEL_AVX512(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { - for (; i + 1 < PARAMS_ML_DSA_44_K; i += 2) { + for (; i + 1 < k; i += 2) { wc_mldsa_encode_w1_88_x2_avx512(w1, w1 + MLDSA_N, w1e, w1e + e); w1 += 2 * MLDSA_N; @@ -2693,7 +2784,7 @@ static void mldsa_vec_encode_w1(const sword32* w1, byte k, sword32 gamma2, } #endif /* Step 2. For each polynomial of vector. */ - for (; i < PARAMS_ML_DSA_44_K; i++) { + for (; i < k; i++) { mldsa_encode_w1_88(w1, w1e); /* Next polynomial. */ w1 += MLDSA_N; @@ -2982,10 +3073,10 @@ static int mldsa_rej_ntt_poly_ex(wc_Shake* shake128, byte* seed, sword32* a, #if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) || \ + defined(WC_MLDSA_CACHE_MATRIX_A) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ - !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) + !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ + defined(MLDSA_SIGN_VEC_HELPERS) /* Generate a random polynomial by rejection. * * @param [in, out] shake128 SHAKE-128 object. @@ -3030,11 +3121,10 @@ static int mldsa_rej_ntt_poly(wc_Shake* shake128, byte* seed, sword32* a, #if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ + defined(WC_MLDSA_CACHE_MATRIX_A) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WC_MLDSA_CACHE_MATRIX_A))) + defined(MLDSA_SIGN_VEC_HELPERS) #if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) #define SHA3_128_BYTES (WC_SHA3_128_COUNT * 8) @@ -4748,7 +4838,8 @@ static int mldsa_expand_s(wc_Shake* shake256, byte* priv_seed, byte eta, #endif /* !WOLFSSL_MLDSA_NO_MAKE_KEY */ #ifndef WOLFSSL_MLDSA_NO_SIGN -#if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) +#if defined(USE_INTEL_SPEEDUP) && !defined(WC_SHA3_NO_ASM) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) #define SHA3_256_BYTES (WC_SHA3_256_COUNT * 8) #ifndef WOLFSSL_NO_ML_DSA_44 @@ -5099,6 +5190,38 @@ static int wc_mldsa_gen_y_avx512(sword32* y, byte* seed, word16 kappa, #endif /* WOLFSSL_MLDSA_HAVE_INTEL_AVX512 */ #endif +/* Expand the private random seed into one polynomial of vector y. + * + * @param [in, out] shake256 SHAKE-256 object. + * @param [in, out] seed Buffer containing seed to expand. + * Has space for two bytes to be appended. + * @param [in] n Value to append to seed. + * @param [in] gamma1_bits Number of bits per value. + * @param [out] y Polynomial. + * @param [out] v Scratch of MLDSA_MAX_V bytes. Holds the + * secret mask, so the caller must zeroize it. + * @return 0 on success. + * @return Negative on hash error. + */ +static int mldsa_expand_mask_poly(wc_Shake* shake256, byte* seed, word16 n, + byte gamma1_bits, sword32* y, byte* v) +{ + int ret; + + /* Step 4: Append to seed and squeeze out data. */ + seed[MLDSA_PRIV_RAND_SEED_SZ + 0] = (byte)n; + seed[MLDSA_PRIV_RAND_SEED_SZ + 1] = (byte)(n >> 8); + ret = mldsa_squeeze256(shake256, seed, MLDSA_Y_SEED_SZ, v, + MLDSA_MAX_V_BLOCKS); + if (ret == 0) { + /* Decode v into polynomial. */ + mldsa_decode_gamma1(v, gamma1_bits, y); + } + + return ret; +} + +#ifndef WOLFSSL_MLDSA_SIGN_SMALLEST_MEM /* Expand the private random seed into vector y. * * FIPS 204 Section 7.3, Algorithm 34 ExpandMask(rho, mu) @@ -5134,19 +5257,10 @@ static int mldsa_vec_expand_mask_c(wc_Shake* shake256, byte* seed, /* Step 2: For each polynomial of vector. */ for (r = 0; (ret == 0) && (r < l); r++) { /* Step 3: Calculate value to append to seed. */ - word16 n = (word16)(kappa + r); - - /* Step 4: Append to seed and squeeze out data. */ - seed[MLDSA_PRIV_RAND_SEED_SZ + 0] = (byte)n; - seed[MLDSA_PRIV_RAND_SEED_SZ + 1] = (byte)(n >> 8); - ret = mldsa_squeeze256(shake256, seed, MLDSA_Y_SEED_SZ, v, - MLDSA_MAX_V_BLOCKS); - if (ret == 0) { - /* Decode v into polynomial. */ - mldsa_decode_gamma1(v, gamma1_bits, y); - /* Next polynomial. */ - y += MLDSA_N; - } + ret = mldsa_expand_mask_poly(shake256, seed, (word16)(kappa + r), + gamma1_bits, y, v); + /* Next polynomial. */ + y += MLDSA_N; } /* v holds the secret mask y. */ @@ -5185,26 +5299,32 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, } else #endif - if (IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + /* Each generator handles one dimension only, so test it before saving + * the vector registers. */ +#ifndef WOLFSSL_NO_ML_DSA_44 + if ((l == 4) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && (SAVE_VECTOR_REGISTERS2() == 0)) { - #ifndef WOLFSSL_NO_ML_DSA_44 - if (l == 4) { - ret = wc_mldsa_gen_y_4_avx2(y, seed, kappa); - } - #endif - #ifndef WOLFSSL_NO_ML_DSA_65 - if (l == 5) { - ret = wc_mldsa_gen_y_5_avx2(y, seed, kappa, shake256); - } - #endif - #ifndef WOLFSSL_NO_ML_DSA_87 - if (l == 7) { - ret = wc_mldsa_gen_y_7_avx2(y, seed, kappa); - } - #endif + ret = wc_mldsa_gen_y_4_avx2(y, seed, kappa); + RESTORE_VECTOR_REGISTERS(); + } + else +#endif +#ifndef WOLFSSL_NO_ML_DSA_65 + if ((l == 5) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + (SAVE_VECTOR_REGISTERS2() == 0)) { + ret = wc_mldsa_gen_y_5_avx2(y, seed, kappa, shake256); + RESTORE_VECTOR_REGISTERS(); + } + else +#endif +#ifndef WOLFSSL_NO_ML_DSA_87 + if ((l == 7) && IS_INTEL_AVX2(cpuid_flags) && IS_INTEL_BMI2(cpuid_flags) && + (SAVE_VECTOR_REGISTERS2() == 0)) { + ret = wc_mldsa_gen_y_7_avx2(y, seed, kappa); RESTORE_VECTOR_REGISTERS(); } else +#endif #endif { ret = mldsa_vec_expand_mask_c(shake256, seed, kappa, gamma1_bits, y, @@ -5213,6 +5333,7 @@ static int mldsa_vec_expand_mask(wc_Shake* shake256, byte* seed, return ret; } +#endif /* !WOLFSSL_MLDSA_SIGN_SMALLEST_MEM */ #endif #if !defined(WOLFSSL_MLDSA_NO_SIGN) || !defined(WOLFSSL_MLDSA_NO_VERIFY) @@ -5666,7 +5787,69 @@ static void mldsa_vec_decompose(const sword32* r, byte k, sword32 gamma2, * Range check operation ******************************************************************************/ -#if !defined(WOLFSSL_MLDSA_NO_SIGN) || !defined(WOLFSSL_MLDSA_NO_VERIFY) +#ifndef WOLFSSL_MLDSA_NO_SIGN +/* Check that the values of the polynomial are in range, in constant time. + * + * Signing range checks run on values derived from the private key, so the + * index of the first failing coefficient must not be observable. Every + * coefficient is examined and no branch depends on a value. + * + * @param [in] a Polynomial. + * @param [in] hi Largest value in range. + * @return 1 when all values are in range, 0 when any is not. + */ +static int mldsa_check_low_ct(const sword32* a, sword32 hi) +{ + unsigned int j; + /* Calculate lowest range value. */ + sword32 nhi = -hi; + word32 good = 1; + + /* For each value of polynomial. */ + for (j = 0; j < MLDSA_N; j++) { + /* Sign bit of the difference is set when the comparison holds. */ + word32 lt = (word32)((word64)((sword64)a[j] - (sword64)hi) >> 63); + word32 gt = (word32)((word64)((sword64)nhi - (sword64)a[j]) >> 63); + + /* Range is -(hi-1)..(hi-1). */ + good &= lt & gt; + } + + return (int)good; +} + +/* The smallest memory signer holds one polynomial at a time and uses the + * scalar form for these checks. */ +#if (defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM)) || \ + (defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) +/* Check that the values of the vector are in range, in constant time. + * + * @param [in] a Vector of polynomials. + * @param [in] l Dimension of vector. + * @param [in] hi Largest value in range. + * @return 1 when all values are in range, 0 when any is not. + */ +static int mldsa_vec_check_low_ct(const sword32* a, byte l, sword32 hi) +{ + unsigned int i; + int good = 1; + + /* For each polynomial of vector. */ + for (i = 0; i < l; i++) { + good &= mldsa_check_low_ct(a, hi); + /* Next polynomial. */ + a += MLDSA_N; + } + + return good; +} +#endif /* (CHECK_Y && !SIGN_SMALLEST_MEM) || + * (CHECK_W0 && !SIGN_SMALL_MEM) */ +#endif /* !WOLFSSL_MLDSA_NO_SIGN */ + +#ifndef WOLFSSL_MLDSA_NO_VERIFY /* Check that the values of the polynomial are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -5706,9 +5889,8 @@ static int mldsa_check_low(const sword32* a, sword32 hi) return (int)(in >> 31); } -#if !defined(WOLFSSL_MLDSA_NO_VERIFY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) +#if !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) /* Check that the values of the vector are in range. * * Many places in FIPS 204. One example from Algorithm 2: @@ -5733,7 +5915,6 @@ static int mldsa_vec_check_low_c(const sword32* a, byte l, sword32 hi) return ret; } -#endif /* Check that the values of the vector are in range. * @@ -5765,6 +5946,7 @@ static int mldsa_vec_check_low(const sword32* a, byte l, sword32 hi) return ret; } #endif +#endif /****************************************************************************** * Hint operations @@ -6886,9 +7068,7 @@ static void mldsa_ntt(sword32* r) #endif #if !defined(WOLFSSL_MLDSA_NO_VERIFY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC))) || \ + !defined(WOLFSSL_MLDSA_NO_SIGN) || \ (defined(WOLFSSL_MLDSA_SMALL) && \ (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) @@ -6952,10 +7132,10 @@ static void mldsa_vec_ntt(sword32* r, byte l) #endif #endif -#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - (!defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) || \ - defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC)))) || \ +#if (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + (!defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) || \ + defined(WC_MLDSA_CACHE_PUB_VECTORS))) || \ + defined(MLDSA_SIGN_VEC_HELPERS) || \ (defined(WOLFSSL_MLDSA_SMALL) && \ (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) @@ -7941,12 +8121,12 @@ static void mldsa_invntt_full(sword32* r) } } -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + defined(MLDSA_SIGN_VEC_HELPERS) /* Inverse Number-Theoretic Transform. * * @param [in, out] r Vector of polynomials to transform. @@ -7974,12 +8154,12 @@ static void mldsa_vec_invntt_full(sword32* r, byte l) } #endif -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + defined(MLDSA_SIGN_VEC_HELPERS) /* Matrix multiplication. * * @param [out] r Vector of polynomials that is result. @@ -8424,13 +8604,43 @@ static void mldsa_poly_red(sword32* a) } } +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && \ + defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) +/* Checksum of a polynomial. + * + * Binds two computations of a value that is not kept between them, so a fault + * in the second is detectable. Rotates and exclusive-ors rather than + * multiplying: the coefficients are secret, and a multiply chain over them is + * not constant time on cores with an operand dependent multiplier. + * + * @param [in] a Polynomial to checksum. + * @return Checksum of the polynomial. + */ +static sword32 mldsa_poly_checksum(const sword32* a) +{ + unsigned int i; + word32 chk = 0; + + for (i = 0; i < MLDSA_N; i++) { + /* Rotate so the position of a changed coefficient matters, and mix in + * the index: rotation alone repeats every 32 of the 256 coefficients, + * so a fault applying the same delta 32 apart would cancel. */ + chk = ((chk << 1) | (chk >> 31)) ^ ((word32)a[i] + i); + } + + return (sword32)chk; +} +#endif + #if (defined(WOLFSSL_MLDSA_SMALL) && \ - (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ + ((!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ (!defined(WOLFSSL_MLDSA_NO_VERIFY) && \ !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY))) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) || \ + (defined(MLDSA_SIGN_VEC_HELPERS) && defined(WOLFSSL_MLDSA_SMALL)) /* Modulo reduce values in polynomials of vector. Range (-2^31)..(2^31-1). * * @param [in, out] a Vector of polynomials. @@ -8579,7 +8789,8 @@ static void mldsa_add(sword32* r, const sword32* a) } } -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) @@ -8657,10 +8868,10 @@ static void mldsa_make_pos(sword32* a) } } -#if !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) || \ +#if (!defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ + !defined(WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM)) || \ defined(WOLFSSL_MLDSA_CHECK_KEY) || \ - (!defined(WOLFSSL_MLDSA_NO_SIGN) && \ - !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM)) + defined(MLDSA_SIGN_VEC_HELPERS) /* Make values in polynomials of vector be in positive range. * * @param [in, out] a Vector of polynomials. @@ -8935,8 +9146,10 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) /* Allocate memory for large intermediates. */ if (ret == 0) { - /* s1-l, s2-k, t-k, a-1 */ - allocSz = (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz + + /* s1-l, s2-k, a-1. t is encoded a polynomial at a time, so it and the + * one decoded s2 polynomial share the s2 vector, which is dead once + * s2 has been encoded into the private key. */ + allocSz = (unsigned int)params->s1Sz + params->s2Sz + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE + (unsigned int)MLDSA_POLY_SIZE; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 @@ -8949,8 +9162,8 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) } else { s2 = s1 + params->s1Sz / sizeof(*s1); - t = s2 + params->s2Sz / sizeof(*s2); - h = (byte*)(t + params->s2Sz / sizeof(*t)); + t = s2; + h = (byte*)(s2 + params->s2Sz / sizeof(*s2)); a = (sword32*)(h + MLDSA_REJ_NTT_POLY_H_SIZE); #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 t64 = (sword64*)(a + MLDSA_N); @@ -8997,8 +9210,11 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) byte* t0 = s2p + params->s2EncSz; byte* t1 = key->p + MLDSA_PUB_SEED_SZ; byte aseed[MLDSA_GEN_A_SEED_SZ]; - sword32* s2t = s2; + /* One decoded s2 polynomial, held after t in the dead s2 vector. */ + sword32* s2t = t + MLDSA_N; sword32* tt = t; + const byte* s2pt = s2p; + word32 s2Stride = (word32)params->s2EncSz / params->k; /* Step 9: Move k down to after public seed. */ XMEMCPY(k, k + MLDSA_PRIV_SEED_SZ, MLDSA_K_SZ); @@ -9110,26 +9326,43 @@ static int mldsa_make_key_from_seed(wc_MlDsaKey* key, const byte* seed) } #endif mldsa_invntt_full(tt); + /* s2 was overwritten by t, so recover this polynomial from the + * copy already encoded into the private key. */ + mldsa_vec_decode_eta_bits(s2pt, params->eta, s2t, 1); mldsa_add(tt, s2t); /* Make positive for decomposing. */ mldsa_make_pos(tt); - tt += MLDSA_N; - s2t += MLDSA_N; - } + /* Step 6, Step 7, Step 9. Alg 22 Steps 2-4, Alg 24 Steps 8-10. + * Decompose t in t0 and t1 and encode into public and private + * key. One polynomial at a time trades the AVX-512 paired encoder + * for the s2Sz saving this implementation exists to make. */ + mldsa_vec_encode_t0_t1(tt, 1, t0, t1); - /* Step 6, Step 7, Step 9. Alg 22 Steps 2-4, Alg 24 Steps 8-10. - * Decompose t in t0 and t1 and encode into public and private key. - */ - mldsa_vec_encode_t0_t1(t, params->k, t0, t1); - /* Step 8. Alg 24, Step 1: Hash public key into private key. */ - ret = mldsa_shake256(&key->shake, key->p, params->pkSz, tr, - MLDSA_TR_SZ); + s2pt += s2Stride; + t0 += MLDSA_D * MLDSA_N / 8; + t1 += MLDSA_U * MLDSA_N / 8; + } + if (ret == 0) { + /* Step 8. Alg 24, Step 1: Hash public key into private key. */ + ret = mldsa_shake256(&key->shake, key->p, params->pkSz, tr, + MLDSA_TR_SZ); + } } if (ret == 0) { /* Public key and private key are available. */ key->prvKeySet = 1; key->pubKeySet = 1; + /* Any cached matrix or vectors belong to the key this replaced. */ +#ifdef WC_MLDSA_CACHE_MATRIX_A + key->aSet = 0; +#endif +#ifdef WC_MLDSA_CACHE_PRIV_VECTORS + key->privVecsSet = 0; +#endif +#ifdef WC_MLDSA_CACHE_PUB_VECTORS + key->pubVecSet = 0; +#endif } /* Zeroize the whole buffer before freeing. It holds the private vectors @@ -9519,7 +9752,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_expand_mask(&key->shake, priv_rand_seed, kappa, params->gamma1_bits, y, params->l, key->heap); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_vec_check_low(y, params->l, + valid = mldsa_vec_check_low_ct(y, params->l, ((sword32)1 << params->gamma1_bits) - params->beta); if (valid) #endif @@ -9544,7 +9777,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_make_pos(w, params->k); mldsa_vec_decompose(w, params->k, params->gamma2, w0, w1); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_vec_check_low(w0, params->k, + valid = mldsa_vec_check_low_ct(w0, params->k, params->gamma2 - params->beta); } if (valid) { @@ -9580,7 +9813,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Step 22: w0 - cs2 */ mldsa_sub(w0 + i * MLDSA_N, cs2 + i * MLDSA_N); /* Step 23: Check w0 - cs2 has low enough values. */ - valid = mldsa_vec_check_low(w0 + i * MLDSA_N, 1, hi); + valid = mldsa_check_low_ct(w0 + i * MLDSA_N, hi); } hi = ((sword32)1 << params->gamma1_bits) - params->beta; for (i = 0; valid && i < params->l; i++) { @@ -9591,15 +9824,21 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_add(z + i * MLDSA_N, y + i * MLDSA_N); mldsa_poly_red(z + i * MLDSA_N); /* Step 23: Check z has low enough values. */ - valid = mldsa_vec_check_low(z + i * MLDSA_N, 1, hi); + valid = mldsa_check_low_ct(z + i * MLDSA_N, hi); } - hi = params->gamma2; - for (i = 0; valid && i < params->k; i++) { - /* Step 25: ct0 = NTT-1(c o t0) */ - mldsa_mul_invntt(ct0 + i * MLDSA_N, c, - t0 + i * MLDSA_N); - /* Step 27: Check ct0 has low enough values. */ - valid = mldsa_vec_check_low(ct0 + i * MLDSA_N, 1, hi); + if (valid) { + hi = params->gamma2; + /* Unlike z and w0-cs2, ct0 carries no uniform mask, so + * the index of a failing polynomial depends on t0. + * Check them all. */ + for (i = 0; i < params->k; i++) { + /* Step 25: ct0 = NTT-1(c o t0) */ + mldsa_mul_invntt(ct0 + i * MLDSA_N, c, + t0 + i * MLDSA_N); + /* Step 27: Check ct0 has low enough values. */ + valid &= mldsa_check_low_ct(ct0 + i * MLDSA_N, + hi); + } } if (valid) { /* Step 26: ct0 = ct0 + w0 */ @@ -9642,6 +9881,12 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #ifdef WOLFSSL_CHECK_MEM_ZERO wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); #endif + /* Parts of the commit and z are written into the caller's buffer as they + * are produced. Do not leave them there on failure. The length is only + * set once the buffer was known to be big enough. */ + if ((ret != 0) && (*sigLen == params->sigSz)) { + ForceZero(sig, params->sigSz); + } if (y != NULL) { word32 zeroSz = allocSz; #ifndef WC_MLDSA_CACHE_MATRIX_A @@ -9654,7 +9899,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } XFREE(y, key->heap, DYNAMIC_TYPE_MLDSA); return ret; -#else +#elif !defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) int ret = 0; const wc_MlDsaParams* params = key->params; const byte* pub_seed = key->k; @@ -9673,17 +9918,17 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, sword32* y = NULL; sword32* y_ntt = NULL; sword32* w0 = NULL; - sword32* w1 = NULL; + sword32* w = NULL; sword32* c = NULL; sword32* z = NULL; sword32* ct0 = NULL; -#ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - sword64* t64 = NULL; -#endif byte* blocks = NULL; + byte* w1e = NULL; byte priv_rand_seed[MLDSA_Y_SEED_SZ]; byte* h = sig + params->lambda / 4 + params->zEncSz; unsigned int allocSz = 0; + /* Bytes of encoded w1 per polynomial. */ + unsigned int w1Stride = (unsigned int)params->w1EncSz / params->k; #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A byte maxK = (byte)min(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A, params->k); @@ -9711,20 +9956,25 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Allocate memory for large intermediates. */ if (ret == 0) { - /* y-l, w0-k, w1-k, blocks, c-1, z-1, A-1 */ - allocSz = (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz + - (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE + + /* y-l, w0-k, w-1, c-1, z-1, A-1, w1e, blocks. + * Walking A a column at a time makes w0 the vector of accumulators, + * w1 is only kept encoded, and w is the scratch for the hints. */ + allocSz = (unsigned int)params->s1Sz + params->s2Sz + (unsigned int)MLDSA_POLY_SIZE + (unsigned int)MLDSA_POLY_SIZE + - (unsigned int)MLDSA_POLY_SIZE; + (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)params->w1EncSz + + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE; #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC allocSz += (unsigned int)params->s1Sz + params->s2Sz + params->s2Sz; #elif defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) - allocSz += (unsigned int)maxK * params->l * + /* w is decomposed and encoded as a full k vector regardless of how + * many rows of A are pre-calculated, so it must hold k polynomials. */ + allocSz += (unsigned int)(params->k - 1) * + (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)maxK * params->l * (unsigned int)MLDSA_POLY_SIZE; - #endif - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - allocSz += (unsigned int)MLDSA_POLY_SIZE * 2U; #endif y = (sword32*)XMALLOC(allocSz, key->heap, DYNAMIC_TYPE_MLDSA); if (y == NULL) { @@ -9735,9 +9985,20 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, y_check = y; #endif w0 = y + params->s1Sz / sizeof(*y_ntt); - w1 = w0 + params->s2Sz / sizeof(*w0); - blocks = (byte*)(w1 + params->s2Sz / sizeof(*w1)); - c = (sword32*)(blocks + MLDSA_REJ_NTT_POLY_H_SIZE); + w = w0 + params->s2Sz / sizeof(*w0); + #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A) + /* Only maxK rows of w are multiplied into, but the ML-DSA-44 + * decompose kernels take no dimension and touch all k. One + * zeroing covers every rejection round: decompose of zero is + * (0, 0), so the rows from maxK up keep the value they hold. */ + if (maxK < params->k) { + XMEMSET(w + (unsigned int)maxK * MLDSA_N, 0, + (size_t)(params->k - maxK) * MLDSA_POLY_SIZE); + } + c = w + (unsigned int)params->k * MLDSA_N; + #else + c = w + MLDSA_N; + #endif z = c + MLDSA_N; a = z + MLDSA_N; ct0 = z; @@ -9746,25 +10007,22 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, s1 = z; s2 = z; t0 = z; - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(a + (1 + maxK * params->l) * MLDSA_N); - #endif + w1e = (byte*)(a + (1 + maxK * params->l) * MLDSA_N); + blocks = w1e + params->w1EncSz; #elif defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) y_ntt = z; s1 = a + MLDSA_N; s2 = s1 + params->s1Sz / sizeof(*s1); t0 = s2 + params->s2Sz / sizeof(*s2); - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(t0 + params->s2Sz / sizeof(*t0)); - #endif + w1e = (byte*)(t0 + params->s2Sz / sizeof(*t0)); + blocks = w1e + params->w1EncSz; #else y_ntt = z; s1 = z; s2 = z; t0 = z; - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - t64 = (sword64*)(a + MLDSA_N); - #endif + w1e = (byte*)(a + MLDSA_N); + blocks = w1e + params->w1EncSz; #endif } } @@ -9794,23 +10052,24 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, /* Step 11: Start rejection sampling loop */ do { byte aseed[MLDSA_GEN_A_SEED_SZ]; - WC_DECLARE_VAR(w1e, byte, MLDSA_MAX_W1_ENC_SZ, 0); - sword32* w = w1; byte* commit = sig; byte r; byte s; + byte rStart; sword32 hi; - sword32* wt = w; + sword32* wt; sword32* w0t = w0; - sword32* w1t = w1; + byte* w1et = w1e; sword32* at = a; + sword32* y_ntt_t; + #ifdef WC_MLDSA_FAULT_HARDEN + const sword32* yc = y; + #endif #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - w0t += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * MLDSA_N; - w1t += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * MLDSA_N; - wt += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * MLDSA_N; - at += WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A * params->l * - MLDSA_N; + w0t += (unsigned int)maxK * MLDSA_N; + w1et += (unsigned int)maxK * w1Stride; + at += (unsigned int)maxK * params->l * MLDSA_N; #endif valid = 1; @@ -9818,7 +10077,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_expand_mask(&key->shake, priv_rand_seed, kappa, params->gamma1_bits, y, params->l, key->heap); #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y - valid = mldsa_vec_check_low(y, params->l, + valid = mldsa_vec_check_low_ct(y, params->l, ((sword32)1 << params->gamma1_bits) - params->beta); #endif @@ -9828,63 +10087,72 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_vec_ntt_full(y_ntt, params->l); mldsa_matrix_mul(w, a, y_ntt, maxK, params->l); #ifdef WOLFSSL_MLDSA_SMALL - mldsa_vec_red(w, params->k); + mldsa_vec_red(w, maxK); #endif mldsa_vec_invntt_full(w, maxK); /* Step 14, Step 22: Make values positive and decompose. */ mldsa_vec_make_pos(w, maxK); - mldsa_vec_decompose(w, maxK, params->gamma2, w0, w1); + mldsa_vec_decompose(w, maxK, params->gamma2, w0, w); + /* Step 15: Encode the w1 polynomials just computed. */ + mldsa_vec_encode_w1(w, maxK, params->gamma2, w1e); #endif /* Step 5: Create the matrix A from the public seed. */ /* Copy the seed into a buffer that has space for s and r. */ XMEMCPY(aseed, pub_seed, MLDSA_PUB_SEED_SZ); #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - r = WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A; + rStart = (byte)maxK; + y_ntt_t = z; #else - r = 0; + rStart = 0; + y_ntt_t = y_ntt; #endif - /* Alg 26. Step 1: Loop over first dimension of matrix. */ - for (; (ret == 0) && valid && (r < params->k); r++) { + /* Alg 26. Step 2: Loop over second dimension of matrix. + * A column at a time transforms each polynomial of y once. With + * every row pre-calculated there is nothing left to stream. */ + for (s = 0; (ret == 0) && valid && (rStart < params->k) && + (s < params->l); s++) { unsigned int e; - sword32* yt = y; - #ifdef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A - sword32* y_ntt_t = z; - #else - sword32* y_ntt_t = y_ntt; - #endif - #ifdef WC_MLDSA_FAULT_HARDEN - sword32* yt_check = yt; - #endif + #ifdef WC_MLDSA_FAULT_HARDEN if (y_check != y) { valid = 0; ret = BAD_COND_E; break; } + /* yc walks y independently of s: a fault in either the index + * or the pointer breaks the agreement. */ + if (yc != y + (unsigned int)s * MLDSA_N) { + valid = 0; + ret = BAD_COND_E; + break; + } #endif + /* Step 13: NTT(y) for this column of the matrix. */ + XMEMCPY(y_ntt_t, y + (unsigned int)s * MLDSA_N, + MLDSA_POLY_SIZE); + mldsa_ntt_full(y_ntt_t); - /* Put r/i into buffer to be hashed. */ - aseed[MLDSA_PUB_SEED_SZ + 1] = r; - /* Alg 26. Step 2: Loop over second dimension of matrix. */ - for (s = 0; s < params->l; s++) { - /* Put s into buffer to be hashed. */ - aseed[MLDSA_PUB_SEED_SZ + 0] = s; + /* Put s into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 0] = s; + wt = w0t; + /* Alg 26. Step 1: Loop over first dimension of matrix. */ + for (r = rStart; r < params->k; r++) { + #ifdef WC_MLDSA_FAULT_HARDEN + if (wt != w0t + (unsigned int)(r - rStart) * MLDSA_N) { + valid = 0; + ret = BAD_COND_E; + break; + } + #endif + /* Put r/i into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 1] = r; /* Alg 26. Step 3: Create polynomial from hashing seed. */ ret = mldsa_rej_ntt_poly_ex(&key->shake, aseed, at, blocks); if (ret != 0) { break; } - XMEMCPY(y_ntt_t, yt, MLDSA_POLY_SIZE); - #ifdef WC_MLDSA_FAULT_HARDEN - if (yt_check + s * MLDSA_N != yt) { - ret = BAD_COND_E; - break; - } - #endif - mldsa_ntt_full(y_ntt_t); - /* Matrix multiply. */ - #ifndef WOLFSSL_MLDSA_SMALL_MEM_POLY64 + /* Step 13: A o NTT(y), accumulated down the column. */ if (s == 0) { #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { @@ -9939,83 +10207,51 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, } #endif } - #else - if (s == 0) { - #ifdef WOLFSSL_MLDSA_SMALL - for (e = 0; e < MLDSA_N; e++) { - t64[e] = (sword64)at[e] * y_ntt_t[e]; - } - #else - for (e = 0; e < MLDSA_N; e += 8) { - t64[e+0] = (sword64)at[e+0] * y_ntt_t[e+0]; - t64[e+1] = (sword64)at[e+1] * y_ntt_t[e+1]; - t64[e+2] = (sword64)at[e+2] * y_ntt_t[e+2]; - t64[e+3] = (sword64)at[e+3] * y_ntt_t[e+3]; - t64[e+4] = (sword64)at[e+4] * y_ntt_t[e+4]; - t64[e+5] = (sword64)at[e+5] * y_ntt_t[e+5]; - t64[e+6] = (sword64)at[e+6] * y_ntt_t[e+6]; - t64[e+7] = (sword64)at[e+7] * y_ntt_t[e+7]; - } - #endif - } - else { - #ifdef WOLFSSL_MLDSA_SMALL - for (e = 0; e < MLDSA_N; e++) { - t64[e] += (sword64)at[e] * y_ntt_t[e]; - } - #else - for (e = 0; e < MLDSA_N; e += 8) { - t64[e+0] += (sword64)at[e+0] * y_ntt_t[e+0]; - t64[e+1] += (sword64)at[e+1] * y_ntt_t[e+1]; - t64[e+2] += (sword64)at[e+2] * y_ntt_t[e+2]; - t64[e+3] += (sword64)at[e+3] * y_ntt_t[e+3]; - t64[e+4] += (sword64)at[e+4] * y_ntt_t[e+4]; - t64[e+5] += (sword64)at[e+5] * y_ntt_t[e+5]; - t64[e+6] += (sword64)at[e+6] * y_ntt_t[e+6]; - t64[e+7] += (sword64)at[e+7] * y_ntt_t[e+7]; - } - #endif - } - #endif - /* Next polynomial. */ - yt += MLDSA_N; - } - if (ret != 0) { - break; - } - #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 - for (e = 0; e < MLDSA_N; e++) { - wt[e] = mldsa_mont_red(t64[e]); + wt += MLDSA_N; } + #ifdef WC_MLDSA_FAULT_HARDEN + yc += MLDSA_N; #endif - mldsa_invntt_full(wt); + } + + /* Steps 13-15: Invert transform, decompose and encode each row of + * w now that every column has been accumulated into it. */ + for (r = rStart; (ret == 0) && (r < params->k); r++) { + unsigned int e; + + /* Step 13: w = NTT-1(A o NTT(y)) */ + mldsa_poly_red(w0t); + mldsa_invntt_full(w0t); /* Step 14, Step 22: Make values positive and decompose. */ - mldsa_make_pos(wt); + mldsa_make_pos(w0t); #ifndef WOLFSSL_NO_ML_DSA_44 if (params->gamma2 == MLDSA_Q_LOW_88) { /* For each value of polynomial. */ for (e = 0; e < MLDSA_N; e++) { - /* Decompose value into two vectors. */ - mldsa_decompose_q88(wt[e], &w0t[e], &w1t[e]); + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q88(w0t[e], &w0t[e], &at[e]); } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_88(at, w1et); } #endif #if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) if (params->gamma2 == MLDSA_Q_LOW_32) { /* For each value of polynomial. */ for (e = 0; e < MLDSA_N; e++) { - /* Decompose value into two vectors. */ - mldsa_decompose_q32(wt[e], &w0t[e], &w1t[e]); + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q32(w0t[e], &w0t[e], &at[e]); } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_32(at, w1et); } #endif #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 - valid = mldsa_vec_check_low(w0t, + valid &= mldsa_check_low_ct(w0t, params->gamma2 - params->beta); #endif - wt += MLDSA_N; w0t += MLDSA_N; - w1t += MLDSA_N; + w1et += w1Stride; } if ((ret == 0) && valid) { sword32* yt = y; @@ -10024,18 +10260,10 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #endif byte* ze = sig + params->lambda / 4; - /* Step 15: Encode w1. */ - WC_ALLOC_VAR_EX(w1e, byte, MLDSA_MAX_W1_ENC_SZ, - key->heap, DYNAMIC_TYPE_MLDSA, ret=MEMORY_E); - if (WC_VAR_OK(w1e)) { - mldsa_vec_encode_w1(w1, params->k, params->gamma2, - w1e); - /* Step 15: Hash mu and encoded w1. - * Step 32: Hash is stored in signature. */ - ret = mldsa_hash256(&key->shake, mu, MLDSA_MU_SZ, - w1e, params->w1EncSz, commit, params->lambda / 4); - } - WC_FREE_VAR_EX(w1e, key->heap, DYNAMIC_TYPE_MLDSA); + /* Step 15: Hash mu and encoded w1. + * Step 32: Hash is stored in signature. */ + ret = mldsa_hash256(&key->shake, mu, MLDSA_MU_SZ, + w1e, params->w1EncSz, commit, params->lambda / 4); if (ret == 0) { /* Step 17: Compute c from first 256 bits of commit. */ ret = mldsa_sample_in_ball_ex(params->level, @@ -10076,7 +10304,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_poly_red(z); /* Step 23: Check z has low enough values. */ hi = ((sword32)1 << params->gamma1_bits) - params->beta; - valid = mldsa_check_low(z, hi); + valid = mldsa_check_low_ct(z, hi); if (valid) { /* Step 32: Encode z into signature. * Commit (c) and h already encoded into signature. */ @@ -10109,8 +10337,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #endif sword32* cs2 = ct0; byte idx = 0; + int ct0Valid = 1; w0t = w0; - w1t = w1; + w1et = w1e; for (r = 0; valid && (r < params->k); r++) { #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC @@ -10142,7 +10371,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_poly_red(w0t); /* Step 23: Check w0 - cs2 has low enough values. */ hi = params->gamma2 - params->beta; - valid = mldsa_check_low(w0t, hi); + valid = mldsa_check_low_ct(w0t, hi); if (valid) { #ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC mldsa_decode_t0(t0pt, t0); @@ -10155,21 +10384,24 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, mldsa_mul(ct0, c, t0 + r * MLDSA_N); #endif mldsa_invntt(ct0); - /* Step 27: Check ct0 has low enough values. */ - valid = mldsa_check_low(ct0, params->gamma2); - } - if (valid) { + /* Step 27: Check ct0 has low enough values. Every + * row: ct0 has no mask, so a failing index leaks t0. */ + ct0Valid &= mldsa_check_low_ct(ct0, params->gamma2); /* Step 26: ct0 = ct0 + w0 */ mldsa_add(ct0, w0t); mldsa_poly_red(ct0); + /* w1 is only kept encoded, so recover the polynomial + * the hint needs. */ + mldsa_decode_w1(w1et, params->gamma2, w); + /* Step 26, 27: Make hint from ct0 and w1 and check * number of hints is valid. * Step 32: h is encoded into signature. */ #ifndef WOLFSSL_NO_ML_DSA_44 if (params->gamma2 == MLDSA_Q_LOW_88) { - valid = (mldsa_make_hint_88(ct0, w1t, h, + valid = (mldsa_make_hint_88(ct0, w, h, &idx) == 0); /* Alg 14, Step 10: Store count of hints for * polynomial at end of list. */ @@ -10179,7 +10411,7 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #if !defined(WOLFSSL_NO_ML_DSA_65) || \ !defined(WOLFSSL_NO_ML_DSA_87) if (params->gamma2 == MLDSA_Q_LOW_32) { - valid = (mldsa_make_hint_32(ct0, w1t, + valid = (mldsa_make_hint_32(ct0, w, params->omega, h, &idx) == 0); /* Alg 14, Step 10: Store count of hints for * polynomial at end of list. */ @@ -10192,8 +10424,9 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, t0pt += MLDSA_D * MLDSA_N / 8; #endif w0t += MLDSA_N; - w1t += MLDSA_N; + w1et += w1Stride; } + valid &= ct0Valid; /* Set remaining hints to zero. */ XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); } @@ -10217,11 +10450,426 @@ static int mldsa_sign_with_seed_mu(wc_MlDsaKey* key, #ifdef WOLFSSL_CHECK_MEM_ZERO wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); #endif + /* Parts of the commit and z are written into the caller's buffer as they + * are produced. Do not leave them there on failure. The length is only + * set once the buffer was known to be big enough. */ + if ((ret != 0) && (*sigLen == params->sigSz)) { + ForceZero(sig, params->sigSz); + } if (y != NULL) { ForceZero(y, allocSz); } XFREE(y, key->heap, DYNAMIC_TYPE_MLDSA); return ret; +#else + int ret = 0; + const wc_MlDsaParams* params = key->params; + const byte* pub_seed = key->k; + const byte* k = pub_seed + MLDSA_PUB_SEED_SZ; + const byte* tr = k + MLDSA_K_SZ; + const byte* s1p = tr + MLDSA_TR_SZ; + const byte* s2p = s1p + params->s1EncSz; + const byte* t0p = s2p + params->s2EncSz; + const byte* mu = seedMu + MLDSA_RND_SZ; + sword32* w = NULL; + sword32* y = NULL; + sword32* a = NULL; + sword32* c = NULL; + sword32* z = NULL; + byte* w1e = NULL; + byte* blocks = NULL; + byte priv_rand_seed[MLDSA_Y_SEED_SZ]; + byte* h = sig + params->lambda / 4 + params->zEncSz; + unsigned int allocSz = 0; + /* Bytes of encoded w1 per polynomial. */ + unsigned int w1Stride = (unsigned int)params->w1EncSz / params->k; + /* Bytes of encoded s1 or s2 per polynomial. */ + unsigned int sStride = (unsigned int)params->s2EncSz / params->k; + /* Checksum of each polynomial of y, to bind the two derivations. */ + sword32 yChk[MLDSA_MAX_L_VECTOR_COUNT / MLDSA_N]; +#ifdef WC_MLDSA_FAULT_HARDEN + sword32* w_check; +#endif + + /* priv_rand_seed will hold the secret signing seed (rho'') derived below + * and yChk a checksum over each polynomial of the secret mask y; + * baseline-zero and register both up front (single-exit function) so any + * later exit before the ForceZero is covered. */ +#ifdef WOLFSSL_CHECK_MEM_ZERO + XMEMSET(priv_rand_seed, 0, sizeof(priv_rand_seed)); + wc_MemZero_Add("mldsa sign priv_rand_seed", priv_rand_seed, + sizeof(priv_rand_seed)); + XMEMSET(yChk, 0, sizeof(yChk)); + wc_MemZero_Add("mldsa sign yChk", yChk, sizeof(yChk)); +#endif + /* Check the signature buffer isn't too small. */ + if (*sigLen < params->sigSz) { + ret = BUFFER_E; + } + if (ret == 0) { + /* Return the size of the signature. */ + *sigLen = params->sigSz; + } + + /* Allocate memory for large intermediates. */ + if (ret == 0) { + /* w-k, y-1, a-1, c-1, z-1, w1e, blocks. + * w0 replaces w in place and w1 is only kept encoded, so w is the + * only vector held. */ + allocSz = (unsigned int)params->s2Sz + + 4U * (unsigned int)MLDSA_POLY_SIZE + + (unsigned int)params->w1EncSz + + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE; + w = (sword32*)XMALLOC(allocSz, key->heap, DYNAMIC_TYPE_MLDSA); + if (w == NULL) { + ret = MEMORY_E; + } + else { + #ifdef WC_MLDSA_FAULT_HARDEN + w_check = w; + #endif + y = w + params->s2Sz / sizeof(*w); + a = y + MLDSA_N; + c = a + MLDSA_N; + z = c + MLDSA_N; + w1e = (byte*)(z + MLDSA_N); + blocks = w1e + params->w1EncSz; + } + } + + if (ret == 0) { + /* Step 9: Compute private random using hash. */ + ret = mldsa_hash256(&key->shake, k, MLDSA_K_SZ, seedMu, + MLDSA_RND_SZ + MLDSA_MU_SZ, priv_rand_seed, + MLDSA_PRIV_RAND_SEED_SZ); + } + if (ret == 0) { + word16 kappa = 0; + int valid; + + /* Step 11: Start rejection sampling loop */ + do { + byte aseed[MLDSA_GEN_A_SEED_SZ]; + byte* commit = sig; + byte* w1et = w1e; + const byte* sp; + sword32* wt; + unsigned int e; + sword32 hi; + byte r; + byte s; + + valid = 1; + /* Copy the seed into a buffer that has space for s and r. */ + XMEMCPY(aseed, pub_seed, MLDSA_PUB_SEED_SZ); + + /* Alg 26. Step 2: Loop over second dimension of matrix. Working + * down the columns keeps one polynomial of y and transforms it + * once rather than once per row. */ + for (s = 0; (ret == 0) && (s < params->l); s++) { + #ifdef WC_MLDSA_FAULT_HARDEN + if (w_check != w) { + valid = 0; + ret = BAD_COND_E; + break; + } + #endif + /* Step 12: Compute polynomial of y from seed and kappa. */ + /* z is not live yet, so it doubles as the expansion scratch + * rather than allocating one per polynomial. */ + ret = mldsa_expand_mask_poly(&key->shake, priv_rand_seed, + (word16)(kappa + s), params->gamma1_bits, y, (byte*)z); + if (ret != 0) { + break; + } + /* Bind this polynomial of y to the one regenerated for z. */ + yChk[s] = mldsa_poly_checksum(y); + #ifdef WOLFSSL_MLDSA_SIGN_CHECK_Y + /* Accumulate rather than leave the loop: an early exit would + * make the number of polynomials of matrix A generated depend + * on the secret mask y. */ + valid &= mldsa_check_low_ct(y, + ((sword32)1 << params->gamma1_bits) - params->beta); + #endif + /* Step 13: NTT(y) */ + mldsa_ntt_full(y); + + /* Put s into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 0] = s; + /* Alg 26. Step 1: Loop over first dimension of matrix. */ + for (r = 0; r < params->k; r++) { + /* Put r/i into buffer to be hashed. */ + aseed[MLDSA_PUB_SEED_SZ + 1] = r; + /* Alg 26. Step 3: Create polynomial from hashing seed. */ + ret = mldsa_rej_ntt_poly_ex(&key->shake, aseed, a, blocks); + if (ret != 0) { + break; + } + wt = w + (unsigned int)r * MLDSA_N; + /* Step 13: A o NTT(y), accumulated down the column. */ + if (s == 0) { + #ifdef WOLFSSL_MLDSA_SMALL + for (e = 0; e < MLDSA_N; e++) { + wt[e] = mldsa_mont_red((sword64)a[e] * y[e]); + } + #else + for (e = 0; e < MLDSA_N; e += 8) { + wt[e+0] = mldsa_mont_red((sword64)a[e+0] * y[e+0]); + wt[e+1] = mldsa_mont_red((sword64)a[e+1] * y[e+1]); + wt[e+2] = mldsa_mont_red((sword64)a[e+2] * y[e+2]); + wt[e+3] = mldsa_mont_red((sword64)a[e+3] * y[e+3]); + wt[e+4] = mldsa_mont_red((sword64)a[e+4] * y[e+4]); + wt[e+5] = mldsa_mont_red((sword64)a[e+5] * y[e+5]); + wt[e+6] = mldsa_mont_red((sword64)a[e+6] * y[e+6]); + wt[e+7] = mldsa_mont_red((sword64)a[e+7] * y[e+7]); + } + #endif + } + else { + #ifdef WOLFSSL_MLDSA_SMALL + for (e = 0; e < MLDSA_N; e++) { + wt[e] += mldsa_mont_red((sword64)a[e] * y[e]); + } + #else + for (e = 0; e < MLDSA_N; e += 8) { + wt[e+0] += mldsa_mont_red((sword64)a[e+0] * y[e+0]); + wt[e+1] += mldsa_mont_red((sword64)a[e+1] * y[e+1]); + wt[e+2] += mldsa_mont_red((sword64)a[e+2] * y[e+2]); + wt[e+3] += mldsa_mont_red((sword64)a[e+3] * y[e+3]); + wt[e+4] += mldsa_mont_red((sword64)a[e+4] * y[e+4]); + wt[e+5] += mldsa_mont_red((sword64)a[e+5] * y[e+5]); + wt[e+6] += mldsa_mont_red((sword64)a[e+6] * y[e+6]); + wt[e+7] += mldsa_mont_red((sword64)a[e+7] * y[e+7]); + } + #endif + } + } + } + + wt = w; + for (r = 0; (ret == 0) && (r < params->k); r++) { + /* Step 13: w = NTT-1(A o NTT(y)) */ + mldsa_poly_red(wt); + mldsa_invntt_full(wt); + /* Step 14, Step 22: Make values positive and decompose. */ + mldsa_make_pos(wt); + #ifndef WOLFSSL_NO_ML_DSA_44 + if (params->gamma2 == MLDSA_Q_LOW_88) { + /* For each value of polynomial. */ + for (e = 0; e < MLDSA_N; e++) { + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q88(wt[e], &wt[e], &a[e]); + } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_88(a, w1et); + } + #endif + #if !defined(WOLFSSL_NO_ML_DSA_65) || !defined(WOLFSSL_NO_ML_DSA_87) + if (params->gamma2 == MLDSA_Q_LOW_32) { + /* For each value of polynomial. */ + for (e = 0; e < MLDSA_N; e++) { + /* w0 replaces w, w1 goes to the scratch polynomial. */ + mldsa_decompose_q32(wt[e], &wt[e], &a[e]); + } + /* Step 15: Encode this polynomial of w1. */ + mldsa_encode_w1_32(a, w1et); + } + #endif + #ifdef WOLFSSL_MLDSA_SIGN_CHECK_W0 + /* Accumulate: w0 comes from the secret mask, so an early exit + * would leak which row rejected. */ + valid &= mldsa_check_low_ct(wt, + params->gamma2 - params->beta); + #endif + wt += MLDSA_N; + w1et += w1Stride; + } + + if ((ret == 0) && valid) { + /* Step 15: Hash mu and encoded w1. + * Step 32: Hash is stored in signature. */ + ret = mldsa_hash256(&key->shake, mu, MLDSA_MU_SZ, w1e, + params->w1EncSz, commit, params->lambda / 4); + } + if ((ret == 0) && valid) { + /* Step 17: Compute c from first 256 bits of commit. */ + ret = mldsa_sample_in_ball_ex(params->level, &key->shake, + commit, params->lambda / 4, params->tau, c, blocks); + } + if ((ret == 0) && valid) { + /* Step 18: NTT(c). */ + mldsa_ntt_small(c); + } + + if ((ret == 0) && valid) { + byte* ze = sig + params->lambda / 4; + + sp = s1p; + hi = ((sword32)1 << params->gamma1_bits) - params->beta; + /* z and w0 - cs2 stop at the first rejecting polynomial of a + * discarded round; y masks which one it was. */ + for (s = 0; (ret == 0) && valid && (s < params->l); s++) { + /* Vector y is not kept, so regenerate this polynomial. */ + /* z is overwritten by the multiply below, so it doubles + * as the expansion scratch. */ + ret = mldsa_expand_mask_poly(&key->shake, priv_rand_seed, + (word16)(kappa + s), params->gamma1_bits, y, (byte*)z); + if (ret != 0) { + break; + } + /* y must match the derivation w was built from. */ + if (yChk[s] != mldsa_poly_checksum(y)) { + valid = 0; + ret = BAD_COND_E; + break; + } + mldsa_vec_decode_eta_bits(sp, params->eta, a, 1); + mldsa_ntt_small(a); + /* Step 19: cs1 = NTT-1(c o s1) */ + mldsa_mul(z, c, a); + mldsa_invntt(z); + /* Step 21: z = y + cs1 */ + mldsa_add(z, y); + mldsa_poly_red(z); + /* Step 23: Check z has low enough values. */ + valid = mldsa_check_low_ct(z, hi); + if (valid) { + /* Step 32: Encode z into signature. + * Commit (c) and h already encoded into signature. */ + #if !defined(WOLFSSL_NO_ML_DSA_44) + if (params->gamma1_bits == MLDSA_GAMMA1_BITS_17) { + mldsa_encode_gamma1_17_bits(z, ze); + /* Move to next place to encode to. */ + ze += MLDSA_GAMMA1_17_ENC_BITS / 2 * MLDSA_N / 4; + } + #endif + #if !defined(WOLFSSL_NO_ML_DSA_65) || \ + !defined(WOLFSSL_NO_ML_DSA_87) + if (params->gamma1_bits == MLDSA_GAMMA1_BITS_19) { + mldsa_encode_gamma1_19_bits(z, ze); + /* Move to next place to encode to. */ + ze += MLDSA_GAMMA1_19_ENC_BITS / 2 * MLDSA_N / 4; + } + #endif + } + sp += sStride; + } + } + if ((ret == 0) && valid) { + const byte* t0pt = t0p; + byte idx = 0; + int ct0Valid = 1; + + sp = s2p; + wt = w; + w1et = w1e; + for (r = 0; (ret == 0) && valid && (r < params->k); r++) { + mldsa_vec_decode_eta_bits(sp, params->eta, a, 1); + mldsa_ntt_small(a); + /* Step 20: cs2 = NTT-1(c o s2) */ + mldsa_mul(z, c, a); + mldsa_invntt(z); + /* Step 22: w0 - cs2 */ + mldsa_sub(wt, z); + mldsa_poly_red(wt); + /* Step 23: Check w0 - cs2 has low enough values. */ + valid = mldsa_check_low_ct(wt, + params->gamma2 - params->beta); + if (valid) { + mldsa_decode_t0(t0pt, a); + mldsa_ntt(a); + /* Step 25: ct0 = NTT-1(c o t0) */ + mldsa_mul(z, c, a); + mldsa_invntt(z); + /* Step 27: Check ct0 has low enough values. Every + * row: ct0 has no mask, so a failing index leaks t0. */ + ct0Valid &= mldsa_check_low_ct(z, params->gamma2); + /* Step 26: ct0 = ct0 + w0 */ + mldsa_add(z, wt); + mldsa_poly_red(z); + + /* w1 is only kept encoded, so recover the polynomial + * the hint needs. */ + mldsa_decode_w1(w1et, params->gamma2, a); + + /* Step 26, 27: Make hint from ct0 and w1 and check + * number of hints is valid. + * Step 32: h is encoded into signature. + */ + #ifndef WOLFSSL_NO_ML_DSA_44 + if (params->gamma2 == MLDSA_Q_LOW_88) { + valid = (mldsa_make_hint_88(z, a, h, &idx) == 0); + /* Alg 14, Step 10: Store count of hints for + * polynomial at end of list. */ + h[PARAMS_ML_DSA_44_OMEGA + r] = idx; + } + #endif + #if !defined(WOLFSSL_NO_ML_DSA_65) || \ + !defined(WOLFSSL_NO_ML_DSA_87) + if (params->gamma2 == MLDSA_Q_LOW_32) { + valid = (mldsa_make_hint_32(z, a, params->omega, + h, &idx) == 0); + /* Alg 14, Step 10: Store count of hints for + * polynomial at end of list. */ + h[params->omega + r] = idx; + } + #endif + } + + sp += sStride; + t0pt += MLDSA_D * MLDSA_N / 8; + wt += MLDSA_N; + w1et += w1Stride; + } + valid &= ct0Valid; + /* Set remaining hints to zero. Only for a valid attempt: a + * rejected one leaves a partial count that is never + * published, and the next attempt rebuilds from index 0. */ + if (valid) { + XMEMSET(h + idx, 0, (size_t)(params->omega - idx)); + } + } + + if (!valid) { + /* Too many attempts - something wrong with implementation. */ + if ((kappa > (word16)(kappa + params->l))) { + ret = BAD_COND_E; + } + + /* Step 30: increment value to append to seed to unique value. + */ + kappa = (word16)(kappa + params->l); + } + } + /* Step 11: Check we have a valid signature. */ + while ((ret == 0) && (!valid)); + } + + ForceZero(priv_rand_seed, sizeof(priv_rand_seed)); + /* The last expansion of y leaves rho'' recoverable from key->shake. */ + ForceZero(&key->shake, sizeof(key->shake)); +#ifdef WOLF_CRYPTO_CB + key->shake.devId = INVALID_DEVID; +#endif +#ifdef WOLFSSL_CHECK_MEM_ZERO + wc_MemZero_Check(priv_rand_seed, sizeof(priv_rand_seed)); +#endif + /* Parts of the commit and z are written into the caller's buffer as they + * are produced. Do not leave them there on failure. The length is only + * set once the buffer was known to be big enough. */ + if ((ret != 0) && (*sigLen == params->sigSz)) { + ForceZero(sig, params->sigSz); + } + /* Checksums are derived from the secret mask y. */ + ForceZero(yChk, sizeof(yChk)); +#ifdef WOLFSSL_CHECK_MEM_ZERO + wc_MemZero_Check(yChk, sizeof(yChk)); +#endif + if (w != NULL) { + ForceZero(w, allocSz); + } + XFREE(w, key->heap, DYNAMIC_TYPE_MLDSA); + return ret; #endif } @@ -10936,6 +11584,14 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, #ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM /* Bytes of encoded z per polynomial - z is streamed one poly at a time. */ word32 zStride = (word32)(MLDSA_N / 8) * (word32)(params->gamma1_bits + 1); +#endif +#ifndef WOLFSSL_MLDSA_VERIFY_NO_MALLOC +#ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM + /* Only one polynomial of z is held at a time in this mode. */ + unsigned int zSz = (unsigned int)MLDSA_POLY_SIZE; +#else + unsigned int zSz = (unsigned int)params->s1Sz; +#endif #endif /* Ensure the signature is the right size for the parameters. */ @@ -10953,7 +11609,7 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, /* z, c, w, t1, w1e. */ unsigned int allocSz; - allocSz = (unsigned int)params->s1Sz + params->w1EncSz + + allocSz = zSz + params->w1EncSz + 3U * (unsigned int)MLDSA_POLY_SIZE + (unsigned int)MLDSA_REJ_NTT_POLY_H_SIZE; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 @@ -10965,7 +11621,7 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, } else { XMEMSET(z, 0, allocSz); - c = z + params->s1Sz / sizeof(*t1); + c = z + zSz / sizeof(*t1); w = c + MLDSA_N; t1 = w + MLDSA_N; block = (byte*)(t1 + MLDSA_N); @@ -10981,7 +11637,7 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, z = key->z; c = key->c; w = key->w; - t1 = key->t1; + t1 = key->vt1; w1e = key->w1e; aBuf = t1; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 @@ -11041,46 +11697,58 @@ static int mldsa_verify_with_mu(wc_MlDsaKey* key, const byte* mu, unsigned int s; unsigned int e; const sword32* zt = z; + /* Source of this polynomial of t1 for the multiply below. The + * result goes to w either way, so a cached polynomial is read in + * place rather than copied. */ + const sword32* t1v = w; - /* Step 1: Decode and NTT vector t1. */ - mldsa_decode_t1(t1p, w); +#ifdef WC_MLDSA_CACHE_PUB_VECTORS + if (key->pubVecSet) { + /* Cached vector is already decoded and transformed. */ + t1v = key->t1 + (size_t)r * MLDSA_N; + } + else +#endif + { + /* Step 1: Decode and NTT vector t1. */ + mldsa_decode_t1(t1p, w); + mldsa_ntt_full(w); + } /* Next polynomial. */ t1p += MLDSA_U * MLDSA_N / 8; - /* Step 10: - NTT(c) o NTT(t1)) */ - mldsa_ntt_full(w); #ifndef WOLFSSL_MLDSA_SMALL_MEM_POLY64 #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { - w[e] = -mldsa_mont_red((sword64)c[e] * w[e]); + w[e] = -mldsa_mont_red((sword64)c[e] * t1v[e]); } #else for (e = 0; e < MLDSA_N; e += 8) { - w[e+0] = -mldsa_mont_red((sword64)c[e+0] * w[e+0]); - w[e+1] = -mldsa_mont_red((sword64)c[e+1] * w[e+1]); - w[e+2] = -mldsa_mont_red((sword64)c[e+2] * w[e+2]); - w[e+3] = -mldsa_mont_red((sword64)c[e+3] * w[e+3]); - w[e+4] = -mldsa_mont_red((sword64)c[e+4] * w[e+4]); - w[e+5] = -mldsa_mont_red((sword64)c[e+5] * w[e+5]); - w[e+6] = -mldsa_mont_red((sword64)c[e+6] * w[e+6]); - w[e+7] = -mldsa_mont_red((sword64)c[e+7] * w[e+7]); + w[e+0] = -mldsa_mont_red((sword64)c[e+0] * t1v[e+0]); + w[e+1] = -mldsa_mont_red((sword64)c[e+1] * t1v[e+1]); + w[e+2] = -mldsa_mont_red((sword64)c[e+2] * t1v[e+2]); + w[e+3] = -mldsa_mont_red((sword64)c[e+3] * t1v[e+3]); + w[e+4] = -mldsa_mont_red((sword64)c[e+4] * t1v[e+4]); + w[e+5] = -mldsa_mont_red((sword64)c[e+5] * t1v[e+5]); + w[e+6] = -mldsa_mont_red((sword64)c[e+6] * t1v[e+6]); + w[e+7] = -mldsa_mont_red((sword64)c[e+7] * t1v[e+7]); } #endif #else #ifdef WOLFSSL_MLDSA_SMALL for (e = 0; e < MLDSA_N; e++) { - t64[e] = -(sword64)c[e] * w[e]; + t64[e] = -(sword64)c[e] * t1v[e]; } #else for (e = 0; e < MLDSA_N; e += 8) { - t64[e+0] = -(sword64)c[e+0] * w[e+0]; - t64[e+1] = -(sword64)c[e+1] * w[e+1]; - t64[e+2] = -(sword64)c[e+2] * w[e+2]; - t64[e+3] = -(sword64)c[e+3] * w[e+3]; - t64[e+4] = -(sword64)c[e+4] * w[e+4]; - t64[e+5] = -(sword64)c[e+5] * w[e+5]; - t64[e+6] = -(sword64)c[e+6] * w[e+6]; - t64[e+7] = -(sword64)c[e+7] * w[e+7]; + t64[e+0] = -(sword64)c[e+0] * t1v[e+0]; + t64[e+1] = -(sword64)c[e+1] * t1v[e+1]; + t64[e+2] = -(sword64)c[e+2] * t1v[e+2]; + t64[e+3] = -(sword64)c[e+3] * t1v[e+3]; + t64[e+4] = -(sword64)c[e+4] * t1v[e+4]; + t64[e+5] = -(sword64)c[e+5] * t1v[e+5]; + t64[e+6] = -(sword64)c[e+6] * t1v[e+6]; + t64[e+7] = -(sword64)c[e+7] * t1v[e+7]; } #endif #endif diff --git a/wolfcrypt/test/test.c b/wolfcrypt/test/test.c index 09cd3d50e30..b51c394c2d0 100644 --- a/wolfcrypt/test/test.c +++ b/wolfcrypt/test/test.c @@ -66473,6 +66473,190 @@ static wc_test_ret_t mldsa_param_test(int param, WC_RNG* rng) } #endif +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) +/* Seed the key pair is generated from. */ +static const byte mldsa_kat_key_seed[MLDSA_SEED_SZ] = { + 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, + 0x08, 0x09, 0x0a, 0x0b, 0x0c, 0x0d, 0x0e, 0x0f, + 0x10, 0x11, 0x12, 0x13, 0x14, 0x15, 0x16, 0x17, + 0x18, 0x19, 0x1a, 0x1b, 0x1c, 0x1d, 0x1e, 0x1f +}; +/* Seed used in place of the per-signature randomness. */ +static const byte mldsa_kat_sig_seed[MLDSA_RND_SZ] = { + 0x20, 0x21, 0x22, 0x23, 0x24, 0x25, 0x26, 0x27, + 0x28, 0x29, 0x2a, 0x2b, 0x2c, 0x2d, 0x2e, 0x2f, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x3a, 0x3b, 0x3c, 0x3d, 0x3e, 0x3f +}; +static const byte mldsa_kat_msg[] = { + 0x77, 0x6f, 0x6c, 0x66, 0x53, 0x53, 0x4c, 0x20, /* "wolfSSL " */ + 0x4d, 0x4c, 0x2d, 0x44, 0x53, 0x41, 0x20, 0x4b, /* "ML-DSA K" */ + 0x41, 0x54 /* "AT" */ +}; + +/* Signers deliberately not FIPS 204 conformant have no known answer. */ +#if !defined(WOLFSSL_MLDSA_FIPS204_DRAFT) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_Y) && \ + !defined(WOLFSSL_MLDSA_SIGN_CHECK_W0) +#define MLDSA_KAT_DIGEST(d) (d) +/* SHAKE-256 digests of the signature the default signer produces. */ +#ifndef WOLFSSL_NO_ML_DSA_44 +static const byte mldsa_kat_digest_44[32] = { + 0x17, 0xc1, 0xa1, 0x07, 0x90, 0xe6, 0xce, 0xc3, + 0x38, 0x17, 0x18, 0x02, 0x41, 0xaf, 0x0a, 0x3f, + 0xbd, 0x2c, 0xb9, 0x0d, 0xbc, 0x3f, 0x5d, 0x8b, + 0x07, 0x98, 0xc6, 0xe3, 0x75, 0x66, 0x8b, 0x3c +}; +#endif +#ifndef WOLFSSL_NO_ML_DSA_65 +static const byte mldsa_kat_digest_65[32] = { + 0x10, 0xb5, 0x77, 0xb1, 0x8f, 0xf0, 0x21, 0x0c, + 0x17, 0x31, 0x54, 0xe9, 0x3e, 0x79, 0xc8, 0x05, + 0x22, 0xdf, 0x27, 0x03, 0xfb, 0x99, 0xc0, 0x8b, + 0xf8, 0x25, 0x4c, 0xda, 0x36, 0xf7, 0x6f, 0xb1 +}; +#endif +#ifndef WOLFSSL_NO_ML_DSA_87 +static const byte mldsa_kat_digest_87[32] = { + 0xbe, 0xf0, 0xb7, 0xe5, 0x5f, 0x86, 0x4a, 0xdb, + 0x48, 0xfc, 0x56, 0x80, 0x93, 0x13, 0xdd, 0x96, + 0x08, 0x2d, 0x0f, 0x86, 0x1b, 0xf1, 0x89, 0x52, + 0x9f, 0x97, 0xb9, 0xca, 0xd3, 0x8f, 0xc3, 0xbf +}; +#endif +#else +#define MLDSA_KAT_DIGEST(d) NULL +#endif + +/* A key generated into an object that already held a key must sign and verify + * exactly like one generated into a fresh object, whatever the key caches. */ +static wc_test_ret_t mldsa_make_key_reuse_test(int param, const byte* expDigest) +{ + wc_test_ret_t ret; + wc_MlDsaKey* key = NULL; + wc_MlDsaKey* freshKey = NULL; + byte* sig = NULL; + byte* freshSig = NULL; + word32 sigLen; + word32 freshSigLen; + int sigSz = 0; + byte digest[32]; + wc_Shake shake; + int keyInit = 0; + int freshKeyInit = 0; + int shakeInit = 0; +#ifndef WOLFSSL_MLDSA_NO_VERIFY + int res = 0; +#endif + + key = (wc_MlDsaKey*)XMALLOC(sizeof(wc_MlDsaKey), HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + freshKey = (wc_MlDsaKey*)XMALLOC(sizeof(wc_MlDsaKey), HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + sig = (byte*)XMALLOC(MLDSA_MAX_SIG_SIZE, HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + freshSig = (byte*)XMALLOC(MLDSA_MAX_SIG_SIZE, HEAP_HINT, + DYNAMIC_TYPE_TMP_BUFFER); + if ((key == NULL) || (freshKey == NULL) || (sig == NULL) || + (freshSig == NULL)) + ERROR_OUT(WC_TEST_RET_ENC_NC, out); + + ret = wc_MlDsaKey_Init(freshKey, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + freshKeyInit = 1; + ret = wc_MlDsaKey_SetParams(freshKey, param); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_MlDsaKey_GetSigLen(freshKey, &sigSz); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_MlDsaKey_MakeKeyFromSeed(freshKey, mldsa_kat_key_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + freshSigLen = (word32)sigSz; + ret = wc_MlDsaKey_SignCtxWithSeed(freshKey, NULL, 0, freshSig, + &freshSigLen, mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), + mldsa_kat_sig_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (expDigest != NULL) { + ret = wc_InitShake256(&shake, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + shakeInit = 1; + ret = wc_Shake256_Update(&shake, freshSig, freshSigLen); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + ret = wc_Shake256_Final(&shake, digest, (word32)sizeof(digest)); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (XMEMCMP(digest, expDigest, sizeof(digest)) != 0) + ERROR_OUT(WC_TEST_RET_ENC_NC, out); + } + + ret = wc_MlDsaKey_Init(key, HEAP_HINT, INVALID_DEVID); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + keyInit = 1; + ret = wc_MlDsaKey_SetParams(key, param); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + + /* Fill the caches from another key before regenerating. */ + ret = wc_MlDsaKey_MakeKeyFromSeed(key, mldsa_kat_sig_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + sigLen = (word32)sigSz; + ret = wc_MlDsaKey_SignCtxWithSeed(key, NULL, 0, sig, &sigLen, + mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), mldsa_kat_sig_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); +#ifndef WOLFSSL_MLDSA_NO_VERIFY + ret = wc_MlDsaKey_VerifyCtx(key, sig, sigLen, NULL, 0, mldsa_kat_msg, + (word32)sizeof(mldsa_kat_msg), &res); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (res != 1) + ERROR_OUT(WC_TEST_RET_ENC_I(res), out); +#endif + + ret = wc_MlDsaKey_MakeKeyFromSeed(key, mldsa_kat_key_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + sigLen = (word32)sigSz; + ret = wc_MlDsaKey_SignCtxWithSeed(key, NULL, 0, sig, &sigLen, + mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), mldsa_kat_sig_seed); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if ((sigLen != freshSigLen) || (XMEMCMP(sig, freshSig, sigLen) != 0)) + ERROR_OUT(WC_TEST_RET_ENC_NC, out); +#ifndef WOLFSSL_MLDSA_NO_VERIFY + res = 0; + ret = wc_MlDsaKey_VerifyCtx(key, freshSig, freshSigLen, NULL, 0, + mldsa_kat_msg, (word32)sizeof(mldsa_kat_msg), &res); + if (ret != 0) + ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); + if (res != 1) + ERROR_OUT(WC_TEST_RET_ENC_I(res), out); +#endif + + ret = 0; +out: + if (shakeInit) + wc_Shake256_Free(&shake); + if (keyInit) + wc_MlDsaKey_Free(key); + if (freshKeyInit) + wc_MlDsaKey_Free(freshKey); + XFREE(freshSig, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + XFREE(sig, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + XFREE(freshKey, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + XFREE(key, HEAP_HINT, DYNAMIC_TYPE_TMP_BUFFER); + return ret; +} +#endif /* !WOLFSSL_MLDSA_NO_SIGN && !WOLFSSL_MLDSA_NO_MAKE_KEY */ + #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ @@ -66547,13 +66731,17 @@ static wc_test_ret_t mldsa_sign_cache_alloc_test(int param, WC_RNG* rng) if (ret != 0) ERROR_OUT(WC_TEST_RET_ENC_EC(ret), out); +#ifndef WOLFSSL_MLDSA_SIGN_SMALL_MEM /* With the fix, signing must populate key->a (allocated buffer is owned * by the key, not leaked to a local). Without the fix, key->a remains - * NULL because the XMALLOC result was assigned to a local variable. */ + * NULL because the XMALLOC result was assigned to a local variable. + * The small memory implementations stream matrix A and never populate + * key->a, so only the round trip below applies to them. */ if (key->a == NULL) ERROR_OUT(WC_TEST_RET_ENC_NC, out); if (key->aSet != 1) ERROR_OUT(WC_TEST_RET_ENC_NC, out); +#endif ret = wc_MlDsaKey_VerifyCtx(key, sig, sigLen, NULL, 0, msg, (word32)sizeof(msg), &res); @@ -67560,6 +67748,27 @@ WOLFSSL_TEST_SUBROUTINE wc_test_ret_t mldsa_test(void) #endif #endif +#if !defined(WOLFSSL_MLDSA_NO_SIGN) && !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) +#ifndef WOLFSSL_NO_ML_DSA_44 + ret = mldsa_make_key_reuse_test(WC_ML_DSA_44, + MLDSA_KAT_DIGEST(mldsa_kat_digest_44)); + if (ret != 0) + ERROR_OUT(ret, out); +#endif +#ifndef WOLFSSL_NO_ML_DSA_65 + ret = mldsa_make_key_reuse_test(WC_ML_DSA_65, + MLDSA_KAT_DIGEST(mldsa_kat_digest_65)); + if (ret != 0) + ERROR_OUT(ret, out); +#endif +#ifndef WOLFSSL_NO_ML_DSA_87 + ret = mldsa_make_key_reuse_test(WC_ML_DSA_87, + MLDSA_KAT_DIGEST(mldsa_kat_digest_87)); + if (ret != 0) + ERROR_OUT(ret, out); +#endif +#endif + #if defined(WC_MLDSA_CACHE_MATRIX_A) && \ !defined(WC_MLDSA_FIXED_ARRAY) && \ !defined(WOLFSSL_MLDSA_NO_MAKE_KEY) && \ diff --git a/wolfssl/wolfcrypt/dilithium.h b/wolfssl/wolfcrypt/dilithium.h index 7d39cf378f9..a8d249e1bae 100644 --- a/wolfssl/wolfcrypt/dilithium.h +++ b/wolfssl/wolfcrypt/dilithium.h @@ -83,6 +83,11 @@ #ifndef WOLF_CRYPT_DILITHIUM_H #define WOLF_CRYPT_DILITHIUM_H +/* Read the build configuration before deriving any option: an application + * that reaches this header before settings.h would otherwise translate an + * empty option set and get a different wc_MlDsaKey layout than the library. */ +#include + /* === Sub-config build-gate translations =============================== */ /* The two sub-gates that (auto-generated, no @@ -150,17 +155,6 @@ #define WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM #endif #endif -#ifdef WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM - /* Smallest verify RAM: on top of the small-mem, no-malloc path, stream the - * signature's z vector one polynomial at a time instead of pinning the whole - * l-vector (~6 KB for ML-DSA-87) at the cost of a per-row z decode+NTT. */ - #ifndef WOLFSSL_MLDSA_VERIFY_SMALL_MEM - #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM - #endif - #ifndef WOLFSSL_MLDSA_VERIFY_NO_MALLOC - #define WOLFSSL_MLDSA_VERIFY_NO_MALLOC - #endif -#endif #ifdef WOLFSSL_DILITHIUM_MAKE_KEY_SMALL_MEM #ifndef WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM #define WOLFSSL_MLDSA_MAKE_KEY_SMALL_MEM @@ -361,6 +355,8 @@ !defined(WOLFSSL_MLDSA_NO_SIGN) #define WOLFSSL_MLDSA_PRIVATE_KEY #endif +/* wc_CheckPrivateKey() needs this to match an ML-DSA certificate to its + * private key. */ #if defined(WOLFSSL_MLDSA_PUBLIC_KEY) && \ defined(WOLFSSL_MLDSA_PRIVATE_KEY) && \ !defined(WOLFSSL_MLDSA_NO_CHECK_KEY) && \ @@ -428,6 +424,9 @@ #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) && !defined(WOLFSSL_DILITHIUM_SIGN_SMALL_MEM) #define WOLFSSL_DILITHIUM_SIGN_SMALL_MEM #endif +#if defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) && !defined(WOLFSSL_DILITHIUM_VERIFY_SMALLEST_MEM) + #define WOLFSSL_DILITHIUM_VERIFY_SMALLEST_MEM +#endif #if defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) && !defined(WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC) #define WOLFSSL_DILITHIUM_SIGN_SMALL_MEM_PRECALC #endif diff --git a/wolfssl/wolfcrypt/settings.h b/wolfssl/wolfcrypt/settings.h index 80151de2464..49344fc3d25 100644 --- a/wolfssl/wolfcrypt/settings.h +++ b/wolfssl/wolfcrypt/settings.h @@ -3760,7 +3760,7 @@ (defined(HAVE_CURVE25519) && defined(HAVE_CURVE25519_KEY_EXPORT)) || \ (defined(HAVE_ED448) && defined(HAVE_ED448_KEY_EXPORT)) || \ (defined(HAVE_CURVE448) && defined(HAVE_CURVE448_KEY_EXPORT)) || \ - defined(HAVE_FALCON) || defined(HAVE_DILITHIUM) || \ + defined(HAVE_FALCON) || defined(WOLFSSL_HAVE_MLDSA) || \ defined(WOLFSSL_HAVE_FRODOKEM) || \ defined(WOLFSSL_HAVE_SLHDSA) || \ (defined(WOLFSSL_HAVE_LMS) && !defined(WOLFSSL_LMS_VERIFY_ONLY)) || \ @@ -3773,7 +3773,7 @@ (defined(HAVE_CURVE25519) && defined(HAVE_CURVE25519_KEY_IMPORT)) || \ (defined(HAVE_ED448) && defined(HAVE_ED448_KEY_IMPORT)) || \ (defined(HAVE_CURVE448) && defined(HAVE_CURVE448_KEY_IMPORT)) || \ - defined(HAVE_FALCON) || defined(HAVE_DILITHIUM) || \ + defined(HAVE_FALCON) || defined(WOLFSSL_HAVE_MLDSA) || \ defined(WOLFSSL_HAVE_FRODOKEM) || \ defined(WOLFSSL_HAVE_SLHDSA) || \ (defined(WOLFSSL_HAVE_LMS) && !defined(WOLFSSL_LMS_VERIFY_ONLY)) || \ @@ -5348,10 +5348,17 @@ blinding by defining WC_BLINDING_NO_RNG_ACKNOWLEDGE_WEAKNESS." #error Experimental settings without WOLFSSL_EXPERIMENTAL_SETTINGS #endif -/* If no malloc then make sure the valid Dilithium settings are used */ -#if defined(HAVE_DILITHIUM) && defined(WOLFSSL_NO_MALLOC) - #undef WOLFSSL_DILITHIUM_VERIFY_NO_MALLOC - #define WOLFSSL_DILITHIUM_VERIFY_NO_MALLOC +/* If no malloc then make sure the valid ML-DSA settings are used */ +#if defined(WOLFSSL_HAVE_MLDSA) && defined(WOLFSSL_NO_MALLOC) + #undef WOLFSSL_MLDSA_VERIFY_NO_MALLOC + #define WOLFSSL_MLDSA_VERIFY_NO_MALLOC + /* The pinned buffers only exist under the small memory verify; without it + * every verification fails with MEMORY_E. */ + #if !defined(WOLFSSL_MLDSA_NO_VERIFY) && \ + !defined(WOLFSSL_MLDSA_VERIFY_ALLOW_MALLOC) + #undef WOLFSSL_MLDSA_VERIFY_SMALL_MEM + #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM + #endif #endif #if defined(WOLFSSL_HAVE_MLKEM) && \ diff --git a/wolfssl/wolfcrypt/wc_mldsa.h b/wolfssl/wolfcrypt/wc_mldsa.h index 34ba33efc26..31ecc255680 100644 --- a/wolfssl/wolfcrypt/wc_mldsa.h +++ b/wolfssl/wolfcrypt/wc_mldsa.h @@ -82,6 +82,19 @@ * legacy compatibility shim is dropped. */ #include +/* Unconditional so the key layout does not depend on include order. */ +#if (defined(WOLFSSL_MLDSA_SIGN_SMALLEST_MEM) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC) || \ + defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM_PRECALC_A)) && \ + !defined(WOLFSSL_MLDSA_SIGN_SMALL_MEM) + #define WOLFSSL_MLDSA_SIGN_SMALL_MEM +#endif +/* Allocates its buffers unless WOLFSSL_MLDSA_VERIFY_NO_MALLOC is also set. */ +#if defined(WOLFSSL_MLDSA_VERIFY_SMALLEST_MEM) && \ + !defined(WOLFSSL_MLDSA_VERIFY_SMALL_MEM) + #define WOLFSSL_MLDSA_VERIFY_SMALL_MEM +#endif + #if defined(WOLFSSL_HAVE_MLDSA) #include @@ -684,7 +697,9 @@ struct wc_MlDsaKey { #endif sword32 c[MLDSA_N]; sword32 w[MLDSA_N]; - sword32 t1[MLDSA_N]; + /* One t1 polynomial, also used for a polynomial of A. Not named t1 as + * WC_MLDSA_CACHE_PUB_VECTORS already has a member of that name. */ + sword32 vt1[MLDSA_N]; byte w1e[MLDSA_MAX_W1_ENC_SZ]; #ifdef WOLFSSL_MLDSA_SMALL_MEM_POLY64 sword64 t64[MLDSA_N];