diff --git a/.wolfssl_known_macro_extras b/.wolfssl_known_macro_extras index 688602cdc0b..d65ba6ed972 100644 --- a/.wolfssl_known_macro_extras +++ b/.wolfssl_known_macro_extras @@ -890,6 +890,8 @@ WOLFSSL_ALT_NAMES_NO_REV WOLFSSL_ARM32_BUILD WOLFSSL_ARM32_PRIVILEGE_MODE WOLFSSL_ARMASM_NEON_NO_TABLE_LOOKUP +WOLFSSL_ARMASM_SHA3_NO_NEON +WOLFSSL_ARMASM_SHA512_NO_NEON WOLFSSL_ARM_ARCH_NEON_64BIT WOLFSSL_ASCON_UNROLL WOLFSSL_ASN_EXTRA @@ -1287,6 +1289,7 @@ __ARCH_STRNCAT_NO_REDIRECT __ARCH_STRNCMP_NO_REDIRECT __ARCH_STRNCPY_NO_REDIRECT __ARCH_STRSTR_NO_REDIRECT +__ARMEB__ __ARM_ARCH __ARM_ARCH_7M__ __ARM_ARCH_PROFILE diff --git a/configure.ac b/configure.ac index ffb92d5e842..05efe91d82c 100644 --- a/configure.ac +++ b/configure.ac @@ -4242,6 +4242,12 @@ then ENABLED_AESSIV=yes fi +if test "$ENABLED_FIPS" != "no" && test "$HAVE_FIPS_VERSION" -ge 7 && \ + { test "$ENABLED_AESSIV" = "yes" || test "$ENABLED_AESEAX" = "yes"; } +then + AC_MSG_ERROR([AES-SIV and AES-EAX are not in the FIPS module boundary: drop --enable-aessiv, --enable-aeseax, --enable-chrony or --enable-all-nonfips-crypto from a FIPS v7 build.]) +fi + # AES-GCM-SIV (RFC 8452) AC_ARG_ENABLE([aesgcm-siv], [AS_HELP_STRING([--enable-aesgcm-siv],[Enable AES-GCM-SIV (RFC 8452) (default: disabled)])], diff --git a/doc/ASM_AND_MATH_DEFINES.md b/doc/ASM_AND_MATH_DEFINES.md index b579a324763..ad996babbac 100644 --- a/doc/ASM_AND_MATH_DEFINES.md +++ b/doc/ASM_AND_MATH_DEFINES.md @@ -420,6 +420,8 @@ kernel-module builds, `WC_C_DYNAMIC_FALLBACK` keeps a C path available. | `WOLFSSL_ARMASM_INLINE` | Use the `*_asm_c.c` inline-assembly variants instead of the `.S` files. Needed when the toolchain will not assemble `.S`, and used for FIPS on ARMv7. | | `WOLFSSL_ARMASM_NO_HW_CRYPTO` | No ARMv8 Crypto Extensions (no AES/SHA instructions) | | `WOLFSSL_ARMASM_NO_NEON` | No NEON | +| `WOLFSSL_ARMASM_SHA512_NO_NEON` | 32-bit Arm: SHA-512 uses its scalar body; implied by `WOLFSSL_ARMASM_NO_NEON` | +| `WOLFSSL_ARMASM_SHA3_NO_NEON` | 32-bit Arm: SHA-3 uses its scalar body; implied by `WOLFSSL_ARMASM_NO_NEON` | | `WOLFSSL_ARMASM_THUMB2` | Thumb-2 encoding (Cortex-M, ARMv7-M) | | `WOLFSSL_ARM_ARCH=` | Architecture level: `4`, `6`, `7`… Gates instruction availability. | | `WOLFSSL_AARCH64_BUILD` | Aarch64 target | diff --git a/linuxkm/Kbuild b/linuxkm/Kbuild index bd90f9477bf..cc2603ec4fe 100644 --- a/linuxkm/Kbuild +++ b/linuxkm/Kbuild @@ -154,16 +154,34 @@ ifeq "$(ENABLED_LINUXKM_PIE)" "yes" ifndef NO_PIE_FLAG ifeq ($(KERNEL_ARCH),arm) - ifneq "$(HAVE_INTCMP)" "yes" - $(error $$NO_PIE_FLAG is unset -- supply NO_PIE_FLAG=1 for target kernel <5.11, else supply NO_PIE_FLAG=0.) + ifeq "$(and $(VERSION),$(PATCHLEVEL))" "" + $(error $$VERSION or $$PATCHLEVEL is unset; supply NO_PIE_FLAG=1 for a target kernel before 5.11, else NO_PIE_FLAG=0.) endif - ifeq ($(intcmp $(VERSION),5,1,0,0),1) - NO_PIE_FLAG := 1 - $(info Note: disabling -fPIE to avoid R_ARM_REL32 on pre-5.11 target kernel.) + ifeq "$(HAVE_INTCMP)" "yes" + ifeq ($(intcmp $(VERSION),5,1,0,0),1) + NO_PIE_FLAG := 1 + $(info Note: disabling -fPIE to avoid R_ARM_REL32 on pre-5.11 target kernel.) + else + ifeq ($(intcmp $(VERSION),5,0,1,0)-$(intcmp $(PATCHLEVEL),11,1,0,0),1-1) + NO_PIE_FLAG := 1 + $(info Note: disabling -fPIE to avoid R_ARM_REL32 on pre-5.11 target kernel.) + else + NO_PIE_FLAG := 0 + endif + endif else - ifeq ($(intcmp $(VERSION),5,0,1,0)-$(intcmp $(PATCHLEVEL),11,1,0,0),1-1) + # GNU make before 4.4 has no $(intcmp), so the pre-5.11 versions are + # enumerated: major 0 to 4, or major 5 with minor 0 to 10. + ifneq ($(filter 0 1 2 3 4,$(VERSION)),) NO_PIE_FLAG := 1 $(info Note: disabling -fPIE to avoid R_ARM_REL32 on pre-5.11 target kernel.) + else ifeq "$(VERSION)" "5" + ifneq ($(filter 0 1 2 3 4 5 6 7 8 9 10,$(PATCHLEVEL)),) + NO_PIE_FLAG := 1 + $(info Note: disabling -fPIE to avoid R_ARM_REL32 on pre-5.11 target kernel.) + else + NO_PIE_FLAG := 0 + endif else NO_PIE_FLAG := 0 endif @@ -275,6 +293,19 @@ $(obj)/wolfcrypt/src/port/arm/%-asm_c.o: ccflags-remove-y += -mgeneral-regs-only $(obj)/wolfcrypt/src/port/arm/%.o: OBJECT_FILES_NON_STANDARD := y endif +# The 32-bit Arm kernel builds soft-float with a baseline -march; the port asm +# needs the ARMv8-A crypto and NEON instructions accepted, as above. +ifeq ($(CONFIG_ARM),y) +# --noexecstack keeps the GNU-stack note on an object whose whole body is +# compiled out, such as the SHA-3 assembly under WC_SHA3_NO_ASM. +$(obj)/wolfcrypt/src/port/arm/%.o: asflags-y := $(WOLFSSL_ASFLAGS) -march=armv8-a -mfpu=crypto-neon-fp-armv8 -Wa,--noexecstack +# The kernel builds with -msoft-float; NEON in a C file needs softfp, as +# arch/arm/lib/Makefile does for xor-neon.o. +$(obj)/wolfcrypt/src/port/arm/%-asm_c.o: ccflags-y += -march=armv8-a -mfpu=crypto-neon-fp-armv8 -mfloat-abi=softfp +$(obj)/wolfcrypt/src/port/arm/%-asm_c.o: ccflags-remove-y += -mgeneral-regs-only +$(obj)/wolfcrypt/src/port/arm/%.o: OBJECT_FILES_NON_STANDARD := y +endif + ifndef READELF READELF := readelf endif @@ -291,6 +322,8 @@ ifndef OBJCOPY OBJCOPY := objcopy endif +# 32-bit Arm asm keeps its constant tables in .text, so the OBJECT check below +# skips wolfCrypt's text sections; x86 and arm64 keep theirs in rodata. RENAME_PIE_TEXT_AND_DATA_SECTIONS := \ if [[ "$(quiet)" != "silent_" ]]; then \ echo -n ' Checking wolfCrypt for unresolved symbols and forbidden relocations... '; \ @@ -383,7 +416,8 @@ RENAME_PIE_TEXT_AND_DATA_SECTIONS := \ next; \ } \ else if ($$4 == "OBJECT") { \ - if (! ($$7 in wolfcrypt_data_sections)) { \ + if (! ($$7 in wolfcrypt_data_sections) && \ + ! ($$7 in wolfcrypt_text_sections)) { \ if ((other_sections[$$7] == ".printk_index") || \ (($$8 ~ /^_entry\.[0-9]+$$|^kernel_read_file_str$$/) && \ (other_sections[$$7] == ".data.rel.ro.local"))) \ diff --git a/linuxkm/arm64_vector_register_glue.c b/linuxkm/arm64_vector_register_glue.c index ff86c6df345..ccee214ea1b 100644 --- a/linuxkm/arm64_vector_register_glue.c +++ b/linuxkm/arm64_vector_register_glue.c @@ -1,5 +1,5 @@ /* arm64_vector_register_glue.c: glue logic to claim and release the FPSIMD - * and NEON registers on arm64 + * and NEON registers on arm64, and the NEON registers on 32-bit Arm * * Copyright (C) 2006-2026 wolfSSL Inc. * @@ -23,22 +23,35 @@ /* included by linuxkm/module_hooks.c */ #ifndef WC_SKIP_INCLUDED_C_FILES -#if !defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) || !defined(CONFIG_ARM64) - #error arm64 vector register glue included in non-vectorized or non-arm64 project. +#if !defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) || \ + (!defined(CONFIG_ARM64) && !defined(CONFIG_ARM)) + #error arm vector register glue included in non-vectorized or non-arm project. #endif #ifndef CONFIG_KERNEL_MODE_NEON /* Without this option the kernel exports no kernel_neon_begin() and * may_use_simd() is always false (linux-6.6.99 fpsimd.c:1904, simd.h:46). */ - #error wolfSSL linuxkm on arm64 requires CONFIG_KERNEL_MODE_NEON. + #error wolfSSL linuxkm on arm requires CONFIG_KERNEL_MODE_NEON. #endif -#if LINUX_VERSION_CODE >= KERNEL_VERSION(6, 19, 0) +#if defined(CONFIG_ARM64) && LINUX_VERSION_CODE >= KERNEL_VERSION(6, 19, 0) /* Written against the void kernel_neon_begin() of linux-6.6.99. 6.19 * changes that signature and has not been read here. */ #error arm64 vector register glue does not yet support kernel_neon_begin() with a state buffer (6.19+). #endif +#if defined(CONFIG_ARM) && LINUX_VERSION_CODE >= KERNEL_VERSION(6, 4, 0) + /* From 6.4 may_use_simd() tests only in_hardirq(), but kernel_neon_begin() + * also BUGs on irqs_disabled() (linux-6.16.12 vfpmodule.c); test it too. */ + #define wc_svr_arm_neon_usable() (may_use_simd() && !irqs_disabled()) +#elif defined(CONFIG_ARM) + /* Before 6.4 kernel_neon_begin() BUGs on in_interrupt() (linux-6.1.62 + * vfpmodule.c): NEON is refused in softirq, allowed with interrupts off. */ + #define wc_svr_arm_neon_usable() may_use_simd() +#else + #define wc_svr_arm_neon_usable() may_use_simd() +#endif + #ifdef DEBUG_VECTOR_REGISTER_ACCESS_FUZZING #error DEBUG_VECTOR_REGISTER_ACCESS_FUZZING is not implemented by the arm64 vector register glue. #endif @@ -119,7 +132,7 @@ __must_check int wc_can_save_vector_registers_x86(void) if (st->depth > 0) ret = (st->inhibit_at == 0); else - ret = may_use_simd() ? 1 : 0; + ret = wc_svr_arm_neon_usable() ? 1 : 0; preempt_enable(); return ret; @@ -160,8 +173,8 @@ __must_check int wc_save_vector_registers_x86(enum wc_svr_flags flags) return 0; } - /* Outermost claim. The bottom-half disable below is what holds the CPU, - * taken by kernel_neon_begin() or wc_svr_arm64_pin_bh(). */ + /* Outermost claim: kernel_neon_begin() or wc_svr_arm64_pin_bh() holds the + * CPU with bottom halves off; 32-bit Arm before 6.4 disables preemption. */ if (flags & WC_SVR_FLAG_INHIBIT) { wc_svr_arm64_pin_bh(st); st->depth = 1; @@ -172,7 +185,7 @@ __must_check int wc_save_vector_registers_x86(enum wc_svr_flags flags) return 0; } - if (! may_use_simd()) { + if (! wc_svr_arm_neon_usable()) { if (flags & WC_SVR_FLAG_MAYBE_INHIBIT) { /* Held without the registers: nested claims are refused. */ wc_svr_arm64_pin_bh(st); diff --git a/linuxkm/linuxkm_memory.c b/linuxkm/linuxkm_memory.c index 2f1b75e1125..510d79a7690 100644 --- a/linuxkm/linuxkm_memory.c +++ b/linuxkm/linuxkm_memory.c @@ -64,6 +64,10 @@ static const struct reloc_layout_ent { [WC_R_AARCH64_LDST64_ABS_LO12_NC] = { "R_AARCH64_LDST64_ABS_LO12_NC", 0b00000000001111111111110000000000, 32, .is_signed = 0, .is_relative = 0, .is_pages = 0, .is_pair_lo = 1, .is_pair_hi = 0 }, [WC_R_AARCH64_PREL32] = { "R_AARCH64_PREL32", ~0UL, 32, .is_signed = 1, .is_relative = 1, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, [WC_R_ARM_ABS32] = { "R_ARM_ABS32", ~0UL, 32, .is_signed = 0, .is_relative = 0, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, + /* ARM-mode BL and B: signed 24-bit word offset in bits 23 to 0, emitted + * by a 32-bit Arm kernel built without CONFIG_THUMB2_KERNEL. */ + [WC_R_ARM_CALL] = { "R_ARM_CALL", 0b00000000111111111111111111111111, 32, .is_signed = 1, .is_relative = 1, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, + [WC_R_ARM_JUMP24] = { "R_ARM_JUMP24", 0b00000000111111111111111111111111, 32, .is_signed = 1, .is_relative = 1, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, [WC_R_ARM_PREL31] = { "R_ARM_PREL31", 0b01111111111111111111111111111111, 32, .is_signed = 1, .is_relative = 1, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, [WC_R_ARM_REL32] = { "R_ARM_REL32", ~0UL, 32, .is_signed = 1, .is_relative = 1, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, [WC_R_ARM_THM_CALL] = { "R_ARM_THM_CALL", 0b00000111111111110010111111111111, 32, .is_signed = 1, .is_relative = 1, .is_pages = 0, .is_pair_lo = 0, .is_pair_hi = 0 }, @@ -413,6 +417,8 @@ ssize_t wc_reloc_normalize_segment( break; case WC_R_ARM_ABS32: + case WC_R_ARM_CALL: + case WC_R_ARM_JUMP24: case WC_R_ARM_PREL31: case WC_R_ARM_REL32: case WC_R_ARM_THM_CALL: diff --git a/linuxkm/linuxkm_memory.h b/linuxkm/linuxkm_memory.h index 76e681da805..705b1f262aa 100644 --- a/linuxkm/linuxkm_memory.h +++ b/linuxkm/linuxkm_memory.h @@ -52,6 +52,8 @@ enum wc_reloc_type { WC_R_AARCH64_LDST64_ABS_LO12_NC, WC_R_AARCH64_PREL32, WC_R_ARM_ABS32, + WC_R_ARM_CALL, + WC_R_ARM_JUMP24, WC_R_ARM_PREL31, WC_R_ARM_REL32, WC_R_ARM_THM_CALL, diff --git a/linuxkm/linuxkm_wc_port.h b/linuxkm/linuxkm_wc_port.h index f40cbf74b0a..a792319b69f 100644 --- a/linuxkm/linuxkm_wc_port.h +++ b/linuxkm/linuxkm_wc_port.h @@ -201,6 +201,12 @@ #define WOLFSSL_DEBUG_PRINTF_FN printk #endif +#ifdef CONFIG_ARM + /* Exported by arch/arm/kernel/traps.c; asmlinkage is empty for C here and + * linux/linkage.h is not included yet at this point in the header. */ + void __div0(void); +#endif + #ifndef WOLFSSL_LINUXKM_USE_MUTEXES struct wolfSSL_Mutex; extern int wc_lkm_LockMutex(struct wolfSSL_Mutex* m); @@ -256,6 +262,11 @@ #if !defined(CONFIG_ARM) && !defined(CONFIG_ARM64) #error ARM SIMD extensions requested, but CONFIG_ARM* is not set. #endif + /* A kernel module runs privileged, so the 32-bit Arm feature test reads + * ID_ISAR5 directly instead of the userspace getauxval() path. */ + #if defined(CONFIG_ARM) && !defined(WOLFSSL_ARM32_PRIVILEGE_MODE) + #define WOLFSSL_ARM32_PRIVILEGE_MODE + #endif #define WOLFSSL_LINUXKM_SIMD #define WOLFSSL_LINUXKM_SIMD_ARM #ifndef WOLFSSL_USE_SAVE_VECTOR_REGISTERS @@ -816,7 +827,7 @@ /* x86 and arm64 share one interface: both glue files keep the wc_*_x86 * names, so callers and the PIE redirect table are the same. */ #if defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && \ - (defined(CONFIG_X86) || defined(CONFIG_ARM64)) + (defined(CONFIG_X86) || defined(CONFIG_ARM64) || defined(CONFIG_ARM)) extern __must_check int wc_linuxkm_allocate_svr_states(void); extern void wc_linuxkm_free_svr_states(void); @@ -843,10 +854,11 @@ #include #endif #endif - #else /* CONFIG_ARM64 */ + #else /* CONFIG_ARM64 || CONFIG_ARM */ /* arch/arm64/include/asm/simd.h, and may_use_simd() with it, * arrived in 4.14; the module otherwise accepts 3.16 and up. */ - #if LINUX_VERSION_CODE < KERNEL_VERSION(4, 14, 0) && \ + #if defined(CONFIG_ARM64) && \ + LINUX_VERSION_CODE < KERNEL_VERSION(4, 14, 0) && \ !defined(WC_DEBUG_FORCE_KERNEL_SETTINGS) #error arm64 vector registers need may_use_simd(), added in 4.14. #endif @@ -978,10 +990,6 @@ #define RESTORE_VECTOR_REGISTERS_MAYBE_INHIBITED() wc_restore_vector_registers_x86(WC_SVR_FLAG_MAYBE_INHIBIT) #endif - #elif defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && defined(CONFIG_ARM) - - #error kernel module 32-bit ARM SIMD is not yet tested or usable. - #elif (defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && \ (!defined(SAVE_VECTOR_REGISTERS) || \ !defined(SAVE_VECTOR_REGISTERS2) || \ @@ -1234,6 +1242,11 @@ #ifndef __ARCH_MEMCMP_NO_REDIRECT typeof(memcmp) *memcmp; #endif + #ifdef CONFIG_ARM + /* The kernel's divide-by-zero reporter, for the EABI division + * helpers the container defines (arch/arm/kernel/traps.c). */ + void (*__div0)(void); + #endif #ifndef __ARCH_MEMCPY_NO_REDIRECT typeof(memcpy) *memcpy; #endif @@ -1361,13 +1374,14 @@ #ifdef WOLFSSL_USE_SAVE_VECTOR_REGISTERS - #if defined(CONFIG_X86) || defined(CONFIG_ARM64) + #if defined(CONFIG_X86) || defined(CONFIG_ARM64) || \ + defined(CONFIG_ARM) typeof(wc_linuxkm_allocate_svr_states) *wc_linuxkm_allocate_svr_states; typeof(wc_can_save_vector_registers_x86) *wc_can_save_vector_registers_x86; typeof(wc_linuxkm_free_svr_states) *wc_linuxkm_free_svr_states; typeof(wc_restore_vector_registers_x86) *wc_restore_vector_registers_x86; typeof(wc_save_vector_registers_x86) *wc_save_vector_registers_x86; - #elif !defined(WC_DEBUG_FORCE_KERNEL_SETTINGS) /* !CONFIG_X86 && !CONFIG_ARM64 */ + #elif !defined(WC_DEBUG_FORCE_KERNEL_SETTINGS) /* !CONFIG_X86 && !CONFIG_ARM64 && !CONFIG_ARM */ #error WOLFSSL_USE_SAVE_VECTOR_REGISTERS is set for an unimplemented architecture. #endif /* arch */ @@ -1744,7 +1758,7 @@ #define get_current WC_PIE_INDIRECT_SYM(get_current) #if defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && \ - (defined(CONFIG_X86) || defined(CONFIG_ARM64)) + (defined(CONFIG_X86) || defined(CONFIG_ARM64) || defined(CONFIG_ARM)) #define wc_linuxkm_allocate_svr_states WC_PIE_INDIRECT_SYM(wc_linuxkm_allocate_svr_states) #define wc_can_save_vector_registers_x86 WC_PIE_INDIRECT_SYM(wc_can_save_vector_registers_x86) #define wc_linuxkm_free_svr_states WC_PIE_INDIRECT_SYM(wc_linuxkm_free_svr_states) @@ -2091,7 +2105,8 @@ #if !defined(BUILDING_WOLFSSL) /* some caller code needs these. */ #if defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) - #if defined(CONFIG_X86) || defined(CONFIG_ARM64) + #if defined(CONFIG_X86) || defined(CONFIG_ARM64) || \ + defined(CONFIG_ARM) WOLFSSL_API __must_check int wc_can_save_vector_registers_x86(void); WOLFSSL_API __must_check int wc_save_vector_registers_x86(enum wc_svr_flags flags); WOLFSSL_API void wc_restore_vector_registers_x86(enum wc_svr_flags flags); @@ -2101,9 +2116,9 @@ #ifndef REENABLE_VECTOR_REGISTERS #define REENABLE_VECTOR_REGISTERS() wc_restore_vector_registers_x86(WC_SVR_FLAG_INHIBIT) #endif - #elif !defined(WC_DEBUG_FORCE_KERNEL_SETTINGS) /* !CONFIG_X86 && !CONFIG_ARM64 */ + #elif !defined(WC_DEBUG_FORCE_KERNEL_SETTINGS) /* !CONFIG_X86 && !CONFIG_ARM64 && !CONFIG_ARM */ #error WOLFSSL_USE_SAVE_VECTOR_REGISTERS is set for an unimplemented architecture. - #endif /* !CONFIG_X86 && !CONFIG_ARM64 */ + #endif /* !CONFIG_X86 && !CONFIG_ARM64 && !CONFIG_ARM */ #endif /* WOLFSSL_USE_SAVE_VECTOR_REGISTERS */ #ifdef WC_LINUXKM_USE_HEAP_WRAPPERS WOLFSSL_API extern void *wc_linuxkm_malloc(size_t size); @@ -2276,7 +2291,6 @@ { return wc_lkm_UnlockMutex(m); } - #endif /* !WC_CONTAINERIZE_THIS */ #endif diff --git a/linuxkm/lkcapi_sha_glue.c b/linuxkm/lkcapi_sha_glue.c index 4acead3bf00..d147e286381 100644 --- a/linuxkm/lkcapi_sha_glue.c +++ b/linuxkm/lkcapi_sha_glue.c @@ -4275,7 +4275,7 @@ static ssize_t wc_get_random_bytes_user(struct iov_iter *iter) { ret = wc_rng_bank_default_checkout(¤t_default_wc_rng_bank); if (ret) { pr_emerg_ratelimited("ERROR: wc_rng_bank_default_checkout() in " - "wc_get_random_bytes_user() returned %ld.\n", ret); + "wc_get_random_bytes_user() returned %zd.\n", ret); return -EIO; /* no fallthrough to native randomness */ } @@ -4322,7 +4322,7 @@ static ssize_t wc_get_random_bytes_user(struct iov_iter *iter) { if (unlikely(ret != 0)) { if (ret != -WC_NO_ERR_TRACE(EINTR)) { pr_emerg_ratelimited( - "ERROR: %s: wc_linuxkm_drbg_generate() returned %ld.\n", + "ERROR: %s: wc_linuxkm_drbg_generate() returned %zd.\n", __func__, ret); } break; @@ -4378,7 +4378,7 @@ static ssize_t wc_extract_crng_user(void __user *buf, size_t nbytes) { ret = wc_rng_bank_default_checkout(¤t_default_wc_rng_bank); if (ret) { pr_emerg_ratelimited("ERROR: wc_rng_bank_default_checkout() in " - "wc_extract_crng_user() returned %ld.\n", ret); + "wc_extract_crng_user() returned %zd.\n", ret); return -EIO; /* no fallthrough to native randomness */ } @@ -4424,7 +4424,7 @@ static ssize_t wc_extract_crng_user(void __user *buf, size_t nbytes) { if (unlikely(ret != 0)) { if (ret != -WC_NO_ERR_TRACE(EINTR)) { pr_emerg_ratelimited( - "ERROR: %s: wc_linuxkm_drbg_generate() returned %ld.\n", + "ERROR: %s: wc_linuxkm_drbg_generate() returned %zd.\n", __func__, ret); } break; diff --git a/linuxkm/module_hooks.c b/linuxkm/module_hooks.c index e1807d28a55..df879104076 100644 --- a/linuxkm/module_hooks.c +++ b/linuxkm/module_hooks.c @@ -760,7 +760,8 @@ int wc_linuxkm_GenerateSeed_IntelRD(struct OS_Seed* os, byte* output, word32 sz) #if defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && defined(CONFIG_X86) #include "linuxkm/x86_vector_register_glue.c" -#elif defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && defined(CONFIG_ARM64) +#elif defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && \ + (defined(CONFIG_ARM64) || defined(CONFIG_ARM)) #include "linuxkm/arm64_vector_register_glue.c" #endif @@ -1808,6 +1809,9 @@ static int set_up_wolfssl_linuxkm_pie_redirect_table(void) { #ifndef __ARCH_MEMCMP_NO_REDIRECT wolfssl_linuxkm_pie_redirect_table.memcmp = memcmp; #endif +#ifdef CONFIG_ARM + wolfssl_linuxkm_pie_redirect_table.__div0 = __div0; +#endif #ifndef CONFIG_FORTIFY_SOURCE #ifndef __ARCH_MEMCPY_NO_REDIRECT #ifdef CONFIG_ARM64 @@ -1954,7 +1958,7 @@ static int set_up_wolfssl_linuxkm_pie_redirect_table(void) { wolfssl_linuxkm_pie_redirect_table.get_current = my_get_current_thread; #if defined(WOLFSSL_USE_SAVE_VECTOR_REGISTERS) && \ - (defined(CONFIG_X86) || defined(CONFIG_ARM64)) + (defined(CONFIG_X86) || defined(CONFIG_ARM64) || defined(CONFIG_ARM)) wolfssl_linuxkm_pie_redirect_table.wc_linuxkm_allocate_svr_states = wc_linuxkm_allocate_svr_states; wolfssl_linuxkm_pie_redirect_table.wc_can_save_vector_registers_x86 = wc_can_save_vector_registers_x86; wolfssl_linuxkm_pie_redirect_table.wc_linuxkm_free_svr_states = wc_linuxkm_free_svr_states; diff --git a/linuxkm/pie_redirect_table.c b/linuxkm/pie_redirect_table.c index d3c0999284b..1ab9da320dc 100644 --- a/linuxkm/pie_redirect_table.c +++ b/linuxkm/pie_redirect_table.c @@ -56,7 +56,7 @@ const struct wolfssl_linuxkm_pie_redirect_table /* The container may hold no undefined symbol (linuxkm/Kbuild:301), so define * these here. arm64 forwards to the kernel's __memcpy()/__memset() * (arch/arm64/lib/memcpy.S:243, memset.S:206); the loops run before that. */ -#if defined(CONFIG_MIPS) || defined(CONFIG_ARM64) +#if defined(CONFIG_MIPS) || defined(CONFIG_ARM64) || defined(CONFIG_ARM) #undef memcpy void *memcpy(void *dest, const void *src, size_t n) { char *dest_i = (char *)dest; @@ -84,3 +84,89 @@ const struct wolfssl_linuxkm_pie_redirect_table return dest; } #endif + +#if defined(CONFIG_ARM) + /* 32-bit Arm code calls the EABI division helpers and the container cannot + * reach the kernel's, so they live here. */ + + /* Report a zero divisor through the kernel's hook, as its own helpers do + * (arch/arm/lib/lib1funcs.S Ldiv0), then return zero as the EABI says. */ + static void wc_lkm_div0(void) { + if (wolfssl_linuxkm_pie_redirect_table.__div0 != NULL) + wolfssl_linuxkm_pie_redirect_table.__div0(); + } + unsigned int __aeabi_uidiv(unsigned int n, unsigned int d); + unsigned int __aeabi_uidiv(unsigned int n, unsigned int d) { + unsigned int q = 0, r = 0; + int i; + if (d == 0) { + wc_lkm_div0(); + return 0u; + } + for (i = 31; i >= 0; i--) { + /* Restoring division with a mask instead of a branch. */ + unsigned int mask; + r = (r << 1) | ((n >> i) & 1u); + mask = 0u - (unsigned int)(r >= d); + r -= d & mask; + q |= (1u << i) & mask; + } + return q; + } + + /* Quotient in the low word, remainder in the high word, as the EABI + * expects in r0 and r1. */ + /* The EABI pair is r0 = quotient, r1 = remainder; a 64-bit return puts its + * low word in r0 only on little-endian, so pack by byte order. */ + #ifdef __ARMEB__ + #define WC_AEABI_PACK(q, r) (((unsigned long long)(q) << 32) | (r)) + #define WC_AEABI_Q(v) ((unsigned int)((v) >> 32)) + #define WC_AEABI_R(v) ((unsigned int)(v)) + #else + #define WC_AEABI_PACK(q, r) (((unsigned long long)(r) << 32) | (q)) + #define WC_AEABI_Q(v) ((unsigned int)(v)) + #define WC_AEABI_R(v) ((unsigned int)((v) >> 32)) + #endif + unsigned long long __aeabi_uidivmod(unsigned int n, unsigned int d); + unsigned long long __aeabi_uidivmod(unsigned int n, unsigned int d) { + unsigned int q = 0, r = 0; + int i; + if (d == 0) { + wc_lkm_div0(); + return 0ULL; + } + for (i = 31; i >= 0; i--) { + unsigned int mask; + r = (r << 1) | ((n >> i) & 1u); + mask = 0u - (unsigned int)(r >= d); + r -= d & mask; + q |= (1u << i) & mask; + } + return WC_AEABI_PACK(q, r); + } + + /* Signed forms work on unsigned magnitudes so INT_MIN is well defined; + * INT_MIN / -1 returns INT_MIN, as SDIV does. */ + int __aeabi_idiv(int n, int d); + int __aeabi_idiv(int n, int d) { + int neg = (n < 0) ^ (d < 0); + unsigned int un = (n < 0) ? (0u - (unsigned int)n) : (unsigned int)n; + unsigned int ud = (d < 0) ? (0u - (unsigned int)d) : (unsigned int)d; + unsigned int uq = __aeabi_uidiv(un, ud); + return neg ? (int)(0u - uq) : (int)uq; + } + + unsigned long long __aeabi_idivmod(int n, int d); + unsigned long long __aeabi_idivmod(int n, int d) { + int nneg = (n < 0); + int qneg = (n < 0) ^ (d < 0); + unsigned int un = nneg ? (0u - (unsigned int)n) : (unsigned int)n; + unsigned int ud = (d < 0) ? (0u - (unsigned int)d) : (unsigned int)d; + unsigned long long um = __aeabi_uidivmod(un, ud); + unsigned int uq = WC_AEABI_Q(um); + unsigned int ur = WC_AEABI_R(um); + int q = qneg ? (int)(0u - uq) : (int)uq; + int r = nneg ? (int)(0u - ur) : (int)ur; + return WC_AEABI_PACK((unsigned int)q, (unsigned int)r); + } +#endif /* CONFIG_ARM */ diff --git a/src/include.am b/src/include.am index c962d19e44b..6fce19e8762 100644 --- a/src/include.am +++ b/src/include.am @@ -114,7 +114,8 @@ endif if BUILD_AESNI src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_asm.S if BUILD_X86_ASM -src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_gcm_x86_asm.S +# 32-bit GCM asm has text relocations, so it is left out; 32-bit AES-GCM uses +# the C GHASH over AES-NI blocks. See WC_AESNI_GCM in aes.c. if BUILD_AESXTS src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_xts_x86_asm.S endif @@ -275,7 +276,8 @@ endif BUILD_PPC32_ASM if BUILD_AESNI src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_asm.S if BUILD_X86_ASM -src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_gcm_x86_asm.S +# 32-bit GCM asm has text relocations, so it is left out; 32-bit AES-GCM uses +# the C GHASH over AES-NI blocks. See WC_AESNI_GCM in aes.c. if BUILD_AESXTS src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_xts_x86_asm.S endif @@ -623,7 +625,8 @@ endif BUILD_PPC32_ASM if BUILD_AESNI src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_asm.S if BUILD_X86_ASM -src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_gcm_x86_asm.S +# 32-bit GCM asm has text relocations, so it is left out; 32-bit AES-GCM uses +# the C GHASH over AES-NI blocks. See WC_AESNI_GCM in aes.c. if BUILD_AESXTS src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_xts_x86_asm.S endif @@ -1014,7 +1017,8 @@ if BUILD_AESNI src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_asm.S src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_x86_64_asm.S if BUILD_X86_ASM -src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_gcm_x86_asm.S +# 32-bit GCM asm has text relocations, so it is left out; 32-bit AES-GCM uses +# the C GHASH over AES-NI blocks. See WC_AESNI_GCM in aes.c. if BUILD_AESXTS src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_xts_x86_asm.S endif @@ -2061,7 +2065,8 @@ if BUILD_AESNI src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_asm.S src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_x86_64_asm.S if BUILD_X86_ASM -src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_gcm_x86_asm.S +# 32-bit GCM asm has text relocations, so it is left out; 32-bit AES-GCM uses +# the C GHASH over AES-NI blocks. See WC_AESNI_GCM in aes.c. if BUILD_AESXTS src_libwolfssl@LIBSUFFIX@_la_SOURCES += wolfcrypt/src/aes_xts_x86_asm.S endif diff --git a/tests/api/test_aes.c b/tests/api/test_aes.c index 9ed82692e73..0b3e10cc0d7 100644 --- a/tests/api/test_aes.c +++ b/tests/api/test_aes.c @@ -5214,6 +5214,77 @@ int test_wc_GmacUpdate(void) return EXPECT_RESULT(); } /* END test_wc_GmacUpdate */ +/* A failed AES-GCM decrypt must not hand back the plaintext it computed. */ +int test_wc_AesGcmDecrypt_WipeOnAuthFail(void) +{ + EXPECT_DECLS; +/* Only the software, AES-NI and Arm lanes wipe; the offload back ends and the + * ACVP harness build return the computed plaintext. A FIPS build before v7, + * and a --enable-selftest build, compile an older boundary aes.c that has no + * wipe. */ +#if (!defined(HAVE_FIPS) || FIPS_VERSION3_GE(7,0,0)) && \ + !defined(HAVE_SELFTEST) && \ + !defined(NO_AES) && defined(HAVE_AESGCM) && defined(HAVE_AES_DECRYPT) && \ + defined(WOLFSSL_AES_256) && !defined(WOLFSSL_AFALG) && \ + !defined(WOLFSSL_KCAPI) && !defined(WOLFSSL_DEVCRYPTO_AES) && \ + !defined(WOLFSSL_ASYNC_CRYPT) && !defined(WOLFSSL_SILABS_SE_ACCEL) && \ + !defined(WOLFSSL_MICROCHIP_TA100) && !defined(WOLFSSL_STM32_BARE) && \ + !defined(STM32_CRYPTO_AES_GCM) && !defined(WOLFSSL_PSOC6_CRYPTO) && \ + !defined(WOLFSSL_RISCV_ASM) && \ + !defined(WOLFSSL_RISCV_VECTOR_CRYPTO_ASM) && \ + !defined(FREESCALE_LTC_AES_GCM) && !defined(ACVP_VECTOR_TESTING) && \ + !defined(WOLFSSL_XILINX_CRYPT) && !defined(WOLFSSL_AFALG_XILINX_AES) && \ + !defined(WOLFSSL_TI_CRYPT) + static const byte key[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + static const byte iv[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62 + }; + Aes aes; + byte pt[WC_AES_BLOCK_SIZE * 5]; + byte ct[sizeof(pt)]; + byte dec[sizeof(pt)]; + byte zeros[sizeof(pt)]; + byte tag[WC_AES_BLOCK_SIZE]; + word32 i; + + /* Zeroed first: the tag is only filled by a call inside ExpectIntEQ, which + * does not run once an earlier expectation has failed. */ + XMEMSET(tag, 0, sizeof(tag)); + for (i = 0; i < (word32)sizeof(pt); i++) { + pt[i] = (byte)(0x40 + i); + } + XMEMSET(zeros, 0, sizeof(zeros)); + XMEMSET(&aes, 0, sizeof(aes)); + + ExpectIntEQ(wc_AesInit(&aes, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesGcmSetKey(&aes, key, sizeof(key)), 0); + ExpectIntEQ(wc_AesGcmEncrypt(&aes, ct, pt, sizeof(pt), iv, sizeof(iv), + tag, sizeof(tag), NULL, 0), 0); + + /* Corrupt the tag: the decrypt must fail and dec must be all zero. */ + tag[0] ^= 0x01; + XMEMSET(dec, 0xff, sizeof(dec)); + ExpectIntEQ(wc_AesGcmDecrypt(&aes, dec, ct, sizeof(ct), iv, sizeof(iv), + tag, sizeof(tag), NULL, 0), WC_NO_ERR_TRACE(AES_GCM_AUTH_E)); + ExpectBufEQ(dec, zeros, sizeof(dec)); + + /* Control: the good tag recovers the plaintext, so the zeros were a wipe. */ + tag[0] ^= 0x01; + XMEMSET(dec, 0, sizeof(dec)); + ExpectIntEQ(wc_AesGcmDecrypt(&aes, dec, ct, sizeof(ct), iv, sizeof(iv), + tag, sizeof(tag), NULL, 0), 0); + ExpectBufEQ(dec, pt, sizeof(pt)); + wc_AesFree(&aes); +#endif + return EXPECT_RESULT(); +} + /******************************************************************************* * AES-CCM ******************************************************************************/ @@ -6523,10 +6594,297 @@ int test_wc_AesXtsStream_ReinitAfterFinal(void) * AES-XTS sector APIs ******************************************************************************/ +/* A streaming request that would overflow the per-tweak byte counter is + * refused before any data is touched, and the counter is left as it was. */ +int test_wc_AesXtsStream_CounterOverflow(void) +{ + EXPECT_DECLS; +#if !defined(NO_AES) && defined(WOLFSSL_AES_XTS) && \ + defined(WOLFSSL_AES_256) && defined(WOLFSSL_AESXTS_STREAM) && \ + !defined(WC_AESXTS_STREAM_NO_REQUEST_ACCOUNTING) && \ + !defined(WOLFSSL_AFALG) && !defined(WOLFSSL_KCAPI) + static const byte key32[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + static const byte tweak[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + XtsAes aes; + XtsAesStreamData xs; + byte buf[WC_AES_BLOCK_SIZE * 2]; + + XMEMSET(&aes, 0, sizeof(aes)); + XMEMSET(&xs, 0, sizeof(xs)); + XMEMSET(buf, 0x5a, sizeof(buf)); + ExpectIntEQ(wc_AesXtsSetKey(&aes, key32, sizeof(key32), + AES_ENCRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsEncryptInit(&aes, tweak, sizeof(tweak), &xs), 0); + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, buf, buf, WC_AES_BLOCK_SIZE, + &xs), 0); + /* One block is counted, so this size overflows a 32-bit total. */ + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, buf, buf, + 0xFFFFFFF0U, &xs), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + /* The refused request did not touch the counter: a block still fits. */ + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, buf, buf, WC_AES_BLOCK_SIZE, + &xs), 0); + ExpectIntEQ(wc_AesXtsEncryptFinal(&aes, NULL, NULL, 0, &xs), 0); + wc_AesXtsFree(&aes); +#endif + return EXPECT_RESULT(); +} + +/* Callers hand AES plain byte buffers, so every entry must accept any + * alignment; the 32-bit Arm bulk block routine once required word alignment. */ +int test_wc_AesUnalignedBuffers(void) +{ + EXPECT_DECLS; +#if !defined(NO_AES) && defined(WOLFSSL_AES_128) && \ + (defined(HAVE_AES_ECB) || defined(WOLFSSL_AES_XTS)) && \ + !defined(WOLFSSL_AFALG) && !defined(WOLFSSL_KCAPI) + /* XTS needs two distinct keys, so the halves differ. */ + static const byte key32[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x66, 0x65, 0x64, 0x63, 0x62, 0x61, 0x39, 0x38, + 0x37, 0x36, 0x35, 0x34, 0x33, 0x32, 0x31, 0x30 + }; + /* Eight blocks: enough for the four-block bulk path to run twice. */ + const word32 sz = WC_AES_BLOCK_SIZE * 8; + byte in[WC_AES_BLOCK_SIZE * 8 + 4]; + byte out[WC_AES_BLOCK_SIZE * 8 + 4]; + byte ref[WC_AES_BLOCK_SIZE * 8]; + word32 i, offIn, offOut; + + for (i = 0; i < sizeof(in); i++) + in[i] = (byte)(i * 7 + 3); + +#ifdef HAVE_AES_ECB + { + Aes aes; + XMEMSET(&aes, 0, sizeof(aes)); + ExpectIntEQ(wc_AesInit(&aes, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesSetKey(&aes, key32, 16, NULL, AES_ENCRYPTION), 0); + ExpectIntEQ(wc_AesEcbEncrypt(&aes, ref, in, sz), 0); + for (offIn = 0; offIn < 4; offIn++) { + for (offOut = 0; offOut < 4; offOut++) { + XMEMMOVE(in + offIn, in, sz); + XMEMSET(out, 0, sizeof(out)); + ExpectIntEQ(wc_AesEcbEncrypt(&aes, out + offOut, in + offIn, + sz), 0); + ExpectBufEQ(out + offOut, ref, sz); + XMEMMOVE(in, in + offIn, sz); + } + } + wc_AesFree(&aes); + } +#endif +#if defined(HAVE_AES_ECB) && defined(HAVE_AES_DECRYPT) + { + /* The bulk block routine has a separate decrypt entry, so cover it too. */ + Aes aes; + XMEMSET(&aes, 0, sizeof(aes)); + ExpectIntEQ(wc_AesInit(&aes, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesSetKey(&aes, key32, 16, NULL, AES_DECRYPTION), 0); + ExpectIntEQ(wc_AesEcbDecrypt(&aes, ref, in, sz), 0); + for (offIn = 0; offIn < 4; offIn++) { + for (offOut = 0; offOut < 4; offOut++) { + XMEMMOVE(in + offIn, in, sz); + XMEMSET(out, 0, sizeof(out)); + ExpectIntEQ(wc_AesEcbDecrypt(&aes, out + offOut, in + offIn, + sz), 0); + ExpectBufEQ(out + offOut, ref, sz); + XMEMMOVE(in, in + offIn, sz); + } + } + wc_AesFree(&aes); + } +#endif +#ifdef WOLFSSL_AES_XTS + { + static const byte tweak[WC_AES_BLOCK_SIZE] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + XtsAes xaes; + XMEMSET(&xaes, 0, sizeof(xaes)); + ExpectIntEQ(wc_AesXtsSetKey(&xaes, key32, sizeof(key32), + AES_ENCRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsEncrypt(&xaes, ref, in, sz, tweak, + sizeof(tweak)), 0); + for (offIn = 0; offIn < 4; offIn++) { + for (offOut = 0; offOut < 4; offOut++) { + XMEMMOVE(in + offIn, in, sz); + XMEMSET(out, 0, sizeof(out)); + ExpectIntEQ(wc_AesXtsEncrypt(&xaes, out + offOut, in + offIn, + sz, tweak, sizeof(tweak)), 0); + ExpectBufEQ(out + offOut, ref, sz); + XMEMMOVE(in, in + offIn, sz); + } + } + wc_AesXtsFree(&xaes); + } +#endif +#if defined(WOLFSSL_AES_XTS) && defined(HAVE_AES_DECRYPT) + { + static const byte tweak2[WC_AES_BLOCK_SIZE] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + XtsAes xaes; + XMEMSET(&xaes, 0, sizeof(xaes)); + ExpectIntEQ(wc_AesXtsSetKey(&xaes, key32, sizeof(key32), + AES_DECRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsDecrypt(&xaes, ref, in, sz, tweak2, + sizeof(tweak2)), 0); + for (offIn = 0; offIn < 4; offIn++) { + for (offOut = 0; offOut < 4; offOut++) { + XMEMMOVE(in + offIn, in, sz); + XMEMSET(out, 0, sizeof(out)); + ExpectIntEQ(wc_AesXtsDecrypt(&xaes, out + offOut, in + offIn, + sz, tweak2, sizeof(tweak2)), 0); + ExpectBufEQ(out + offOut, ref, sz); + XMEMMOVE(in, in + offIn, sz); + } + } + wc_AesXtsFree(&xaes); + } +#endif +#endif + return EXPECT_RESULT(); +} + +/* SP 800-38E section 4: a data unit (one tweak) is at most 2^20 AES blocks. */ +int test_wc_AesXtsDataUnitLimit(void) +{ + EXPECT_DECLS; +#if !defined(NO_AES) && defined(WOLFSSL_AES_XTS) && \ + defined(WOLFSSL_AES_256) && FIPS_VERSION3_GE(6,0,0) && \ + !defined(WOLFSSL_AFALG) && !defined(WOLFSSL_KCAPI) + static const byte key32[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + static const byte tweak[] = { + 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, + 0x38, 0x39, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66 + }; + const word32 tweakLen = (word32)sizeof(tweak); + const word32 limit = FIPS_AES_XTS_MAX_BYTES_PER_TWEAK; + XtsAes aes; + byte buf[WC_AES_BLOCK_SIZE * 2]; + + XMEMSET(&aes, 0, sizeof(aes)); + XMEMSET(buf, 0, sizeof(buf)); + + /* One shot: one block over the limit is refused before any data is + * touched, so a small buffer is fine. */ + ExpectIntEQ(wc_AesXtsSetKey(&aes, key32, sizeof(key32), + AES_ENCRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsEncrypt(&aes, buf, buf, limit + WC_AES_BLOCK_SIZE, + tweak, tweakLen), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + wc_AesXtsFree(&aes); +#ifdef HAVE_AES_DECRYPT + ExpectIntEQ(wc_AesXtsSetKey(&aes, key32, sizeof(key32), + AES_DECRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsDecrypt(&aes, buf, buf, limit + WC_AES_BLOCK_SIZE, + tweak, tweakLen), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + wc_AesXtsFree(&aes); +#endif + +#ifdef WOLFSSL_AESXTS_STREAM + { + /* Streaming: near the limit a request that would overflow is refused + * and a smaller one still fits, proving the refusal was not counted. */ + XtsAesStreamData xs; + const word32 chunk = 64 * 1024; + word32 done = 0; + byte* big = (byte*)XMALLOC(chunk * 2, NULL, DYNAMIC_TYPE_TMP_BUFFER); + + ExpectNotNull(big); + if (big != NULL) { + XMEMSET(big, 0x5a, chunk * 2); + XMEMSET(&xs, 0, sizeof(xs)); + ExpectIntEQ(wc_AesXtsSetKey(&aes, key32, sizeof(key32), + AES_ENCRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsEncryptInit(&aes, tweak, tweakLen, + &xs), 0); + while (EXPECT_SUCCESS() && done + chunk < limit) { + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, big, big, chunk, + &xs), 0); + done += chunk; + } + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, big, big, chunk * 2, + &xs), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, big, big, chunk, + &xs), 0); + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, buf, buf, + WC_AES_BLOCK_SIZE, &xs), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + ExpectIntEQ(wc_AesXtsEncryptFinal(&aes, NULL, NULL, 0, + &xs), 0); + wc_AesXtsFree(&aes); + + /* A new tweak is a new data unit. */ + XMEMSET(&xs, 0, sizeof(xs)); + ExpectIntEQ(wc_AesXtsSetKey(&aes, key32, sizeof(key32), + AES_ENCRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsEncryptInit(&aes, tweak, tweakLen, + &xs), 0); + ExpectIntEQ(wc_AesXtsEncryptUpdate(&aes, big, big, chunk, + &xs), 0); + ExpectIntEQ(wc_AesXtsEncryptFinal(&aes, NULL, NULL, 0, + &xs), 0); + wc_AesXtsFree(&aes); + +/* v6.0.0 limits the one-shot decrypt but not the streaming one. */ +#if FIPS_VERSION3_GE(7,0,0) && defined(HAVE_AES_DECRYPT) + done = 0; + XMEMSET(&xs, 0, sizeof(xs)); + ExpectIntEQ(wc_AesXtsSetKey(&aes, key32, sizeof(key32), + AES_DECRYPTION, NULL, INVALID_DEVID), 0); + ExpectIntEQ(wc_AesXtsDecryptInit(&aes, tweak, tweakLen, + &xs), 0); + while (EXPECT_SUCCESS() && done + chunk < limit) { + ExpectIntEQ(wc_AesXtsDecryptUpdate(&aes, big, big, chunk, + &xs), 0); + done += chunk; + } + ExpectIntEQ(wc_AesXtsDecryptUpdate(&aes, big, big, chunk * 2, + &xs), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + ExpectIntEQ(wc_AesXtsDecryptUpdate(&aes, big, big, chunk, + &xs), 0); + ExpectIntEQ(wc_AesXtsDecryptUpdate(&aes, buf, buf, + WC_AES_BLOCK_SIZE, &xs), WC_NO_ERR_TRACE(BAD_FUNC_ARG)); + ExpectIntEQ(wc_AesXtsDecryptFinal(&aes, NULL, NULL, 0, + &xs), 0); + wc_AesXtsFree(&aes); +#endif + XFREE(big, NULL, DYNAMIC_TYPE_TMP_BUFFER); + } + } +#endif /* WOLFSSL_AESXTS_STREAM */ +#endif + return EXPECT_RESULT(); +} + /* * test function for wc_AesXtsEncryptSector, wc_AesXtsDecryptSector, * wc_AesXtsEncryptConsecutiveSectors, and wc_AesXtsDecryptConsecutiveSectors */ + int test_wc_AesXtsEncryptDecryptSector(void) { EXPECT_DECLS; diff --git a/tests/api/test_aes.h b/tests/api/test_aes.h index 2bdf86eda15..f0b39d3f27d 100644 --- a/tests/api/test_aes.h +++ b/tests/api/test_aes.h @@ -59,6 +59,7 @@ int test_wc_AesGcmStream(void); int test_wc_AesGcmStream_MidStreamState(void); int test_wc_AesGcmStream_ReinitAfterFinal(void); int test_wc_AesGcmStream_BadAuthTag(void); +int test_wc_AesGcmDecrypt_WipeOnAuthFail(void); int test_wc_AesKeyWrapVectors(void); int test_wc_AesKeyWrapDecisionCoverage(void); int test_wc_AesGcmDecisionCoverage(void); @@ -88,6 +89,9 @@ int test_wc_AesXtsEncryptDecryptSector(void); int test_wc_AesXtsStream(void); int test_wc_AesXtsStream_MidStreamState(void); int test_wc_AesXtsStream_ReinitAfterFinal(void); +int test_wc_AesXtsStream_CounterOverflow(void); +int test_wc_AesUnalignedBuffers(void); +int test_wc_AesXtsDataUnitLimit(void); #if defined(WOLFSSL_AES_EAX) && defined(WOLFSSL_AES_256) && \ (!defined(HAVE_FIPS) || FIPS_VERSION_GE(5, 3)) && !defined(HAVE_SELFTEST) int test_wc_AesEaxVectors(void); @@ -222,6 +226,7 @@ int test_wc_CryptoCb_AesKeyWrapEcbCompose(void); TEST_DECL_GROUP("aes", test_wc_AesGcmStream_MidStreamState), \ TEST_DECL_GROUP("aes", test_wc_AesGcmStream_ReinitAfterFinal), \ TEST_DECL_GROUP("aes", test_wc_AesGcmStream_BadAuthTag), \ + TEST_DECL_GROUP("aes", test_wc_AesGcmDecrypt_WipeOnAuthFail), \ TEST_DECL_GROUP("aes", test_wc_AesKeyWrapVectors), \ TEST_DECL_GROUP("aes", test_wc_AesKeyWrapDecisionCoverage), \ TEST_DECL_GROUP("aes", test_wc_AesGcmDecisionCoverage), \ @@ -248,6 +253,9 @@ int test_wc_CryptoCb_AesKeyWrapEcbCompose(void); TEST_DECL_GROUP("aes", test_wc_AesXtsStream), \ TEST_DECL_GROUP("aes", test_wc_AesXtsStream_MidStreamState), \ TEST_DECL_GROUP("aes", test_wc_AesXtsStream_ReinitAfterFinal), \ + TEST_DECL_GROUP("aes", test_wc_AesXtsStream_CounterOverflow), \ + TEST_DECL_GROUP("aes", test_wc_AesUnalignedBuffers), \ + TEST_DECL_GROUP("aes", test_wc_AesXtsDataUnitLimit), \ TEST_DECL_GROUP("aes", test_wc_AesCbc_MonteCarlo), \ TEST_DECL_GROUP("aes", test_wc_AesCtr_MonteCarlo), \ TEST_DECL_GROUP("aes", test_wc_AesGcm_MonteCarlo), \ diff --git a/tests/unit-mcdc/test_aes_whitebox.c b/tests/unit-mcdc/test_aes_whitebox.c index b402a3148cd..6183d24d82b 100644 --- a/tests/unit-mcdc/test_aes_whitebox.c +++ b/tests/unit-mcdc/test_aes_whitebox.c @@ -281,13 +281,13 @@ static void wb_aesnew_common(void) * * AES_set_encrypt_key_AESNI / AES_set_decrypt_key_AESNI * line 1068 / 1099: if (!userKey || !aes) -> idx0 !userKey, idx1 !aes - * AesGcmAadUpdate_aesni + * AesGcmAadUpdate_asm * line 12136: if (aSz != 0 && a != NULL) -> idx1 (a != NULL) - * AesGcmEncryptUpdate_aesni - * line 12305: AesGcmAadUpdate_aesni(..., (cSz > 0) && (c != NULL)) + * AesGcmEncryptUpdate_asm + * line 12305: AesGcmAadUpdate_asm(..., (cSz > 0) && (c != NULL)) * -> idx1 (c != NULL) * line 12310: if (cSz != 0 && c != NULL) -> idx1 (c != NULL) - * AesGcmDecryptUpdate_aesni + * AesGcmDecryptUpdate_asm * line 12635: if (cSz != 0 && p != NULL) -> idx1 (p != NULL) * * The key setters return BAD_FUNC_ARG before any AES-NI work. For the GCM @@ -331,7 +331,8 @@ static void wb_aesni(void) } } -#if defined(HAVE_AESGCM) && defined(WOLFSSL_AESGCM_STREAM) +/* The key-expansion pairs above are plain AES-NI; only this group is GCM. */ +#if defined(WC_AESNI_GCM) && defined(HAVE_AESGCM) && defined(WOLFSSL_AESGCM_STREAM) { /* AES-NI GCM streaming ptr guards; both halves within this binary */ Aes aes; byte key[16], iv[12], in[16], out[16]; @@ -342,19 +343,19 @@ static void wb_aesni(void) wc_AesGcmInit(&aes, key, sizeof(key), iv, sizeof(iv)) == 0) { /* 12136: if (aSz != 0 && a != NULL) -- hold aSz!=0, flip a!=NULL */ aes.aOver = 0; - (void)AesGcmAadUpdate_aesni(&aes, in, 16, 0); /* a!=NULL T */ + (void)AesGcmAadUpdate_asm(&aes, in, 16, 0); /* a!=NULL T */ aes.aOver = 0; - (void)AesGcmAadUpdate_aesni(&aes, NULL, 16, 0); /* a!=NULL F */ + (void)AesGcmAadUpdate_asm(&aes, NULL, 16, 0); /* a!=NULL F */ /* 12305 + 12310: c!=NULL -- hold cSz!=0, flip c (out) */ aes.aOver = 0; aes.cOver = 0; - (void)AesGcmEncryptUpdate_aesni(&aes, out, in, 16, in, 16); /* c T */ + (void)AesGcmEncryptUpdate_asm(&aes, out, in, 16, in, 16); /* c T */ aes.aOver = 0; aes.cOver = 0; - (void)AesGcmEncryptUpdate_aesni(&aes, NULL, in, 16, in, 16); /* c F */ + (void)AesGcmEncryptUpdate_asm(&aes, NULL, in, 16, in, 16); /* c F */ /* 12635: p!=NULL -- hold cSz!=0, flip p (out) */ aes.aOver = 0; aes.cOver = 0; - (void)AesGcmDecryptUpdate_aesni(&aes, out, in, 16, in, 16); /* p T */ + (void)AesGcmDecryptUpdate_asm(&aes, out, in, 16, in, 16); /* p T */ aes.aOver = 0; aes.cOver = 0; - (void)AesGcmDecryptUpdate_aesni(&aes, NULL, in, 16, in, 16); /* p F */ + (void)AesGcmDecryptUpdate_asm(&aes, NULL, in, 16, in, 16); /* p F */ wc_AesFree(&aes); } else { diff --git a/wolfcrypt/src/aes.c b/wolfcrypt/src/aes.c index f7aa53f9eb1..5c294833b5d 100644 --- a/wolfcrypt/src/aes.c +++ b/wolfcrypt/src/aes.c @@ -64,6 +64,7 @@ block cipher mechanism that uses n-bit binary string parameter key with 128-bits * HAVE_AESGCM_DECRYPT: Enable AES-GCM decryption default: on * (when HAVE_AESGCM is enabled) * WOLFSSL_AESGCM_STREAM: Enable streaming AES-GCM API default: off + * WC_AESNI_GCM: x86 AES-GCM asm in use, set by aes.c (internal) * WC_AES_GCM_DEC_AUTH_EARLY: Authenticate tag before decryption default: off * GCM_SMALL: Small GCM table, saves memory default: off * GCM_TABLE: Full 4-bit GCM lookup table, faster default: off @@ -166,6 +167,37 @@ block cipher mechanism that uses n-bit binary string parameter key with 128-bits #define WC_AES_ARM64_SVR_END() WC_DO_NOTHING #endif +/* The 32-bit x86 GCM asm has text relocations the integrity hash cannot cover, + * so 32-bit builds run the C GCM over AES-NI blocks. */ +#if defined(WOLFSSL_AESNI) && !defined(WOLFSSL_X86_BUILD) + #define WC_AESNI_GCM +#endif +/* The C GCM streaming path reaches vector code on aarch64 and on 32-bit x86 + * with AES-NI, so it holds the registers itself there. */ +#if defined(__aarch64__) && defined(WOLFSSL_ARMASM) + #define WC_AES_GCM_C_SVR_BEGIN() WC_AES_ARM64_SVR_BEGIN() + #define WC_AES_GCM_C_SVR_END() WC_AES_ARM64_SVR_END() +#elif defined(WOLFSSL_AESNI) && !defined(WC_AESNI_GCM) + #ifdef WC_C_DYNAMIC_FALLBACK + /* A refused claim drops this object to the C block and leaves it there; the + * C path keeps its own key copy. */ + #define WC_AES_GCM_C_SVR_BEGIN() \ + do { if (aes->use_aesni) { \ + if (SAVE_VECTOR_REGISTERS2() != 0) \ + aes->use_aesni = 0; \ + } } while (0) + #else + #define WC_AES_GCM_C_SVR_BEGIN() \ + do { if (aes->use_aesni) { SAVE_VECTOR_REGISTERS(return _svr_ret;); } \ + } while (0) + #endif + #define WC_AES_GCM_C_SVR_END() \ + do { if (aes->use_aesni) { RESTORE_VECTOR_REGISTERS(); } } while (0) +#else + #define WC_AES_GCM_C_SVR_BEGIN() WC_DO_NOTHING + #define WC_AES_GCM_C_SVR_END() WC_DO_NOTHING +#endif + #ifdef WOLF_CRYPTO_CB #include #endif @@ -1171,6 +1203,134 @@ static void Check_CPU_support_HwCrypto(Aes* aes) #endif /* (__aarch64__ && !WOLFSSL_ARMASM_NO_HW_CRYPTO) || * WOLFSSL_ARM32_AES_DISPATCH */ +/* A kernel build must hold the vector registers around the 32-bit Arm asm. */ +#if !defined(__aarch64__) && !defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) + /* Every 32-bit Arm asm call returns a status, so a failed vector-register + * save is reported instead of the work being skipped silently. */ + #define WC_AES32_SVR_BEGIN() \ + do { \ + int _svr = SAVE_VECTOR_REGISTERS2(); \ + if (_svr != 0) return _svr; \ + } while (0) + #define WC_AES32_SVR_END() RESTORE_VECTOR_REGISTERS() + static WC_INLINE int wc_svr_AES_set_key_AARCH32(const byte* userKey, + int keylen, byte* key, int dir) { + WC_AES32_SVR_BEGIN(); + AES_set_key_AARCH32(userKey, keylen, key, dir); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_encrypt_AARCH32(const byte* inBlock, + byte* outBlock, byte* key, int nr) { + WC_AES32_SVR_BEGIN(); + AES_encrypt_AARCH32(inBlock, outBlock, key, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_decrypt_AARCH32(const byte* inBlock, + byte* outBlock, byte* key, int nr) { + WC_AES32_SVR_BEGIN(); + AES_decrypt_AARCH32(inBlock, outBlock, key, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_encrypt_blocks_AARCH32(const byte* in, + byte* out, word32 sz, byte* key, int nr) { + WC_AES32_SVR_BEGIN(); + AES_encrypt_blocks_AARCH32(in, out, sz, key, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_decrypt_blocks_AARCH32(const byte* in, + byte* out, word32 sz, byte* key, int nr) { + WC_AES32_SVR_BEGIN(); + AES_decrypt_blocks_AARCH32(in, out, sz, key, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_CBC_encrypt_AARCH32(const byte* in, + byte* out, word32 sz, byte* reg, byte* key, int rounds) { + WC_AES32_SVR_BEGIN(); + AES_CBC_encrypt_AARCH32(in, out, sz, reg, key, rounds); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_CBC_decrypt_AARCH32(const byte* in, + byte* out, word32 sz, byte* reg, byte* key, int rounds) { + WC_AES32_SVR_BEGIN(); + AES_CBC_decrypt_AARCH32(in, out, sz, reg, key, rounds); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_CTR_encrypt_AARCH32(const byte* in, + byte* out, word32 sz, byte* reg, byte* key, byte* tmp, word32* left, + word32 rounds) { + WC_AES32_SVR_BEGIN(); + AES_CTR_encrypt_AARCH32(in, out, sz, reg, key, tmp, left, rounds); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_GCM_set_key_AARCH32(const byte* nonce, + const byte* key, byte* gcm_h, int nr) { + WC_AES32_SVR_BEGIN(); + AES_GCM_set_key_AARCH32(nonce, key, gcm_h, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_GCM_encrypt_AARCH32(const byte* in, + byte* out, word32 sz, const byte* nonce, word32 nonceSz, byte* tag, + word32 tagSz, const byte* aad, word32 aadSz, byte* key, byte* gcm_h, + byte* tmp, byte* reg, int nr) { + WC_AES32_SVR_BEGIN(); + AES_GCM_encrypt_AARCH32(in, out, sz, nonce, nonceSz, tag, tagSz, aad, + aadSz, key, gcm_h, tmp, reg, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_GCM_decrypt_AARCH32(const byte* in, + byte* out, word32 sz, const byte* nonce, word32 nonceSz, const byte* tag, + word32 tagSz, const byte* aad, word32 aadSz, byte* key, byte* gcm_h, + byte* tmp, byte* reg, int nr) { + int _ret; + WC_AES32_SVR_BEGIN(); + _ret = AES_GCM_decrypt_AARCH32(in, out, sz, nonce, nonceSz, tag, tagSz, + aad, aadSz, key, gcm_h, tmp, reg, nr); + WC_AES32_SVR_END(); + return _ret; + } + #define AES_set_key_AARCH32 wc_svr_AES_set_key_AARCH32 + #define AES_encrypt_AARCH32 wc_svr_AES_encrypt_AARCH32 + #define AES_decrypt_AARCH32 wc_svr_AES_decrypt_AARCH32 + #define AES_encrypt_blocks_AARCH32 wc_svr_AES_encrypt_blocks_AARCH32 + #define AES_decrypt_blocks_AARCH32 wc_svr_AES_decrypt_blocks_AARCH32 + #define AES_CBC_encrypt_AARCH32 wc_svr_AES_CBC_encrypt_AARCH32 + #define AES_CBC_decrypt_AARCH32 wc_svr_AES_CBC_decrypt_AARCH32 + #define AES_CTR_encrypt_AARCH32 wc_svr_AES_CTR_encrypt_AARCH32 + #define AES_GCM_set_key_AARCH32 wc_svr_AES_GCM_set_key_AARCH32 + #define AES_GCM_encrypt_AARCH32 wc_svr_AES_GCM_encrypt_AARCH32 + #define AES_GCM_decrypt_AARCH32 wc_svr_AES_GCM_decrypt_AARCH32 + #ifdef WOLFSSL_AES_XTS + static WC_INLINE int wc_svr_AES_XTS_encrypt_AARCH32(const byte* in, + byte* out, word32 sz, const byte* i, byte* key, byte* key2, byte* tmp, + int nr) { + WC_AES32_SVR_BEGIN(); + AES_XTS_encrypt_AARCH32(in, out, sz, i, key, key2, tmp, nr); + WC_AES32_SVR_END(); + return 0; + } + static WC_INLINE int wc_svr_AES_XTS_decrypt_AARCH32(const byte* in, + byte* out, word32 sz, const byte* i, byte* key, byte* key2, byte* tmp, + int nr) { + WC_AES32_SVR_BEGIN(); + AES_XTS_decrypt_AARCH32(in, out, sz, i, key, key2, tmp, nr); + WC_AES32_SVR_END(); + return 0; + } + #define AES_XTS_encrypt_AARCH32 wc_svr_AES_XTS_encrypt_AARCH32 + #define AES_XTS_decrypt_AARCH32 wc_svr_AES_XTS_decrypt_AARCH32 + #endif /* WOLFSSL_AES_XTS */ +#endif /* !__aarch64__ && !WOLFSSL_ARMASM_NO_HW_CRYPTO */ + #if defined(WOLFSSL_AES_DIRECT) || defined(HAVE_AESCCM) || \ defined(WOLFSSL_AESGCM_STREAM) || defined(WOLFSSL_AESGCM_SIV) static WARN_UNUSED_RESULT int wc_AesEncrypt(Aes* aes, const byte* inBlock, @@ -1180,12 +1340,21 @@ static WARN_UNUSED_RESULT int wc_AesEncrypt(Aes* aes, const byte* inBlock, #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + int _svr_ret = AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } } else #else - AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + { + int _svr_ret = AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -1221,12 +1390,21 @@ static WARN_UNUSED_RESULT int wc_AesDecrypt(Aes* aes, const byte* inBlock, #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + int _svr_ret = AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } } else #else - AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + { + int _svr_ret = AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -3645,12 +3823,21 @@ WC_ALL_ARGS_NOT_NULL static WARN_UNUSED_RESULT int wc_AesEncrypt( #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + int _svr_ret = AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } } else #else - AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + { + int _svr_ret = AES_encrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -4499,12 +4686,21 @@ WC_ALL_ARGS_NOT_NULL static WARN_UNUSED_RESULT int wc_AesDecrypt( #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + int _svr_ret = AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } } else #else - AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, (int)aes->rounds); + { + int _svr_ret = AES_decrypt_AARCH32(inBlock, outBlock, (byte*)aes->key, + (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -5049,11 +5245,23 @@ static WARN_UNUSED_RESULT int wc_AesDecrypt(Aes* aes, const byte* inBlock, #ifdef WOLFSSL_ARM32_AES_DISPATCH Check_CPU_support_HwCrypto(aes); if (aes->use_aes_hw_crypto) { - AES_set_key_AARCH32(userKey, keylen, (byte*)aes->key, dir); + int _svr_ret = AES_set_key_AARCH32(userKey, keylen, + (byte*)aes->key, dir); + if (_svr_ret != 0) { + aes->keyInstalled = 0; + return _svr_ret; + } } else #else - AES_set_key_AARCH32(userKey, keylen, (byte*)aes->key, dir); + { + int _svr_ret = AES_set_key_AARCH32(userKey, keylen, + (byte*)aes->key, dir); + if (_svr_ret != 0) { + aes->keyInstalled = 0; + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #endif /* !WOLFSSL_ARMASM_NO_HW_CRYPTO */ #if defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) || defined(WOLFSSL_ARM32_AES_DISPATCH) @@ -5989,11 +6197,29 @@ static void AesSetKey_C(Aes* aes, const byte* key, word32 keySz, int dir) #ifdef WOLFSSL_ARM32_AES_DISPATCH Check_CPU_support_HwCrypto(aes); if (aes->use_aes_hw_crypto) { - AES_set_key_AARCH32(userKey, keylen, (byte*)aes->key, dir); + int _svr_ret = AES_set_key_AARCH32(userKey, keylen, + (byte*)aes->key, dir); + if (_svr_ret != 0) { + #ifdef WOLFSSL_IMX6_CAAM_BLOB + /* local[] holds the raw key; wipe it on this early return. */ + ForceZero(local, sizeof(local)); + #endif + return _svr_ret; + } } else #else - AES_set_key_AARCH32(userKey, keylen, (byte*)aes->key, dir); + { + int _svr_ret = AES_set_key_AARCH32(userKey, keylen, + (byte*)aes->key, dir); + if (_svr_ret != 0) { + #ifdef WOLFSSL_IMX6_CAAM_BLOB + /* local[] holds the raw key; wipe it on this early return. */ + ForceZero(local, sizeof(local)); + #endif + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else Check_CPU_support_HwCrypto(aes); @@ -6031,6 +6257,9 @@ static void AesSetKey_C(Aes* aes, const byte* key, word32 keySz, int dir) (void)dir; #endif } + #endif + #ifdef WOLFSSL_IMX6_CAAM_BLOB + ForceZero(local, sizeof(local)); #endif return 0; #else @@ -7404,13 +7633,21 @@ int wc_AesCbcEncrypt(Aes* aes, byte* out, const byte* in, word32 sz) #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_CBC_encrypt_AARCH32(in, out, sz, (byte*)aes->reg, - (byte*)aes->key, (int)aes->rounds); + int _svr_ret = AES_CBC_encrypt_AARCH32(in, out, sz, + (byte*)aes->reg, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } } else #else - AES_CBC_encrypt_AARCH32(in, out, sz, (byte*)aes->reg, (byte*)aes->key, - (int)aes->rounds); + { + int _svr_ret = AES_CBC_encrypt_AARCH32(in, out, sz, + (byte*)aes->reg, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -7662,13 +7899,21 @@ int wc_AesCbcEncrypt(Aes* aes, byte* out, const byte* in, word32 sz) #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_CBC_decrypt_AARCH32(in, out, sz, (byte*)aes->reg, - (byte*)aes->key, (int)aes->rounds); + int _svr_ret = AES_CBC_decrypt_AARCH32(in, out, sz, + (byte*)aes->reg, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } } else #else - AES_CBC_decrypt_AARCH32(in, out, sz, (byte*)aes->reg, (byte*)aes->key, - (int)aes->rounds); + { + int _svr_ret = AES_CBC_decrypt_AARCH32(in, out, sz, + (byte*)aes->reg, (byte*)aes->key, (int)aes->rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -7819,8 +8064,10 @@ int wc_AesCbcEncrypt(Aes* aes, byte* out, const byte* in, word32 sz) while (blocks--) { XMEMCPY(aes->tmp, in, WC_AES_BLOCK_SIZE); ret = AesDecrypt_preFetchOpt(aes, in, out, &did_prefetches); + /* break, not return: a return here would skip + * VECTOR_REGISTERS_POP and leave the vector registers held. */ if (ret != 0) - return ret; + break; xorbuf(out, (byte*)aes->reg, WC_AES_BLOCK_SIZE); /* store iv for next call */ XMEMCPY(aes->reg, aes->tmp, WC_AES_BLOCK_SIZE); @@ -8141,6 +8388,16 @@ int wc_AesCbcEncrypt(Aes* aes, byte* out, const byte* in, word32 sz) SAVE_VECTOR_REGISTERS(return _svr_ret;); } #endif + #if defined(WOLFSSL_ARMASM) && !defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) && \ + !defined(__aarch64__) + /* Same reason: the 32-bit Arm wrapper claims after the drain. */ + #ifdef WOLFSSL_ARM32_AES_DISPATCH + if (aes->use_aes_hw_crypto) + #endif + { + SAVE_VECTOR_REGISTERS(return _svr_ret;); + } + #endif /* consume any unused bytes left in aes->tmp */ processed = min(aes->left, sz); @@ -8167,14 +8424,26 @@ int wc_AesCbcEncrypt(Aes* aes, byte* out, const byte* in, word32 sz) #ifndef __aarch64__ #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_CTR_encrypt_AARCH32(in, out, sz, (byte*)aes->reg, - (byte*)aes->key, (byte*)aes->tmp, &aes->left, aes->rounds); + int _svr_ret = AES_CTR_encrypt_AARCH32(in, out, sz, + (byte*)aes->reg, (byte*)aes->key, (byte*)aes->tmp, + &aes->left, aes->rounds); + RESTORE_VECTOR_REGISTERS(); + if (_svr_ret != 0) { + return _svr_ret; + } return 0; } else #else - AES_CTR_encrypt_AARCH32(in, out, sz, (byte*)aes->reg, - (byte*)aes->key, (byte*)aes->tmp, &aes->left, aes->rounds); + { + int _svr_ret = AES_CTR_encrypt_AARCH32(in, out, sz, + (byte*)aes->reg, (byte*)aes->key, (byte*)aes->tmp, + &aes->left, aes->rounds); + RESTORE_VECTOR_REGISTERS(); + if (_svr_ret != 0) { + return _svr_ret; + } + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto) { @@ -8757,7 +9026,7 @@ void GenerateM0(Gcm* gcm) #endif #endif -#if defined(WOLFSSL_AESNI) && defined(GCM_TABLE_4BIT) && \ +#if defined(WC_AESNI_GCM) && defined(GCM_TABLE_4BIT) && \ defined(WC_C_DYNAMIC_FALLBACK) void GCM_generate_m0_aesni(const unsigned char *h, unsigned char *m) XASM_LINK("GCM_generate_m0_aesni"); @@ -8880,11 +9149,18 @@ int wc_AesGcmSetKey(Aes* aes, const byte* key, word32 len) #if !defined(__aarch64__) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto && aes->use_pmull_hw_crypto) { - AES_GCM_set_key_AARCH32(iv, (byte*)aes->key, aes->gcm.H, - aes->rounds); - /* Undo the reflection the assembly applied, so the stored H is - * plain H for the portable streaming GHASH and for GenerateM0 - * below. Each bulk assembly call reflects its own copy. */ + int _svr_ret = AES_GCM_set_key_AARCH32(iv, (byte*)aes->key, + aes->gcm.H, aes->rounds); + if (_svr_ret != 0) { + #ifdef WOLFSSL_IMX6_CAAM_BLOB + /* local[] holds the raw key; wipe it on this early return. */ + ForceZero(local, sizeof(local)); + #endif + WC_AES_GCM_UNKEY(aes); + return _svr_ret; + } + /* Undo the assembly's reflection so the stored H is plain H for the + * streaming GHASH and GenerateM0; bulk calls reflect their own. */ GcmReflectH(aes->gcm.H); #if defined(GCM_TABLE) || defined(GCM_TABLE_4BIT) GenerateM0(&aes->gcm); @@ -8892,14 +9168,24 @@ int wc_AesGcmSetKey(Aes* aes, const byte* key, word32 len) } else #else - AES_GCM_set_key_AARCH32(iv, (byte*)aes->key, aes->gcm.H, aes->rounds); - /* Undo the reflection the assembly applied, so the stored H is plain - * H for the portable streaming GHASH and for GenerateM0 below. Each - * bulk assembly call reflects its own copy. */ - GcmReflectH(aes->gcm.H); + { + int _svr_ret = AES_GCM_set_key_AARCH32(iv, (byte*)aes->key, + aes->gcm.H, aes->rounds); + if (_svr_ret != 0) { + #ifdef WOLFSSL_IMX6_CAAM_BLOB + /* local[] holds the raw key; wipe it on this early return. */ + ForceZero(local, sizeof(local)); + #endif + WC_AES_GCM_UNKEY(aes); + return _svr_ret; + } + /* Undo the assembly's reflection so the stored H is plain H for the + * streaming GHASH and GenerateM0; bulk calls reflect their own. */ + GcmReflectH(aes->gcm.H); #if defined(GCM_TABLE) || defined(GCM_TABLE_4BIT) - GenerateM0(&aes->gcm); + GenerateM0(&aes->gcm); #endif + } #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto && aes->use_pmull_hw_crypto) { @@ -8958,7 +9244,7 @@ int wc_AesGcmSetKey(Aes* aes, const byte* key, word32 len) if (ret == 0) { #if defined(GCM_TABLE) || defined(GCM_TABLE_4BIT) - #if defined(WOLFSSL_AESNI) && defined(GCM_TABLE_4BIT) + #if defined(WC_AESNI_GCM) && defined(GCM_TABLE_4BIT) if (aes->use_aesni) { #if defined(WC_C_DYNAMIC_FALLBACK) #ifdef HAVE_INTEL_AVX2 @@ -8982,7 +9268,7 @@ int wc_AesGcmSetKey(Aes* aes, const byte* key, word32 len) #endif /* WC_C_DYNAMIC_FALLBACK */ } else - #endif /* AESNI */ + #endif /* WC_AESNI_GCM && GCM_TABLE_4BIT */ { GenerateM0(&aes->gcm); } @@ -9026,7 +9312,8 @@ int wc_AesGcmSetKey(Aes* aes, const byte* key, word32 len) } -#ifdef WOLFSSL_AESNI + +#ifdef WC_AESNI_GCM void AES_GCM_encrypt_aesni(const unsigned char *in, unsigned char *out, const unsigned char* addt, const unsigned char* ivec, @@ -9112,7 +9399,7 @@ void AES_GCM_decrypt_vaes(const unsigned char *in, unsigned char *out, #endif /* HAVE_INTEL_AVX1 */ #endif /* HAVE_AES_DECRYPT */ -#endif /* WOLFSSL_AESNI */ +#endif /* WC_AESNI_GCM */ #if defined(WOLFSSL_RISCV_SCALAR_CRYPTO_ASM) && defined(HAVE_AESGCM) && \ !defined(WOLFSSL_RISCV_VECTOR_CRYPTO_ASM) @@ -11705,14 +11992,16 @@ int wc_AesGcmEncrypt(Aes* aes, byte* out, const byte* in, word32 sz, /* Reflect a copy of H into the form the PMULL assembly wants - the * stored H must stay un-reflected for the portable GHASH. */ byte h[WC_AES_BLOCK_SIZE]; + int _svr_ret; XMEMCPY(h, aes->gcm.H, WC_AES_BLOCK_SIZE); GcmReflectH(h); - AES_GCM_encrypt_AARCH32(in, out, sz, iv, ivSz, authTag, authTagSz, - authIn, authInSz, (byte*)aes->key, h, (byte*)aes->tmp, + _svr_ret = AES_GCM_encrypt_AARCH32(in, out, sz, iv, ivSz, authTag, + authTagSz, authIn, authInSz, (byte*)aes->key, h, (byte*)aes->tmp, (byte*)aes->reg, aes->rounds); ForceZero(h, sizeof(h)); - ret = 0; + /* Carry the status to VECTOR_REGISTERS_POP; do not return here. */ + ret = _svr_ret; } else #else @@ -11720,15 +12009,16 @@ int wc_AesGcmEncrypt(Aes* aes, byte* out, const byte* in, word32 sz, /* Reflect a copy of H into the form the PMULL assembly wants - the * stored H must stay un-reflected for the portable GHASH. */ byte h[WC_AES_BLOCK_SIZE]; + int _svr_ret; XMEMCPY(h, aes->gcm.H, WC_AES_BLOCK_SIZE); GcmReflectH(h); - AES_GCM_encrypt_AARCH32(in, out, sz, iv, ivSz, authTag, authTagSz, - authIn, authInSz, (byte*)aes->key, h, (byte*)aes->tmp, + _svr_ret = AES_GCM_encrypt_AARCH32(in, out, sz, iv, ivSz, authTag, + authTagSz, authIn, authInSz, (byte*)aes->key, h, (byte*)aes->tmp, (byte*)aes->reg, aes->rounds); ForceZero(h, sizeof(h)); + ret = _svr_ret; } - ret = 0; #endif /* WOLFSSL_ARM32_AES_DISPATCH */ #else if (aes->use_aes_hw_crypto && aes->use_pmull_hw_crypto) { @@ -11762,7 +12052,7 @@ int wc_AesGcmEncrypt(Aes* aes, byte* out, const byte* in, word32 sz, ret = AES_GCM_encrypt_ASM(aes, out, in, sz, iv, ivSz, authTag, authTagSz, authIn, authInSz); #else -#ifdef WOLFSSL_AESNI +#ifdef WC_AESNI_GCM if (aes->use_aesni) { #ifdef HAVE_INTEL_AVX512 if ((sz >= WC_AES_BLOCK_SIZE * WC_VAES_GCM_MIN_BLOCKS) && @@ -11804,7 +12094,7 @@ int wc_AesGcmEncrypt(Aes* aes, byte* out, const byte* in, word32 sz, } } else -#endif /* WOLFSSL_AESNI */ +#endif /* WC_AESNI_GCM */ { ret = AES_GCM_encrypt_C(aes, out, in, sz, iv, ivSz, authTag, authTagSz, authIn, authInSz); @@ -12414,7 +12704,7 @@ int wc_AesGcmDecrypt(Aes* aes, byte* out, const byte* in, word32 sz, const byte* authIn, word32 authInSz) { int ret; -#ifdef WOLFSSL_AESNI +#ifdef WC_AESNI_GCM int res = WC_NO_ERR_TRACE(AES_GCM_AUTH_E); #endif @@ -12621,7 +12911,7 @@ int wc_AesGcmDecrypt(Aes* aes, byte* out, const byte* in, word32 sz, authTagSz, authIn, authInSz); } #else -#ifdef WOLFSSL_AESNI +#ifdef WC_AESNI_GCM if (aes->use_aesni) { #ifdef HAVE_INTEL_AVX512 if ((sz >= WC_AES_BLOCK_SIZE * WC_VAES_GCM_MIN_BLOCKS) && @@ -12679,7 +12969,7 @@ int wc_AesGcmDecrypt(Aes* aes, byte* out, const byte* in, word32 sz, } } else -#endif /* WOLFSSL_AESNI */ +#endif /* WC_AESNI_GCM */ { ret = AES_GCM_decrypt_C(aes, out, in, sz, iv, ivSz, authTag, authTagSz, authIn, authInSz); @@ -12688,6 +12978,15 @@ int wc_AesGcmDecrypt(Aes* aes, byte* out, const byte* in, word32 sz, VECTOR_REGISTERS_POP; + /* Wipe the output on a failed tag so unauthenticated plaintext is not + * handed back. The hardware backends above return before this point. */ +#if !(defined(HAVE_FIPS_VERSION) && (HAVE_FIPS_VERSION >= 2) && \ + defined(ACVP_VECTOR_TESTING)) + if (ret == WC_NO_ERR_TRACE(AES_GCM_AUTH_E) && out != NULL && sz > 0) { + ForceZero(out, sz); + } +#endif + return ret; } #endif @@ -12860,7 +13159,7 @@ static WARN_UNUSED_RESULT int AesGcmFinal_C( return 0; } -#ifdef WOLFSSL_AESNI +#ifdef WC_AESNI_GCM #ifdef __cplusplus extern "C" { @@ -12978,13 +13277,13 @@ extern void AES_GCM_encrypt_final_aesni(unsigned char* tag, } /* extern "C" */ #endif -/* Initialize the AES GCM cipher with an IV. AES-NI implementations. +/* Initialize the AES GCM cipher with an IV. x86 assembly back end. * * @param [in, out] aes AES object. * @param [in] iv IV/nonce buffer. * @param [in] ivSz Length of IV/nonce data. */ -static WARN_UNUSED_RESULT int AesGcmInit_aesni( +static WARN_UNUSED_RESULT int AesGcmInit_asm( Aes* aes, const byte* iv, word32 ivSz) { ASSERT_SAVED_VECTOR_REGISTERS(); @@ -13037,14 +13336,14 @@ static WARN_UNUSED_RESULT int AesGcmInit_aesni( /* Update the AES GCM for encryption with authentication data. * - * Implementation uses AVX2, AVX1 or straight AES-NI optimized assembly code. + * Runs on the x86 AES-GCM assembly, whichever width the CPU supports. * * @param [in, out] aes AES object. * @param [in] a Buffer holding authentication data. * @param [in] aSz Length of authentication data in bytes. * @param [in] endA Whether no more authentication data is expected. */ -static WARN_UNUSED_RESULT int AesGcmAadUpdate_aesni( +static WARN_UNUSED_RESULT int AesGcmAadUpdate_asm( Aes* aes, const byte* a, word32 aSz, int endA) { word32 blocks; @@ -13202,7 +13501,7 @@ static WARN_UNUSED_RESULT int AesGcmAadUpdate_aesni( /* Update the AES GCM for encryption with data and/or authentication data. * - * Implementation uses AVX2, AVX1 or straight AES-NI optimized assembly code. + * Runs on the x86 AES-GCM assembly, whichever width the CPU supports. * * @param [in, out] aes AES object. * @param [out] c Buffer to hold cipher text. @@ -13211,7 +13510,7 @@ static WARN_UNUSED_RESULT int AesGcmAadUpdate_aesni( * @param [in] a Buffer holding authentication data. * @param [in] aSz Length of authentication data in bytes. */ -static WARN_UNUSED_RESULT int AesGcmEncryptUpdate_aesni( +static WARN_UNUSED_RESULT int AesGcmEncryptUpdate_asm( Aes* aes, byte* c, const byte* p, word32 cSz, const byte* a, word32 aSz) { word32 blocks; @@ -13221,7 +13520,7 @@ static WARN_UNUSED_RESULT int AesGcmEncryptUpdate_aesni( ASSERT_SAVED_VECTOR_REGISTERS(); /* Hash in A, the Authentication Data */ - ret = AesGcmAadUpdate_aesni(aes, a, aSz, (cSz > 0) && (c != NULL)); + ret = AesGcmAadUpdate_asm(aes, a, aSz, (cSz > 0) && (c != NULL)); if (ret != 0) return ret; @@ -13380,14 +13679,14 @@ static WARN_UNUSED_RESULT int AesGcmEncryptUpdate_aesni( /* Finalize the AES GCM for encryption and calculate the authentication tag. * - * Calls AVX2, AVX1 or straight AES-NI optimized assembly code. + * Runs on the x86 AES-GCM assembly, whichever width the CPU supports. * * @param [in, out] aes AES object. * @param [in] authTag Buffer to hold authentication tag. * @param [in] authTagSz Length of authentication tag in bytes. * @return 0 on success. */ -static WARN_UNUSED_RESULT int AesGcmEncryptFinal_aesni( +static WARN_UNUSED_RESULT int AesGcmEncryptFinal_asm( Aes* aes, byte* authTag, word32 authTagSz) { /* AAD block incomplete when > 0 */ @@ -13536,7 +13835,7 @@ extern void AES_GCM_decrypt_final_aesni(unsigned char* tag, * @param [in] a Buffer holding authentication data. * @param [in] aSz Length of authentication data in bytes. */ -static WARN_UNUSED_RESULT int AesGcmDecryptUpdate_aesni( +static WARN_UNUSED_RESULT int AesGcmDecryptUpdate_asm( Aes* aes, byte* p, const byte* c, word32 cSz, const byte* a, word32 aSz) { word32 blocks; @@ -13546,7 +13845,7 @@ static WARN_UNUSED_RESULT int AesGcmDecryptUpdate_aesni( ASSERT_SAVED_VECTOR_REGISTERS(); /* Hash in A, the Authentication Data */ - ret = AesGcmAadUpdate_aesni(aes, a, aSz, cSz > 0); + ret = AesGcmAadUpdate_asm(aes, a, aSz, cSz > 0); if (ret != 0) return ret; @@ -13717,7 +14016,7 @@ static WARN_UNUSED_RESULT int AesGcmDecryptUpdate_aesni( * @return AES_GCM_AUTH_E when authentication tag doesn't match calculated * value. */ -static WARN_UNUSED_RESULT int AesGcmDecryptFinal_aesni( +static WARN_UNUSED_RESULT int AesGcmDecryptFinal_asm( Aes* aes, const byte* authTag, word32 authTagSz) { int ret = 0; @@ -13806,7 +14105,7 @@ static WARN_UNUSED_RESULT int AesGcmDecryptFinal_aesni( return ret; } #endif /* HAVE_AES_DECRYPT || HAVE_AESGCM_DECRYPT */ -#endif /* WOLFSSL_AESNI */ +#endif /* WC_AESNI_GCM */ #if defined(__aarch64__) && defined(WOLFSSL_ARMASM) && \ !defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) @@ -14218,7 +14517,7 @@ static WARN_UNUSED_RESULT int AesGcmDecryptUpdate_AARCH64(Aes* aes, byte* p, /* Finalize the AES GCM for decryption and check the authentication tag. * - * Calls AVX2, AVX1 or straight AES-NI optimized assembly code. + * Runs on the AArch64 crypto-extension assembly. * * @param [in, out] aes AES object. * @param [in] authTag Buffer holding authentication tag. @@ -14738,11 +15037,11 @@ int wc_AesGcmInit(Aes* aes, const byte* key, word32 len, const byte* iv, if (iv != NULL) { /* Initialize with the IV. */ - #ifdef WOLFSSL_AESNI + #ifdef WC_AESNI_GCM if (aes->use_aesni) { ret = SAVE_VECTOR_REGISTERS2(); if (ret == 0) { - ret = AesGcmInit_aesni(aes, iv, ivSz); + ret = AesGcmInit_asm(aes, iv, ivSz); RESTORE_VECTOR_REGISTERS(); } else { @@ -14766,11 +15065,11 @@ int wc_AesGcmInit(Aes* aes, const byte* key, word32 len, const byte* iv, #elif defined(WOLFSSL_RISCV_ASM) ret = AesGcmInit_RISCV64(aes, iv, ivSz); if (0) - #endif /* WOLFSSL_AESNI */ + #endif /* WC_AESNI_GCM */ { - WC_AES_ARM64_SVR_BEGIN(); + WC_AES_GCM_C_SVR_BEGIN(); ret = AesGcmInit_C(aes, iv, ivSz); - WC_AES_ARM64_SVR_END(); + WC_AES_GCM_C_SVR_END(); } if (ret == 0) @@ -14900,10 +15199,10 @@ int wc_AesGcmEncryptUpdate(Aes* aes, byte* out, const byte* in, word32 sz, if (ret == 0) { /* Encrypt with AAD and/or plaintext. */ - #ifdef WOLFSSL_AESNI + #ifdef WC_AESNI_GCM if (aes->use_aesni) { SAVE_VECTOR_REGISTERS(return _svr_ret;); - ret = AesGcmEncryptUpdate_aesni(aes, out, in, sz, authIn, authInSz); + ret = AesGcmEncryptUpdate_asm(aes, out, in, sz, authIn, authInSz); RESTORE_VECTOR_REGISTERS(); } else @@ -14921,7 +15220,7 @@ int wc_AesGcmEncryptUpdate(Aes* aes, byte* out, const byte* in, word32 sz, if (0) #endif { - WC_AES_ARM64_SVR_BEGIN(); + WC_AES_GCM_C_SVR_BEGIN(); /* Encrypt the plaintext. */ ret = AesGcmCryptUpdate_C(aes, out, in, sz); if (ret == 0) { @@ -14929,7 +15228,7 @@ int wc_AesGcmEncryptUpdate(Aes* aes, byte* out, const byte* in, word32 sz, * new cipher text. */ GHASH_UPDATE(aes, authIn, authInSz, out, sz); } - WC_AES_ARM64_SVR_END(); + WC_AES_GCM_C_SVR_END(); } } @@ -14974,10 +15273,10 @@ int wc_AesGcmEncryptFinal(Aes* aes, byte* authTag, word32 authTagSz) if (ret == 0) { /* Calculate authentication tag. */ - #ifdef WOLFSSL_AESNI + #ifdef WC_AESNI_GCM if (aes->use_aesni) { SAVE_VECTOR_REGISTERS(return _svr_ret;); - ret = AesGcmEncryptFinal_aesni(aes, authTag, authTagSz); + ret = AesGcmEncryptFinal_asm(aes, authTag, authTagSz); RESTORE_VECTOR_REGISTERS(); } else @@ -14994,9 +15293,9 @@ int wc_AesGcmEncryptFinal(Aes* aes, byte* authTag, word32 authTagSz) if (0) #endif { - WC_AES_ARM64_SVR_BEGIN(); + WC_AES_GCM_C_SVR_BEGIN(); ret = AesGcmFinal_C(aes, authTag, authTagSz); - WC_AES_ARM64_SVR_END(); + WC_AES_GCM_C_SVR_END(); } } @@ -15070,10 +15369,10 @@ int wc_AesGcmDecryptUpdate(Aes* aes, byte* out, const byte* in, word32 sz, if (ret == 0) { /* Decrypt with AAD and/or cipher text. */ - #ifdef WOLFSSL_AESNI + #ifdef WC_AESNI_GCM if (aes->use_aesni) { SAVE_VECTOR_REGISTERS(return _svr_ret;); - ret = AesGcmDecryptUpdate_aesni(aes, out, in, sz, authIn, authInSz); + ret = AesGcmDecryptUpdate_asm(aes, out, in, sz, authIn, authInSz); RESTORE_VECTOR_REGISTERS(); } else @@ -15091,13 +15390,13 @@ int wc_AesGcmDecryptUpdate(Aes* aes, byte* out, const byte* in, word32 sz, if (0) #endif { + WC_AES_GCM_C_SVR_BEGIN(); /* Update the authentication tag with any authentication data and * cipher text. */ - WC_AES_ARM64_SVR_BEGIN(); GHASH_UPDATE(aes, authIn, authInSz, in, sz); /* Decrypt the cipher text. */ ret = AesGcmCryptUpdate_C(aes, out, in, sz); - WC_AES_ARM64_SVR_END(); + WC_AES_GCM_C_SVR_END(); } } @@ -15137,10 +15436,10 @@ int wc_AesGcmDecryptFinal(Aes* aes, const byte* authTag, word32 authTagSz) if (ret == 0) { /* Calculate authentication tag and compare with one passed in.. */ - #ifdef WOLFSSL_AESNI + #ifdef WC_AESNI_GCM if (aes->use_aesni) { SAVE_VECTOR_REGISTERS(return _svr_ret;); - ret = AesGcmDecryptFinal_aesni(aes, authTag, authTagSz); + ret = AesGcmDecryptFinal_asm(aes, authTag, authTagSz); RESTORE_VECTOR_REGISTERS(); } else @@ -15158,10 +15457,10 @@ int wc_AesGcmDecryptFinal(Aes* aes, const byte* authTag, word32 authTagSz) #endif { ALIGN32 byte calcTag[WC_AES_BLOCK_SIZE]; + WC_AES_GCM_C_SVR_BEGIN(); /* Calculate authentication tag. */ - WC_AES_ARM64_SVR_BEGIN(); ret = AesGcmFinal_C(aes, calcTag, WC_AES_BLOCK_SIZE); - WC_AES_ARM64_SVR_END(); + WC_AES_GCM_C_SVR_END(); if (ret == 0) { /* Check calculated tag matches the one passed in. */ if (ConstantCompare(authTag, calcTag, (int)authTagSz) != 0) { @@ -15171,6 +15470,8 @@ int wc_AesGcmDecryptFinal(Aes* aes, const byte* authTag, word32 authTagSz) } } + /* Final cannot see earlier Update output; on AES_GCM_AUTH_E the caller + * must discard it (Security Policy operational rules). */ return ret; } #endif /* HAVE_AES_DECRYPT || HAVE_AESGCM_DECRYPT */ @@ -15576,7 +15877,8 @@ int wc_AesCcmDecrypt(Aes* aes, byte* out, const byte* in, word32 inSz, wolfSSL_CryptHwMutexUnLock(); if (status != kStatus_Success) { - XMEMSET(out, 0, inSz); + /* SP 800-38C section 6.2: do not reveal the payload on a failed tag. */ + ForceZero(out, inSz); return AES_CCM_AUTH_E; } return 0; @@ -16061,15 +16363,14 @@ int wc_AesCcmDecrypt(Aes* aes, byte* out, const byte* in, word32 inSz, if (ret == 0) { if (ConstantCompare(A, authTag, (int)authTagSz) != 0) { - /* If the authTag check fails, don't keep the decrypted data. - * Unfortunately, you need the decrypted data to calculate the - * check value. */ + /* SP 800-38C section 6.2: do not reveal the payload on a failed + * tag. */ #if defined(HAVE_FIPS_VERSION) && (HAVE_FIPS_VERSION >= 2) && \ defined(ACVP_VECTOR_TESTING) WOLFSSL_MSG("Preserve output for vector responses"); #else if (inSz > 0) - XMEMSET(out, 0, inSz); + ForceZero(out, inSz); #endif ret = AES_CCM_AUTH_E; } @@ -16679,7 +16980,7 @@ static WARN_UNUSED_RESULT int _AesEcbEncrypt( #elif !defined(__aarch64__) && defined(WOLFSSL_ARMASM) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_encrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, + ret = AES_encrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, (int)aes->rounds); } else { @@ -16687,7 +16988,10 @@ static WARN_UNUSED_RESULT int _AesEcbEncrypt( aes->rounds); } #elif !defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) - AES_encrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, (int)aes->rounds); + { + ret = AES_encrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, + (int)aes->rounds); + } #else AES_ECB_encrypt(in, out, sz, (const unsigned char*)aes->key, aes->rounds); #endif @@ -16795,7 +17099,7 @@ static WARN_UNUSED_RESULT int _AesEcbDecrypt( #elif !defined(__aarch64__) && defined(WOLFSSL_ARMASM) #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_decrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, + ret = AES_decrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, (int)aes->rounds); } else { @@ -16803,7 +17107,10 @@ static WARN_UNUSED_RESULT int _AesEcbDecrypt( aes->rounds); } #elif !defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) - AES_decrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, (int)aes->rounds); + { + ret = AES_decrypt_blocks_AARCH32(in, out, sz, (byte*)aes->key, + (int)aes->rounds); + } #else AES_ECB_decrypt(in, out, sz, (const unsigned char*)aes->key, aes->rounds); #endif @@ -17943,17 +18250,24 @@ int wc_AesKeyUnWrap_ex(Aes *aes, const byte* in, word32 inSz, byte* out, ret = AesKeyUnWrapRaw(aes, in, inSz, out, a); if (ret != 0) { + /* out and a may hold decrypted blocks; wipe them (ISO/IEC 19790 7.9). */ + ForceZero(out, inSz - KEYWRAP_BLOCK_SIZE); + ForceZero(a, sizeof(a)); return ret; } /* verify IV */ if (ConstantCompare(a, expIv, KEYWRAP_BLOCK_SIZE) != 0) { - /* IV check failed: wipe the recovered plaintext key material left in - * out before returning so it is not leaked to the caller */ + /* Wipe the unauthenticated key material in out and a + * (ISO/IEC 19790 7.9). */ ForceZero(out, inSz - KEYWRAP_BLOCK_SIZE); + ForceZero(a, sizeof(a)); return BAD_KEYWRAP_IV_E; } + /* Wipe the last decrypted block on the success path too. */ + ForceZero(a, sizeof(a)); + return (int)(inSz - KEYWRAP_BLOCK_SIZE); } @@ -18180,6 +18494,9 @@ int wc_AesKeyUnWrap_Pad_ex(Aes* aes, const byte* in, word32 inSz, byte* out, ret = AesKeyUnWrapRaw(aes, in, inSz, out, a); } if (ret != 0) { + /* out and a may hold decrypted blocks; wipe them (ISO/IEC 19790 7.9). */ + ForceZero(out, inSz - KEYWRAP_BLOCK_SIZE); + ForceZero(a, sizeof(a)); return ret; } @@ -18246,12 +18563,18 @@ int wc_AesKeyUnWrap_Pad_ex(Aes* aes, const byte* in, word32 inSz, byte* out, } } + /* Wipe the AIV scratch on the success path too (ISO/IEC 19790 7.9). */ + ForceZero(a, sizeof(a)); + ForceZero(expConst, sizeof(expConst)); + return (int)mli; badIv: /* integrity check failed: wipe the recovered plaintext in out so it is - * not leaked to the caller */ + * not leaked to the caller, and the AIV scratch with it */ ForceZero(out, inSz - KEYWRAP_BLOCK_SIZE); + ForceZero(a, sizeof(a)); + ForceZero(expConst, sizeof(expConst)); return BAD_KEYWRAP_IV_E; } @@ -18934,10 +19257,7 @@ int wc_AesXtsEncrypt(XtsAes* xaes, byte* out, const byte* in, word32 sz, } #if FIPS_VERSION3_GE(6,0,0) - /* SP800-38E - Restrict data unit to 2^20 blocks per key. A block is - * WC_AES_BLOCK_SIZE or 16-bytes (128-bits). So each key may only be used to - * protect up to 1,048,576 blocks of WC_AES_BLOCK_SIZE (16,777,216 bytes) - */ + /* SP 800-38E section 4: a data unit is at most 2^20 AES blocks. */ if (sz > FIPS_AES_XTS_MAX_BYTES_PER_TWEAK) { WOLFSSL_MSG("Request exceeds allowed bytes per SP800-38E"); return BAD_FUNC_ARG; @@ -18973,16 +19293,26 @@ int wc_AesXtsEncrypt(XtsAes* xaes, byte* out, const byte* in, word32 sz, * wc_AesEncrypt). */ #ifdef WOLFSSL_ARM32_AES_DISPATCH if (xaes->aes.use_aes_hw_crypto) { - AES_XTS_encrypt_AARCH32(in, out, sz, i, (byte*)xaes->aes.key, - (byte*)xaes->tweak.key, (byte*)xaes->aes.tmp, xaes->aes.rounds); + int _svr_ret = AES_XTS_encrypt_AARCH32(in, out, sz, i, + (byte*)xaes->aes.key, (byte*)xaes->tweak.key, + (byte*)xaes->aes.tmp, xaes->aes.rounds); + if (_svr_ret != 0) { + return _svr_ret; + } ret = 0; } else { ret = AesXtsEncrypt_sw(xaes, out, in, sz, i); } #else - AES_XTS_encrypt_AARCH32(in, out, sz, i, (byte*)xaes->aes.key, - (byte*)xaes->tweak.key, (byte*)xaes->aes.tmp, xaes->aes.rounds); + { + int _svr_ret = AES_XTS_encrypt_AARCH32(in, out, sz, i, + (byte*)xaes->aes.key, (byte*)xaes->tweak.key, (byte*)xaes->aes.tmp, + xaes->aes.rounds); + if (_svr_ret != 0) { + return _svr_ret; + } + } ret = 0; #endif #elif defined(WOLFSSL_AESNI) @@ -19227,10 +19557,7 @@ static int AesXtsEncryptUpdate(XtsAes* xaes, byte* out, const byte* in, word32 s } #endif #if FIPS_VERSION3_GE(6,0,0) - /* SP800-38E - Restrict data unit to 2^20 blocks per key. A block is - * WC_AES_BLOCK_SIZE or 16-bytes (128-bits). So each key may only be used to - * protect up to 1,048,576 blocks of WC_AES_BLOCK_SIZE (16,777,216 bytes) - */ + /* SP 800-38E section 4: a data unit is at most 2^20 AES blocks. */ if (newTweakBytes > FIPS_AES_XTS_MAX_BYTES_PER_TWEAK) { WOLFSSL_MSG("Request exceeds allowed bytes per SP800-38E"); @@ -19554,10 +19881,7 @@ int wc_AesXtsDecrypt(XtsAes* xaes, byte* out, const byte* in, word32 sz, #endif #if FIPS_VERSION3_GE(6,0,0) - /* SP800-38E - Restrict data unit to 2^20 blocks per key. A block is - * WC_AES_BLOCK_SIZE or 16-bytes (128-bits). So each key may only be used to - * protect up to 1,048,576 blocks of WC_AES_BLOCK_SIZE (16,777,216 bytes) - */ + /* SP 800-38E section 4: a data unit is at most 2^20 AES blocks. */ if (sz > FIPS_AES_XTS_MAX_BYTES_PER_TWEAK) { WOLFSSL_MSG("Request exceeds allowed bytes per SP800-38E"); return BAD_FUNC_ARG; @@ -19594,17 +19918,15 @@ int wc_AesXtsDecrypt(XtsAes* xaes, byte* out, const byte* in, word32 sz, * wc_AesDecrypt). */ #ifdef WOLFSSL_ARM32_AES_DISPATCH if (aes->use_aes_hw_crypto) { - AES_XTS_decrypt_AARCH32(in, out, sz, i, (byte*)aes->key, + ret = AES_XTS_decrypt_AARCH32(in, out, sz, i, (byte*)aes->key, (byte*)xaes->tweak.key, (byte*)aes->tmp, aes->rounds); - ret = 0; } else { ret = AesXtsDecrypt_sw(xaes, out, in, sz, i); } #else - AES_XTS_decrypt_AARCH32(in, out, sz, i, (byte*)aes->key, + ret = AES_XTS_decrypt_AARCH32(in, out, sz, i, (byte*)aes->key, (byte*)xaes->tweak.key, (byte*)aes->tmp, aes->rounds); - ret = 0; #endif #elif defined(WOLFSSL_AESNI) if (aes->use_aesni) { @@ -19854,10 +20176,7 @@ static int AesXtsDecryptUpdate(XtsAes* xaes, byte* out, const byte* in, word32 s } #endif #if FIPS_VERSION3_GE(6,0,0) - /* SP800-38E - Restrict data unit to 2^20 blocks per key. A block is - * WC_AES_BLOCK_SIZE or 16-bytes (128-bits). So each key may only be used to - * protect up to 1,048,576 blocks of WC_AES_BLOCK_SIZE (16,777,216 bytes) - */ + /* SP 800-38E section 4: a data unit is at most 2^20 AES blocks. */ if (newTweakBytes > FIPS_AES_XTS_MAX_BYTES_PER_TWEAK) { WOLFSSL_MSG("Request exceeds allowed bytes per SP800-38E"); @@ -19982,6 +20301,7 @@ int wc_AesXtsDecryptFinal(XtsAes* xaes, byte* out, const byte* in, word32 sz, * * returns 0 on success */ +/* Each sector is its own data unit; wc_AesXtsEncrypt() applies the limit. */ int wc_AesXtsEncryptConsecutiveSectors(XtsAes* aes, byte* out, const byte* in, word32 sz, word64 sector, word32 sectorSz) { @@ -20033,6 +20353,7 @@ int wc_AesXtsEncryptConsecutiveSectors(XtsAes* aes, byte* out, const byte* in, * * returns 0 on success */ +/* Per-sector data unit, as on the encrypt path above. */ int wc_AesXtsDecryptConsecutiveSectors(XtsAes* aes, byte* out, const byte* in, word32 sz, word64 sector, word32 sectorSz) { @@ -20110,6 +20431,12 @@ int wc_local_CmacUpdateAes(struct Cmac *cmac, const byte* in, word32 inSz) { #endif /* WOLFSSL_CMAC */ +/* AES-SIV and AES-EAX are not Approved; keep them out of the FIPS image. */ +#if FIPS_VERSION3_GE(7,0,0) && \ + (defined(WOLFSSL_AES_SIV) || defined(WOLFSSL_AES_EAX)) + #error "WOLFSSL_AES_SIV and WOLFSSL_AES_EAX are not in the FIPS module boundary." +#endif + #ifdef WOLFSSL_AES_SIV /* diff --git a/wolfcrypt/src/ge_operations.c b/wolfcrypt/src/ge_operations.c index c4e72b32ea7..fda375ba2e3 100644 --- a/wolfcrypt/src/ge_operations.c +++ b/wolfcrypt/src/ge_operations.c @@ -10287,10 +10287,12 @@ void ge_tobytes_nct(unsigned char *s,const ge_p2 *h) /* if HAVE_ED25519 but not HAVE_CURVE25519, and an asm implementation is built, * then curve25519() won't get its WOLFSSL_LOCAL attribute unless we dummy-call - * it here. + * it here. The 32-bit Arm asm only emits curve25519() under HAVE_CURVE25519, + * so there the call would be an undefined symbol. */ #if defined(CURVED25519_ASM) && defined(WOLFSSL_API_PREFIX_MAP) && \ - !defined(HAVE_CURVE25519) && !defined(FREESCALE_LTC_ECC) + !defined(HAVE_CURVE25519) && !defined(FREESCALE_LTC_ECC) && \ + (!defined(WOLFSSL_ARMASM) || defined(__aarch64__)) WOLFSSL_LOCAL void _wc_curve25519_dummy(void); WOLFSSL_LOCAL void _wc_curve25519_dummy(void) { (void)curve25519((byte *)0, (byte *)0, (const byte *)0); diff --git a/wolfcrypt/src/include.am b/wolfcrypt/src/include.am index 6bc3bb40cee..4aede8d7d75 100644 --- a/wolfcrypt/src/include.am +++ b/wolfcrypt/src/include.am @@ -20,6 +20,7 @@ EXTRA_DIST += wolfcrypt/src/aes_asm.asm EXTRA_DIST += wolfcrypt/src/aes_x86_64_asm.asm EXTRA_DIST += wolfcrypt/src/aes_gcm_asm.asm EXTRA_DIST += wolfcrypt/src/aes_gcm_x86_asm.asm +EXTRA_DIST += wolfcrypt/src/aes_gcm_x86_asm.S EXTRA_DIST += wolfcrypt/src/aes_xts_asm.asm EXTRA_DIST += wolfcrypt/src/aes_xts_x86_asm.asm EXTRA_DIST += wolfcrypt/src/chacha_asm.asm diff --git a/wolfcrypt/src/port/arm/armv8-32-aes-asm.S b/wolfcrypt/src/port/arm/armv8-32-aes-asm.S index 83f00a5ae6f..30b52c11754 100644 --- a/wolfcrypt/src/port/arm/armv8-32-aes-asm.S +++ b/wolfcrypt/src/port/arm/armv8-32-aes-asm.S @@ -23435,7 +23435,8 @@ AES_encrypt_blocks_AARCH32: L_aes_encrypt_blocks_arm32_crypto_192_start_4: cmp r2, #4 blt L_aes_encrypt_blocks_arm32_crypto_192_start_2 - vldm.8 r0!, {q12-q15} + vld1.8 {q12-q13}, [r0]! + vld1.8 {q14-q15}, [r0]! aese.8 q12, q0 aesmc.8 q12, q12 aese.8 q13, q0 @@ -23537,7 +23538,8 @@ L_aes_encrypt_blocks_arm32_crypto_192_start_4: veor.32 q15, q15, q10 sub r3, r3, #48 sub r2, r2, #4 - vstm.8 r1!, {q12-q15} + vst1.8 {q12-q13}, [r1]! + vst1.8 {q14-q15}, [r1]! cmp r2, #4 bge L_aes_encrypt_blocks_arm32_crypto_192_start_4 L_aes_encrypt_blocks_arm32_crypto_192_start_2: @@ -23643,7 +23645,8 @@ L_aes_encrypt_blocks_arm32_crypto_start_256: L_aes_encrypt_blocks_arm32_crypto_256_start_4: cmp r2, #4 blt L_aes_encrypt_blocks_arm32_crypto_256_start_2 - vldm.8 r0!, {q12-q15} + vld1.8 {q12-q13}, [r0]! + vld1.8 {q14-q15}, [r0]! aese.8 q12, q0 aesmc.8 q12, q12 aese.8 q13, q0 @@ -23763,7 +23766,8 @@ L_aes_encrypt_blocks_arm32_crypto_256_start_4: veor.32 q15, q15, q10 sub r3, r3, #0x50 sub r2, r2, #4 - vstm.8 r1!, {q12-q15} + vst1.8 {q12-q13}, [r1]! + vst1.8 {q14-q15}, [r1]! cmp r2, #4 bge L_aes_encrypt_blocks_arm32_crypto_256_start_4 L_aes_encrypt_blocks_arm32_crypto_256_start_2: @@ -23885,7 +23889,8 @@ L_aes_encrypt_blocks_arm32_crypto_start_128: L_aes_encrypt_blocks_arm32_crypto_128_start_4: cmp r2, #4 blt L_aes_encrypt_blocks_arm32_crypto_128_start_2 - vldm.8 r0!, {q12-q15} + vld1.8 {q12-q13}, [r0]! + vld1.8 {q14-q15}, [r0]! aese.8 q12, q0 aesmc.8 q12, q12 aese.8 q13, q0 @@ -23967,7 +23972,8 @@ L_aes_encrypt_blocks_arm32_crypto_128_start_4: aese.8 q15, q9 veor.32 q15, q15, q10 sub r2, r2, #4 - vstm.8 r1!, {q12-q15} + vst1.8 {q12-q13}, [r1]! + vst1.8 {q14-q15}, [r1]! cmp r2, #4 bge L_aes_encrypt_blocks_arm32_crypto_128_start_4 L_aes_encrypt_blocks_arm32_crypto_128_start_2: @@ -24070,7 +24076,8 @@ AES_decrypt_blocks_AARCH32: cmp r2, #4 blt L_aes_decrypt_blocks_arm32_crypto_192_start_2 L_aes_decrypt_blocks_arm32_crypto_192_start_4: - vldm.8 r0!, {q12-q15} + vld1.8 {q12-q13}, [r0]! + vld1.8 {q14-q15}, [r0]! aesd.8 q12, q0 aesimc.8 q12, q12 aesd.8 q13, q0 @@ -24172,7 +24179,8 @@ L_aes_decrypt_blocks_arm32_crypto_192_start_4: veor.32 q15, q15, q10 sub r3, r3, #48 sub r2, r2, #4 - vstm.8 r1!, {q12-q15} + vst1.8 {q12-q13}, [r1]! + vst1.8 {q14-q15}, [r1]! cmp r2, #4 bge L_aes_decrypt_blocks_arm32_crypto_192_start_4 L_aes_decrypt_blocks_arm32_crypto_192_start_2: @@ -24278,7 +24286,8 @@ L_aes_decrypt_blocks_arm32_crypto_start_256: cmp r2, #4 blt L_aes_decrypt_blocks_arm32_crypto_256_start_2 L_aes_decrypt_blocks_arm32_crypto_256_start_4: - vldm.8 r0!, {q12-q15} + vld1.8 {q12-q13}, [r0]! + vld1.8 {q14-q15}, [r0]! aesd.8 q12, q0 aesimc.8 q12, q12 aesd.8 q13, q0 @@ -24398,7 +24407,8 @@ L_aes_decrypt_blocks_arm32_crypto_256_start_4: veor.32 q15, q15, q10 sub r3, r3, #0x50 sub r2, r2, #4 - vstm.8 r1!, {q12-q15} + vst1.8 {q12-q13}, [r1]! + vst1.8 {q14-q15}, [r1]! cmp r2, #4 bge L_aes_decrypt_blocks_arm32_crypto_256_start_4 L_aes_decrypt_blocks_arm32_crypto_256_start_2: @@ -24520,7 +24530,8 @@ L_aes_decrypt_blocks_arm32_crypto_start_128: cmp r2, #4 blt L_aes_decrypt_blocks_arm32_crypto_128_start_2 L_aes_decrypt_blocks_arm32_crypto_128_start_4: - vldm.8 r0!, {q12-q15} + vld1.8 {q12-q13}, [r0]! + vld1.8 {q14-q15}, [r0]! aesd.8 q12, q0 aesimc.8 q12, q12 aesd.8 q13, q0 @@ -24602,7 +24613,8 @@ L_aes_decrypt_blocks_arm32_crypto_128_start_4: aesd.8 q15, q9 veor.32 q15, q15, q10 sub r2, r2, #4 - vstm.8 r1!, {q12-q15} + vst1.8 {q12-q13}, [r1]! + vst1.8 {q14-q15}, [r1]! cmp r2, #4 bge L_aes_decrypt_blocks_arm32_crypto_128_start_4 L_aes_decrypt_blocks_arm32_crypto_128_start_2: diff --git a/wolfcrypt/src/port/arm/armv8-32-aes-asm_c.c b/wolfcrypt/src/port/arm/armv8-32-aes-asm_c.c index da717f8837d..70587149417 100644 --- a/wolfcrypt/src/port/arm/armv8-32-aes-asm_c.c +++ b/wolfcrypt/src/port/arm/armv8-32-aes-asm_c.c @@ -23853,7 +23853,8 @@ WC_OMIT_FRAME_POINTER void AES_encrypt_blocks_AARCH32(const byte* in, byte* out, "L_aes_encrypt_blocks_arm32_crypto_192_start_4_%=:\n\t" "cmp %[sz], #4\n\t" "blt L_aes_encrypt_blocks_arm32_crypto_192_start_2_%=\n\t" - "vldm.8 %[in]!, {q12-q15}\n\t" + "vld1.8 {q12-q13}, [%[in]]!\n\t" + "vld1.8 {q14-q15}, [%[in]]!\n\t" "aese.8 q12, q0\n\t" "aesmc.8 q12, q12\n\t" "aese.8 q13, q0\n\t" @@ -23955,7 +23956,8 @@ WC_OMIT_FRAME_POINTER void AES_encrypt_blocks_AARCH32(const byte* in, byte* out, "veor.32 q15, q15, q10\n\t" "sub %[key], %[key], #48\n\t" "sub %[sz], %[sz], #4\n\t" - "vstm.8 %[out]!, {q12-q15}\n\t" + "vst1.8 {q12-q13}, [%[out]]!\n\t" + "vst1.8 {q14-q15}, [%[out]]!\n\t" "cmp %[sz], #4\n\t" "bge L_aes_encrypt_blocks_arm32_crypto_192_start_4_%=\n\t" "\n" @@ -24066,7 +24068,8 @@ WC_OMIT_FRAME_POINTER void AES_encrypt_blocks_AARCH32(const byte* in, byte* out, "L_aes_encrypt_blocks_arm32_crypto_256_start_4_%=:\n\t" "cmp %[sz], #4\n\t" "blt L_aes_encrypt_blocks_arm32_crypto_256_start_2_%=\n\t" - "vldm.8 %[in]!, {q12-q15}\n\t" + "vld1.8 {q12-q13}, [%[in]]!\n\t" + "vld1.8 {q14-q15}, [%[in]]!\n\t" "aese.8 q12, q0\n\t" "aesmc.8 q12, q12\n\t" "aese.8 q13, q0\n\t" @@ -24186,7 +24189,8 @@ WC_OMIT_FRAME_POINTER void AES_encrypt_blocks_AARCH32(const byte* in, byte* out, "veor.32 q15, q15, q10\n\t" "sub %[key], %[key], #0x50\n\t" "sub %[sz], %[sz], #4\n\t" - "vstm.8 %[out]!, {q12-q15}\n\t" + "vst1.8 {q12-q13}, [%[out]]!\n\t" + "vst1.8 {q14-q15}, [%[out]]!\n\t" "cmp %[sz], #4\n\t" "bge L_aes_encrypt_blocks_arm32_crypto_256_start_4_%=\n\t" "\n" @@ -24313,7 +24317,8 @@ WC_OMIT_FRAME_POINTER void AES_encrypt_blocks_AARCH32(const byte* in, byte* out, "L_aes_encrypt_blocks_arm32_crypto_128_start_4_%=:\n\t" "cmp %[sz], #4\n\t" "blt L_aes_encrypt_blocks_arm32_crypto_128_start_2_%=\n\t" - "vldm.8 %[in]!, {q12-q15}\n\t" + "vld1.8 {q12-q13}, [%[in]]!\n\t" + "vld1.8 {q14-q15}, [%[in]]!\n\t" "aese.8 q12, q0\n\t" "aesmc.8 q12, q12\n\t" "aese.8 q13, q0\n\t" @@ -24395,7 +24400,8 @@ WC_OMIT_FRAME_POINTER void AES_encrypt_blocks_AARCH32(const byte* in, byte* out, "aese.8 q15, q9\n\t" "veor.32 q15, q15, q10\n\t" "sub %[sz], %[sz], #4\n\t" - "vstm.8 %[out]!, {q12-q15}\n\t" + "vst1.8 {q12-q13}, [%[out]]!\n\t" + "vst1.8 {q14-q15}, [%[out]]!\n\t" "cmp %[sz], #4\n\t" "bge L_aes_encrypt_blocks_arm32_crypto_128_start_4_%=\n\t" "\n" @@ -24525,7 +24531,8 @@ WC_OMIT_FRAME_POINTER void AES_decrypt_blocks_AARCH32(const byte* in, byte* out, "blt L_aes_decrypt_blocks_arm32_crypto_192_start_2_%=\n\t" "\n" "L_aes_decrypt_blocks_arm32_crypto_192_start_4_%=:\n\t" - "vldm.8 %[in]!, {q12-q15}\n\t" + "vld1.8 {q12-q13}, [%[in]]!\n\t" + "vld1.8 {q14-q15}, [%[in]]!\n\t" "aesd.8 q12, q0\n\t" "aesimc.8 q12, q12\n\t" "aesd.8 q13, q0\n\t" @@ -24627,7 +24634,8 @@ WC_OMIT_FRAME_POINTER void AES_decrypt_blocks_AARCH32(const byte* in, byte* out, "veor.32 q15, q15, q10\n\t" "sub %[key], %[key], #48\n\t" "sub %[sz], %[sz], #4\n\t" - "vstm.8 %[out]!, {q12-q15}\n\t" + "vst1.8 {q12-q13}, [%[out]]!\n\t" + "vst1.8 {q14-q15}, [%[out]]!\n\t" "cmp %[sz], #4\n\t" "bge L_aes_decrypt_blocks_arm32_crypto_192_start_4_%=\n\t" "\n" @@ -24738,7 +24746,8 @@ WC_OMIT_FRAME_POINTER void AES_decrypt_blocks_AARCH32(const byte* in, byte* out, "blt L_aes_decrypt_blocks_arm32_crypto_256_start_2_%=\n\t" "\n" "L_aes_decrypt_blocks_arm32_crypto_256_start_4_%=:\n\t" - "vldm.8 %[in]!, {q12-q15}\n\t" + "vld1.8 {q12-q13}, [%[in]]!\n\t" + "vld1.8 {q14-q15}, [%[in]]!\n\t" "aesd.8 q12, q0\n\t" "aesimc.8 q12, q12\n\t" "aesd.8 q13, q0\n\t" @@ -24858,7 +24867,8 @@ WC_OMIT_FRAME_POINTER void AES_decrypt_blocks_AARCH32(const byte* in, byte* out, "veor.32 q15, q15, q10\n\t" "sub %[key], %[key], #0x50\n\t" "sub %[sz], %[sz], #4\n\t" - "vstm.8 %[out]!, {q12-q15}\n\t" + "vst1.8 {q12-q13}, [%[out]]!\n\t" + "vst1.8 {q14-q15}, [%[out]]!\n\t" "cmp %[sz], #4\n\t" "bge L_aes_decrypt_blocks_arm32_crypto_256_start_4_%=\n\t" "\n" @@ -24985,7 +24995,8 @@ WC_OMIT_FRAME_POINTER void AES_decrypt_blocks_AARCH32(const byte* in, byte* out, "blt L_aes_decrypt_blocks_arm32_crypto_128_start_2_%=\n\t" "\n" "L_aes_decrypt_blocks_arm32_crypto_128_start_4_%=:\n\t" - "vldm.8 %[in]!, {q12-q15}\n\t" + "vld1.8 {q12-q13}, [%[in]]!\n\t" + "vld1.8 {q14-q15}, [%[in]]!\n\t" "aesd.8 q12, q0\n\t" "aesimc.8 q12, q12\n\t" "aesd.8 q13, q0\n\t" @@ -25067,7 +25078,8 @@ WC_OMIT_FRAME_POINTER void AES_decrypt_blocks_AARCH32(const byte* in, byte* out, "aesd.8 q15, q9\n\t" "veor.32 q15, q15, q10\n\t" "sub %[sz], %[sz], #4\n\t" - "vstm.8 %[out]!, {q12-q15}\n\t" + "vst1.8 {q12-q13}, [%[out]]!\n\t" + "vst1.8 {q14-q15}, [%[out]]!\n\t" "cmp %[sz], #4\n\t" "bge L_aes_decrypt_blocks_arm32_crypto_128_start_4_%=\n\t" "\n" diff --git a/wolfcrypt/src/port/arm/armv8-32-sha3-asm.S b/wolfcrypt/src/port/arm/armv8-32-sha3-asm.S index c5be47e5692..717179939f2 100644 --- a/wolfcrypt/src/port/arm/armv8-32-sha3-asm.S +++ b/wolfcrypt/src/port/arm/armv8-32-sha3-asm.S @@ -40,7 +40,7 @@ #if !defined(__aarch64__) && !defined(WOLFSSL_ARMASM_THUMB2) #ifndef WOLFSSL_ARMASM_INLINE #ifdef WOLFSSL_SHA3 -#ifndef WOLFSSL_ARMASM_NO_NEON +#ifndef WOLFSSL_ARMASM_SHA3_NO_NEON #ifndef __APPLE__ .text .type L_sha3_arm32_neon_rt, %object @@ -334,8 +334,8 @@ L_sha3_arm32_neon_begin: vpop {d8-d15} bx lr .size BlockSha3,.-BlockSha3 -#endif /* WOLFSSL_ARMASM_NO_NEON */ -#ifdef WOLFSSL_ARMASM_NO_NEON +#endif /* WOLFSSL_ARMASM_SHA3_NO_NEON */ +#ifdef WOLFSSL_ARMASM_SHA3_NO_NEON #ifndef __APPLE__ .text .type L_sha3_arm32_rt, %object @@ -2334,7 +2334,7 @@ L_sha3_arm32_begin: add sp, sp, #0xcc pop {r4, r5, r6, r7, r8, r9, r10, r11, pc} .size BlockSha3,.-BlockSha3 -#endif /* WOLFSSL_ARMASM_NO_NEON */ +#endif /* WOLFSSL_ARMASM_SHA3_NO_NEON */ #endif /* WOLFSSL_SHA3 */ #if defined(__linux__) && defined(__ELF__) diff --git a/wolfcrypt/src/port/arm/armv8-32-sha3-asm_c.c b/wolfcrypt/src/port/arm/armv8-32-sha3-asm_c.c index da28f24fb4d..635d5b4e203 100644 --- a/wolfcrypt/src/port/arm/armv8-32-sha3-asm_c.c +++ b/wolfcrypt/src/port/arm/armv8-32-sha3-asm_c.c @@ -57,7 +57,7 @@ #endif /* __ghs__ */ #ifdef WOLFSSL_SHA3 -#ifndef WOLFSSL_ARMASM_NO_NEON +#ifndef WOLFSSL_ARMASM_SHA3_NO_NEON XALIGNED(16) static const word64 L_sha3_arm32_neon_rt[] = { 0x0000000000000001UL, 0x0000000000008082UL, 0x800000000000808aUL, 0x8000000080008000UL, @@ -364,8 +364,8 @@ WC_OMIT_FRAME_POINTER void BlockSha3(word64* state) ); } -#endif /* WOLFSSL_ARMASM_NO_NEON */ -#ifdef WOLFSSL_ARMASM_NO_NEON +#endif /* WOLFSSL_ARMASM_SHA3_NO_NEON */ +#ifdef WOLFSSL_ARMASM_SHA3_NO_NEON XALIGNED(16) static const word64 L_sha3_arm32_rt[] = { 0x0000000000000001UL, 0x0000000000008082UL, 0x800000000000808aUL, 0x8000000080008000UL, @@ -2374,7 +2374,7 @@ WC_OMIT_FRAME_POINTER void BlockSha3(word64* state) ); } -#endif /* WOLFSSL_ARMASM_NO_NEON */ +#endif /* WOLFSSL_ARMASM_SHA3_NO_NEON */ #endif /* WOLFSSL_SHA3 */ #endif /* WOLFSSL_ARMASM_INLINE */ diff --git a/wolfcrypt/src/port/arm/armv8-32-sha512-asm.S b/wolfcrypt/src/port/arm/armv8-32-sha512-asm.S index 8b98d9f4f1d..df9f4d0712b 100644 --- a/wolfcrypt/src/port/arm/armv8-32-sha512-asm.S +++ b/wolfcrypt/src/port/arm/armv8-32-sha512-asm.S @@ -34,7 +34,7 @@ #if !defined(__aarch64__) && !defined(WOLFSSL_ARMASM_THUMB2) #ifndef WOLFSSL_ARMASM_INLINE #if defined(WOLFSSL_SHA512) || defined(WOLFSSL_SHA384) -#ifdef WOLFSSL_ARMASM_NO_NEON +#ifdef WOLFSSL_ARMASM_SHA512_NO_NEON #ifndef __APPLE__ .text .type L_SHA512_transform_len_k, %object @@ -7509,8 +7509,8 @@ L_SHA512_transform_len_start: add sp, sp, #0xc0 pop {r4, r5, r6, r7, r8, r9, r10, r11, pc} .size Transform_Sha512_Len_base,.-Transform_Sha512_Len_base -#endif /* WOLFSSL_ARMASM_NO_NEON */ -#ifndef WOLFSSL_ARMASM_NO_NEON +#endif /* WOLFSSL_ARMASM_SHA512_NO_NEON */ +#ifndef WOLFSSL_ARMASM_SHA512_NO_NEON #ifndef __APPLE__ .text .type L_SHA512_transform_neon_len_k, %object @@ -9063,7 +9063,7 @@ L_SHA512_transform_neon_len_start: vpop {d8-d15} bx lr .size Transform_Sha512_Len_neon,.-Transform_Sha512_Len_neon -#endif /* !WOLFSSL_ARMASM_NO_NEON */ +#endif /* !WOLFSSL_ARMASM_SHA512_NO_NEON */ #endif /* WOLFSSL_SHA512 || WOLFSSL_SHA384 */ #if defined(__linux__) && defined(__ELF__) diff --git a/wolfcrypt/src/port/arm/armv8-32-sha512-asm_c.c b/wolfcrypt/src/port/arm/armv8-32-sha512-asm_c.c index c0ad7515142..8f7e166de62 100644 --- a/wolfcrypt/src/port/arm/armv8-32-sha512-asm_c.c +++ b/wolfcrypt/src/port/arm/armv8-32-sha512-asm_c.c @@ -53,7 +53,7 @@ #if defined(WOLFSSL_SHA512) || defined(WOLFSSL_SHA384) #include -#ifdef WOLFSSL_ARMASM_NO_NEON +#ifdef WOLFSSL_ARMASM_SHA512_NO_NEON XALIGNED(16) static const word64 L_SHA512_transform_len_k[] = { 0x428a2f98d728ae22UL, 0x7137449123ef65cdUL, 0xb5c0fbcfec4d3b2fUL, 0xe9b5dba58189dbbcUL, @@ -7546,10 +7546,10 @@ WC_OMIT_FRAME_POINTER void Transform_Sha512_Len_base(wc_Sha512* sha512, ); } -#endif /* WOLFSSL_ARMASM_NO_NEON */ +#endif /* WOLFSSL_ARMASM_SHA512_NO_NEON */ #include -#ifndef WOLFSSL_ARMASM_NO_NEON +#ifndef WOLFSSL_ARMASM_SHA512_NO_NEON XALIGNED(16) static const word64 L_SHA512_transform_neon_len_k[] = { 0x428a2f98d728ae22UL, 0x7137449123ef65cdUL, 0xb5c0fbcfec4d3b2fUL, 0xe9b5dba58189dbbcUL, @@ -9119,7 +9119,7 @@ WC_OMIT_FRAME_POINTER void Transform_Sha512_Len_neon(wc_Sha512* sha512, ); } -#endif /* !WOLFSSL_ARMASM_NO_NEON */ +#endif /* !WOLFSSL_ARMASM_SHA512_NO_NEON */ #endif /* WOLFSSL_SHA512 || WOLFSSL_SHA384 */ #endif /* WOLFSSL_ARMASM_INLINE */ diff --git a/wolfcrypt/src/sha256.c b/wolfcrypt/src/sha256.c index 5f0510f7c89..12e95b6aef8 100644 --- a/wolfcrypt/src/sha256.c +++ b/wolfcrypt/src/sha256.c @@ -1478,6 +1478,50 @@ int wc_InitSha256_ex(wc_Sha256* sha256, void* heap, int devId) #define SHA256_ARM32_DISPATCH #endif +/* The crypto and NEON transforms run on the q registers, so a kernel module + * must hold the vector registers around them; the base transform needs none. */ +#if !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_NO_NEON) +#ifdef WOLFSSL_USE_SAVE_VECTOR_REGISTERS + #define WC_SHA256_ARM32_SVR_BEGIN() \ + do { int _svr_ret = SAVE_VECTOR_REGISTERS2(); \ + if (_svr_ret != 0) return _svr_ret; } while (0) + #define WC_SHA256_ARM32_SVR_END() RESTORE_VECTOR_REGISTERS() +#else + #define WC_SHA256_ARM32_SVR_BEGIN() WC_DO_NOTHING + #define WC_SHA256_ARM32_SVR_END() WC_DO_NOTHING +#endif +#ifndef WOLFSSL_ARMASM_NO_HW_CRYPTO +static WC_INLINE int Transform_Sha256_Len_crypto_arm32(wc_Sha256* sha256, + const byte* data, word32 len) +{ + WC_SHA256_ARM32_SVR_BEGIN(); + Transform_Sha256_Len_crypto(sha256, data, len); + WC_SHA256_ARM32_SVR_END(); + return 0; +} +#endif +/* Guarded exactly as armv8-32-sha256-asm_c.c guards the assembly it calls, so + * a wrapper exists only where its implementation does. */ +#if !defined(WOLFSSL_ARMASM_NO_NEON_IMPL) || defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) +static WC_INLINE int Transform_Sha256_Len_neon_arm32(wc_Sha256* sha256, + const byte* data, word32 len) +{ + WC_SHA256_ARM32_SVR_BEGIN(); + Transform_Sha256_Len_neon(sha256, data, len); + WC_SHA256_ARM32_SVR_END(); + return 0; +} +#endif +#endif +#if !defined(WOLFSSL_ARMASM_NO_BASE_IMPL) || defined(WOLFSSL_ARMASM_NO_NEON) +static WC_INLINE int Transform_Sha256_Len_base_arm32(wc_Sha256* sha256, + const byte* data, word32 len) +{ + Transform_Sha256_Len_base(sha256, data, len); + return 0; +} +#endif + #ifdef SHA256_ARM32_DISPATCH static int sha256_transform_check = 0; @@ -1487,12 +1531,12 @@ static cpuid_flags_atomic_t sha256_cpuid_flags = WC_CPUID_ATOMIC_INITIALIZER; * requires no extension at all - so the pointer is safe to use even if read * before Sha256_SetTransform() runs. */ #ifndef WOLFSSL_ARMASM_NO_BASE_IMPL - #define SHA256_ARM32_TRANSFORM_INIT Transform_Sha256_Len_base + #define SHA256_ARM32_TRANSFORM_INIT Transform_Sha256_Len_base_arm32 #else - #define SHA256_ARM32_TRANSFORM_INIT Transform_Sha256_Len_neon + #define SHA256_ARM32_TRANSFORM_INIT Transform_Sha256_Len_neon_arm32 #endif -static void (*Transform_Sha256_Len_p)(wc_Sha256* sha256, const byte* data, +static int (*Transform_Sha256_Len_p)(wc_Sha256* sha256, const byte* data, word32 len) = SHA256_ARM32_TRANSFORM_INIT; /* Select the crypto-extension transform when the CPU implements FEAT_SHA256, @@ -1508,25 +1552,25 @@ static void Sha256_SetTransform(void) cpuid_get_flags_atomic(&sha256_cpuid_flags); if (IS_ARM32_SHA256(sha256_cpuid_flags)) { - Transform_Sha256_Len_p = Transform_Sha256_Len_crypto; + Transform_Sha256_Len_p = Transform_Sha256_Len_crypto_arm32; } #if !defined(WOLFSSL_ARMASM_NO_NEON_IMPL) && \ !defined(WOLFSSL_ARMASM_NO_BASE_IMPL) else if (IS_ARM32_ASIMD(sha256_cpuid_flags)) { - Transform_Sha256_Len_p = Transform_Sha256_Len_neon; + Transform_Sha256_Len_p = Transform_Sha256_Len_neon_arm32; } else { - Transform_Sha256_Len_p = Transform_Sha256_Len_base; + Transform_Sha256_Len_p = Transform_Sha256_Len_base_arm32; } #elif !defined(WOLFSSL_ARMASM_NO_NEON_IMPL) /* Base dropped - a NEON build always implements Advanced SIMD. */ else { - Transform_Sha256_Len_p = Transform_Sha256_Len_neon; + Transform_Sha256_Len_p = Transform_Sha256_Len_neon_arm32; } #else /* NEON implementation dropped - the base transform needs no extension. */ else { - Transform_Sha256_Len_p = Transform_Sha256_Len_base; + Transform_Sha256_Len_p = Transform_Sha256_Len_base_arm32; } #endif @@ -1572,15 +1616,15 @@ int wc_InitSha256_ex(wc_Sha256* sha256, void* heap, int devId) static WC_INLINE int Transform_Sha256(wc_Sha256* sha256, const byte* data) { #ifdef SHA256_ARM32_DISPATCH - (*Transform_Sha256_Len_p)(sha256, data, WC_SHA256_BLOCK_SIZE); + return (*Transform_Sha256_Len_p)(sha256, data, WC_SHA256_BLOCK_SIZE); #elif defined(WOLFSSL_ARMASM_THUMB2) || defined(WOLFSSL_ARMASM_NO_NEON) - Transform_Sha256_Len_base(sha256, data, WC_SHA256_BLOCK_SIZE); + return Transform_Sha256_Len_base_arm32(sha256, data, WC_SHA256_BLOCK_SIZE); #elif defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) - Transform_Sha256_Len_neon(sha256, data, WC_SHA256_BLOCK_SIZE); + return Transform_Sha256_Len_neon_arm32(sha256, data, WC_SHA256_BLOCK_SIZE); #else - Transform_Sha256_Len_crypto(sha256, data, WC_SHA256_BLOCK_SIZE); + return Transform_Sha256_Len_crypto_arm32(sha256, data, + WC_SHA256_BLOCK_SIZE); #endif - return 0; } /* Multi-block form of Transform_Sha256() - see there for the selection. */ @@ -1588,15 +1632,14 @@ static WC_INLINE int Transform_Sha256_Len(wc_Sha256* sha256, const byte* data, word32 len) { #ifdef SHA256_ARM32_DISPATCH - (*Transform_Sha256_Len_p)(sha256, data, len); + return (*Transform_Sha256_Len_p)(sha256, data, len); #elif defined(WOLFSSL_ARMASM_THUMB2) || defined(WOLFSSL_ARMASM_NO_NEON) - Transform_Sha256_Len_base(sha256, data, len); + return Transform_Sha256_Len_base_arm32(sha256, data, len); #elif defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) - Transform_Sha256_Len_neon(sha256, data, len); + return Transform_Sha256_Len_neon_arm32(sha256, data, len); #else - Transform_Sha256_Len_crypto(sha256, data, len); + return Transform_Sha256_Len_crypto_arm32(sha256, data, len); #endif - return 0; } #define XTRANSFORM Transform_Sha256 diff --git a/wolfcrypt/src/sha3.c b/wolfcrypt/src/sha3.c index c40c704dab5..0f5c5ae8573 100644 --- a/wolfcrypt/src/sha3.c +++ b/wolfcrypt/src/sha3.c @@ -173,9 +173,9 @@ #endif #if defined(WOLFSSL_ARMASM) && !defined(__aarch64__) && \ - !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_NO_NEON) + !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_SHA3_NO_NEON) /* armv8-32-sha3-asm.S has a NEON block (vpush d8-d15) and an integer-only - * one under WOLFSSL_ARMASM_NO_NEON; only the NEON block needs the save. */ + * one under WOLFSSL_ARMASM_SHA3_NO_NEON; only NEON needs the save. */ #define SHA3_BLOCK_VREGS(f) 1 #define SHA3_NEEDS_VREG_CLAIM #endif diff --git a/wolfcrypt/src/sha512.c b/wolfcrypt/src/sha512.c index bac5fc9ef57..6ff622620b5 100644 --- a/wolfcrypt/src/sha512.c +++ b/wolfcrypt/src/sha512.c @@ -1688,33 +1688,60 @@ static void Sha512_SetTransform(void) static int transform_check = 0; -#if !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_NO_NEON) -static void Transform_Sha512_neon(wc_Sha512* sha512, const byte* data) +#if !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_SHA512_NO_NEON) +/* The 32-bit Arm NEON transform runs on the d registers, so a kernel module + * must hold the vector registers around it. */ +#ifdef WOLFSSL_USE_SAVE_VECTOR_REGISTERS + #define WC_SHA512_ARM32_SVR_BEGIN() \ + do { int _svr_ret = SAVE_VECTOR_REGISTERS2(); \ + if (_svr_ret != 0) return _svr_ret; } while (0) + #define WC_SHA512_ARM32_SVR_END() RESTORE_VECTOR_REGISTERS() +#else + #define WC_SHA512_ARM32_SVR_BEGIN() WC_DO_NOTHING + #define WC_SHA512_ARM32_SVR_END() WC_DO_NOTHING +#endif +static int Transform_Sha512_neon(wc_Sha512* sha512, const byte* data) { + WC_SHA512_ARM32_SVR_BEGIN(); Transform_Sha512_Len_neon(sha512, data, WC_SHA512_BLOCK_SIZE); + WC_SHA512_ARM32_SVR_END(); + return 0; +} +static int Transform_Sha512_Len_neon_arm32(wc_Sha512* sha512, + const byte* data, word32 len) +{ + WC_SHA512_ARM32_SVR_BEGIN(); + Transform_Sha512_Len_neon(sha512, data, len); + WC_SHA512_ARM32_SVR_END(); + return 0; } #endif -#if defined(WOLFSSL_ARMASM_THUMB2) || defined(WOLFSSL_ARMASM_NO_NEON) -static void Transform_Sha512_base(wc_Sha512* sha512, const byte* data) +#if defined(WOLFSSL_ARMASM_THUMB2) || defined(WOLFSSL_ARMASM_SHA512_NO_NEON) +static int Transform_Sha512_base(wc_Sha512* sha512, const byte* data) { Transform_Sha512_Len_base(sha512, data, WC_SHA512_BLOCK_SIZE); + return 0; +} +static int Transform_Sha512_Len_base_arm32(wc_Sha512* sha512, + const byte* data, word32 len) +{ + Transform_Sha512_Len_base(sha512, data, len); + return 0; } #endif -static void (*Transform_Sha512_p)(wc_Sha512* sha512, const byte* data) = NULL; -static void (*Transform_Sha512_Len_p)(wc_Sha512* sha512, const byte* data, +static int (*Transform_Sha512_p)(wc_Sha512* sha512, const byte* data) = NULL; +static int (*Transform_Sha512_Len_p)(wc_Sha512* sha512, const byte* data, word32 len) = NULL; static WC_INLINE int Transform_Sha512(wc_Sha512 *sha512, const byte* data) { - (*Transform_Sha512_p)(sha512, data); - return 0; + return (*Transform_Sha512_p)(sha512, data); } static WC_INLINE int Transform_Sha512_Len(wc_Sha512 *sha512, const byte* data, word32 len) { - (*Transform_Sha512_Len_p)(sha512, data, len); - return 0; + return (*Transform_Sha512_Len_p)(sha512, data, len); } static void Sha512_SetTransform(void) @@ -1722,15 +1749,15 @@ static void Sha512_SetTransform(void) if (transform_check) return; -#if !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_NO_NEON) +#if !defined(WOLFSSL_ARMASM_THUMB2) && !defined(WOLFSSL_ARMASM_SHA512_NO_NEON) { Transform_Sha512_p = Transform_Sha512_neon; - Transform_Sha512_Len_p = Transform_Sha512_Len_neon; + Transform_Sha512_Len_p = Transform_Sha512_Len_neon_arm32; } #else { Transform_Sha512_p = Transform_Sha512_base; - Transform_Sha512_Len_p = Transform_Sha512_Len_base; + Transform_Sha512_Len_p = Transform_Sha512_Len_base_arm32; } #endif diff --git a/wolfcrypt/src/wc_lms_impl.c b/wolfcrypt/src/wc_lms_impl.c index 74535332084..cd77409dbf7 100644 --- a/wolfcrypt/src/wc_lms_impl.c +++ b/wolfcrypt/src/wc_lms_impl.c @@ -3722,7 +3722,7 @@ static int wc_lms_treehash_update(LmsState* state, LmsPrivState* privState, byte* left = dp + LMS_D_LEN; byte* temp = left + params->hash_len; WC_DECLARE_VAR(stack, byte, (LMS_MAX_HEIGHT + 1) * LMS_MAX_NODE_LEN, 0); - byte* sp; + byte* sp = NULL; byte* spEnd; word32 max_cb = (word32)1 << params->cacheBits; word32 i; diff --git a/wolfssl/wolfcrypt/aes.h b/wolfssl/wolfcrypt/aes.h index 6b428203811..dd3d48a3f7e 100644 --- a/wolfssl/wolfcrypt/aes.h +++ b/wolfssl/wolfcrypt/aes.h @@ -546,11 +546,14 @@ struct Aes { #ifdef WOLFSSL_AES_XTS #if FIPS_VERSION3_GE(6,0,0) - /* SP800-38E - Restrict data unit to 2^20 blocks per key. A block is - * WC_AES_BLOCK_SIZE or 16-bytes (128-bits). So each key may only be used to - * protect up to 1,048,576 blocks of WC_AES_BLOCK_SIZE (16,777,216 bytes) - */ + /* SP 800-38E section 4: a data unit (one tweak) is at most 2^20 AES + * blocks, 16,777,216 bytes. Per data unit, not per key. */ #define FIPS_AES_XTS_MAX_BYTES_PER_TWEAK 16777216 + #if defined(WOLFSSL_AESXTS_STREAM) && \ + defined(WC_AESXTS_STREAM_NO_REQUEST_ACCOUNTING) + /* Without the accounting the streaming limit is never enforced. */ + #error "SP 800-38E per-data-unit limit would go unenforced." + #endif #endif struct XtsAes { Aes aes; diff --git a/wolfssl/wolfcrypt/settings.h b/wolfssl/wolfcrypt/settings.h index 85f98891574..11548875d1e 100644 --- a/wolfssl/wolfcrypt/settings.h +++ b/wolfssl/wolfcrypt/settings.h @@ -4925,6 +4925,41 @@ #define WOLFSSL_CURVE25519_BLINDING #endif +#if ((defined(HAVE_FIPS) && FIPS_VERSION3_GE(7,0,0)) || \ + defined(WOLFSSL_FIPS_READY) || defined(WOLFSSL_FIPS_DEV)) && \ + defined(WOLFSSL_ARMASM) && !defined(__aarch64__) && \ + !defined(WOLFSSL_ARMASM_THUMB2) + /* One AES and one SHA-256 per build, nothing picked at run time; the asm + * files still keep the body a no-crypto or no-NEON build needs. */ + #ifndef WOLFSSL_ARMASM_NO_BASE_IMPL + #define WOLFSSL_ARMASM_NO_BASE_IMPL + #endif + #if !defined(WOLFSSL_ARMASM_NO_HW_CRYPTO) && \ + !defined(WOLFSSL_ARMASM_NO_NEON_IMPL) + #define WOLFSSL_ARMASM_NO_NEON_IMPL + #endif + /* The kernel module's DRBG runs SHA-512, and reseeds through wolfEntropy's + * SHA-3, from hardirq, where NEON is never usable. */ + #ifdef WOLFSSL_LINUXKM + #ifndef WOLFSSL_ARMASM_SHA512_NO_NEON + #define WOLFSSL_ARMASM_SHA512_NO_NEON + #endif + #ifndef WOLFSSL_ARMASM_SHA3_NO_NEON + #define WOLFSSL_ARMASM_SHA3_NO_NEON + #endif + #endif +#endif + +/* WOLFSSL_ARMASM_NO_NEON drops every NEON body, SHA-512's and SHA-3's too. */ +#ifdef WOLFSSL_ARMASM_NO_NEON + #ifndef WOLFSSL_ARMASM_SHA512_NO_NEON + #define WOLFSSL_ARMASM_SHA512_NO_NEON + #endif + #ifndef WOLFSSL_ARMASM_SHA3_NO_NEON + #define WOLFSSL_ARMASM_SHA3_NO_NEON + #endif +#endif + /* curve25519/ed25519 implementation selection. * * These are derived here rather than in fe_operations.h because the generated diff --git a/wolfssl/wolfcrypt/sha512.h b/wolfssl/wolfcrypt/sha512.h index 8de7266446c..10494a50cb7 100644 --- a/wolfssl/wolfcrypt/sha512.h +++ b/wolfssl/wolfcrypt/sha512.h @@ -230,7 +230,7 @@ struct wc_Sha512 { #if defined(WOLFSSL_SHA512) || defined(WOLFSSL_SHA384) #ifdef WOLFSSL_ARMASM -#if !defined(WOLFSSL_ARMASM_NO_NEON) +#if !defined(WOLFSSL_ARMASM_SHA512_NO_NEON) WOLFSSL_LOCAL void Transform_Sha512_Len_neon(wc_Sha512* sha512, const byte* data, word32 len); #ifdef WOLFSSL_ARMASM_CRYPTO_SHA512