fix(esp_wifi): Optimize crypto operations for SAE

- Montgomery multiplication fast path for P-256 mulmod
- Jacobi symbol for legendre (replacing exp_mod)
- Software Jacobian point multiplication for MPI-only chips
- ECC hardware acceleration for supported chips
- ECDH fast path for P-256
This commit is contained in:
Kapil Gupta
2026-04-25 17:00:15 +05:30
parent d8116f2fa5
commit 28ef80d57c
4 changed files with 1623 additions and 126 deletions
+12
View File
@@ -630,6 +630,18 @@ menu "Wi-Fi"
help
Select this option to enable WiFi Easy Connect Support.
config ESP_WIFI_P256_ACCEL
bool "Enable P-256 crypto acceleration"
depends on ESP_WIFI_MBEDTLS_CRYPTO
default y
help
Enable Espressif-specific P-256 acceleration in the WPA supplicant
crypto layer. This reduces SAE and DPP latency on supported targets
at the cost of additional code size.
If disabled, the supplicant falls back to the generic Mbed TLS
implementation.
config ESP_WIFI_11R_SUPPORT
bool "Enable 802.11R (Fast Transition) Support"
default n
@@ -1,5 +1,5 @@
/*
* SPDX-FileCopyrightText: 2015-2025 Espressif Systems (Shanghai) CO LTD
* SPDX-FileCopyrightText: 2015-2026 Espressif Systems (Shanghai) CO LTD
*
* SPDX-License-Identifier: Apache-2.0
*/
@@ -16,6 +16,253 @@
#include "random.h"
#include "sha256.h"
#include "mbedtls/pk.h"
#include "p256_common.h"
#if CONFIG_ESP_WIFI_P256_ACCEL
static int mpi_is_secp256r1_prime(const mbedtls_mpi *p)
{
u8 p_be[P256_LEN_BYTES];
if (!p || mbedtls_mpi_size(p) != P256_LEN_BYTES) {
return 0;
}
if (mbedtls_mpi_write_binary(p, p_be, sizeof(p_be)) != 0) {
return 0;
}
return os_memcmp(p_be, p256_p_be, sizeof(p_be)) == 0;
}
static int p256_words_is_one(const u32 *a)
{
size_t i;
if (a[0] != 1) {
return 0;
}
for (i = 1; i < P256_WORDS; i++) {
if (a[i] != 0) {
return 0;
}
}
return 1;
}
static int p256_words_cmp(const u32 *a, const u32 *b)
{
int i;
for (i = P256_WORDS - 1; i >= 0; i--) {
if (a[i] < b[i]) {
return -1;
}
if (a[i] > b[i]) {
return 1;
}
}
return 0;
}
static size_t p256_words_ctz(const u32 *a)
{
size_t i;
for (i = 0; i < P256_WORDS; i++) {
if (a[i] != 0) {
return i * 32 + __builtin_ctz(a[i]);
}
}
return P256_WORDS * 32;
}
static void p256_words_rshift(u32 *a, size_t count)
{
size_t word_shift = count / 32;
size_t bit_shift = count % 32;
size_t i;
if (word_shift >= P256_WORDS) {
os_memset(a, 0, sizeof(u32) * P256_WORDS);
return;
}
if (word_shift > 0) {
for (i = 0; i + word_shift < P256_WORDS; i++) {
a[i] = a[i + word_shift];
}
for (; i < P256_WORDS; i++) {
a[i] = 0;
}
}
if (bit_shift > 0) {
for (i = 0; i < P256_WORDS - 1; i++) {
a[i] = (a[i] >> bit_shift) |
(a[i + 1] << (32 - bit_shift));
}
a[P256_WORDS - 1] >>= bit_shift;
}
}
static void p256_words_lshift(const u32 *in, size_t count, u32 *out)
{
size_t word_shift = count / 32;
size_t bit_shift = count % 32;
size_t i;
os_memset(out, 0, sizeof(u32) * P256_WORDS);
if (word_shift >= P256_WORDS) {
return;
}
for (i = 0; i < P256_WORDS; i++) {
u64 val;
size_t dst;
if (in[i] == 0) {
continue;
}
dst = i + word_shift;
if (dst >= P256_WORDS) {
break;
}
val = (u64) in[i] << bit_shift;
out[dst] |= (u32) val;
if (bit_shift > 0 && dst + 1 < P256_WORDS) {
out[dst + 1] |= (u32)(val >> 32);
}
}
}
static void p256_words_sub(u32 *a, const u32 *b)
{
size_t i;
u64 borrow = 0;
for (i = 0; i < P256_WORDS; i++) {
u64 ai = a[i];
u64 bi = b[i];
u64 res = ai - bi - borrow;
a[i] = (u32) res;
borrow = (ai < bi + borrow) ? 1 : 0;
}
}
static void p256_words_swap(u32 *a, u32 *b)
{
u32 tmp[P256_WORDS];
os_memcpy(tmp, a, sizeof(tmp));
os_memcpy(a, b, sizeof(tmp));
os_memcpy(b, tmp, sizeof(tmp));
}
static void p256_words_mod(u32 *a, const u32 *n)
{
u32 tmp[P256_WORDS];
while (p256_words_cmp(a, n) >= 0) {
size_t a_bits = p256_words_bitlen(a);
size_t n_bits = p256_words_bitlen(n);
size_t shift = a_bits - n_bits;
p256_words_lshift(n, shift, tmp);
if (p256_words_cmp(a, tmp) < 0) {
shift--;
p256_words_lshift(n, shift, tmp);
}
p256_words_sub(a, tmp);
}
}
static int crypto_bignum_mulmod_secp256r1(const mbedtls_mpi *a,
const mbedtls_mpi *b,
const mbedtls_mpi *mod,
mbedtls_mpi *out)
{
u32 a_words[P256_WORDS];
u32 b_words[P256_WORDS];
u32 b_mont[P256_WORDS];
u32 result[P256_WORDS];
if (!mpi_is_secp256r1_prime(mod) ||
p256_words_from_mpi_reduced(a, a_words) != 0 ||
p256_words_from_mpi_reduced(b, b_words) != 0) {
return -2;
}
p256_mont_mul(b_mont, b_words, p256_r2_le);
p256_mont_mul(result, a_words, b_mont);
return p256_words_to_mpi(result, out) == 0 ? 0 : -1;
}
static int crypto_bignum_legendre_secp256r1(const mbedtls_mpi *a,
const mbedtls_mpi *p)
{
u32 A[P256_WORDS];
u32 N[P256_WORDS];
unsigned int n_mod8;
unsigned int a_mod4;
unsigned int n_mod4;
size_t two_power;
int sign = 1;
if (!mpi_is_secp256r1_prime(p) ||
p256_words_from_mpi_reduced(a, A) != 0) {
return -2;
}
os_memcpy(N, p256_p_le, sizeof(N));
if (p256_words_is_zero(A)) {
return 0;
}
while (!p256_words_is_zero(A)) {
if (p256_words_is_one(A)) {
return sign;
}
n_mod8 = N[0] & 0x7;
two_power = p256_words_ctz(A);
if (two_power > 0) {
p256_words_rshift(A, two_power);
if ((n_mod8 == 3 || n_mod8 == 5) && (two_power & 1U)) {
sign = -sign;
}
}
p256_words_swap(A, N);
a_mod4 = A[0] & 0x3;
n_mod4 = N[0] & 0x3;
if (a_mod4 == 3 && n_mod4 == 3) {
sign = -sign;
}
if (p256_words_cmp(A, N) >= 0) {
p256_words_mod(A, N);
}
if (p256_words_is_one(N)) {
return sign;
}
}
return p256_words_is_one(N) ? sign : 0;
}
#endif
struct crypto_bignum *crypto_bignum_init(void)
{
@@ -119,7 +366,43 @@ int crypto_bignum_exptmod(const struct crypto_bignum *a,
const struct crypto_bignum *c,
struct crypto_bignum *d)
{
return mbedtls_mpi_exp_mod((mbedtls_mpi *) d, (const mbedtls_mpi *) a, (const mbedtls_mpi *) b, (const mbedtls_mpi *) c, NULL) ? -1 : 0;
int ret;
/* Fast path for small public exponents frequently used in SAE math. */
if (mbedtls_mpi_cmp_int((const mbedtls_mpi *) b, 0) >= 0 &&
mbedtls_mpi_cmp_int((const mbedtls_mpi *) b, 3) <= 0) {
if (mbedtls_mpi_cmp_int((const mbedtls_mpi *) b, 0) == 0) {
ret = mbedtls_mpi_lset((mbedtls_mpi *) d, 1) ||
mbedtls_mpi_mod_mpi((mbedtls_mpi *) d, (mbedtls_mpi *) d,
(const mbedtls_mpi *) c);
return ret ? -1 : 0;
}
if (mbedtls_mpi_cmp_int((const mbedtls_mpi *) b, 1) == 0) {
ret = mbedtls_mpi_copy((mbedtls_mpi *) d, (const mbedtls_mpi *) a) ||
mbedtls_mpi_mod_mpi((mbedtls_mpi *) d, (mbedtls_mpi *) d,
(const mbedtls_mpi *) c);
return ret ? -1 : 0;
}
if (mbedtls_mpi_cmp_int((const mbedtls_mpi *) b, 2) == 0) {
return crypto_bignum_mulmod(a, a, c, d);
} else {
mbedtls_mpi tmp;
mbedtls_mpi_init(&tmp);
ret = crypto_bignum_mulmod(a, a, c, (struct crypto_bignum *) &tmp);
if (ret == 0) {
ret = crypto_bignum_mulmod((struct crypto_bignum *) &tmp,
a, c, d);
}
mbedtls_mpi_free(&tmp);
return ret ? -1 : 0;
}
}
return mbedtls_mpi_exp_mod((mbedtls_mpi *) d, (const mbedtls_mpi *) a,
(const mbedtls_mpi *) b,
(const mbedtls_mpi *) c, NULL) ? -1 : 0;
}
@@ -152,6 +435,15 @@ int crypto_bignum_mulmod(const struct crypto_bignum *a,
const struct crypto_bignum *c,
struct crypto_bignum *d)
{
#if CONFIG_ESP_WIFI_P256_ACCEL
int fast_ret = crypto_bignum_mulmod_secp256r1((const mbedtls_mpi *) a,
(const mbedtls_mpi *) b,
(const mbedtls_mpi *) c,
(mbedtls_mpi *) d);
if (fast_ret != -2) {
return fast_ret;
}
#endif
return mbedtls_mpi_mul_mpi((mbedtls_mpi *)d, (const mbedtls_mpi *)a, (const mbedtls_mpi *)b) ||
mbedtls_mpi_mod_mpi((mbedtls_mpi *)d, (mbedtls_mpi *)d, (const mbedtls_mpi *)c) ? -1 : 0;
}
@@ -160,17 +452,7 @@ int crypto_bignum_sqrmod(const struct crypto_bignum *a,
const struct crypto_bignum *b,
struct crypto_bignum *c)
{
int res;
struct crypto_bignum *tmp = crypto_bignum_init();
if (!tmp) {
return -1;
}
res = mbedtls_mpi_copy((mbedtls_mpi *) tmp, (const mbedtls_mpi *) a);
res = crypto_bignum_mulmod(a, tmp, b, c);
crypto_bignum_deinit(tmp, 0);
return res ? -1 : 0;
return crypto_bignum_mulmod(a, a, b, c);
}
int crypto_bignum_rshift(const struct crypto_bignum *a, int n,
@@ -219,8 +501,8 @@ int crypto_bignum_rand(struct crypto_bignum *r, const struct crypto_bignum *m)
mbedtls_esp_random, NULL) != 0) ? -1 : 0);
}
int crypto_bignum_legendre(const struct crypto_bignum *a,
const struct crypto_bignum *p)
static int mbedtls_bignum_legendre(const struct crypto_bignum *a,
const struct crypto_bignum *p)
{
mbedtls_mpi exp, tmp;
int res = -2, ret;
@@ -252,6 +534,22 @@ cleanup:
return res;
}
int crypto_bignum_legendre(const struct crypto_bignum *a,
const struct crypto_bignum *p)
{
#if CONFIG_ESP_WIFI_P256_ACCEL
int legendre_res;
legendre_res = crypto_bignum_legendre_secp256r1((const mbedtls_mpi *) a,
(const mbedtls_mpi *) p);
if (legendre_res != -2) {
return legendre_res;
}
#endif
return mbedtls_bignum_legendre(a, p);
}
int crypto_bignum_addmod(const struct crypto_bignum *a,
const struct crypto_bignum *b,
const struct crypto_bignum *c,
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,236 @@
/*
* SPDX-FileCopyrightText: 2026 Espressif Systems (Shanghai) CO LTD
*
* SPDX-License-Identifier: Apache-2.0
*/
/*
* Shared P-256 (secp256r1) word-level and Montgomery arithmetic used by
* both the bignum and EC fast paths. All helpers are static inline so
* each translation unit gets its own copy without linkage issues.
*
* Prerequisites: the including .c file must already provide u8/u32/u64
* typedefs, os_memcmp/os_memset/os_memcpy (via utils/common.h), and
* mbedtls/bignum.h.
*/
#pragma once
#define P256_WORDS 8
#define P256_LEN_BYTES 32
static const u8 p256_p_be[P256_LEN_BYTES] = {
0xff, 0xff, 0xff, 0xff, 0x00, 0x00, 0x00, 0x01,
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
0x00, 0x00, 0x00, 0x00, 0xff, 0xff, 0xff, 0xff,
0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff
};
static const u32 p256_p_le[P256_WORDS] = {
0xffffffffU, 0xffffffffU, 0xffffffffU, 0x00000000U,
0x00000000U, 0x00000000U, 0x00000001U, 0xffffffffU
};
/* R^2 mod p in little-endian word order, for Montgomery domain entry. */
static const u32 p256_r2_le[P256_WORDS] = {
0x00000003U, 0x00000000U, 0xffffffffU, 0xfffffffbU,
0xfffffffeU, 0xffffffffU, 0xfffffffdU, 0x00000004U
};
static inline int p256_words_is_zero(const u32 *a)
{
size_t i;
for (i = 0; i < P256_WORDS; i++) {
if (a[i] != 0) {
return 0;
}
}
return 1;
}
static inline size_t p256_words_bitlen(const u32 *a)
{
int i;
for (i = P256_WORDS - 1; i >= 0; i--) {
if (a[i] != 0) {
return (size_t) i * 32 + 32 - __builtin_clz(a[i]);
}
}
return 0;
}
static inline int p256_words_from_mpi(const mbedtls_mpi *in, u32 *out)
{
u8 in_be[P256_LEN_BYTES];
size_t i;
if (!in || in->MBEDTLS_PRIVATE(s) < 0 ||
mbedtls_mpi_size(in) > P256_LEN_BYTES ||
mbedtls_mpi_write_binary(in, in_be, sizeof(in_be)) != 0) {
return -1;
}
for (i = 0; i < P256_WORDS; i++) {
size_t off = P256_LEN_BYTES - (i + 1) * 4;
out[i] = ((u32) in_be[off] << 24) |
((u32) in_be[off + 1] << 16) |
((u32) in_be[off + 2] << 8) |
(u32) in_be[off + 3];
}
return 0;
}
static inline int p256_words_from_mpi_reduced(const mbedtls_mpi *in, u32 *out)
{
u8 in_be[P256_LEN_BYTES];
size_t i;
size_t in_size;
if (!in || in->MBEDTLS_PRIVATE(s) < 0) {
return -1;
}
in_size = mbedtls_mpi_size(in);
if (in_size > P256_LEN_BYTES) {
return -1;
}
if (mbedtls_mpi_write_binary(in, in_be, sizeof(in_be)) != 0) {
return -1;
}
if (in_size == P256_LEN_BYTES &&
os_memcmp(in_be, p256_p_be, sizeof(in_be)) >= 0) {
return -1;
}
for (i = 0; i < P256_WORDS; i++) {
size_t off = P256_LEN_BYTES - (i + 1) * 4;
out[i] = ((u32) in_be[off] << 24) |
((u32) in_be[off + 1] << 16) |
((u32) in_be[off + 2] << 8) |
(u32) in_be[off + 3];
}
return 0;
}
static inline int p256_words_to_mpi(const u32 *in, mbedtls_mpi *out)
{
u8 out_be[P256_LEN_BYTES];
size_t i;
for (i = 0; i < P256_WORDS; i++) {
size_t off = P256_LEN_BYTES - (i + 1) * 4;
out_be[off] = (u8)(in[i] >> 24);
out_be[off + 1] = (u8)(in[i] >> 16);
out_be[off + 2] = (u8)(in[i] >> 8);
out_be[off + 3] = (u8) in[i];
}
return mbedtls_mpi_read_binary(out, out_be, sizeof(out_be));
}
static inline u32 p256_words_sub_borrow(u32 *z, const u32 *x, const u32 *y)
{
size_t i;
u32 borrow = 0;
for (i = 0; i < P256_WORDS; i++) {
u64 diff = (u64) x[i] - y[i] - borrow;
z[i] = (u32) diff;
borrow = -(u32)(diff >> 32);
}
return borrow;
}
static inline void p256_words_cmov(u32 *z, const u32 *x, u32 c)
{
size_t i;
u32 mask = (u32) - (int) c;
for (i = 0; i < P256_WORDS; i++) {
z[i] = (z[i] & ~mask) | (x[i] & mask);
}
}
static inline u64 p256_u32_muladd64(u32 x, u32 y, u32 z, u32 t)
{
return (u64) x * y + z + t;
}
static inline u32 p256_u288_muladd(u32 z[P256_WORDS + 1], u32 x,
const u32 y[P256_WORDS])
{
size_t i;
u32 carry = 0;
for (i = 0; i < P256_WORDS; i++) {
u64 prod = p256_u32_muladd64(x, y[i], z[i], carry);
z[i] = (u32) prod;
carry = (u32)(prod >> 32);
}
{
u64 sum = (u64) z[P256_WORDS] + carry;
z[P256_WORDS] = (u32) sum;
carry = (u32)(sum >> 32);
}
return carry;
}
static inline void p256_u288_rshift32(u32 z[P256_WORDS + 1], u32 c)
{
size_t i;
for (i = 0; i < P256_WORDS; i++) {
z[i] = z[i + 1];
}
z[P256_WORDS] = c;
}
/*
* CIOS Montgomery multiplication for secp256r1.
*
* The Montgomery constant mu = -p^{-1} mod 2^{32} equals 1 for this prime
* because p[0] = 0xFFFFFFFF, i.e. p ≡ -1 (mod 2^{32}). That simplifies
* the reduction factor to u = new_a[0] * 1 = new_a[0], which we compute
* early as a[0] + x[i]*y[0] (the low word of the partial accumulator after
* the multiply step) to break the data dependency.
*/
static inline void p256_mont_mul(u32 z[P256_WORDS],
const u32 x[P256_WORDS],
const u32 y[P256_WORDS])
{
u32 a[P256_WORDS + 1] = {0};
u32 reduced[P256_WORDS];
size_t i;
for (i = 0; i < P256_WORDS; i++) {
u32 u = a[0] + x[i] * y[0];
u32 c = p256_u288_muladd(a, x[i], y);
c += p256_u288_muladd(a, u, p256_p_le);
p256_u288_rshift32(a, c);
}
{
u32 carry_add = a[P256_WORDS];
u32 carry_sub = p256_words_sub_borrow(reduced, a, p256_p_le);
u32 use_sub = carry_add | (1U - carry_sub);
os_memcpy(z, a, sizeof(u32) * P256_WORDS);
p256_words_cmov(z, reduced, use_sub);
}
}