ProtoCore v1.0.16
Deterministic, zero-heap network stack for embedded targets
Loading...
Searching...
No Matches
fe25519.h
Go to the documentation of this file.
1// ProtoCore v1.0.16 - Copyright (C) 2026 Douglas Quigg (dstroy0) <dquigg123@gmail.com>
2// SPDX-License-Identifier: AGPL-3.0-or-later
3
4/**
5 * @file fe25519.h
6 * @brief Per-variant GF(2^255-19) field layer on the RSA/MPI hardware accelerator (X25519 + Ed25519).
7 *
8 * Field elements are canonical `uint32[8]` (< p = 2^255-19) so every field multiply is a single
9 * 256-bit modular multiply on the RSA accelerator (S3: ~1,386 cycles vs 7,955 for the software SIMD
10 * `protocore_gf_mul`; P4: ~2,118 cycles / 5.9 us). add/sub are native 32-bit (carry + one conditional subtract
11 * of p); bytes<->fe is a per-scalar-mult conversion, not per multiply. This is the shared engine behind
12 * both the X25519 KEX (`protocore_curve25519.cpp`) and the Ed25519 host-key signature (`protocore_ed25519.cpp`) on
13 * every die with a single-shot hardware MODMULT (S3 hw_ver1, P4 and newer hw_ver3 - see the gate below);
14 * the radix-2^16 `protocore_gf` path is the native / classic-ESP32 fallback in both.
15 *
16 * The accelerator (and its lock) are shared with mbedTLS RSA/DH, so a scalar-mult brackets itself with
17 * `protocore_fe_hw_enable()` / `protocore_fe_hw_disable()` (mbedTLS's own `esp_mpi_{enable,disable}_hardware_hw_op`,
18 * i.e. acquire the MPI lock + clock/power the peripheral) and holds the lock for its whole run.
19 *
20 * `static inline` on purpose: the cheap ops (add/sub/cswap) inline into the ladder in each translation
21 * unit with no cross-TU call overhead, and the whole layer stays one source of truth.
22 *
23 * Two surfaces, and they are not the same kind of thing: the field ops below are shared internals that
24 * X25519 and Ed25519 are written in and call directly, and ::Fe25519 is the namespace this module
25 * exports over those same ops.
26 *
27 * @author Douglas Quigg (dstroy0)
28 * @date 2026
29 */
30
31#ifndef PROTOCORE_FE25519_H
32#define PROTOCORE_FE25519_H
33
34#include "protocore_config.h" // the entry point: protocore_types.h for the widths
35
36#if PROTOCORE_ENABLE_FE25519
37
38#include "crypto/ct_eq/ct_eq.h" // protocore_ct_eq
39
41
42// 25519 has no dedicated ECC accelerator on any ESP32 die, so the RSA MODMULT is the field-layer win wherever
43// it exists - track the HAL's PROTOCORE_RSA_MODMUL_HW (S3, P4, ...). Classic ESP32 / native keep the software ladder.
44#if PROTOCORE_RSA_MODMUL_HW
45#define PROTOCORE_FE25519_MPI_HW 1
46#endif
47
48#if PROTOCORE_FE25519_MPI_HW
49
50/** @brief A field element of GF(2^255-19): canonical, eight little-endian 32-bit limbs (< p). */
51typedef uint32_t fe[8];
52
53// Constants for the 256-bit modular multiply mod p = 2^255-19 (scratchpad/montconst.py): Montgomery m'
54// and R^2 mod p (= 38^2 = 1444 = 0x5a4). Preloading R^2 into the result block makes the accelerator
55// return a plain residue X*Y mod p rather than a Montgomery form (the esp_mpi_mul_mpi_mod convention).
56static const uint32_t FE_MOD_MPRIME = 0x286bca1bu;
57static const uint32_t FE_MOD_P[8] = {0xffffffedu, 0xffffffffu, 0xffffffffu, 0xffffffffu,
58 0xffffffffu, 0xffffffffu, 0xffffffffu, 0x7fffffffu};
59static const uint32_t FE_MOD_R2[8] = {0x000005a4u, 0, 0, 0, 0, 0, 0, 0};
60
61// Acquire the accelerator (lock + power) for a scalar-mult, and drop it after. Bracket every run. Thin names
62// over the HAL so the X25519 ladder / Ed25519 point arithmetic read as before.
63static inline void protocore_fe_hw_enable(void)
64{
65 protocore_rsa_hw_acquire();
66}
67static inline void protocore_fe_hw_disable(void)
68{
69 protocore_rsa_hw_release();
70}
71
72// z = x*y mod p (8 words / 256-bit) on the RSA MODMULT. Requires protocore_fe_hw_enable() first. Canonical (< p),
73// safe if z aliases x/y. Delegates to the HAL modmul with this domain's constants; the crypto TUs that pull
74// this in build at -O2 (the OPT the module states in its CMakeLists), where the always_inline HAL folds FE_MOD_P / the
75// mostly-zero FE_MOD_R2 into immediate stores - the hand-tuned ~1,380-cyc path.
76static inline void fe_mul(fe z, const fe x, const fe y)
77{
78 protocore_rsa_modmul(z, x, y, FE_MOD_P, FE_MOD_MPRIME, FE_MOD_R2, 8);
79}
80static inline void fe_sq(fe o, const fe x)
81{
82 fe_mul(o, x, x);
83}
84
85static inline void fe_copy(fe o, const fe a)
86{
87 for (int i = 0; i < 8; i++)
88 {
89 o[i] = a[i];
90 }
91}
92static inline void fe_0(fe o)
93{
94 for (int i = 0; i < 8; i++)
95 {
96 o[i] = 0;
97 }
98}
99static inline void fe_1(fe o)
100{
101 o[0] = 1;
102 for (int i = 1; i < 8; i++)
103 {
104 o[i] = 0;
105 }
106}
107// If o >= p (o is in [p, 2p)), subtract p. Constant-time: the borrow out of o-p selects o or o-p.
108static inline void fe_reduce_once(fe o)
109{
110 uint32_t t[8];
111 int64_t b = 0;
112 for (int i = 0; i < 8; i++)
113 {
114 b += (int64_t)o[i] - (int64_t)FE_MOD_P[i];
115 t[i] = (uint32_t)b;
116 b >>= 32;
117 }
118 uint32_t keep = (uint32_t)b; // 0 if o>=p (take t=o-p), 0xffffffff if o<p (keep o)
119 for (int i = 0; i < 8; i++)
120 {
121 o[i] = (o[i] & keep) | (t[i] & ~keep);
122 }
123}
124static inline void fe_add(fe o, const fe x, const fe y) // x,y < p -> o = x+y mod p
125{
126 uint64_t c = 0;
127 for (int i = 0; i < 8; i++)
128 {
129 c += (uint64_t)x[i] + y[i];
130 o[i] = (uint32_t)c;
131 c >>= 32;
132 }
133 fe_reduce_once(o); // x+y < 2p, one conditional subtract
134}
135static inline void fe_sub(fe o, const fe x, const fe y) // x,y < p -> o = x-y mod p
136{
137 int64_t b = 0;
138 uint32_t t[8];
139 for (int i = 0; i < 8; i++)
140 {
141 b += (int64_t)x[i] - (int64_t)y[i];
142 t[i] = (uint32_t)b;
143 b >>= 32;
144 }
145 uint32_t borrow = (uint32_t)b; // 0xffffffff if x<y -> add p back
146 uint64_t c = 0;
147 for (int i = 0; i < 8; i++)
148 {
149 c += (uint64_t)t[i] + (FE_MOD_P[i] & borrow);
150 o[i] = (uint32_t)c;
151 c >>= 32;
152 }
153}
154static inline void fe_cswap(fe x, fe y, uint32_t swap) // constant-time swap of x,y when swap==1
155{
156 uint32_t mask = (uint32_t)(-(int32_t)swap);
157 for (int i = 0; i < 8; i++)
158 {
159 uint32_t t = mask & (x[i] ^ y[i]);
160 x[i] ^= t;
161 y[i] ^= t;
162 }
163}
164static inline void fe_frombytes(fe o, const uint8_t b[32])
165{
166 for (int i = 0; i < 8; i++)
167 {
168 o[i] = (uint32_t)b[4 * i] | ((uint32_t)b[4 * i + 1] << 8) | ((uint32_t)b[4 * i + 2] << 16) |
169 ((uint32_t)b[4 * i + 3] << 24);
170 }
171 o[7] &= 0x7fffffffu; // Ed25519/X25519 both ignore bit 255 of the y/u coordinate
172 fe_reduce_once(o); // the masked value can still be in [p, 2^255) -> canonicalize
173}
174static inline void fe_tobytes(uint8_t b[32], const fe a)
175{
176 fe t;
177 fe_copy(t, a);
178 fe_reduce_once(t); // freeze to the canonical residue
179 for (int i = 0; i < 8; i++)
180 {
181 b[4 * i] = (uint8_t)t[i];
182 b[4 * i + 1] = (uint8_t)(t[i] >> 8);
183 b[4 * i + 2] = (uint8_t)(t[i] >> 16);
184 b[4 * i + 3] = (uint8_t)(t[i] >> 24);
185 }
186}
187// o = a^(p-2) = a^-1 mod p (tweetnacl square-and-multiply chain for the exponent 2^255-21).
188static inline void fe_invert(fe o, const fe a)
189{
190 fe c;
191 fe_copy(c, a);
192 for (int i = 253; i >= 0; i--)
193 {
194 fe_sq(c, c);
195 if (i != 2 && i != 4)
196 {
197 fe_mul(c, c, a);
198 }
199 }
200 fe_copy(o, c);
201}
202// o = a^((p-5)/8) = a^(2^252-3) - the square-root exponent for Ed25519 point decompression.
203static inline void fe_pow2523(fe o, const fe a)
204{
205 fe c;
206 fe_copy(c, a);
207 for (int i = 250; i >= 0; i--)
208 {
209 fe_sq(c, c);
210 if (i != 1)
211 {
212 fe_mul(c, c, a);
213 }
214 }
215 fe_copy(o, c);
216}
217// Low bit of the canonical encoding (Ed25519 x-coordinate sign).
218static inline int fe_parity(const fe a)
219{
220 uint8_t d[32];
221 fe_tobytes(d, a);
222 return d[0] & 1;
223}
224// 0 if a and b encode the same field element, -1 otherwise (constant-time over the 32 bytes).
225static inline int fe_neq(const fe a, const fe b)
226{
227 uint8_t c[32];
228 uint8_t d[32];
229 fe_tobytes(c, a);
230 fe_tobytes(d, b);
231 return protocore_ct_eq(c, d, 32) ? 0 : -1;
232}
233
234// --- the namespace over the field layer ------------------------------------
235//
236// Every field element an entry reads or writes is the caller's own, and no entry carries anything to
237// the next one, so none of them needs a borrow and this module states no BORROW constant.
238
239/** @brief The product a multiply writes and the two factors it reads. */
240typedef struct
241{
242 uint32_t *z; ///< the destination fe, eight limbs; may alias @c x or @c y
243 const uint32_t *x; ///< the first factor, canonical
244 const uint32_t *y; ///< the second factor, canonical
245} Fe25519MulArgs;
246
247/** @brief The square a squaring writes and the element it reads. */
248typedef struct
249{
250 uint32_t *o; ///< the destination fe, eight limbs
251 const uint32_t *x; ///< the element squared
252} Fe25519SqArgs;
253
254/** @brief The destination and source of a field-element copy. */
255typedef struct
256{
257 uint32_t *o; ///< the destination fe, eight limbs
258 const uint32_t *a; ///< the source fe
259} Fe25519CopyArgs;
260
261/** @brief The element set to zero. */
262typedef struct
263{
264 uint32_t *o; ///< the destination fe, eight limbs
265} Fe25519ZeroArgs;
266
267/** @brief The element set to one. */
268typedef struct
269{
270 uint32_t *o; ///< the destination fe, eight limbs
271} Fe25519OneArgs;
272
273/** @brief The element canonicalized in place. */
274typedef struct
275{
276 uint32_t *o; ///< the fe reduced in place, in [p, 2p) on entry
277} Fe25519ReduceArgs;
278
279/** @brief The sum an addition writes and the two terms it reads. */
280typedef struct
281{
282 uint32_t *o; ///< the destination fe, eight limbs
283 const uint32_t *x; ///< the first term, < p
284 const uint32_t *y; ///< the second term, < p
285} Fe25519AddArgs;
286
287/** @brief The difference a subtraction writes and the two terms it reads. */
288typedef struct
289{
290 uint32_t *o; ///< the destination fe, eight limbs
291 const uint32_t *x; ///< the minuend, < p
292 const uint32_t *y; ///< the subtrahend, < p
293} Fe25519SubArgs;
294
295/** @brief The two elements a conditional swap exchanges, and the bit selecting it. */
296typedef struct
297{
298 uint32_t *x; ///< the first fe, exchanged in place
299 uint32_t *y; ///< the second fe, exchanged in place
300 uint32_t swap; ///< 1 swaps, 0 leaves both alone
301} Fe25519CswapArgs;
302
303/** @brief The 32 bytes a decode reads and the element it writes. */
304typedef struct
305{
306 uint32_t *o; ///< the destination fe, eight limbs
307 const uint8_t *b; ///< 32 little-endian bytes; bit 255 is ignored
308} Fe25519FromBytesArgs;
309
310/** @brief The element an encode reads and the 32 bytes it writes. */
311typedef struct
312{
313 uint8_t *b; ///< 32 little-endian bytes of the canonical residue
314 const uint32_t *a; ///< the source fe
315} Fe25519ToBytesArgs;
316
317/** @brief The inverse an inversion writes and the element it reads. */
318typedef struct
319{
320 uint32_t *o; ///< the destination fe, eight limbs
321 const uint32_t *a; ///< the element inverted
322} Fe25519InvertArgs;
323
324/** @brief The power a^((p-5)/8) writes and the element it reads. */
325typedef struct
326{
327 uint32_t *o; ///< the destination fe, eight limbs
328 const uint32_t *a; ///< the base
329} Fe25519Pow2523Args;
330
331/** @brief The element a parity read is taken over. */
332typedef struct
333{
334 const uint32_t *a; ///< the fe whose canonical encoding supplies the low bit
335} Fe25519ParityArgs;
336
337/** @brief The two elements an equality test compares. */
338typedef struct
339{
340 const uint32_t *a; ///< the first fe
341 const uint32_t *b; ///< the second fe
342} Fe25519NeqArgs;
343
344/**
345 * @brief GF(2^255-19) on the RSA/MPI accelerator.
346 *
347 * A caller sets the members a call takes, invokes it through ::Fe25519, and reads the outcome off the
348 * same handle.
349 *
350 * Fe25519.hw_enable(work);
351 * Fe25519.mul_args.z = z;
352 * Fe25519.mul_args.x = x;
353 * Fe25519.mul_args.y = y;
354 * Fe25519.mul(work);
355 * Fe25519.hw_disable(work);
356 *
357 * @var Fe25519Ns::mul_args the product a multiply writes and the two factors it reads
358 * @var Fe25519Ns::sq_args the square a squaring writes and the element it reads
359 * @var Fe25519Ns::copy_args the destination and source of a field-element copy
360 * @var Fe25519Ns::zero_args the element set to zero
361 * @var Fe25519Ns::one_args the element set to one
362 * @var Fe25519Ns::reduce_args the element canonicalized in place
363 * @var Fe25519Ns::add_args the sum an addition writes and the two terms it reads
364 * @var Fe25519Ns::sub_args the difference a subtraction writes and the two terms it reads
365 * @var Fe25519Ns::cswap_args the two elements a conditional swap exchanges, and the bit selecting it
366 * @var Fe25519Ns::frombytes_args the 32 bytes a decode reads and the element it writes
367 * @var Fe25519Ns::tobytes_args the element an encode reads and the 32 bytes it writes
368 * @var Fe25519Ns::invert_args the inverse an inversion writes and the element it reads
369 * @var Fe25519Ns::pow2523_args the power a^((p-5)/8) writes and the element it reads
370 * @var Fe25519Ns::parity_args the element a parity read is taken over
371 * @var Fe25519Ns::neq_args the two elements an equality test compares
372 * @var Fe25519Ns::ok a call's true/false outcome; false on a null operand
373 * @var Fe25519Ns::parity the low bit of the canonical encoding the last parity read recovered
374 * @var Fe25519Ns::neq 0 when the last compare found the same element, -1 otherwise
375 * @var Fe25519Ns::hw_enable take the accelerator lock and power it for a run
376 * @var Fe25519Ns::hw_disable drop the lock and power the accelerator down
377 * @var Fe25519Ns::mul z = x*y mod p, one 256-bit MODMULT
378 * @var Fe25519Ns::sq o = x^2 mod p
379 * @var Fe25519Ns::copy o = a
380 * @var Fe25519Ns::zero o = 0
381 * @var Fe25519Ns::one o = 1
382 * @var Fe25519Ns::reduce_once subtract p once when o >= p, in constant time
383 * @var Fe25519Ns::add o = x+y mod p
384 * @var Fe25519Ns::sub o = x-y mod p
385 * @var Fe25519Ns::cswap exchange x and y when @c swap is 1, in constant time
386 * @var Fe25519Ns::frombytes decode 32 little-endian bytes to a canonical element
387 * @var Fe25519Ns::tobytes encode a canonical element to 32 little-endian bytes
388 * @var Fe25519Ns::invert o = a^(p-2) = a^-1 mod p
389 * @var Fe25519Ns::pow2523 o = a^((p-5)/8), the Ed25519 decompression square-root exponent
390 * @var Fe25519Ns::get_parity read the low bit of a's canonical encoding into @ref Fe25519Ns::parity
391 * @var Fe25519Ns::get_neq compare a and b in constant time into @ref Fe25519Ns::neq
392 *
393 * @ref Fe25519Ns::mul, @ref Fe25519Ns::sq, @ref Fe25519Ns::invert and @ref Fe25519Ns::pow2523 run on the
394 * accelerator, so @ref Fe25519Ns::hw_enable brackets them and @ref Fe25519Ns::hw_disable closes the run.
395 *
396 * @c work goes unread by every entry: each one works on the caller's own field elements and carries
397 * nothing to the next call, so this module needs no borrow, holds no context, and names no BORROW
398 * constant. The ops themselves stay @c static @c inline above, where the X25519 ladder and the Ed25519
399 * point arithmetic call them directly with no cross-TU call in the loop.
400 *
401 * No storage member and no context: a caller sets operands and reads @ref Fe25519Ns::ok, and that is
402 * all the surface there is.
403 */
404typedef struct
405{
406 Fe25519MulArgs mul_args;
407 Fe25519SqArgs sq_args;
408 Fe25519CopyArgs copy_args;
409 Fe25519ZeroArgs zero_args;
410 Fe25519OneArgs one_args;
411 Fe25519ReduceArgs reduce_args;
412 Fe25519AddArgs add_args;
413 Fe25519SubArgs sub_args;
414 Fe25519CswapArgs cswap_args;
415 Fe25519FromBytesArgs frombytes_args;
416 Fe25519ToBytesArgs tobytes_args;
417 Fe25519InvertArgs invert_args;
418 Fe25519Pow2523Args pow2523_args;
419 Fe25519ParityArgs parity_args;
420 Fe25519NeqArgs neq_args;
421 proto_bool ok;
422 int parity;
423 int neq;
424} Fe25519Vars;
425
426/** @brief The operands and the outcome. */
427extern Fe25519Vars Fe25519V;
428
429/** @brief The entries. */
430typedef struct
431{
432 void (*const hw_enable)(uint8_t *work);
433 void (*const hw_disable)(uint8_t *work);
434 void (*const mul)(uint8_t *work);
435 void (*const sq)(uint8_t *work);
436 void (*const copy)(uint8_t *work);
437 void (*const zero)(uint8_t *work);
438 void (*const one)(uint8_t *work);
439 void (*const reduce_once)(uint8_t *work);
440 void (*const add)(uint8_t *work);
441 void (*const sub)(uint8_t *work);
442 void (*const cswap)(uint8_t *work);
443 void (*const frombytes)(uint8_t *work);
444 void (*const tobytes)(uint8_t *work);
445 void (*const invert)(uint8_t *work);
446 void (*const pow2523)(uint8_t *work);
447 void (*const get_parity)(uint8_t *work);
448 void (*const get_neq)(uint8_t *work);
449} Fe25519Ns;
450
451// What the table binds, defined once in the .c and taking one parameter each: everything
452// else an entry needs is an operand in Fe25519V or a region of the borrow at a fixed offset.
453void protocore_fe25519_hw_enable(uint8_t *work);
454void protocore_fe25519_hw_disable(uint8_t *work);
455void protocore_fe25519_mul(uint8_t *work);
456void protocore_fe25519_sq(uint8_t *work);
457void protocore_fe25519_copy(uint8_t *work);
458void protocore_fe25519_zero(uint8_t *work);
459void protocore_fe25519_one(uint8_t *work);
460void protocore_fe25519_reduce_once(uint8_t *work);
461void protocore_fe25519_add(uint8_t *work);
462void protocore_fe25519_sub(uint8_t *work);
463void protocore_fe25519_cswap(uint8_t *work);
464void protocore_fe25519_frombytes(uint8_t *work);
465void protocore_fe25519_tobytes(uint8_t *work);
466void protocore_fe25519_invert(uint8_t *work);
467void protocore_fe25519_pow2523(uint8_t *work);
468void protocore_fe25519_get_parity(uint8_t *work);
469void protocore_fe25519_get_neq(uint8_t *work);
470
471// `static const`, initialised HERE rather than `extern` against a definition in the .c: a
472// const object whose initializer every translation unit can see is a COMPILE-TIME FACT, so
473// `Fe25519.hw_enable(work)` resolves to a named function and becomes a DIRECT call. An extern table
474// leaves the call indirect and the symbol live at every level, -O2 -flto included.
475static const Fe25519Ns Fe25519 __attribute__((unused)) = {
476 .hw_enable = protocore_fe25519_hw_enable,
477 .hw_disable = protocore_fe25519_hw_disable,
478 .mul = protocore_fe25519_mul,
479 .sq = protocore_fe25519_sq,
480 .copy = protocore_fe25519_copy,
481 .zero = protocore_fe25519_zero,
482 .one = protocore_fe25519_one,
483 .reduce_once = protocore_fe25519_reduce_once,
484 .add = protocore_fe25519_add,
485 .sub = protocore_fe25519_sub,
486 .cswap = protocore_fe25519_cswap,
487 .frombytes = protocore_fe25519_frombytes,
488 .tobytes = protocore_fe25519_tobytes,
489 .invert = protocore_fe25519_invert,
490 .pow2523 = protocore_fe25519_pow2523,
491 .get_parity = protocore_fe25519_get_parity,
492 .get_neq = protocore_fe25519_get_neq,
493};
494
495#endif // PROTOCORE_FE25519_MPI_HW
496
498
499#endif // PROTOCORE_ENABLE_FE25519
500
501#endif // PROTOCORE_FE25519_H
Constant-time comparison for secret-dependent checks.
#define PROTOCORE_BEGIN_DECLS
Give a header's declarations C linkage, so their symbol names carry no parameter types.
Definition types.h:96
_Bool proto_bool
The truth value.
Definition types.h:64
#define PROTOCORE_END_DECLS
Definition types.h:97