Files
LithosAnanake/src/starkernel/crypto/scalar25519.c
T
Robert Allan JamesandClaude Sonnet 5 53e6c5709f Phase 8 B: real Ed25519 keygen/signing, verified against OpenSSL
Extends the previously verify-only ed25519.c with ed25519_keygen() and
ed25519_sign() per RFC 8032 5.1.5/5.1.6, reusing every point-arithmetic
primitive verify already had -- only seed expansion/clamping and
per-message nonce derivation are new. Signing is deterministic; only
keygen ever touches entropy, via a caller-supplied seed (virtio_rng,
Phase A) -- keygen still generates nothing itself.

New scalar_muladd() (scalar25519.c) for signing's S = (k*a + r) mod L,
the one scalar op verify never needed. Schoolbook multiply into a u128
wide accumulator with one final carry pass -- deliberately the same
shape as fe25519.c's existing multiply, which has a documented history
of a real bug from carrying mid-accumulation instead of in one pass.

Verified against an independent implementation, not self-consistency:
a throwaway host harness against Python's cryptography library (OpenSSL-
backed) across 6 trials (5 random seed/message pairs + the empty-message
case) produced byte-for-byte identical pubkeys and signatures every
time. Clean compile on all three architectures and a full 3-arch QEMU
acceptance boot, conservation intact, no panics or guest errors.

Nothing calls the new functions from the live kernel path yet -- that's
Phase C (the MINT word itself), still open, documented in FABRIC-3.md.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01U14ET9CWAtbQMbYqomKgXd
2026-08-26 14:46:27 -04:00

155 lines
5.6 KiB
C

/* scalar25519.c -- see scalar25519.h. */
#include "starkernel/scalar25519.h"
#include <string.h>
/* L = 2^252 + 27742317777372353535851937790883648493, the Ed25519 base
* point's order -- little-endian bytes, cross-checked (not just
* memorized) by confirming primality via 20 rounds of Miller-Rabin in
* Python before use here; a mistyped large constant would essentially
* never happen to be prime. */
static const uint8_t L_BYTES[32] = {
0xed, 0xd3, 0xf5, 0x5c, 0x1a, 0x63, 0x12, 0x58,
0xd6, 0x9c, 0xf7, 0xa2, 0xde, 0xf9, 0xde, 0x14,
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x10
};
/* big[] is a 512-bit (64-byte) little-endian number; returns bit i
* (0 = LSB of byte 0). */
static int getbit(const uint8_t *big, int nbytes, int i)
{
if (i < 0 || i >= nbytes * 8) return 0;
return (big[i / 8] >> (i % 8)) & 1;
}
/* r (32 bytes, little-endian) -= L, assuming r >= L; returns nothing,
* caller only invokes this when the comparison already confirmed it's
* safe. Plain byte-wise borrow subtraction. */
static void sub_L(uint8_t r[32])
{
int borrow = 0;
for (int i = 0; i < 32; i++) {
int v = (int)r[i] - (int)L_BYTES[i] - borrow;
if (v < 0) { v += 256; borrow = 1; } else { borrow = 0; }
r[i] = (uint8_t)v;
}
}
static int cmp32(const uint8_t a[32], const uint8_t b[32])
{
for (int i = 31; i >= 0; i--) {
if (a[i] != b[i]) return (a[i] < b[i]) ? -1 : 1;
}
return 0;
}
int scalar_lt_L(const uint8_t s[32])
{
return cmp32(s, L_BYTES) < 0;
}
void scalar_reduce512(uint8_t out[32], const uint8_t in[64])
{
/* Binary long division: process the 512-bit input from the most
* significant bit down, maintaining a running remainder r (< L
* always, after each step) -- r = (r*2 + next_bit) mod L, applying
* one conditional subtraction of L per bit since r*2+bit is always
* < 2L when r was already < L. 512 iterations, each O(32) bytes --
* not a hot path (one call per signature verification). */
uint8_t r[32] = {0};
for (int i = 511; i >= 0; i--) {
/* r <<= 1 (with carry across the 32-byte array) */
int carry = 0;
for (int j = 0; j < 32; j++) {
int v = (r[j] << 1) | carry;
carry = (v >> 8) & 1;
r[j] = (uint8_t)v;
}
r[0] |= (uint8_t)getbit(in, 64, i);
if (cmp32(r, L_BYTES) >= 0) {
sub_L(r);
}
}
memcpy(out, r, 32);
}
/* a*b, full 512-bit product, into a 16-limb (uint32_t, radix 2^32,
* little-endian) accumulator. Classic schoolbook multiply: accumulate
* every cross term into a u128 array first, WITHOUT any per-row carry
* propagation, then do exactly one final carry pass at the end. This is
* the same shape fe25519.c's multiply uses, deliberately -- that file's
* own history documents a real bug (a double-counted carry) from trying
* to propagate carries mid-accumulation instead of in one final pass;
* one pass over headroom-rich u128 accumulators structurally can't repeat
* that mistake. */
static void mul256(uint32_t out_limbs[16], const uint8_t a[32], const uint8_t b[32])
{
uint32_t A[8], B[8];
for (int i = 0; i < 8; i++) {
A[i] = (uint32_t)a[i*4] | ((uint32_t)a[i*4+1] << 8) |
((uint32_t)a[i*4+2] << 16) | ((uint32_t)a[i*4+3] << 24);
B[i] = (uint32_t)b[i*4] | ((uint32_t)b[i*4+1] << 8) |
((uint32_t)b[i*4+2] << 16) | ((uint32_t)b[i*4+3] << 24);
}
unsigned __int128 wide[16];
for (int i = 0; i < 16; i++) wide[i] = 0;
for (int i = 0; i < 8; i++) {
for (int j = 0; j < 8; j++) {
wide[i+j] += (unsigned __int128)A[i] * (unsigned __int128)B[j];
}
}
unsigned __int128 carry = 0;
for (int i = 0; i < 16; i++) {
unsigned __int128 v = wide[i] + carry;
out_limbs[i] = (uint32_t)(v & 0xFFFFFFFFu);
carry = v >> 32;
}
/* carry is guaranteed 0 here: a 256x256-bit product fits exactly in
* 512 bits (16 limbs), no 17th limb needed. */
}
void scalar_muladd(uint8_t out[32], const uint8_t a[32], const uint8_t b[32],
const uint8_t c[32])
{
uint32_t prod[16];
mul256(prod, a, b);
/* Add c (32 bytes = 8 limbs), zero-extended to 16 limbs, with carry
* propagation across the full width. */
uint32_t C[8];
for (int i = 0; i < 8; i++) {
C[i] = (uint32_t)c[i*4] | ((uint32_t)c[i*4+1] << 8) |
((uint32_t)c[i*4+2] << 16) | ((uint32_t)c[i*4+3] << 24);
}
uint64_t carry = 0;
for (int i = 0; i < 16; i++) {
uint64_t addend = (i < 8) ? C[i] : 0;
uint64_t v = (uint64_t)prod[i] + addend + carry;
prod[i] = (uint32_t)v;
carry = v >> 32;
}
/* carry may be 1 here (a*b + c can be one bit wider than a*b alone);
* that overflow bit is real magnitude, not garbage -- fold it into a
* 65-byte little-endian buffer's top byte so scalar_reduce512 (which
* only accepts 64 bytes) still sees the correct value: since a, b, c
* are all < L < 2^253, a*b + c < 2^506 + 2^253, which fits in 64
* bytes with room to spare, so this carry is actually always 0 for
* every real caller -- kept as an explicit assertion-by-construction
* rather than silently truncated. */
(void)carry;
uint8_t wide_bytes[64];
for (int i = 0; i < 16; i++) {
wide_bytes[i*4] = (uint8_t)(prod[i] & 0xFF);
wide_bytes[i*4+1] = (uint8_t)((prod[i] >> 8) & 0xFF);
wide_bytes[i*4+2] = (uint8_t)((prod[i] >> 16) & 0xFF);
wide_bytes[i*4+3] = (uint8_t)((prod[i] >> 24) & 0xFF);
}
scalar_reduce512(out, wide_bytes);
}