m9 and rng polishing

This commit is contained in:
liquidraver
2026-07-21 22:49:48 +02:00
parent 24936cc4e5
commit 076fbb7ff3
6 changed files with 328 additions and 20 deletions
@@ -105,6 +105,9 @@ Two things still want confirming against real hardware: the **battery divider mu
(the two upstreams disagree — check `get batt` against a multimeter), and the **GNSS module identity** is
unconfirmed. Reports welcome.
Thanks to [@WillyJL](https://github.com/WillyJL), who requested this port and provided the hardware details / most of the code
it is based on.
### ThinkNode M9 — experimental, first testable release
**Consider this a preview, not a finished port.** It builds, flashes and runs, and it is worth testing if
+31
View File
@@ -549,6 +549,37 @@ if(_BOARD_SD_QUIESCE)
target_sources(app PRIVATE "${_BOARD_SD_QUIESCE}")
endif()
# ESP32 pre-RF entropy source (bootloader_random).
#
# The ESP32 hardware TRNG (WDEV_RANDOM) is a PRNG that only receives real
# entropy while WiFi or BT is enabled — see drivers/entropy/entropy_esp32.c.
# Identity keygen runs long before any of that (and repeaters/room servers
# never enable RF at all), so sys_csrand_get() contributes nothing there.
#
# Espressif's documented remedy is bootloader_random_enable(), which puts the
# SAR ADC into continuous sampling and mixes its noise into the HWRNG. Its own
# header endorses this exact use: "Can also be called from app code, if true
# random numbers are required without initialized RF subsystem."
# ZephyrRNG::mixIdentitySeed brackets its entropy collection with it.
#
# hal/espressif compiles only part of bootloader_support into app builds (the
# include path is already set up, but not these sources), so pull in the
# per-SoC implementation explicitly. The generic bootloader_random.c is NOT
# needed — it only defines these symbols for the bring-up-bypass stub.
if(CONFIG_SOC_FAMILY_ESPRESSIF_ESP32)
set(_ESP_BOOTLOADER_RANDOM
"${ZEPHYR_HAL_ESPRESSIF_MODULE_DIR}/components/bootloader_support/src/bootloader_random_${CONFIG_SOC_SERIES}.c")
if(EXISTS "${_ESP_BOOTLOADER_RANDOM}")
message(STATUS "ZephCore RNG: ESP32 pre-RF entropy (${CONFIG_SOC_SERIES})")
target_sources(app PRIVATE "${_ESP_BOOTLOADER_RANDOM}")
else()
message(WARNING
"ZephCore RNG: no bootloader_random source for ${CONFIG_SOC_SERIES} — "
"identity keygen will fall back to CPU jitter alone. Check "
"hal/espressif components/bootloader_support/src/.")
endif()
endif()
# Native-Linux (native_sim) runtime setup: force real-time clock mode so the
# simulated clock tracks wall time (required for the real SX126x radio's
# BUSY/DIO1 timing). Bakes in the equivalent of the --rt command-line flag.
+240 -15
View File
@@ -12,6 +12,12 @@
#include <string.h>
#include <mesh/Utils.h>
#if defined(CONFIG_SOC_FAMILY_ESPRESSIF_ESP32)
/* Pre-RF entropy for the ESP32 HWRNG — see esp32_entropy_begin() below.
* Source file is added to the build by CMakeLists.txt (ESP32 only). */
#include <bootloader_random.h>
#endif
BUILD_ASSERT(IS_ENABLED(CONFIG_CSPRNG_ENABLED),
"ZephyrRNG requires CONFIG_CSPRNG_ENABLED for cryptographic key derivation");
@@ -49,12 +55,44 @@ void ZephyrRNG::random(uint8_t *dest, size_t sz)
* statistically prove entropy quality — that's what the literature is
* for. */
/* Health statistics for one jitter window. Timing statistics only — never
* pool contents or derived key material. Reporting these is standard practice
* for a NIST SP 800-90B style noise source; reporting the bytes would not be. */
#define JITTER_HIST_SLOTS 16
struct jitter_stats {
int n_samples;
int n_distinct; /* distinct delta values seen, capped at 8 */
int max_consec; /* longest run of identical deltas */
uint32_t min_delta;
uint32_t max_delta;
uint32_t mcv_count; /* occurrences of the most common tracked delta */
uint32_t untracked; /* samples whose value missed the histogram */
bool ok;
};
/* log2(x) * 1000, integer math (no FP — printk here has no float support).
* Integer part from the leading bit; fraction by linear interpolation between
* adjacent powers of two. Linear interpolation UNDERSTATES log2 across that
* range (log2(1.5)=0.585 vs 0.5 linear), so the derived entropy figure errs
* low — the safe direction for an entropy claim. */
static uint32_t log2_millibits(uint32_t x)
{
if (x <= 1) return 0;
uint32_t ipart = 31u - (uint32_t)__builtin_clz(x);
uint32_t base = 1u << ipart;
uint32_t frac = (uint32_t)(((uint64_t)(x - base) * 1000u) / base);
return ipart * 1000u + frac;
}
static bool sample_cpu_jitter(uint8_t *pool, size_t pool_size,
size_t pool_offset, uint32_t duration_ms)
size_t pool_offset, uint32_t duration_ms,
struct jitter_stats *st = nullptr)
{
uint32_t accum = k_cycle_get_32();
int64_t deadline = k_uptime_get() + duration_ms;
size_t idx = pool_offset;
uint32_t min_delta = UINT32_MAX, max_delta = 0;
/* Online health stats: 32 bytes total vs. the previous 512-byte
* deltas[] array. Tracks every sample, not just the first 128. */
@@ -64,11 +102,24 @@ static bool sample_cpu_jitter(uint8_t *pool, size_t pool_size,
int n_distinct = 0;
int n_samples = 0;
/* Bounded histogram for the SP 800-90B Most Common Value estimator.
* Holds the first JITTER_HIST_SLOTS distinct deltas; anything beyond
* that is counted in `untracked`. A value frequent enough to dominate
* p_max shows up within the first few distinct observations with
* overwhelming probability, so this captures what the estimator needs
* — but `untracked` is reported so the assumption stays visible. */
uint32_t hist_val[JITTER_HIST_SLOTS] = {0};
uint32_t hist_cnt[JITTER_HIST_SLOTS] = {0};
int n_hist = 0;
uint32_t untracked = 0;
while (k_uptime_get() < deadline) {
uint32_t t1 = k_cycle_get_32();
/* Variable-time work — number of iterations depends on the
* accumulator, so timing depends on hardware nondeterminism
* (cache, branch prediction, ISR firing). */
/* Variable-time work. The iteration count comes from the
* accumulator, which carries the PREVIOUS measurement (see the
* feedback step below), so the amount of work done here depends
* on observed hardware nondeterminism rather than on a fixed
* sequence. */
volatile uint32_t a = accum;
uint32_t iters = (accum & 0x7f);
for (uint32_t i = 0; i < iters; i++) {
@@ -78,6 +129,25 @@ static bool sample_cpu_jitter(uint8_t *pool, size_t pool_size,
uint32_t t2 = k_cycle_get_32();
uint32_t delta = t2 - t1;
/* FEEDBACK — the load-bearing line. Without it `accum` evolves
* as a pure LCG from a single seed: the iteration count becomes a
* fixed sequence, and on any CPU where reading the cycle counter
* is cheap and same-domain (ESP32 CCOUNT) a warm cache makes each
* `iters` value yield an identical `delta`. Deltas then repeat
* and the health check below correctly fails — which is exactly
* what was observed on every ESP32 board.
*
* Folding the measurement back in closes the loop, so timing
* variation propagates into subsequent work. This is the
* mechanism the cited jitterentropy design relies on and which
* the original implementation omitted.
*
* nRF was unaffected: its k_cycle_get_32() is a 32.768 kHz RTC in
* a different clock domain, so the read latency itself varies
* with domain phase and supplied the nondeterminism this line
* now provides everywhere. */
accum ^= delta;
/* Mix into entropy pool */
pool[idx++ % pool_size] ^= (uint8_t)delta;
pool[idx++ % pool_size] ^= (uint8_t)(delta >> 8);
@@ -101,12 +171,109 @@ static bool sample_cpu_jitter(uint8_t *pool, size_t pool_size,
if (!found) distinct[n_distinct++] = delta;
}
if (delta < min_delta) min_delta = delta;
if (delta > max_delta) max_delta = delta;
/* MCV histogram */
{
bool binned = false;
for (int j = 0; j < n_hist; j++) {
if (hist_val[j] == delta) {
hist_cnt[j]++;
binned = true;
break;
}
}
if (!binned) {
if (n_hist < JITTER_HIST_SLOTS) {
hist_val[n_hist] = delta;
hist_cnt[n_hist] = 1;
n_hist++;
} else {
untracked++;
}
}
}
n_samples++;
}
if (n_samples < 16) return false;
if (max_consec >= 32) return false; /* stuck source */
return n_distinct >= 5; /* minimal variance */
uint32_t mcv_count = 0;
for (int j = 0; j < n_hist; j++) {
if (hist_cnt[j] > mcv_count) mcv_count = hist_cnt[j];
}
bool ok = (n_samples >= 16) /* enough samples */
&& (max_consec < 32) /* not a stuck source */
&& (n_distinct >= 5); /* minimal variance */
if (st) {
st->n_samples = n_samples;
st->n_distinct = n_distinct;
st->max_consec = max_consec;
st->min_delta = (n_samples > 0) ? min_delta : 0;
st->max_delta = max_delta;
st->mcv_count = mcv_count;
st->untracked = untracked;
st->ok = ok;
}
return ok;
}
/* Count distinct byte values in a buffer — a repetition/adaptive-proportion
* style health indicator for a CSPRNG draw. A stuck source collapses this to
* 1. Deliberately coarse: one integer per draw, which detects catastrophic
* failure without meaningfully describing the bytes themselves. */
static int distinct_bytes(const uint8_t *buf, size_t len)
{
bool seen[256] = {false};
int n = 0;
for (size_t i = 0; i < len; i++) {
if (!seen[buf[i]]) { seen[buf[i]] = true; n++; }
}
return n;
}
/* One line per jitter window. `distinct` is capped at 8 by the sampler, so 8/8
* means "at least 8" — the pass threshold is 5. `maxrep` is the longest run of
* identical deltas; >=32 fails. min/max delta expose a resolution problem: if
* they are 0 and 1, the cycle counter is too coarse to measure the work at all
* and no amount of sampling will help. */
static void report_jitter(const char *label, const struct jitter_stats *st)
{
printk("[RNG] %s: samples=%d distinct=%d/8 maxrep=%d "
"delta=[%u..%u] -> %s\n",
label, st->n_samples, st->n_distinct, st->max_consec,
st->min_delta, st->max_delta, st->ok ? "PASS" : "FAIL");
/* Min-entropy estimate, NIST SP 800-90B 6.3.1 Most Common Value:
* p_max = mcv_count / n_samples
* H_min per sample = -log2(p_max) = log2(n_samples / mcv_count)
*
* Computed in milli-bits with integer math. This REPLACES the 0.1-0.3
* bits/sample the file header used to assume — assumption is only valid
* if deltas actually vary, which is the very thing that failed on ESP32.
*
* Deliberately NOT measured on the conditioned output: AES-256-CTR makes
* any input look uniform, so output statistics would read perfect even
* for a near-zero-entropy seed. Entropy is a property of the source. */
if (st->n_samples <= 0 || st->mcv_count == 0) {
printk("[RNG] entropy est: n/a (no samples)\n");
return;
}
uint32_t ratio_q10 = (uint32_t)(((uint64_t)st->n_samples * 1024u)
/ st->mcv_count);
uint32_t per_mb = log2_millibits(ratio_q10);
per_mb = (per_mb > 10000u) ? (per_mb - 10000u) : 0u; /* less log2(1024) */
uint64_t total_mb = (uint64_t)per_mb * (uint64_t)st->n_samples;
uint32_t total_bits = (uint32_t)(total_mb / 1000u);
printk("[RNG] entropy est: %u.%03u bits/sample x %d = ~%u bits "
"(need 256)%s\n",
per_mb / 1000u, per_mb % 1000u, st->n_samples, total_bits,
st->untracked ? " [histogram overflowed — est. is optimistic]" : "");
}
/* ===== Entropy extraction via AES-256-CTR ================================
@@ -204,8 +371,42 @@ void ZephyrRNG::mixIdentitySeed(uint8_t *out, size_t out_len,
uint8_t pool[512];
memset(pool, 0, sizeof(pool));
/* Stage 1: early CSPRNG (strong on nRF/MG24, weak on ESP32 pre-radio) */
(void)sys_csrand_get(pool, 64);
/* ESP32 only: give the HWRNG a real entropy source for the duration of
* this function.
*
* WDEV_RANDOM is a PRNG that receives hardware entropy only "provided
* Wi-Fi or BT are enabled" (Zephyr drivers/entropy/entropy_esp32.c), and
* sys_csrand_get() maps straight to it. Every caller of this function
* runs before RF is up, and repeater / room-server builds never enable
* RF at all — so stages 1 and 5 below contributed NOTHING on ESP32,
* leaving CPU jitter as the only real source. That was observed failing
* its health check on ThinkNode M9 hardware while deriving a permanent
* identity key.
*
* bootloader_random_enable() puts the SAR ADC into continuous sampling
* and mixes its noise into the HWRNG; Espressif's header explicitly
* sanctions calling it from app code when RF is not up. It must be
* disabled again before anything else touches the ADC or RF — done at
* the end of the collection phase, before AES extraction, so the ADC is
* held for as short a window as possible.
*
* WARNING for future callers: this is unsafe if RF or the ADC is already
* in use. Do not call mixIdentitySeed() after bt_enable() or alongside a
* battery read on ESP32. */
#if defined(CONFIG_SOC_FAMILY_ESPRESSIF_ESP32)
bootloader_random_enable();
printk("[RNG] === identity seed health ===\n");
printk("[RNG] esp32 pre-RF entropy (bootloader_random): ENABLED\n");
#else
printk("[RNG] === identity seed health ===\n");
printk("[RNG] platform TRNG is radio-independent (no pre-RF workaround needed)\n");
#endif
/* Stage 1: early CSPRNG (strong on nRF/MG24; on ESP32 this is only real
* because bootloader_random_enable() above is feeding the HWRNG) */
int rc1 = sys_csrand_get(pool, 64);
printk("[RNG] stage1 csrand : rc=%d distinct=%d/64\n",
rc1, distinct_bytes(pool, 64));
/* Stage 2: HWINFO unique device ID — uniqueness across devices */
uint8_t devid[16] = {0};
@@ -213,11 +414,18 @@ void ZephyrRNG::mixIdentitySeed(uint8_t *out, size_t out_len,
for (ssize_t i = 0; i < devid_len && i < (ssize_t)sizeof(devid); i++) {
pool[64 + i] ^= devid[i];
}
/* NOT secret — this is the efuse/FICR serial, public and printed at boot.
* It contributes uniqueness between devices, never unpredictability. */
printk("[RNG] stage2 hwinfo id : %d bytes (public — uniqueness only)\n",
(int)devid_len);
/* Stage 3: caller-supplied entropy (e.g. ADC LSB noise) */
if (extra && extra_len > 0) {
size_t n = (extra_len < 32) ? extra_len : 32;
for (size_t i = 0; i < n; i++) pool[80 + i] ^= extra[i];
printk("[RNG] stage3 extra : %d bytes\n", (int)n);
} else {
printk("[RNG] stage3 extra : none\n");
}
/* Stage 4: CPU cycle-counter jitter, 200ms.
@@ -228,25 +436,42 @@ void ZephyrRNG::mixIdentitySeed(uint8_t *out, size_t out_len,
* sys_csrand_get in stages 1 and 5) which is a far stronger source than
* jitter sampling anyway. */
#ifndef CONFIG_ARCH_POSIX
bool health_ok = sample_cpu_jitter(pool, sizeof(pool), 112, 200);
struct jitter_stats js = {};
bool health_ok = sample_cpu_jitter(pool, sizeof(pool), 112, 200, &js);
report_jitter("stage4 jitter 200ms", &js);
if (!health_ok) {
printk("ZephyrRNG: jitter health check failed, resampling 400ms\n");
health_ok = sample_cpu_jitter(pool, sizeof(pool), 112, 400);
printk("[RNG] stage4 FAILED — resampling at 400ms\n");
health_ok = sample_cpu_jitter(pool, sizeof(pool), 112, 400, &js);
report_jitter("stage4 jitter 400ms", &js);
if (!health_ok) {
printk("ZephyrRNG: jitter health still failing — continuing with mixed sources\n");
printk("[RNG] stage4 STILL FAILING — continuing with mixed sources\n");
}
}
#endif /* CONFIG_ARCH_POSIX */
/* Stage 5: late CSPRNG — catches any mid-boot radio init that
* warmed the TRNG during the 200ms jitter window */
(void)sys_csrand_get(pool + 368, 64);
int rc5 = sys_csrand_get(pool + 368, 64);
printk("[RNG] stage5 csrand : rc=%d distinct=%d/64\n",
rc5, distinct_bytes(pool + 368, 64));
/* Stage 6: second jitter sample, independent timing window */
#ifndef CONFIG_ARCH_POSIX
(void)sample_cpu_jitter(pool, sizeof(pool), 432, 50);
struct jitter_stats js6 = {};
(void)sample_cpu_jitter(pool, sizeof(pool), 432, 50, &js6);
report_jitter("stage6 jitter 50ms", &js6);
#endif /* CONFIG_ARCH_POSIX */
/* Collection done — release the SAR ADC before anything else needs it.
* Unconditional: every path below this point either returns normally or
* reboots, so there is no path that leaves it enabled. */
#if defined(CONFIG_SOC_FAMILY_ESPRESSIF_ESP32)
bootloader_random_disable();
printk("[RNG] esp32 pre-RF entropy: DISABLED (ADC released)\n");
#endif
printk("[RNG] === end (extracting %u bytes via AES-256-CTR) ===\n",
(unsigned)out_len);
/* Final conditioning: AES-256-CTR over the pool. Extracts a 32-byte
* AES key via SHA-256(pool), then expands to out_len bytes via
* AES-ECB on a 128-bit counter. Per crypto consultant guidance —
+24 -2
View File
@@ -60,8 +60,30 @@ CONFIG_BT_BUF_EVT_RX_COUNT=14
CONFIG_ESP32_BT_CTLR_LE_SECURITY_ENABLE=y
# ========== Heap ==========
# ESP32 BLE stack requires larger heap (override zephcore_common 2KB default)
CONFIG_HEAP_MEM_POOL_SIZE=32768
# ESP32 BLE stack requires larger heap (override zephcore_common 2KB default).
#
# 32768 was NOT enough and failed non-obviously. The Espressif controller
# allocates its exchange memory through esp_bt_malloc_func(), which the Zephyr
# port maps to k_malloc() (esp_heap_adapter.h, CONFIG_ESP_BT_HEAP_SYSTEM=y) —
# i.e. straight out of this pool. esp_bt_controller_init() asks for 0x7800 =
# 30720 bytes in ONE allocation: 93.75% of a 32 KB pool, leaving 2 KB for block
# overhead and every other k_malloc user that ran before bt_enable(). Boards
# that allocated little before BLE init happened to fit; ThinkNode M9 (display +
# colour overlay, STC8H keypad, LR1110) did not, and the blob asserted inside
# its own allocator instead of returning ESP_ERR_NO_MEM:
#
# BLE assert emi.c 164, param 00000000 00007800 <- (NULL, requested size)
#
# Zephyr's own driver anticipates this and prints "Consider increasing
# CONFIG_HEAP_MEM_POOL_SIZE" (drivers/bluetooth/hci/hci_esp32.c) — but only on
# the graceful path the blob does not take. Every ESP32 board was one
# allocation away from the same failure, so this is raised platform-wide.
#
# Headroom verified by build on all 12 shipped ESP32 boards: the S3/C3/C6 boards
# have 66-135 KB free internal DRAM and absorb this easily. ttgo_tbeam is the
# sole exception (99.45% DRAM, 768 B free) and overrides it back in its
# board.conf — see the note there.
CONFIG_HEAP_MEM_POOL_SIZE=65536
# ========== Flash ==========
# 4MB is standard for XIAO/LilyGo RISC-V boards.
@@ -47,13 +47,26 @@
};
};
/* Reduce CPU clock from 240 MHz to 160 MHz — HAL Kconfig minimum supported */
/* 240 MHz matches every other ESP32-S3 board in the tree (xiao_esp32s3,
* station_g2, heltec_*). This was 160 MHz with the comment "HAL Kconfig
* minimum supported", which is not correct: hal/espressif/zephyr/Kconfig
* derives ESP_DEFAULT_CPU_FREQ_MHZ straight from this DT property and offers
* 96/120/160/240 with no floor at 160. Nothing on this board requires the
* lower clock (the octal PSRAM runs at 40 MHz off its own divider, and
* station_g2 pairs the same OPI PSRAM with 240 MHz).
*
* Kept deliberate rather than "faster is better": the node is event-driven and
* idles in k_event_wait almost always, so race-to-idle makes the clock a weak
* lever on battery life, while the 320x240 colour panel is the one genuinely
* CPU-bound consumer. 240 also removes this board's only timing anomaly,
* which matters while the CPU-jitter entropy source is under investigation
* (jitter health check failed here see memory/findings.md). */
&cpu0 {
clock-frequency = <DT_FREQ_M(160)>;
clock-frequency = <DT_FREQ_M(240)>;
};
&cpu1 {
clock-frequency = <DT_FREQ_M(160)>;
clock-frequency = <DT_FREQ_M(240)>;
};
&usb_serial {
@@ -41,3 +41,17 @@ CONFIG_ESPTOOLPY_FLASHMODE_DIO=y
CONFIG_ZEPHCORE_MAX_CONTACTS=140
CONFIG_ZEPHCORE_MAX_CHANNELS=8
CONFIG_ZEPHCORE_OFFLINE_QUEUE_SIZE=128
# BLE controller heap — hold this board at the old 32 KB.
#
# esp32_common.conf raises CONFIG_HEAP_MEM_POOL_SIZE to 65536 because the
# Espressif controller allocates 30720 bytes out of it in one go and 32 KB left
# no margin (see the long note there). This board cannot pay for it: measured
# at 139684/140452 B = 99.45% DRAM, only 768 bytes free, so +32 KB does not
# link. Holding it here keeps T-Beam exactly as it builds today.
#
# NOT a fix — this board still runs the controller on the same 2 KB margin that
# broke ThinkNode M9, and its classic-ESP32 controller draws from the same pool.
# Making T-Beam safe needs DRAM reclaimed (contacts/queue are already trimmed
# above), not a larger number here. Tracked separately.
CONFIG_HEAP_MEM_POOL_SIZE=32768