tsb: pure and tricks tiers reach full oracle feature parity

Both tiers gain the features the asm tier already carries — one-wire
half-duplex (via libavr's new .half_duplex), the config-page activation
timeout, and emergency erase (password \0 + double-confirm wipes flash,
EEPROM and the config page) — on top of the watchdog bail, password gate and
config/flash/EEPROM read-write they already had. pure stays idiomatic
(flash_table info block, one function per command) at 950 B; tricks keeps its
compiler trickery (call-saved global-register page walk, unified runtime-flag
paths pinned noinline/noclone, streaming stores, arithmetic command decode)
at 808 B. Both byte-identical across generated and reflect modes; the size
gradient across the three tiers is now 502 / 808 / 950 B.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-07-19 16:47:21 +02:00
parent 3ea8957e69
commit 3a0d3790ee
2 changed files with 180 additions and 116 deletions

View File

@@ -1,12 +1,15 @@
// TinySafeBoot on libavr — tier 2: C++ with compiler trickery.
//
// Same protocol and libavr surface as the pure variant (tsb_pure.cpp), but the
// readable one-handler-per-command shape is traded for size: flash and EEPROM
// share a single code path selected by a *runtime* flag decoded from the
// command byte, so the compiler cannot constant-propagate it into two clones.
// Attributes pin that sharing down (noinline/noclone) and the hot page pointer
// and byte counter are pinned to call-saved registers to erase the prologue
// push/pop that C++ function decomposition otherwise pays. No inline assembly.
// Same protocol, libavr surface and full feature set as the pure variant
// (tsb_pure.cpp) — watchdog bail, one-wire, config-page timeout, password gate,
// emergency erase, config/flash/EEPROM read-write — but the readable
// one-handler-per-command shape is traded for size. Flash and EEPROM share a
// single code path selected by a *runtime* flag decoded from the command byte,
// so the compiler cannot constant-propagate it into two clones; attributes
// (noinline/noclone) pin that sharing down; the hot page address and byte
// counter live in call-saved global registers to erase the prologue push/pop
// that C++ function decomposition otherwise pays; and pages stream straight to
// SPM/EEPROM with no SRAM staging. No inline assembly.
#include <libavr/libavr.hpp>
@@ -17,7 +20,7 @@ namespace spm = avr::spm;
namespace ee = avr::eeprom;
using dev = avr::device<{.clock = 16_MHz}>;
using serial_t = dev::uart0<{.baud = 115200_Bd, .max_baud_error = 3_pct}>;
using serial_t = dev::uart0<{.baud = 115200_Bd, .max_baud_error = 3_pct, .half_duplex = true}>;
inline constexpr serial_t serial{};
namespace tsb {
@@ -26,6 +29,7 @@ constexpr auto off = avr::irq::guard_policy::unused;
constexpr std::uint8_t confirm = '!';
constexpr std::uint8_t request = '?';
constexpr std::uint8_t knock = '@';
constexpr std::uint16_t page = spm::page_bytes;
constexpr std::uint16_t boot_bytes = 1024;
@@ -47,21 +51,16 @@ constexpr std::uint16_t build_date = 26 * 512 + 7 * 32 + 19;
};
// clang-format on
[[gnu::section(".noinit")]] std::uint8_t buffer[page];
// The hot page walk lives in call-saved global registers, TSB-style: g_addr is
// the running flash/EEPROM byte address, g_cnt the byte countdown. Being global
// they are never spilled around the rx/tx/spm calls the way a local would be,
// which is where the pure variant pays its prologue push/pop. r4-r7 are
// call-saved, so the library's UART/SPM helpers preserve them across calls.
// they are never spilled around the rx/tx/spm calls the way a local would be
// r4-r7 are call-saved, so the library's UART and SPM helpers preserve them.
register std::uint16_t g_addr asm("r4");
register std::uint8_t g_cnt asm("r6");
std::uint8_t rx()
{
for (;;)
if (auto byte = serial.read())
return *byte;
return serial.read_blocking();
}
void tx(std::uint8_t byte)
@@ -74,13 +73,8 @@ const std::uint8_t *flash_ptr(std::uint16_t addr)
return reinterpret_cast<const std::uint8_t *>(addr);
}
// One page transfer, memory selected at run time. noinline + noclone keep it a
// single shared body: the `flash` flag arrives from the command byte, so the
// optimiser cannot split it back into a flash copy and an EEPROM copy. All of
// these walk g_addr / g_cnt, set by the caller.
// Stream g_cnt bytes to the host from flash (LPM) or EEPROM, advancing g_addr
// so a caller can send consecutive pages without re-seeding it.
// Stream g_cnt bytes to the host from flash (LPM) or EEPROM, memory chosen at
// run time so the optimiser cannot split the loop into two clones.
[[gnu::noinline, gnu::noclone]] void send(bool flash)
{
do {
@@ -89,36 +83,30 @@ const std::uint8_t *flash_ptr(std::uint16_t addr)
} while (--g_cnt);
}
// Take one page from the host into the SRAM buffer.
[[gnu::noinline]] void get_page()
{
g_cnt = 0;
do {
buffer[g_cnt] = rx();
} while (++g_cnt != page);
}
[[gnu::noinline]] bool request_confirm()
{
tx(request);
return rx() == confirm;
}
// Program the SRAM buffer into the already-erased page at g_addr (flash) or into
// EEPROM. g_addr is left on the page base for the caller to advance.
[[gnu::noinline, gnu::noclone]] void write_page(bool flash)
// Stream one page straight from the host into the already-erased flash page at
// g_addr (SPM word buffer, low byte then high) or into EEPROM — no SRAM staging,
// so receive and store are one loop. The memory is a run-time flag.
[[gnu::noinline, gnu::noclone]] void store_page(bool flash)
{
g_cnt = 0;
if (flash) {
do {
spm::fill<off>(g_addr + g_cnt, static_cast<std::uint16_t>(buffer[g_cnt] | (buffer[g_cnt + 1] << 8)));
std::uint8_t lo = rx();
std::uint8_t hi = rx();
spm::fill<off>(g_addr + g_cnt, static_cast<std::uint16_t>(lo | (hi << 8)));
g_cnt += 2;
} while (g_cnt != page);
spm::write_page<off>(g_addr);
spm::wait();
} else {
do {
ee::write<off>(g_addr + g_cnt, buffer[g_cnt]);
ee::write<off>(g_addr + g_cnt, rx());
} while (++g_cnt != page);
}
}
@@ -130,6 +118,18 @@ const std::uint8_t *flash_ptr(std::uint16_t addr)
__builtin_unreachable();
}
// Erase the whole application, one page at a time.
[[gnu::noinline]] void erase_application()
{
g_addr = 0;
do {
spm::erase_page<off>(g_addr);
spm::wait();
g_addr += page;
} while (g_addr < app_end);
spm::rww_enable<off>();
}
// 'f'/'e': stream memory back one page per host '!'. send advances g_addr, so
// flash self-terminates at the application boundary; EEPROM runs until the host
// stops.
@@ -150,22 +150,13 @@ const std::uint8_t *flash_ptr(std::uint16_t addr)
// the host offers behind '?'.
[[gnu::noinline]] void write_mem(bool flash)
{
if (flash) {
g_addr = 0;
do {
spm::erase_page<off>(g_addr);
spm::wait();
g_addr += page;
} while (g_addr < app_end);
}
if (flash)
erase_application();
g_addr = 0;
while (request_confirm()) {
get_page();
write_page(flash);
store_page(flash);
g_addr += page;
}
if (flash)
spm::rww_enable<off>();
}
// 'C': replace the config page, then echo it back for the host to verify.
@@ -173,46 +164,82 @@ void write_config()
{
if (!request_confirm())
return;
get_page();
g_addr = app_end;
spm::erase_page<off>(g_addr);
spm::wait();
write_page(true);
store_page(true);
spm::rww_enable<off>();
g_addr = app_end;
g_cnt = page;
send(true); // g_addr is still app_end
send(true);
}
// Emergency erase: wipe the application flash, the EEPROM and the config page.
[[gnu::noinline]] void emergency_erase()
{
erase_application();
g_addr = 0;
do {
ee::write<off>(g_addr, 0xff);
} while (++g_addr <= eeprom_end);
spm::erase_page<off>(app_end);
spm::wait();
spm::rww_enable<off>();
}
// The password gate. A byte of 0 requests emergency erase; a wrong byte hangs
// the loader (still draining the line), so it can never fall through to erase.
enum class gate : std::uint8_t { pass, emergency };
[[gnu::noinline]] gate password_gate()
{
for (const std::uint8_t *pw = flash_ptr(app_end + 3);; ++pw) {
std::uint8_t expected = avr::flash_load(pw);
if (expected == 0xff)
return gate::pass;
std::uint8_t got = rx();
if (got == 0)
return gate::emergency;
if (got != expected)
for (;;)
rx();
}
}
[[noreturn]] void run()
{
if (avr::hw::mcusr::read() & avr::hw::mcusr::wdrf(1).value)
if (avr::hw::mcusr::wdrf.test())
appjump();
avr::init<serial_t>();
std::uint32_t idle = static_cast<std::uint32_t>(avr::flash_load(flash_ptr(app_end + 2)) | 16) << 16;
std::uint8_t knocks = 0;
std::uint32_t idle = 4000000;
while (knocks < 3) {
if (auto byte = serial.read())
knocks = *byte == '@' ? knocks + 1 : 0;
knocks = *byte == knock ? knocks + 1 : 0;
else if (--idle == 0)
appjump();
}
for (const std::uint8_t *pw = flash_ptr(app_end + 3); avr::flash_load(pw) != 0xff; ++pw)
if (rx() != avr::flash_load(pw))
for (;;) {
}
g_addr = reinterpret_cast<std::uint16_t>(&info[0]);
g_cnt = sizeof(info);
send(true);
switch (password_gate()) {
case gate::pass:
g_addr = reinterpret_cast<std::uint16_t>(&info[0]);
g_cnt = sizeof(info);
send(true);
break;
case gate::emergency:
if (!request_confirm() || !request_confirm())
appjump();
emergency_erase();
break;
}
for (;;) {
tx(confirm);
// Decode the command arithmetically so `flash`/`write` stay runtime
// values: bit 5 is the case bit (upper = write), and the folded-lower
// letter picks the memory. A single unified path serves f/F/e/E.
tx(confirm); // Mainloop ready
// Decode the command arithmetically so flash/write stay run-time values:
// bit 5 is the case bit (upper = write), the folded-lower letter picks the
// memory. A single unified path serves f/F/e/E.
std::uint8_t cmd = rx();
std::uint8_t lower = cmd | 0x20;
bool write = (cmd & 0x20) == 0;