tsb: reimplement TinySafeBoot on libavr in three size tiers

The native-UART fixed-baud TinySafeBoot protocol, ported onto libavr as a
crt-free boot-section loader, in three variants that trade clarity for size:

  tsb_pure   740 B  idiomatic C++: SRAM page buffer, separate flash/EEPROM
                    leaves, shared framing; the polled `unused` guard posture.
  tsb_tricks 658 B  unified runtime-flag paths (noinline/noclone), call-saved
                    global-register page walk — attributes only, no asm.
  tsb_asm    508 B  streaming store + hand-rolled UART/SPM/EEPROM/erase loops;
                    fits the 512 B boot section (BOOTSZ=11). Trims the optional
                    password gate and WDT-reset bail — unreachable in C++ with
                    both (hand-asm is ~15 % denser). Tiers 1-2 keep them and
                    live in the 1 KB section they fit.

All three are .text byte-identical across libavr's generated and reflect modes.
The CMake build strips the leaked -O3 (a Release build is silently -O3, not the
-Os this loader is measured against) and gates each variant's size against its
section. A simavr harness (test/device.c + test/tsbtest.py) drives the real wire
protocol over a pty and flashes the device; the size and protocol tests run in
ctest. Verified byte-for-byte against the reference tsbloader_adv (C#/mono):
activate, read info, flash write + verify.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-07-19 05:00:51 +02:00
parent 14ac91c815
commit 64c1e484b5
11 changed files with 1231 additions and 102 deletions

245
tsb/tsb_tricks.cpp Normal file
View File

@@ -0,0 +1,245 @@
// TinySafeBoot on libavr — tier 2: C++ with compiler trickery.
//
// Same protocol and libavr surface as the pure variant (tsb_pure.cpp), but the
// readable one-handler-per-command shape is traded for size: flash and EEPROM
// share a single code path selected by a *runtime* flag decoded from the
// command byte, so the compiler cannot constant-propagate it into two clones.
// Attributes pin that sharing down (noinline/noclone) and the hot page pointer
// and byte counter are pinned to call-saved registers to erase the prologue
// push/pop that C++ function decomposition otherwise pays. No inline assembly.
#include <libavr/libavr.hpp>
#include <avr/io.h> // SP / RAMEND for the crt-free boot entry
using namespace avr::literals;
namespace spm = avr::spm;
namespace ee = avr::eeprom;
using dev = avr::device<{.clock = 16_MHz}>;
using serial_t = dev::uart0<{.baud = 115200_Bd, .max_baud_error = 3_pct}>;
inline constexpr serial_t serial{};
namespace tsb {
constexpr auto off = avr::irq::guard_policy::unused;
constexpr std::uint8_t confirm = '!';
constexpr std::uint8_t request = '?';
constexpr std::uint16_t page = spm::page_bytes;
constexpr std::uint16_t boot_bytes = 1024;
constexpr std::uint16_t app_end = spm::flash_bytes - boot_bytes - page;
constexpr std::uint16_t eeprom_end = avr::hw::db.mem.eeprom_size - 1;
constexpr std::uint16_t build_date = 26 * 512 + 7 * 32 + 19;
// clang-format off
[[gnu::progmem]] constexpr std::uint8_t info[16] = {
'T', 'S', 'B',
build_date & 0xFF, build_date >> 8,
0xF3,
0x1E, 0x95, 0x0F,
page / 2,
(app_end / 2) & 0xFF, (app_end / 2) >> 8,
eeprom_end & 0xFF, eeprom_end >> 8,
0xAA, 0xAA,
};
// clang-format on
[[gnu::section(".noinit")]] std::uint8_t buffer[page];
// The hot page walk lives in call-saved global registers, TSB-style: g_addr is
// the running flash/EEPROM byte address, g_cnt the byte countdown. Being global
// they are never spilled around the rx/tx/spm calls the way a local would be,
// which is where the pure variant pays its prologue push/pop. r4-r7 are
// call-saved, so the library's UART/SPM helpers preserve them across calls.
register std::uint16_t g_addr asm("r4");
register std::uint8_t g_cnt asm("r6");
std::uint8_t rx()
{
for (;;)
if (auto byte = serial.read())
return *byte;
}
void tx(std::uint8_t byte)
{
serial.write(byte);
}
const std::uint8_t *flash_ptr(std::uint16_t addr)
{
return reinterpret_cast<const std::uint8_t *>(addr);
}
// One page transfer, memory selected at run time. noinline + noclone keep it a
// single shared body: the `flash` flag arrives from the command byte, so the
// optimiser cannot split it back into a flash copy and an EEPROM copy. All of
// these walk g_addr / g_cnt, set by the caller.
// Stream g_cnt bytes to the host from flash (LPM) or EEPROM, advancing g_addr
// so a caller can send consecutive pages without re-seeding it.
[[gnu::noinline, gnu::noclone]] void send(bool flash)
{
do {
tx(flash ? avr::flash_load(flash_ptr(g_addr)) : ee::read(g_addr));
++g_addr;
} while (--g_cnt);
}
// Take one page from the host into the SRAM buffer.
[[gnu::noinline]] void get_page()
{
g_cnt = 0;
do {
buffer[g_cnt] = rx();
} while (++g_cnt != page);
}
[[gnu::noinline]] bool request_confirm()
{
tx(request);
return rx() == confirm;
}
// Program the SRAM buffer into the already-erased page at g_addr (flash) or into
// EEPROM. g_addr is left on the page base for the caller to advance.
[[gnu::noinline, gnu::noclone]] void write_page(bool flash)
{
g_cnt = 0;
if (flash) {
do {
spm::fill<off>(g_addr + g_cnt, static_cast<std::uint16_t>(buffer[g_cnt] | (buffer[g_cnt + 1] << 8)));
g_cnt += 2;
} while (g_cnt != page);
spm::write_page<off>(g_addr);
spm::wait();
} else {
do {
ee::write<off>(g_addr + g_cnt, buffer[g_cnt]);
} while (++g_cnt != page);
}
}
[[noreturn]] void appjump()
{
spm::wait();
reinterpret_cast<void (*)()>(0)();
__builtin_unreachable();
}
// 'f'/'e': stream memory back one page per host '!'. send advances g_addr, so
// flash self-terminates at the application boundary; EEPROM runs until the host
// stops.
[[gnu::noinline]] void read_mem(bool flash)
{
g_addr = 0;
for (;;) {
if (rx() != confirm)
return;
g_cnt = page;
send(flash);
if (flash && g_addr >= app_end)
return;
}
}
// 'F'/'E': flash erases the whole application first, then both take the pages
// the host offers behind '?'.
[[gnu::noinline]] void write_mem(bool flash)
{
if (flash) {
g_addr = 0;
do {
spm::erase_page<off>(g_addr);
spm::wait();
g_addr += page;
} while (g_addr < app_end);
}
g_addr = 0;
while (request_confirm()) {
get_page();
write_page(flash);
g_addr += page;
}
if (flash)
spm::rww_enable<off>();
}
// 'C': replace the config page, then echo it back for the host to verify.
void write_config()
{
if (!request_confirm())
return;
get_page();
g_addr = app_end;
spm::erase_page<off>(g_addr);
spm::wait();
write_page(true);
spm::rww_enable<off>();
g_cnt = page;
send(true); // g_addr is still app_end
}
[[noreturn]] void run()
{
if (avr::hw::reg<"MCUSR">::read() & avr::hw::field<"MCUSR", "WDRF">{}(1).value)
appjump();
avr::init<serial_t>();
std::uint8_t knocks = 0;
std::uint32_t idle = 4000000;
while (knocks < 3) {
if (auto byte = serial.read())
knocks = *byte == '@' ? knocks + 1 : 0;
else if (--idle == 0)
appjump();
}
for (const std::uint8_t *pw = flash_ptr(app_end + 3); avr::flash_load(pw) != 0xff; ++pw)
if (rx() != avr::flash_load(pw))
for (;;) {
}
g_addr = reinterpret_cast<std::uint16_t>(&info[0]);
g_cnt = sizeof(info);
send(true);
for (;;) {
tx(confirm);
// Decode the command arithmetically so `flash`/`write` stay runtime
// values: bit 5 is the case bit (upper = write), and the folded-lower
// letter picks the memory. A single unified path serves f/F/e/E.
std::uint8_t cmd = rx();
std::uint8_t lower = cmd | 0x20;
bool write = (cmd & 0x20) == 0;
if (lower == 'f' || lower == 'e') {
bool flash = lower == 'f';
if (write)
write_mem(flash);
else
read_mem(flash);
} else if (lower == 'c') {
if (write) {
write_config();
} else {
g_addr = app_end;
g_cnt = page;
send(true);
}
} else {
appjump();
}
}
}
} // namespace tsb
extern "C" [[gnu::naked, gnu::used, gnu::section(".vectors")]] void __boot_entry()
{
SP = RAMEND;
tsb::run();
}