tsb: reimplement TinySafeBoot on libavr in three size tiers
The native-UART fixed-baud TinySafeBoot protocol, ported onto libavr as a
crt-free boot-section loader, in three variants that trade clarity for size:
tsb_pure 740 B idiomatic C++: SRAM page buffer, separate flash/EEPROM
leaves, shared framing; the polled `unused` guard posture.
tsb_tricks 658 B unified runtime-flag paths (noinline/noclone), call-saved
global-register page walk — attributes only, no asm.
tsb_asm 508 B streaming store + hand-rolled UART/SPM/EEPROM/erase loops;
fits the 512 B boot section (BOOTSZ=11). Trims the optional
password gate and WDT-reset bail — unreachable in C++ with
both (hand-asm is ~15 % denser). Tiers 1-2 keep them and
live in the 1 KB section they fit.
All three are .text byte-identical across libavr's generated and reflect modes.
The CMake build strips the leaked -O3 (a Release build is silently -O3, not the
-Os this loader is measured against) and gates each variant's size against its
section. A simavr harness (test/device.c + test/tsbtest.py) drives the real wire
protocol over a pty and flashes the device; the size and protocol tests run in
ctest. Verified byte-for-byte against the reference tsbloader_adv (C#/mono):
activate, read info, flash write + verify.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
297
tsb/tsb_asm.cpp
Normal file
297
tsb/tsb_asm.cpp
Normal file
@@ -0,0 +1,297 @@
|
||||
// TinySafeBoot on libavr — tier 3: C++ with minimal inline assembly.
|
||||
//
|
||||
// The tier-2 structure (unified runtime-flag paths, global-register page walk)
|
||||
// with its hottest primitives — the UART poll/read/write and the SPM word/page
|
||||
// stores — written as small, self-contained inline-asm sequences. Everything
|
||||
// above them (command dispatch, activation, the page loops) stays C++. This is
|
||||
// the ≤512-byte boot-section deliverable.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <avr/io.h> // SP / RAMEND for the crt-free boot entry
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
|
||||
namespace tsb {
|
||||
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t confirm = '!';
|
||||
constexpr std::uint8_t request = '?';
|
||||
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr std::uint16_t boot_bytes = 512;
|
||||
constexpr std::uint16_t app_end = spm::flash_bytes - boot_bytes - page;
|
||||
constexpr std::uint16_t eeprom_end = avr::hw::db.mem.eeprom_size - 1;
|
||||
|
||||
constexpr std::uint16_t build_date = 26 * 512 + 7 * 32 + 19;
|
||||
|
||||
// clang-format off
|
||||
[[gnu::progmem]] constexpr std::uint8_t info[16] = {
|
||||
'T', 'S', 'B',
|
||||
build_date & 0xFF, build_date >> 8,
|
||||
0xF3,
|
||||
0x1E, 0x95, 0x0F,
|
||||
page / 2,
|
||||
(app_end / 2) & 0xFF, (app_end / 2) >> 8,
|
||||
eeprom_end & 0xFF, eeprom_end >> 8,
|
||||
0xAA, 0xAA,
|
||||
};
|
||||
// clang-format on
|
||||
|
||||
// The hot page walk lives in call-saved global registers, TSB-style: g_addr is
|
||||
// the running flash/EEPROM byte address, g_cnt the byte countdown. Being global
|
||||
// they are never spilled around the rx/tx/spm calls the way a local would be,
|
||||
// which is where the pure variant pays its prologue push/pop. r4-r7 are
|
||||
// call-saved, so the library's SPM helpers preserve them across calls.
|
||||
register std::uint16_t g_addr asm("r4");
|
||||
register std::uint8_t g_cnt asm("r6");
|
||||
|
||||
// rx/tx carry fixed assembler names so the hand-rolled loops can `rcall` them;
|
||||
// noinline keeps every caller funneling through the one shared copy (rx also
|
||||
// preserves Z/r0, which the store/send loops rely on across the call).
|
||||
[[gnu::used, gnu::noinline]] std::uint8_t rx() asm("tsb_rx");
|
||||
[[gnu::used, gnu::noinline]] void tx(std::uint8_t) asm("tsb_tx");
|
||||
|
||||
// Blocking receive: spin on RXC0, then take UDR0. The driver's read() returns a
|
||||
// std::optional for non-blocking use; a bootloader only ever blocks, so the tight
|
||||
// poll drops the option's has-value plumbing.
|
||||
std::uint8_t rx()
|
||||
{
|
||||
std::uint8_t byte;
|
||||
asm volatile("%=: lds %0, %[sra] \n\t"
|
||||
" sbrs %0, %[rxc] \n\t"
|
||||
" rjmp %=b \n\t"
|
||||
" lds %0, %[udr] \n\t"
|
||||
: "=&r"(byte)
|
||||
: [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [rxc] "I"(RXC0), [udr] "n"(_SFR_MEM_ADDR(UDR0)));
|
||||
return byte;
|
||||
}
|
||||
|
||||
// Blocking transmit: spin on UDRE0, then store UDR0.
|
||||
void tx(std::uint8_t byte)
|
||||
{
|
||||
asm volatile("%=: lds __tmp_reg__, %[sra] \n\t"
|
||||
" sbrs __tmp_reg__, %[udre] \n\t"
|
||||
" rjmp %=b \n\t"
|
||||
" sts %[udr], %[b] \n\t"
|
||||
:
|
||||
: [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [udre] "I"(UDRE0), [udr] "n"(_SFR_MEM_ADDR(UDR0)), [b] "r"(byte));
|
||||
}
|
||||
|
||||
const std::uint8_t *flash_ptr(std::uint16_t addr)
|
||||
{
|
||||
return reinterpret_cast<const std::uint8_t *>(addr);
|
||||
}
|
||||
|
||||
// One page transfer, memory selected at run time. noinline + noclone keep it a
|
||||
// single shared body: the `flash` flag arrives from the command byte, so the
|
||||
// optimiser cannot split it back into a flash copy and an EEPROM copy. All of
|
||||
// these walk g_addr / g_cnt, set by the caller.
|
||||
|
||||
// Stream g_cnt bytes to the host from flash (LPM) or EEPROM, advancing g_addr so
|
||||
// a caller can send consecutive pages without re-seeding it. GCC's unified loop
|
||||
// (one body, per-byte memory branch) is already smaller than a split asm pair,
|
||||
// so this one stays C++.
|
||||
[[gnu::noinline, gnu::noclone]] void send(bool flash)
|
||||
{
|
||||
do {
|
||||
tx(flash ? avr::flash_load(flash_ptr(g_addr)) : ee::read(g_addr));
|
||||
++g_addr;
|
||||
} while (--g_cnt);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] bool request_confirm()
|
||||
{
|
||||
tx(request);
|
||||
return rx() == confirm;
|
||||
}
|
||||
|
||||
// Stream one page from the host straight into the already-erased flash page at
|
||||
// g_addr (SPM word buffer, low byte then high) or into EEPROM — no SRAM staging,
|
||||
// so the receive and the store are one loop instead of two. The flash fill is
|
||||
// hand-rolled asm: Z the flash word address, the word received into r0:r1 via
|
||||
// the tiny rcall'd rx (which preserves Z), then committed by avr-libc.
|
||||
[[gnu::noinline, gnu::noclone]] void store_page(bool flash)
|
||||
{
|
||||
if (flash) {
|
||||
std::uint8_t words = page / 2;
|
||||
asm volatile(" movw r30, %[base] \n\t"
|
||||
"%=: rcall tsb_rx \n\t"
|
||||
" mov r0, r24 \n\t"
|
||||
" rcall tsb_rx \n\t"
|
||||
" mov r1, r24 \n\t"
|
||||
" ldi r25, %[fill] \n\t"
|
||||
" out %[spmcsr], r25 \n\t"
|
||||
" spm \n\t"
|
||||
" clr r1 \n\t"
|
||||
" adiw r30, 2 \n\t"
|
||||
" dec %[words] \n\t"
|
||||
" brne %=b \n\t"
|
||||
: [words] "+d"(words)
|
||||
: [base] "r"(g_addr), [fill] "M"(_BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR))
|
||||
: "r24", "r25", "r30", "r31", "memory");
|
||||
spm::write_page<off>(g_addr);
|
||||
spm::wait();
|
||||
} else {
|
||||
// EEPROM: rx each byte straight into the cell array, X the running
|
||||
// address. The tight EEMPE→EEPE strobe replaces the library's wider
|
||||
// atomic write (which the interrupt-driven queue and split modes need).
|
||||
std::uint8_t cnt = page;
|
||||
asm volatile(
|
||||
" movw r26, %[a] \n\t"
|
||||
"%=: rcall tsb_rx \n\t"
|
||||
"0: sbic %[eecr], %[eepe] \n\t"
|
||||
" rjmp 0b \n\t"
|
||||
" out %[eedr], r24 \n\t"
|
||||
" out %[earl], r26 \n\t"
|
||||
" out %[earh], r27 \n\t"
|
||||
" sbi %[eecr], %[eempe] \n\t"
|
||||
" sbi %[eecr], %[eepe] \n\t"
|
||||
" adiw r26, 1 \n\t"
|
||||
" dec %[c] \n\t"
|
||||
" brne %=b \n\t"
|
||||
: [c] "+d"(cnt)
|
||||
: [a] "r"(g_addr), [eecr] "I"(_SFR_IO_ADDR(EECR)), [eedr] "I"(_SFR_IO_ADDR(EEDR)),
|
||||
[earl] "I"(_SFR_IO_ADDR(EEARL)), [earh] "I"(_SFR_IO_ADDR(EEARH)), [eepe] "I"(EEPE), [eempe] "I"(EEMPE)
|
||||
: "r24", "r26", "r27");
|
||||
}
|
||||
}
|
||||
|
||||
[[noreturn]] void appjump()
|
||||
{
|
||||
spm::wait();
|
||||
asm volatile("jmp 0"); // hand over to the application reset vector at 0x0000
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
// 'f'/'e': stream memory back one page per host '!'. send advances g_addr, so
|
||||
// flash self-terminates at the application boundary; EEPROM runs until the host
|
||||
// stops.
|
||||
[[gnu::noinline]] void read_mem(bool flash)
|
||||
{
|
||||
g_addr = 0;
|
||||
for (;;) {
|
||||
if (rx() != confirm)
|
||||
return;
|
||||
g_cnt = page;
|
||||
send(flash);
|
||||
if (flash && g_addr >= app_end)
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// 'F'/'E': flash erases the whole application first, then both take the pages
|
||||
// the host offers behind '?'.
|
||||
[[gnu::noinline]] void write_mem(bool flash)
|
||||
{
|
||||
if (flash) {
|
||||
// Erase every application page [0, app_end) with Z the running byte
|
||||
// address and the busy-wait inline — avoids the Y juggling GCC needs to
|
||||
// step the non-adiw'able global address, and its prologue push/pop.
|
||||
asm volatile(" clr r30 \n\t"
|
||||
" clr r31 \n\t"
|
||||
"%=: ldi r25, %[ers] \n\t"
|
||||
" out %[spmcsr], r25 \n\t"
|
||||
" spm \n\t"
|
||||
"0: in r25, %[spmcsr] \n\t"
|
||||
" sbrc r25, 0 \n\t"
|
||||
" rjmp 0b \n\t"
|
||||
" subi r30, 0x80 \n\t"
|
||||
" sbci r31, 0xFF \n\t"
|
||||
" cpi r30, lo8(%[end]) \n\t"
|
||||
" ldi r25, hi8(%[end]) \n\t"
|
||||
" cpc r31, r25 \n\t"
|
||||
" brlo %=b \n\t"
|
||||
:
|
||||
: [ers] "M"(_BV(PGERS) | _BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR)), [end] "i"(app_end)
|
||||
: "r25", "r30", "r31");
|
||||
}
|
||||
g_addr = 0;
|
||||
while (request_confirm()) {
|
||||
store_page(flash);
|
||||
g_addr += page;
|
||||
}
|
||||
if (flash)
|
||||
spm::rww_enable<off>();
|
||||
}
|
||||
|
||||
// 'C': replace the config page, then echo it back for the host to verify.
|
||||
void write_config()
|
||||
{
|
||||
if (!request_confirm())
|
||||
return;
|
||||
g_addr = app_end;
|
||||
spm::erase_page<off>(g_addr);
|
||||
spm::wait();
|
||||
store_page(true);
|
||||
spm::rww_enable<off>();
|
||||
g_cnt = page;
|
||||
send(true); // g_addr is still app_end
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
// Minimal 115200 8N1 bring-up: 8N1 is the UCSR0C reset value, so only U2X0,
|
||||
// UBRR0 (16 at 16 MHz → 2.1 % error) and the RX/TX enables need writing — the
|
||||
// driver's avr::init also programs UCSR0C.
|
||||
avr::hw::reg<"UCSR0A">::write(avr::hw::field<"UCSR0A", "U2X0">{}(1).value);
|
||||
avr::hw::reg<"UBRR0">::write16(16);
|
||||
avr::hw::reg<"UCSR0B">::write(static_cast<std::uint8_t>(avr::hw::field<"UCSR0B", "RXEN0">{}(1).value |
|
||||
avr::hw::field<"UCSR0B", "TXEN0">{}(1).value));
|
||||
|
||||
// The password gate that the canonical loader carries (compare host bytes
|
||||
// against the config page, hang on mismatch) is dropped here: it is optional
|
||||
// (a blank config page means no password, the usual case) and its ~26 bytes
|
||||
// are what a C++ build cannot spare inside the 512-byte boot section. Tiers 1
|
||||
// and 2 keep it; this asm variant trades it for the size budget.
|
||||
std::uint8_t knocks = 0;
|
||||
std::uint16_t idle = 0xFFFF;
|
||||
while (knocks < 3) {
|
||||
if (avr::hw::reg<"UCSR0A">::read() & avr::hw::field<"UCSR0A", "RXC0">{}(1).value)
|
||||
knocks = avr::hw::reg<"UDR0">::read() == '@' ? knocks + 1 : 0;
|
||||
else if (--idle == 0)
|
||||
appjump();
|
||||
}
|
||||
|
||||
g_addr = reinterpret_cast<std::uint16_t>(&info[0]);
|
||||
g_cnt = sizeof(info);
|
||||
send(true);
|
||||
|
||||
for (;;) {
|
||||
tx(confirm);
|
||||
// Decode the command arithmetically so `flash`/`write` stay runtime
|
||||
// values: bit 5 is the case bit (upper = write), and the folded-lower
|
||||
// letter picks the memory. A single unified path serves f/F/e/E.
|
||||
std::uint8_t cmd = rx();
|
||||
std::uint8_t lower = cmd | 0x20;
|
||||
bool write = (cmd & 0x20) == 0;
|
||||
if (lower == 'f' || lower == 'e') {
|
||||
bool flash = lower == 'f';
|
||||
if (write)
|
||||
write_mem(flash);
|
||||
else
|
||||
read_mem(flash);
|
||||
} else if (lower == 'c') {
|
||||
if (write) {
|
||||
write_config();
|
||||
} else {
|
||||
g_addr = app_end;
|
||||
g_cnt = page;
|
||||
send(true);
|
||||
}
|
||||
} else {
|
||||
appjump();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace tsb
|
||||
|
||||
extern "C" [[gnu::naked, gnu::used, gnu::section(".vectors")]] void __boot_entry()
|
||||
{
|
||||
SP = RAMEND;
|
||||
tsb::run();
|
||||
}
|
||||
Reference in New Issue
Block a user