Direct register access now reads through the named surface (hw::mcusr::wdrf.test(), hw::ucsr0b::write(...)) instead of the string form, matching how libavr itself is written. Zero-overhead: pure 740 B, tricks 658 B, asm 508 B unchanged, all byte-identical across modes, protocol green. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
297 lines
9.8 KiB
C++
297 lines
9.8 KiB
C++
// TinySafeBoot on libavr — tier 3: C++ with minimal inline assembly.
|
|
//
|
|
// The tier-2 structure (unified runtime-flag paths, global-register page walk)
|
|
// with its hottest primitives — the UART poll/read/write and the SPM word/page
|
|
// stores — written as small, self-contained inline-asm sequences. Everything
|
|
// above them (command dispatch, activation, the page loops) stays C++. This is
|
|
// the ≤512-byte boot-section deliverable.
|
|
|
|
#include <libavr/libavr.hpp>
|
|
|
|
#include <avr/io.h> // SP / RAMEND for the crt-free boot entry
|
|
|
|
using namespace avr::literals;
|
|
namespace spm = avr::spm;
|
|
namespace ee = avr::eeprom;
|
|
|
|
namespace tsb {
|
|
|
|
constexpr auto off = avr::irq::guard_policy::unused;
|
|
|
|
constexpr std::uint8_t confirm = '!';
|
|
constexpr std::uint8_t request = '?';
|
|
|
|
constexpr std::uint16_t page = spm::page_bytes;
|
|
constexpr std::uint16_t boot_bytes = 512;
|
|
constexpr std::uint16_t app_end = spm::flash_bytes - boot_bytes - page;
|
|
constexpr std::uint16_t eeprom_end = avr::hw::db.mem.eeprom_size - 1;
|
|
|
|
constexpr std::uint16_t build_date = 26 * 512 + 7 * 32 + 19;
|
|
|
|
// clang-format off
|
|
[[gnu::progmem]] constexpr std::uint8_t info[16] = {
|
|
'T', 'S', 'B',
|
|
build_date & 0xFF, build_date >> 8,
|
|
0xF3,
|
|
0x1E, 0x95, 0x0F,
|
|
page / 2,
|
|
(app_end / 2) & 0xFF, (app_end / 2) >> 8,
|
|
eeprom_end & 0xFF, eeprom_end >> 8,
|
|
0xAA, 0xAA,
|
|
};
|
|
// clang-format on
|
|
|
|
// The hot page walk lives in call-saved global registers, TSB-style: g_addr is
|
|
// the running flash/EEPROM byte address, g_cnt the byte countdown. Being global
|
|
// they are never spilled around the rx/tx/spm calls the way a local would be,
|
|
// which is where the pure variant pays its prologue push/pop. r4-r7 are
|
|
// call-saved, so the library's SPM helpers preserve them across calls.
|
|
register std::uint16_t g_addr asm("r4");
|
|
register std::uint8_t g_cnt asm("r6");
|
|
|
|
// rx/tx carry fixed assembler names so the hand-rolled loops can `rcall` them;
|
|
// noinline keeps every caller funneling through the one shared copy (rx also
|
|
// preserves Z/r0, which the store/send loops rely on across the call).
|
|
[[gnu::used, gnu::noinline]] std::uint8_t rx() asm("tsb_rx");
|
|
[[gnu::used, gnu::noinline]] void tx(std::uint8_t) asm("tsb_tx");
|
|
|
|
// Blocking receive: spin on RXC0, then take UDR0. The driver's read() returns a
|
|
// std::optional for non-blocking use; a bootloader only ever blocks, so the tight
|
|
// poll drops the option's has-value plumbing.
|
|
std::uint8_t rx()
|
|
{
|
|
std::uint8_t byte;
|
|
asm volatile("%=: lds %0, %[sra] \n\t"
|
|
" sbrs %0, %[rxc] \n\t"
|
|
" rjmp %=b \n\t"
|
|
" lds %0, %[udr] \n\t"
|
|
: "=&r"(byte)
|
|
: [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [rxc] "I"(RXC0), [udr] "n"(_SFR_MEM_ADDR(UDR0)));
|
|
return byte;
|
|
}
|
|
|
|
// Blocking transmit: spin on UDRE0, then store UDR0.
|
|
void tx(std::uint8_t byte)
|
|
{
|
|
asm volatile("%=: lds __tmp_reg__, %[sra] \n\t"
|
|
" sbrs __tmp_reg__, %[udre] \n\t"
|
|
" rjmp %=b \n\t"
|
|
" sts %[udr], %[b] \n\t"
|
|
:
|
|
: [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [udre] "I"(UDRE0), [udr] "n"(_SFR_MEM_ADDR(UDR0)), [b] "r"(byte));
|
|
}
|
|
|
|
const std::uint8_t *flash_ptr(std::uint16_t addr)
|
|
{
|
|
return reinterpret_cast<const std::uint8_t *>(addr);
|
|
}
|
|
|
|
// One page transfer, memory selected at run time. noinline + noclone keep it a
|
|
// single shared body: the `flash` flag arrives from the command byte, so the
|
|
// optimiser cannot split it back into a flash copy and an EEPROM copy. All of
|
|
// these walk g_addr / g_cnt, set by the caller.
|
|
|
|
// Stream g_cnt bytes to the host from flash (LPM) or EEPROM, advancing g_addr so
|
|
// a caller can send consecutive pages without re-seeding it. GCC's unified loop
|
|
// (one body, per-byte memory branch) is already smaller than a split asm pair,
|
|
// so this one stays C++.
|
|
[[gnu::noinline, gnu::noclone]] void send(bool flash)
|
|
{
|
|
do {
|
|
tx(flash ? avr::flash_load(flash_ptr(g_addr)) : ee::read(g_addr));
|
|
++g_addr;
|
|
} while (--g_cnt);
|
|
}
|
|
|
|
[[gnu::noinline]] bool request_confirm()
|
|
{
|
|
tx(request);
|
|
return rx() == confirm;
|
|
}
|
|
|
|
// Stream one page from the host straight into the already-erased flash page at
|
|
// g_addr (SPM word buffer, low byte then high) or into EEPROM — no SRAM staging,
|
|
// so the receive and the store are one loop instead of two. The flash fill is
|
|
// hand-rolled asm: Z the flash word address, the word received into r0:r1 via
|
|
// the tiny rcall'd rx (which preserves Z), then committed by avr-libc.
|
|
[[gnu::noinline, gnu::noclone]] void store_page(bool flash)
|
|
{
|
|
if (flash) {
|
|
std::uint8_t words = page / 2;
|
|
asm volatile(" movw r30, %[base] \n\t"
|
|
"%=: rcall tsb_rx \n\t"
|
|
" mov r0, r24 \n\t"
|
|
" rcall tsb_rx \n\t"
|
|
" mov r1, r24 \n\t"
|
|
" ldi r25, %[fill] \n\t"
|
|
" out %[spmcsr], r25 \n\t"
|
|
" spm \n\t"
|
|
" clr r1 \n\t"
|
|
" adiw r30, 2 \n\t"
|
|
" dec %[words] \n\t"
|
|
" brne %=b \n\t"
|
|
: [words] "+d"(words)
|
|
: [base] "r"(g_addr), [fill] "M"(_BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR))
|
|
: "r24", "r25", "r30", "r31", "memory");
|
|
spm::write_page<off>(g_addr);
|
|
spm::wait();
|
|
} else {
|
|
// EEPROM: rx each byte straight into the cell array, X the running
|
|
// address. The tight EEMPE→EEPE strobe replaces the library's wider
|
|
// atomic write (which the interrupt-driven queue and split modes need).
|
|
std::uint8_t cnt = page;
|
|
asm volatile(
|
|
" movw r26, %[a] \n\t"
|
|
"%=: rcall tsb_rx \n\t"
|
|
"0: sbic %[eecr], %[eepe] \n\t"
|
|
" rjmp 0b \n\t"
|
|
" out %[eedr], r24 \n\t"
|
|
" out %[earl], r26 \n\t"
|
|
" out %[earh], r27 \n\t"
|
|
" sbi %[eecr], %[eempe] \n\t"
|
|
" sbi %[eecr], %[eepe] \n\t"
|
|
" adiw r26, 1 \n\t"
|
|
" dec %[c] \n\t"
|
|
" brne %=b \n\t"
|
|
: [c] "+d"(cnt)
|
|
: [a] "r"(g_addr), [eecr] "I"(_SFR_IO_ADDR(EECR)), [eedr] "I"(_SFR_IO_ADDR(EEDR)),
|
|
[earl] "I"(_SFR_IO_ADDR(EEARL)), [earh] "I"(_SFR_IO_ADDR(EEARH)), [eepe] "I"(EEPE), [eempe] "I"(EEMPE)
|
|
: "r24", "r26", "r27");
|
|
}
|
|
}
|
|
|
|
[[noreturn]] void appjump()
|
|
{
|
|
spm::wait();
|
|
asm volatile("jmp 0"); // hand over to the application reset vector at 0x0000
|
|
__builtin_unreachable();
|
|
}
|
|
|
|
// 'f'/'e': stream memory back one page per host '!'. send advances g_addr, so
|
|
// flash self-terminates at the application boundary; EEPROM runs until the host
|
|
// stops.
|
|
[[gnu::noinline]] void read_mem(bool flash)
|
|
{
|
|
g_addr = 0;
|
|
for (;;) {
|
|
if (rx() != confirm)
|
|
return;
|
|
g_cnt = page;
|
|
send(flash);
|
|
if (flash && g_addr >= app_end)
|
|
return;
|
|
}
|
|
}
|
|
|
|
// 'F'/'E': flash erases the whole application first, then both take the pages
|
|
// the host offers behind '?'.
|
|
[[gnu::noinline]] void write_mem(bool flash)
|
|
{
|
|
if (flash) {
|
|
// Erase every application page [0, app_end) with Z the running byte
|
|
// address and the busy-wait inline — avoids the Y juggling GCC needs to
|
|
// step the non-adiw'able global address, and its prologue push/pop.
|
|
asm volatile(" clr r30 \n\t"
|
|
" clr r31 \n\t"
|
|
"%=: ldi r25, %[ers] \n\t"
|
|
" out %[spmcsr], r25 \n\t"
|
|
" spm \n\t"
|
|
"0: in r25, %[spmcsr] \n\t"
|
|
" sbrc r25, 0 \n\t"
|
|
" rjmp 0b \n\t"
|
|
" subi r30, 0x80 \n\t"
|
|
" sbci r31, 0xFF \n\t"
|
|
" cpi r30, lo8(%[end]) \n\t"
|
|
" ldi r25, hi8(%[end]) \n\t"
|
|
" cpc r31, r25 \n\t"
|
|
" brlo %=b \n\t"
|
|
:
|
|
: [ers] "M"(_BV(PGERS) | _BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR)), [end] "i"(app_end)
|
|
: "r25", "r30", "r31");
|
|
}
|
|
g_addr = 0;
|
|
while (request_confirm()) {
|
|
store_page(flash);
|
|
g_addr += page;
|
|
}
|
|
if (flash)
|
|
spm::rww_enable<off>();
|
|
}
|
|
|
|
// 'C': replace the config page, then echo it back for the host to verify.
|
|
void write_config()
|
|
{
|
|
if (!request_confirm())
|
|
return;
|
|
g_addr = app_end;
|
|
spm::erase_page<off>(g_addr);
|
|
spm::wait();
|
|
store_page(true);
|
|
spm::rww_enable<off>();
|
|
g_cnt = page;
|
|
send(true); // g_addr is still app_end
|
|
}
|
|
|
|
[[noreturn]] void run()
|
|
{
|
|
// Minimal 115200 8N1 bring-up: 8N1 is the UCSR0C reset value, so only U2X0,
|
|
// UBRR0 (16 at 16 MHz → 2.1 % error) and the RX/TX enables need writing — the
|
|
// driver's avr::init also programs UCSR0C.
|
|
avr::hw::ucsr0a::write(avr::hw::ucsr0a::u2x0(1).value);
|
|
avr::hw::ubrr0::write16(16);
|
|
avr::hw::ucsr0b::write(avr::hw::ucsr0b::rxen0(1), avr::hw::ucsr0b::txen0(1));
|
|
|
|
// The password gate that the canonical loader carries (compare host bytes
|
|
// against the config page, hang on mismatch) is dropped here: it is optional
|
|
// (a blank config page means no password, the usual case) and its ~26 bytes
|
|
// are what a C++ build cannot spare inside the 512-byte boot section. Tiers 1
|
|
// and 2 keep it; this asm variant trades it for the size budget.
|
|
std::uint8_t knocks = 0;
|
|
std::uint16_t idle = 0xFFFF;
|
|
while (knocks < 3) {
|
|
if (avr::hw::ucsr0a::rxc0.test())
|
|
knocks = avr::hw::udr0::read() == '@' ? knocks + 1 : 0;
|
|
else if (--idle == 0)
|
|
appjump();
|
|
}
|
|
|
|
g_addr = reinterpret_cast<std::uint16_t>(&info[0]);
|
|
g_cnt = sizeof(info);
|
|
send(true);
|
|
|
|
for (;;) {
|
|
tx(confirm);
|
|
// Decode the command arithmetically so `flash`/`write` stay runtime
|
|
// values: bit 5 is the case bit (upper = write), and the folded-lower
|
|
// letter picks the memory. A single unified path serves f/F/e/E.
|
|
std::uint8_t cmd = rx();
|
|
std::uint8_t lower = cmd | 0x20;
|
|
bool write = (cmd & 0x20) == 0;
|
|
if (lower == 'f' || lower == 'e') {
|
|
bool flash = lower == 'f';
|
|
if (write)
|
|
write_mem(flash);
|
|
else
|
|
read_mem(flash);
|
|
} else if (lower == 'c') {
|
|
if (write) {
|
|
write_config();
|
|
} else {
|
|
g_addr = app_end;
|
|
g_cnt = page;
|
|
send(true);
|
|
}
|
|
} else {
|
|
appjump();
|
|
}
|
|
}
|
|
}
|
|
|
|
} // namespace tsb
|
|
|
|
extern "C" [[gnu::naked, gnu::used, gnu::section(".vectors")]] void __boot_entry()
|
|
{
|
|
SP = RAMEND;
|
|
tsb::run();
|
|
}
|