Files
bootloader/tsb/tsb_asm.cpp
BlackMark f32a27ff15 tsb: use the named register surface
Direct register access now reads through the named surface
(hw::mcusr::wdrf.test(), hw::ucsr0b::write(...)) instead of the string form,
matching how libavr itself is written. Zero-overhead: pure 740 B, tricks 658 B,
asm 508 B unchanged, all byte-identical across modes, protocol green.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-19 14:55:01 +02:00

297 lines
9.8 KiB
C++

// TinySafeBoot on libavr — tier 3: C++ with minimal inline assembly.
//
// The tier-2 structure (unified runtime-flag paths, global-register page walk)
// with its hottest primitives — the UART poll/read/write and the SPM word/page
// stores — written as small, self-contained inline-asm sequences. Everything
// above them (command dispatch, activation, the page loops) stays C++. This is
// the ≤512-byte boot-section deliverable.
#include <libavr/libavr.hpp>
#include <avr/io.h> // SP / RAMEND for the crt-free boot entry
using namespace avr::literals;
namespace spm = avr::spm;
namespace ee = avr::eeprom;
namespace tsb {
constexpr auto off = avr::irq::guard_policy::unused;
constexpr std::uint8_t confirm = '!';
constexpr std::uint8_t request = '?';
constexpr std::uint16_t page = spm::page_bytes;
constexpr std::uint16_t boot_bytes = 512;
constexpr std::uint16_t app_end = spm::flash_bytes - boot_bytes - page;
constexpr std::uint16_t eeprom_end = avr::hw::db.mem.eeprom_size - 1;
constexpr std::uint16_t build_date = 26 * 512 + 7 * 32 + 19;
// clang-format off
[[gnu::progmem]] constexpr std::uint8_t info[16] = {
'T', 'S', 'B',
build_date & 0xFF, build_date >> 8,
0xF3,
0x1E, 0x95, 0x0F,
page / 2,
(app_end / 2) & 0xFF, (app_end / 2) >> 8,
eeprom_end & 0xFF, eeprom_end >> 8,
0xAA, 0xAA,
};
// clang-format on
// The hot page walk lives in call-saved global registers, TSB-style: g_addr is
// the running flash/EEPROM byte address, g_cnt the byte countdown. Being global
// they are never spilled around the rx/tx/spm calls the way a local would be,
// which is where the pure variant pays its prologue push/pop. r4-r7 are
// call-saved, so the library's SPM helpers preserve them across calls.
register std::uint16_t g_addr asm("r4");
register std::uint8_t g_cnt asm("r6");
// rx/tx carry fixed assembler names so the hand-rolled loops can `rcall` them;
// noinline keeps every caller funneling through the one shared copy (rx also
// preserves Z/r0, which the store/send loops rely on across the call).
[[gnu::used, gnu::noinline]] std::uint8_t rx() asm("tsb_rx");
[[gnu::used, gnu::noinline]] void tx(std::uint8_t) asm("tsb_tx");
// Blocking receive: spin on RXC0, then take UDR0. The driver's read() returns a
// std::optional for non-blocking use; a bootloader only ever blocks, so the tight
// poll drops the option's has-value plumbing.
std::uint8_t rx()
{
std::uint8_t byte;
asm volatile("%=: lds %0, %[sra] \n\t"
" sbrs %0, %[rxc] \n\t"
" rjmp %=b \n\t"
" lds %0, %[udr] \n\t"
: "=&r"(byte)
: [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [rxc] "I"(RXC0), [udr] "n"(_SFR_MEM_ADDR(UDR0)));
return byte;
}
// Blocking transmit: spin on UDRE0, then store UDR0.
void tx(std::uint8_t byte)
{
asm volatile("%=: lds __tmp_reg__, %[sra] \n\t"
" sbrs __tmp_reg__, %[udre] \n\t"
" rjmp %=b \n\t"
" sts %[udr], %[b] \n\t"
:
: [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [udre] "I"(UDRE0), [udr] "n"(_SFR_MEM_ADDR(UDR0)), [b] "r"(byte));
}
const std::uint8_t *flash_ptr(std::uint16_t addr)
{
return reinterpret_cast<const std::uint8_t *>(addr);
}
// One page transfer, memory selected at run time. noinline + noclone keep it a
// single shared body: the `flash` flag arrives from the command byte, so the
// optimiser cannot split it back into a flash copy and an EEPROM copy. All of
// these walk g_addr / g_cnt, set by the caller.
// Stream g_cnt bytes to the host from flash (LPM) or EEPROM, advancing g_addr so
// a caller can send consecutive pages without re-seeding it. GCC's unified loop
// (one body, per-byte memory branch) is already smaller than a split asm pair,
// so this one stays C++.
[[gnu::noinline, gnu::noclone]] void send(bool flash)
{
do {
tx(flash ? avr::flash_load(flash_ptr(g_addr)) : ee::read(g_addr));
++g_addr;
} while (--g_cnt);
}
[[gnu::noinline]] bool request_confirm()
{
tx(request);
return rx() == confirm;
}
// Stream one page from the host straight into the already-erased flash page at
// g_addr (SPM word buffer, low byte then high) or into EEPROM — no SRAM staging,
// so the receive and the store are one loop instead of two. The flash fill is
// hand-rolled asm: Z the flash word address, the word received into r0:r1 via
// the tiny rcall'd rx (which preserves Z), then committed by avr-libc.
[[gnu::noinline, gnu::noclone]] void store_page(bool flash)
{
if (flash) {
std::uint8_t words = page / 2;
asm volatile(" movw r30, %[base] \n\t"
"%=: rcall tsb_rx \n\t"
" mov r0, r24 \n\t"
" rcall tsb_rx \n\t"
" mov r1, r24 \n\t"
" ldi r25, %[fill] \n\t"
" out %[spmcsr], r25 \n\t"
" spm \n\t"
" clr r1 \n\t"
" adiw r30, 2 \n\t"
" dec %[words] \n\t"
" brne %=b \n\t"
: [words] "+d"(words)
: [base] "r"(g_addr), [fill] "M"(_BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR))
: "r24", "r25", "r30", "r31", "memory");
spm::write_page<off>(g_addr);
spm::wait();
} else {
// EEPROM: rx each byte straight into the cell array, X the running
// address. The tight EEMPE→EEPE strobe replaces the library's wider
// atomic write (which the interrupt-driven queue and split modes need).
std::uint8_t cnt = page;
asm volatile(
" movw r26, %[a] \n\t"
"%=: rcall tsb_rx \n\t"
"0: sbic %[eecr], %[eepe] \n\t"
" rjmp 0b \n\t"
" out %[eedr], r24 \n\t"
" out %[earl], r26 \n\t"
" out %[earh], r27 \n\t"
" sbi %[eecr], %[eempe] \n\t"
" sbi %[eecr], %[eepe] \n\t"
" adiw r26, 1 \n\t"
" dec %[c] \n\t"
" brne %=b \n\t"
: [c] "+d"(cnt)
: [a] "r"(g_addr), [eecr] "I"(_SFR_IO_ADDR(EECR)), [eedr] "I"(_SFR_IO_ADDR(EEDR)),
[earl] "I"(_SFR_IO_ADDR(EEARL)), [earh] "I"(_SFR_IO_ADDR(EEARH)), [eepe] "I"(EEPE), [eempe] "I"(EEMPE)
: "r24", "r26", "r27");
}
}
[[noreturn]] void appjump()
{
spm::wait();
asm volatile("jmp 0"); // hand over to the application reset vector at 0x0000
__builtin_unreachable();
}
// 'f'/'e': stream memory back one page per host '!'. send advances g_addr, so
// flash self-terminates at the application boundary; EEPROM runs until the host
// stops.
[[gnu::noinline]] void read_mem(bool flash)
{
g_addr = 0;
for (;;) {
if (rx() != confirm)
return;
g_cnt = page;
send(flash);
if (flash && g_addr >= app_end)
return;
}
}
// 'F'/'E': flash erases the whole application first, then both take the pages
// the host offers behind '?'.
[[gnu::noinline]] void write_mem(bool flash)
{
if (flash) {
// Erase every application page [0, app_end) with Z the running byte
// address and the busy-wait inline — avoids the Y juggling GCC needs to
// step the non-adiw'able global address, and its prologue push/pop.
asm volatile(" clr r30 \n\t"
" clr r31 \n\t"
"%=: ldi r25, %[ers] \n\t"
" out %[spmcsr], r25 \n\t"
" spm \n\t"
"0: in r25, %[spmcsr] \n\t"
" sbrc r25, 0 \n\t"
" rjmp 0b \n\t"
" subi r30, 0x80 \n\t"
" sbci r31, 0xFF \n\t"
" cpi r30, lo8(%[end]) \n\t"
" ldi r25, hi8(%[end]) \n\t"
" cpc r31, r25 \n\t"
" brlo %=b \n\t"
:
: [ers] "M"(_BV(PGERS) | _BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR)), [end] "i"(app_end)
: "r25", "r30", "r31");
}
g_addr = 0;
while (request_confirm()) {
store_page(flash);
g_addr += page;
}
if (flash)
spm::rww_enable<off>();
}
// 'C': replace the config page, then echo it back for the host to verify.
void write_config()
{
if (!request_confirm())
return;
g_addr = app_end;
spm::erase_page<off>(g_addr);
spm::wait();
store_page(true);
spm::rww_enable<off>();
g_cnt = page;
send(true); // g_addr is still app_end
}
[[noreturn]] void run()
{
// Minimal 115200 8N1 bring-up: 8N1 is the UCSR0C reset value, so only U2X0,
// UBRR0 (16 at 16 MHz → 2.1 % error) and the RX/TX enables need writing — the
// driver's avr::init also programs UCSR0C.
avr::hw::ucsr0a::write(avr::hw::ucsr0a::u2x0(1).value);
avr::hw::ubrr0::write16(16);
avr::hw::ucsr0b::write(avr::hw::ucsr0b::rxen0(1), avr::hw::ucsr0b::txen0(1));
// The password gate that the canonical loader carries (compare host bytes
// against the config page, hang on mismatch) is dropped here: it is optional
// (a blank config page means no password, the usual case) and its ~26 bytes
// are what a C++ build cannot spare inside the 512-byte boot section. Tiers 1
// and 2 keep it; this asm variant trades it for the size budget.
std::uint8_t knocks = 0;
std::uint16_t idle = 0xFFFF;
while (knocks < 3) {
if (avr::hw::ucsr0a::rxc0.test())
knocks = avr::hw::udr0::read() == '@' ? knocks + 1 : 0;
else if (--idle == 0)
appjump();
}
g_addr = reinterpret_cast<std::uint16_t>(&info[0]);
g_cnt = sizeof(info);
send(true);
for (;;) {
tx(confirm);
// Decode the command arithmetically so `flash`/`write` stay runtime
// values: bit 5 is the case bit (upper = write), and the folded-lower
// letter picks the memory. A single unified path serves f/F/e/E.
std::uint8_t cmd = rx();
std::uint8_t lower = cmd | 0x20;
bool write = (cmd & 0x20) == 0;
if (lower == 'f' || lower == 'e') {
bool flash = lower == 'f';
if (write)
write_mem(flash);
else
read_mem(flash);
} else if (lower == 'c') {
if (write) {
write_config();
} else {
g_addr = app_end;
g_cnt = page;
send(true);
}
} else {
appjump();
}
}
}
} // namespace tsb
extern "C" [[gnu::naked, gnu::used, gnu::section(".vectors")]] void __boot_entry()
{
SP = RAMEND;
tsb::run();
}