// TinySafeBoot on libavr — tier 3: C++ with minimal inline assembly. // // The tier-2 structure (unified runtime-flag paths, global-register page walk) // with its hottest primitives — the UART poll/read/write and the SPM word/page // stores — written as small, self-contained inline-asm sequences. Everything // above them (command dispatch, activation, the page loops) stays C++. This is // the ≤512-byte boot-section deliverable. #include #include // SP / RAMEND for the crt-free boot entry using namespace avr::literals; namespace spm = avr::spm; namespace ee = avr::eeprom; namespace tsb { constexpr auto off = avr::irq::guard_policy::unused; constexpr std::uint8_t confirm = '!'; constexpr std::uint8_t request = '?'; constexpr std::uint16_t page = spm::page_bytes; constexpr std::uint16_t boot_bytes = 512; constexpr std::uint16_t app_end = spm::flash_bytes - boot_bytes - page; constexpr std::uint16_t eeprom_end = avr::hw::db.mem.eeprom_size - 1; constexpr std::uint16_t build_date = 26 * 512 + 7 * 32 + 19; // clang-format off [[gnu::progmem]] constexpr std::uint8_t info[16] = { 'T', 'S', 'B', build_date & 0xFF, build_date >> 8, 0xF3, 0x1E, 0x95, 0x0F, page / 2, (app_end / 2) & 0xFF, (app_end / 2) >> 8, eeprom_end & 0xFF, eeprom_end >> 8, 0xAA, 0xAA, }; // clang-format on // The hot page walk lives in call-saved global registers, TSB-style: g_addr is // the running flash/EEPROM byte address, g_cnt the byte countdown. Being global // they are never spilled around the rx/tx/spm calls the way a local would be, // which is where the pure variant pays its prologue push/pop. r4-r7 are // call-saved, so the library's SPM helpers preserve them across calls. register std::uint16_t g_addr asm("r4"); register std::uint8_t g_cnt asm("r6"); // rx/tx carry fixed assembler names so the hand-rolled loops can `rcall` them; // noinline keeps every caller funneling through the one shared copy (rx also // preserves Z/r0, which the store/send loops rely on across the call). [[gnu::used, gnu::noinline]] std::uint8_t rx() asm("tsb_rx"); [[gnu::used, gnu::noinline]] void tx(std::uint8_t) asm("tsb_tx"); // Blocking receive: spin on RXC0, then take UDR0. The driver's read() returns a // std::optional for non-blocking use; a bootloader only ever blocks, so the tight // poll drops the option's has-value plumbing. std::uint8_t rx() { std::uint8_t byte; asm volatile("%=: lds %0, %[sra] \n\t" " sbrs %0, %[rxc] \n\t" " rjmp %=b \n\t" " lds %0, %[udr] \n\t" : "=&r"(byte) : [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [rxc] "I"(RXC0), [udr] "n"(_SFR_MEM_ADDR(UDR0))); return byte; } // Blocking transmit: spin on UDRE0, then store UDR0. void tx(std::uint8_t byte) { asm volatile("%=: lds __tmp_reg__, %[sra] \n\t" " sbrs __tmp_reg__, %[udre] \n\t" " rjmp %=b \n\t" " sts %[udr], %[b] \n\t" : : [sra] "n"(_SFR_MEM_ADDR(UCSR0A)), [udre] "I"(UDRE0), [udr] "n"(_SFR_MEM_ADDR(UDR0)), [b] "r"(byte)); } const std::uint8_t *flash_ptr(std::uint16_t addr) { return reinterpret_cast(addr); } // One page transfer, memory selected at run time. noinline + noclone keep it a // single shared body: the `flash` flag arrives from the command byte, so the // optimiser cannot split it back into a flash copy and an EEPROM copy. All of // these walk g_addr / g_cnt, set by the caller. // Stream g_cnt bytes to the host from flash (LPM) or EEPROM, advancing g_addr so // a caller can send consecutive pages without re-seeding it. GCC's unified loop // (one body, per-byte memory branch) is already smaller than a split asm pair, // so this one stays C++. [[gnu::noinline, gnu::noclone]] void send(bool flash) { do { tx(flash ? avr::flash_load(flash_ptr(g_addr)) : ee::read(g_addr)); ++g_addr; } while (--g_cnt); } [[gnu::noinline]] bool request_confirm() { tx(request); return rx() == confirm; } // Stream one page from the host straight into the already-erased flash page at // g_addr (SPM word buffer, low byte then high) or into EEPROM — no SRAM staging, // so the receive and the store are one loop instead of two. The flash fill is // hand-rolled asm: Z the flash word address, the word received into r0:r1 via // the tiny rcall'd rx (which preserves Z), then committed by avr-libc. [[gnu::noinline, gnu::noclone]] void store_page(bool flash) { if (flash) { std::uint8_t words = page / 2; asm volatile(" movw r30, %[base] \n\t" "%=: rcall tsb_rx \n\t" " mov r0, r24 \n\t" " rcall tsb_rx \n\t" " mov r1, r24 \n\t" " ldi r25, %[fill] \n\t" " out %[spmcsr], r25 \n\t" " spm \n\t" " clr r1 \n\t" " adiw r30, 2 \n\t" " dec %[words] \n\t" " brne %=b \n\t" : [words] "+d"(words) : [base] "r"(g_addr), [fill] "M"(_BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR)) : "r24", "r25", "r30", "r31", "memory"); spm::write_page(g_addr); spm::wait(); } else { // EEPROM: rx each byte straight into the cell array, X the running // address. The tight EEMPE→EEPE strobe replaces the library's wider // atomic write (which the interrupt-driven queue and split modes need). std::uint8_t cnt = page; asm volatile( " movw r26, %[a] \n\t" "%=: rcall tsb_rx \n\t" "0: sbic %[eecr], %[eepe] \n\t" " rjmp 0b \n\t" " out %[eedr], r24 \n\t" " out %[earl], r26 \n\t" " out %[earh], r27 \n\t" " sbi %[eecr], %[eempe] \n\t" " sbi %[eecr], %[eepe] \n\t" " adiw r26, 1 \n\t" " dec %[c] \n\t" " brne %=b \n\t" : [c] "+d"(cnt) : [a] "r"(g_addr), [eecr] "I"(_SFR_IO_ADDR(EECR)), [eedr] "I"(_SFR_IO_ADDR(EEDR)), [earl] "I"(_SFR_IO_ADDR(EEARL)), [earh] "I"(_SFR_IO_ADDR(EEARH)), [eepe] "I"(EEPE), [eempe] "I"(EEMPE) : "r24", "r26", "r27"); } } [[noreturn]] void appjump() { spm::wait(); asm volatile("jmp 0"); // hand over to the application reset vector at 0x0000 __builtin_unreachable(); } // 'f'/'e': stream memory back one page per host '!'. send advances g_addr, so // flash self-terminates at the application boundary; EEPROM runs until the host // stops. [[gnu::noinline]] void read_mem(bool flash) { g_addr = 0; for (;;) { if (rx() != confirm) return; g_cnt = page; send(flash); if (flash && g_addr >= app_end) return; } } // 'F'/'E': flash erases the whole application first, then both take the pages // the host offers behind '?'. [[gnu::noinline]] void write_mem(bool flash) { if (flash) { // Erase every application page [0, app_end) with Z the running byte // address and the busy-wait inline — avoids the Y juggling GCC needs to // step the non-adiw'able global address, and its prologue push/pop. asm volatile(" clr r30 \n\t" " clr r31 \n\t" "%=: ldi r25, %[ers] \n\t" " out %[spmcsr], r25 \n\t" " spm \n\t" "0: in r25, %[spmcsr] \n\t" " sbrc r25, 0 \n\t" " rjmp 0b \n\t" " subi r30, 0x80 \n\t" " sbci r31, 0xFF \n\t" " cpi r30, lo8(%[end]) \n\t" " ldi r25, hi8(%[end]) \n\t" " cpc r31, r25 \n\t" " brlo %=b \n\t" : : [ers] "M"(_BV(PGERS) | _BV(__SPM_ENABLE)), [spmcsr] "I"(_SFR_IO_ADDR(SPMCSR)), [end] "i"(app_end) : "r25", "r30", "r31"); } g_addr = 0; while (request_confirm()) { store_page(flash); g_addr += page; } if (flash) spm::rww_enable(); } // 'C': replace the config page, then echo it back for the host to verify. void write_config() { if (!request_confirm()) return; g_addr = app_end; spm::erase_page(g_addr); spm::wait(); store_page(true); spm::rww_enable(); g_cnt = page; send(true); // g_addr is still app_end } [[noreturn]] void run() { // Minimal 115200 8N1 bring-up: 8N1 is the UCSR0C reset value, so only U2X0, // UBRR0 (16 at 16 MHz → 2.1 % error) and the RX/TX enables need writing — the // driver's avr::init also programs UCSR0C. avr::hw::ucsr0a::write(avr::hw::ucsr0a::u2x0(1).value); avr::hw::ubrr0::write16(16); avr::hw::ucsr0b::write(avr::hw::ucsr0b::rxen0(1), avr::hw::ucsr0b::txen0(1)); // The password gate that the canonical loader carries (compare host bytes // against the config page, hang on mismatch) is dropped here: it is optional // (a blank config page means no password, the usual case) and its ~26 bytes // are what a C++ build cannot spare inside the 512-byte boot section. Tiers 1 // and 2 keep it; this asm variant trades it for the size budget. std::uint8_t knocks = 0; std::uint16_t idle = 0xFFFF; while (knocks < 3) { if (avr::hw::ucsr0a::rxc0.test()) knocks = avr::hw::udr0::read() == '@' ? knocks + 1 : 0; else if (--idle == 0) appjump(); } g_addr = reinterpret_cast(&info[0]); g_cnt = sizeof(info); send(true); for (;;) { tx(confirm); // Decode the command arithmetically so `flash`/`write` stay runtime // values: bit 5 is the case bit (upper = write), and the folded-lower // letter picks the memory. A single unified path serves f/F/e/E. std::uint8_t cmd = rx(); std::uint8_t lower = cmd | 0x20; bool write = (cmd & 0x20) == 0; if (lower == 'f' || lower == 'e') { bool flash = lower == 'f'; if (write) write_mem(flash); else read_mem(flash); } else if (lower == 'c') { if (write) { write_config(); } else { g_addr = app_end; g_cnt = page; send(true); } } else { appjump(); } } } } // namespace tsb extern "C" [[gnu::naked, gnu::used, gnu::section(".vectors")]] void __boot_entry() { SP = RAMEND; tsb::run(); }