Compare commits
7 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a4da885e36 | |||
| 335e494a31 | |||
| 799709efcf | |||
| 7ae80087b3 | |||
| b477f53ca5 | |||
| 84d3f679c2 | |||
| b5020a1e20 |
@@ -248,6 +248,24 @@ if(PROJECT_IS_TOP_LEVEL)
|
||||
-DLIMIT=${PUREBOOT_LIMIT} -P ${CMAKE_CURRENT_SOURCE_DIR}/test/check_size.cmake)
|
||||
endfunction()
|
||||
|
||||
# The autobaud loader (pureboot/autobaud.md): one clock-agnostic image per
|
||||
# chip, no clock × baud axis, size-tested against the same per-chip budget on
|
||||
# every chip.
|
||||
#
|
||||
# Only the unified version is built. Its two predecessors —
|
||||
# pureboot_autobaud_pure.cpp at 508 B and pureboot_autobaud_reg.cpp at 512 —
|
||||
# had 4 B and 0 B of margin on the 1284P, and the fix for the activation hang
|
||||
# costs ~22, which puts them at 530 and 534. Neither can ship, so the choice
|
||||
# the branch existed to offer is settled by measurement rather than taste.
|
||||
# The sources stay for the record; autobaud.md carries the numbers.
|
||||
function(pureboot_autobaud_variant name source)
|
||||
pureboot_add_autobaud(${name} ${source})
|
||||
add_test(NAME ${name}.size
|
||||
COMMAND ${CMAKE_COMMAND} -DSIZE_TOOL=${CMAKE_SIZE} -DELF=$<TARGET_FILE:${name}>
|
||||
-DLIMIT=${PUREBOOT_LIMIT} -P ${CMAKE_CURRENT_SOURCE_DIR}/test/check_size.cmake)
|
||||
endfunction()
|
||||
pureboot_autobaud_variant(pureboot_autobaud_uni pureboot_autobaud_uni.cpp)
|
||||
|
||||
# One point of the exhaustive matrix, named from its resolved parameters
|
||||
# so the enumeration cannot collide with itself. Unreachable rates drop
|
||||
# out here rather than aborting the configure.
|
||||
@@ -378,4 +396,30 @@ if(PROJECT_IS_TOP_LEVEL)
|
||||
${CMAKE_BINARY_DIR}/pbusart1-work usart1)
|
||||
set_tests_properties(pureboot.usart1 PROPERTIES TIMEOUT 180)
|
||||
endif()
|
||||
|
||||
# The autobaud variants driven end to end over the software-UART bridge (both
|
||||
# under review — pureboot/autobaud.md): the host sends the 0xC0 calibration
|
||||
# pulse, the loader times it, locks, and programs. Run on the near-flash 328P
|
||||
# and the word-addressed 1284P — the two flash-addressing classes — and each
|
||||
# at two clocks with the one binary, which is the clock-agnostic property
|
||||
# autobaud exists for (test/pbautobaud.py). The fixture application banners
|
||||
# over the same software link at the first clock's rate.
|
||||
if(LIBAVR_MCU MATCHES "^atmega(328p|1284p)$" AND DEFINED PB_DEVICE)
|
||||
add_executable(pbapp_autobaud test/pbapp.cpp)
|
||||
target_link_libraries(pbapp_autobaud PRIVATE libavr)
|
||||
target_compile_definitions(pbapp_autobaud PRIVATE PUREBOOT_CLOCK_HZ=1000000
|
||||
PUREBOOT_BAUD=9600 PUREBOOT_SOFT_SERIAL PUREBOOT_TX=pb1)
|
||||
add_custom_command(TARGET pbapp_autobaud POST_BUILD
|
||||
COMMAND ${CMAKE_OBJCOPY} -O binary
|
||||
$<TARGET_FILE:pbapp_autobaud> $<TARGET_FILE:pbapp_autobaud>.bin)
|
||||
foreach(_variant uni)
|
||||
add_test(NAME pureboot.autobaud_${_variant}
|
||||
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/test/pbautobaud.py
|
||||
${PB_DEVICE} $<TARGET_FILE:pureboot_autobaud_${_variant}> ${PUREBOOT_SIM_MCU}
|
||||
${PUREBOOT_BASE_HEX} ${PUREBOOT_PAGE} $<TARGET_FILE:pbapp_autobaud>.bin
|
||||
1000000 9600 ${CMAKE_CURRENT_SOURCE_DIR}/pureboot/pureboot.py
|
||||
${CMAKE_BINARY_DIR}/pbautobaud-${_variant}-work)
|
||||
set_tests_properties(pureboot.autobaud_${_variant} PROPERTIES TIMEOUT 240)
|
||||
endforeach()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
2
libavr
2
libavr
Submodule libavr updated: a8ed8c4851...edc77ca43f
@@ -310,3 +310,38 @@ function(pureboot_add_loader name)
|
||||
set_target_properties(${name} PROPERTIES PUREBOOT_HZ ${PB_CLOCK} PUREBOOT_BAUD ${PB_BAUD}
|
||||
PUREBOOT_LINK ${_link})
|
||||
endfunction()
|
||||
|
||||
# pureboot_add_autobaud(<name> <source> [RX <pin>] [TX <pin>])
|
||||
#
|
||||
# An autobaud software-serial loader from <source> (pureboot_autobaud_*.cpp).
|
||||
# Autobaud measures the host's bit timing at runtime, so the image carries no
|
||||
# clock and no baud — one binary per chip runs at any F_CPU. Same per-chip
|
||||
# geometry, link and codegen flags as pureboot_add_loader(); only the clock and
|
||||
# baud axes fall away. Two source files are under review (autobaud.md):
|
||||
# pureboot_autobaud_pure.cpp and pureboot_autobaud_reg.cpp.
|
||||
function(pureboot_add_autobaud name source)
|
||||
cmake_parse_arguments(PB "" "RX;TX" "" ${ARGN})
|
||||
if(NOT PB_RX)
|
||||
set(PB_RX pb0)
|
||||
endif()
|
||||
if(NOT PB_TX)
|
||||
set(PB_TX pb1)
|
||||
endif()
|
||||
foreach(_pin ${PB_RX} ${PB_TX})
|
||||
if(NOT _pin MATCHES "^p[a-h][0-7]$")
|
||||
message(FATAL_ERROR "pureboot_add_autobaud(${name}): pin '${_pin}' is not of the form pb1")
|
||||
endif()
|
||||
endforeach()
|
||||
get_property(_base_hex GLOBAL PROPERTY PUREBOOT_BASE_HEX)
|
||||
get_property(_app GLOBAL PROPERTY PUREBOOT_APP)
|
||||
get_property(_wrap GLOBAL PROPERTY PUREBOOT_WRAP)
|
||||
|
||||
add_executable(${name} ${CMAKE_CURRENT_FUNCTION_LIST_DIR}/${source})
|
||||
target_link_libraries(${name} PRIVATE libavr)
|
||||
target_compile_definitions(${name} PRIVATE PUREBOOT_RX=${PB_RX} PUREBOOT_TX=${PB_TX})
|
||||
target_compile_options(${name} PRIVATE
|
||||
-fno-ivopts -fira-algorithm=priority -fno-move-loop-invariants -fno-tree-ter -fno-split-wide-types)
|
||||
target_link_options(${name} PRIVATE -nostartfiles -Wl,--section-start=.text=${_base_hex}
|
||||
-Wl,--defsym=pureboot_app=${_app} ${_wrap})
|
||||
add_custom_command(TARGET ${name} POST_BUILD COMMAND ${CMAKE_SIZE} $<TARGET_FILE:${name}>)
|
||||
endfunction()
|
||||
|
||||
@@ -98,8 +98,15 @@ speak to the build. This exact deployment runs the full protocol suite in CI
|
||||
|
||||
Reset enters the loader (BOOTRST on the boot-sectioned megas, the patched
|
||||
reset vector elsewhere) — except a watchdog reset, which hands straight to the
|
||||
application, since the application owns its watchdog and must clear WDRF
|
||||
itself.
|
||||
application with no activation window, since the application owns its watchdog.
|
||||
This is deliberate: it lets an application reboot itself instantly rather than
|
||||
sit through the window. The application must clear WDRF itself (libavr's
|
||||
`watchdog::disable()` does). **Gotcha:** WDRF is sticky (cleared only by
|
||||
software, not by a later reset), so an application that watchdog-resets and
|
||||
never clears it diverts *every* subsequent reset — external ones included —
|
||||
past the window too, and the loader becomes reachable only through an external
|
||||
programmer until the flag is cleared. A serial recovery path therefore assumes
|
||||
the application clears WDRF on its own reset path.
|
||||
|
||||
The host then knocks `p` then `b`, each awaited byte under a fresh activation
|
||||
window; any other byte is discarded and awaited again, so line noise can delay
|
||||
@@ -123,6 +130,13 @@ On chips whose flash exceeds 64 KiB (the 1284s — info-block flag bit 1) the
|
||||
addresses (the 644s' 64 KiB is exactly the 16-bit byte space). EEPROM
|
||||
addresses and all counts are bytes.
|
||||
|
||||
The loader trusts the host to keep addresses in range: it does not bound them
|
||||
against the info block. **Gotcha:** a `w` (or `r`) that runs past `E2END` wraps
|
||||
— EEAR is only as wide as the array, so an address past the end truncates onto
|
||||
low EEPROM and the write silently overwrites it. Keeping writes within the
|
||||
advertised sizes is the host's job (the shipped tool does); the flash budget
|
||||
is better spent on features than on re-checking a bound the host already holds.
|
||||
|
||||
| Cmd | Arguments | Reply |
|
||||
|---|---|---|
|
||||
| `b` | — | the 12-byte info block |
|
||||
|
||||
251
pureboot/autobaud.md
Normal file
251
pureboot/autobaud.md
Normal file
@@ -0,0 +1,251 @@
|
||||
# pureboot autobaud — findings, and how the version question settled itself
|
||||
|
||||
The autobaud loader measures the host's bit timing **at runtime** from a
|
||||
calibration pulse, so the image carries no clock: one clock-agnostic binary per
|
||||
chip runs at any F_CPU and locks onto whatever baud the host sends. It exists
|
||||
for the software-serial deployments — the RC-oscillator parts (the tinies,
|
||||
internal-oscillator megas) whose exact clock is uncertain and drifts, so today
|
||||
each needs a per-clock build. Autobaud erases that axis. That is the win —
|
||||
deployment, not bytes.
|
||||
|
||||
This document records how the fit was established, why the two versions that
|
||||
were under review are both dead, and what the loader looks like now.
|
||||
|
||||
## The decision, settled by measurement
|
||||
|
||||
Two autobaud loaders were built for review, differing in one tradeoff — the
|
||||
running-slot write guard against strict purity. `pureboot_autobaud_pure.cpp`
|
||||
landed at **508 B** on the 1284P (4 B spare) and `pureboot_autobaud_reg.cpp` at
|
||||
**512** (zero spare).
|
||||
|
||||
Then hardware testing found a defect that neither could absorb.
|
||||
|
||||
**A single spurious calibration pulse wedged the loader.** `run()` budgeted only
|
||||
the start-edge wait inside `measure()`; the `rx()` that read the knock behind it
|
||||
was unbudgeted and blocked forever. One stray low pulse on an unattended device
|
||||
— EMI, or a host that opens the port and never knocks — held the loader in its
|
||||
activation loop and the application never ran. On a field device that is a hang,
|
||||
not a hiccup, and it is exactly the deployment autobaud is for.
|
||||
|
||||
The fix is to bound the whole activation: an expired knock budget returns a byte
|
||||
that cannot be the knock, so control falls back into the budgeted `measure()`,
|
||||
and a line that stays idle boots the application there. It costs about 22 bytes.
|
||||
|
||||
| 1284P, with the activation fix | size | 512 B budget |
|
||||
|---|---|---|
|
||||
| `pureboot_autobaud_pure.cpp` | 530 | **over by 18** |
|
||||
| `pureboot_autobaud_reg.cpp` | 534 | **over by 22** |
|
||||
| `pureboot_autobaud_uni.cpp` | **464** | **48 B spare** |
|
||||
|
||||
Both candidates were unshippable, and the margin they were competing over was
|
||||
never real — it was the space the missing fix should have occupied. So the
|
||||
choice is not between them. It is the third loader below, which fits with room
|
||||
to spare *and* carries features neither had. The two sources stay in the tree
|
||||
for the record; only the unified one is built.
|
||||
|
||||
## The unified loader
|
||||
|
||||
The insight that paid was the one that had already paid once: **merging command
|
||||
bodies removes cost that moving them around only redistributes.** Folding `R`,
|
||||
`r` and `w` into a single address-and-count path had been worth 14 B earlier.
|
||||
Pushed further — one read command and one write command over *named spaces* —
|
||||
it is worth far more, because four transfer loops collapse into one.
|
||||
|
||||
`pureboot_autobaud_uni.cpp` is pureboot 5. It is strictly pure: no inline
|
||||
assembly, no global register variable, and **no GPIOR either** — the measured
|
||||
unit lives in a plain static, so the loader claims no chip resource an
|
||||
application might want, and the GPIOR-versus-static question disappears along
|
||||
with the chips that have no GPIOR.
|
||||
|
||||
### The protocol
|
||||
|
||||
| command | arguments | |
|
||||
|---|---|---|
|
||||
| `b` | — | version, then the three signature bytes |
|
||||
| `J` | addr16 | ack, then jump (word address) |
|
||||
| `W` | sel8, addr16, page bytes | fill the flash page buffer |
|
||||
| `G` | sel8, addr16, n8 | read n bytes (0 means 256) |
|
||||
| `g` | sel8, addr16, n8, then n bytes | write, each byte acked |
|
||||
|
||||
`sel` is `space | bank << 4`. The low nibble names the space; the high nibble is
|
||||
flash's third address byte, so every transfer speaks a **byte** address inside a
|
||||
64 KiB bank and no command has to carry word addresses. The host must not span a
|
||||
bank boundary in one transfer — it already chunks by page, so nothing it does
|
||||
comes close.
|
||||
|
||||
| space | | |
|
||||
|---|---|---|
|
||||
| 0 | flash | `lpm`/`elpm` |
|
||||
| 1 | EEPROM | |
|
||||
| 2 | data | SRAM — and with it the register file and every I/O register, which share the data address space on AVR |
|
||||
| 3 | fuse and lock | |
|
||||
| 4 | SPM | write-only: the data byte goes to SPMCSR and fires the instruction at the selected address |
|
||||
|
||||
Three things follow from that table that the loader never had:
|
||||
|
||||
- **RAM read and write**, the missing feature. In a per-command design it would
|
||||
have cost a fresh dispatch arm and a fresh loop, ~30 B, on a loader with 4 B
|
||||
spare. As one more space on a shared loop it is a single `ld`/`st`. It also
|
||||
hands the host arbitrary **I/O register access** for free, because AVR maps
|
||||
the peripherals into the same address space.
|
||||
- **Host-driven SPM.** `W` used to end with a hardcoded erase, write and RWW
|
||||
re-enable — 42 B. Those are now three writes to the SPM space, reusing the
|
||||
store path's own address, data byte and ack. The host pays three extra
|
||||
round-trips per page (18 wire bytes against 256 of data) and gains the ability
|
||||
to issue *any* SPM operation, lock bits included.
|
||||
- **`W` on the same footing as everything else.** It takes the same selector and
|
||||
the same byte address instead of a word address of its own, which made flash
|
||||
addressing uniform across the protocol *and* was 20 B cheaper than keeping its
|
||||
private convention.
|
||||
|
||||
Read and write are `G` and `g` — the same letter, one bit apart — so the
|
||||
transfer loop picks its direction with a one-word skip rather than a compare.
|
||||
|
||||
### Why the SPM space has to be one primitive
|
||||
|
||||
It is tempting to go further and expose a generic "poke this I/O register", from
|
||||
which the host could drive SPM itself. The hardware forbids it: `out SPMCSR, x`
|
||||
and the `spm` that follows must issue within four cycles, and EEPROM's
|
||||
EEMPE→EEPE window is the same shape. A host cannot hit a four-cycle window
|
||||
across a serial link. **Atomicity is the floor, and the atomic unit must be
|
||||
resident.** That is the real limit on how low-level a bootloader's primitives
|
||||
can go — not the byte count.
|
||||
|
||||
## Where the bytes went
|
||||
|
||||
The starting point was the 508 B pure build, disassembled and attributed:
|
||||
|
||||
| phase | bytes |
|
||||
|---|---|
|
||||
| `link::rx` + `link::tx` (bit-banged UART) | 102 |
|
||||
| autobaud measure + knock | 62 |
|
||||
| reset vector, pin init, WDRF check, jump, ack, EEPROM wait | 50 |
|
||||
| command loop head + dispatch tree | 42 |
|
||||
| `W` program flash page | 98 |
|
||||
| `R` / `r` / `w` / `F` bodies, plus the shared address decode | 122 |
|
||||
| `b` info, `J` jump | 32 |
|
||||
|
||||
**208 B — 41% — is physical layer and activation**, which no protocol change can
|
||||
touch. The command bodies were the entire addressable surface, and they were
|
||||
four copies of one idea.
|
||||
|
||||
The route from a first attempt to the final loader, all on the 1284P:
|
||||
|
||||
| step | size |
|
||||
|---|---|
|
||||
| unified `G`/`P`, `load`/`store` outlined, `__uint24` cursor | 600 |
|
||||
| …`load`/`store` inlined; unit in `.noinit`, not `.bss` | 534 |
|
||||
| …16-bit cursor with the bank in the selector; direction as a command bit | 510 |
|
||||
| …erase/write/RWW moved out to the SPM space | 484 |
|
||||
| …`W` sharing the selector-and-address decode | **464** |
|
||||
|
||||
Three of those steps are worth keeping as lessons:
|
||||
|
||||
- **Outlining `load`/`store` cost more than the four bodies they replaced.** As
|
||||
functions they were 110 B against the 106 B of inline bodies — the AVR ABI's
|
||||
argument marshalling plus prologue ate the entire saving. Inlined into the one
|
||||
shared loop they cost only their own instructions. The call-site lesson cuts
|
||||
both ways: *merging* call sites pays, *creating* one does not.
|
||||
- **A `.bss` static drags in `__do_clear_bss`** — 18 B of startup code to zero a
|
||||
variable that is always measured before it is read. `.noinit` is correct here
|
||||
and free.
|
||||
- **A three-byte cursor taxes every space.** Widening the shared cursor so flash
|
||||
could reach past 64 KiB put an extra increment on EEPROM and RAM reads that
|
||||
never need it. Moving the bank into the selector byte kept the cursor at
|
||||
sixteen bits and cost nothing on the wire.
|
||||
|
||||
### What did not work
|
||||
|
||||
- **Encoding the space in the command byte** (so dispatch becomes masking rather
|
||||
than a compare tree) cannot carry the bank's four bits alongside a space. The
|
||||
cheap half of the idea survived as the direction bit; the rest lost to the
|
||||
selector byte, which is also more extensible.
|
||||
- **A generic primitive interpreter** — a loader with no logic at all, driven
|
||||
entirely by the host — is not reachable on AVR. Harvard architecture means the
|
||||
program counter cannot fetch from data space, so the classic "upload a flash
|
||||
algorithm into RAM and jump to it" bootstrap is impossible, and on every
|
||||
boot-sectioned part SPM only takes effect from the boot section anyway. What
|
||||
remains is a fixed primitive set: still a protocol, still logic, only at a
|
||||
different granularity. debugWIRE reaches that design point only because its
|
||||
interpreter is *in silicon*; it costs the loader nothing because it is not in
|
||||
the loader.
|
||||
- Below 512 B the saved bytes are largely unspendable on the boot-sectioned
|
||||
chips: the 328P's smallest boot section is exactly 512 B, and the 1284P's is
|
||||
1024 B, of which pureboot already occupies only the top half. The margin
|
||||
matters as headroom for correctness fixes — as this defect showed — not as
|
||||
flash returned to the application. On the patch-vector parts, which have no
|
||||
boot section, it *is* returned: on the ATtiny13 the loader is 43% of a 1 KiB
|
||||
part, and every byte is real.
|
||||
|
||||
## The codegen coupling, still load-bearing
|
||||
|
||||
`count >> 2` is exact only because the calibration pulse's bit-count (7, from
|
||||
the 0xC0 byte) equals the poll loop's cycles per iteration (7 — `sbis` 1,
|
||||
`rjmp` 2, `adiw` 2, `rjmp` 2). The loop shape survived every restructuring here,
|
||||
verified in the disassembly, but a toolchain bump that reshapes it would break
|
||||
the lock silently. `test/pbautobaud.py` is what pins it: a wrong unit fails the
|
||||
flash verify.
|
||||
|
||||
## Sizes — every chip
|
||||
|
||||
Budget 510 B on the patch-vector parts, 512 elsewhere. The 1284P is no longer
|
||||
the tight one: the bank nibble made far flash *cheaper* than the near-flash
|
||||
arithmetic it replaced.
|
||||
|
||||
| size | chips | budget | spare |
|
||||
|---|---|---|---|
|
||||
| 444 | ATtiny13, 13A | 510 | 66 |
|
||||
| 448 | ATmega48, 48A, 48P, 48PA; ATtiny25 | 510 | 62 |
|
||||
| 452 | ATtiny45, 85 | 510 | 58 |
|
||||
| 460 | ATmega644, 644A, 644P, 644PA | 512 | 52 |
|
||||
| 464 | **ATmega1284, 1284P**; ATmega8, 8A, 88, 88A, 88P, 88PA | 512 | 48 |
|
||||
| 466 | ATmega16, 16A, 32, 32A; 164A/P/PA, 168/A/P/PA, 324A/P/PA, 328, 328P | 512 | 46 |
|
||||
|
||||
All 37 chips build and size-test green, plus the 12-preset reflect spot set
|
||||
(guidance rule 4 — the reflect matrix is never run in full), which matches its
|
||||
generated counterpart byte for byte on every chip in the set. Worst case across
|
||||
the whole set is **466 B, 46 under budget**.
|
||||
|
||||
## Host tool and simulation
|
||||
|
||||
- **`pureboot.py`** speaks both generations. `Info.version >= 5` selects the
|
||||
unified path; everything below it keeps the four-command protocol, so the
|
||||
fixed-baud loader is untouched. `--autobaud` sends the 0xC0 pulse and one
|
||||
knock, then derives full geometry from the signature. New: `--peek ADDR[:N]`
|
||||
and `--poke ADDR:HEX` reach the data space.
|
||||
- **`test/pbautobaud.py`** drives the loader over the GPIO⇄pty software-UART
|
||||
bridge through the calibration handshake, a flash + EEPROM + fuse round-trip
|
||||
cross-checked against the simulator's own memory, a RAM read/write round-trip,
|
||||
and a hand-over to the fixture application — then repeats at double the F_CPU
|
||||
with the same binary, which is the clock-agnostic property autobaud exists
|
||||
for. Run on the near-flash 328P and the word-addressed 1284P.
|
||||
- It also **pins the activation hang**: the test sends a lone calibration pulse
|
||||
with no knock behind it and requires the application to boot. Against the
|
||||
unfixed loader that assertion never returns.
|
||||
|
||||
## What remains
|
||||
|
||||
- **Real-hardware acceptance.** A cycle-exact simulator cannot produce what
|
||||
autobaud exists for: a real RC oscillator at ±10% with drift and jitter.
|
||||
simavr proves the arithmetic and the fit at exact clocks; only silicon proves
|
||||
the feature. Drive an internal-oscillator ATtiny at a fixed host baud and
|
||||
confirm lock plus a full flash and verify.
|
||||
- **A generic `spm::command()` in libavr.** The SPM space issues a runtime
|
||||
command through `spm::detail::page_command` where RAMPZ exists, and falls back
|
||||
to a dispatch over the known operations where it does not — the one
|
||||
preprocessor branch in the file. A two-line library addition would make it
|
||||
uniform and save a few bytes on the 36 non-RAMPZ chips, none of which are
|
||||
tight.
|
||||
- **Retire or revive the two dead variants.** They are kept only as the record
|
||||
of the measurement; nothing builds them.
|
||||
|
||||
## Files
|
||||
|
||||
- `pureboot_autobaud_uni.cpp` — the loader. pureboot 5.
|
||||
- `pureboot_autobaud_pure.cpp`, `pureboot_autobaud_reg.cpp` — superseded, not
|
||||
built; 530 and 534 B on the 1284P once the activation hang is fixed.
|
||||
- `pureboot.py` — `--autobaud`, the unified transfer path, `--peek`/`--poke`.
|
||||
- `test/pbautobaud.py` — the end-to-end sim test and the hang regression.
|
||||
- `local/scratch/autobaud/floor_1284.S` (libavr checkout) — the hand-asm floor
|
||||
probe at 506 B, off-tree and gitignored; a size reference only. The unified
|
||||
loader is 42 B under it, with features the probe never had.
|
||||
@@ -67,18 +67,29 @@ constexpr std::uint8_t timeout_seconds = PUREBOOT_TIMEOUT;
|
||||
|
||||
// The loader's one identity number. The protocol carries none of its own —
|
||||
// a version implies it, and the host tool holds that map (README.md).
|
||||
constexpr std::uint8_t version = 3;
|
||||
constexpr std::uint8_t version = 4;
|
||||
|
||||
// The 'b' reply, byte for byte (layout: README.md). Flash-resident because
|
||||
// no crt copies a .data image — and flash_table's storage carries the word
|
||||
// alignment 'b' needs to halve the address on the large chips.
|
||||
// One wire byte per line: this is the reply's layout, not a list.
|
||||
// clang-format off
|
||||
inline constexpr avr::flash_table<std::array<std::uint8_t, 12>{
|
||||
'P', 'B', version, avr::hw::db.signature[0], avr::hw::db.signature[1], avr::hw::db.signature[2],
|
||||
static_cast<std::uint8_t>(page), // 0 means 256
|
||||
wire_base & 0xff, wire_base >> 8, avr::hw::db.mem.eeprom_size & 0xff, avr::hw::db.mem.eeprom_size >> 8,
|
||||
static_cast<std::uint8_t>((boot_section ? 0 : 1) | (word_flash ? 2 : 0)), // patch-vector, word-addressed
|
||||
'P',
|
||||
'B',
|
||||
version,
|
||||
avr::hw::db.signature[0],
|
||||
avr::hw::db.signature[1],
|
||||
avr::hw::db.signature[2],
|
||||
static_cast<std::uint8_t>(page), // 0 means 256
|
||||
wire_base & 0xff,
|
||||
wire_base >> 8,
|
||||
avr::hw::db.mem.eeprom_size & 0xff,
|
||||
avr::hw::db.mem.eeprom_size >> 8,
|
||||
static_cast<std::uint8_t>((boot_section ? 0 : 1) | (word_flash ? 2 : 0)), // patch-vector, word-addressed
|
||||
}>
|
||||
info_data;
|
||||
// clang-format on
|
||||
|
||||
// The serial link, per the build's PUREBOOT_USART / PUREBOOT_SOFT_SERIAL,
|
||||
// defaulting to the chip's USART0 where it has one. The software receiver is
|
||||
@@ -321,8 +332,7 @@ void program_flash(std::uint16_t wire_address, std::uint8_t slot_high)
|
||||
// A page is aligned, so it never crosses 64 KiB: RAMPZ is a per-page
|
||||
// constant and the 16-bit Z's low byte is the whole in-page offset.
|
||||
const std::uint8_t rampz = static_cast<std::uint8_t>(wire_address >> 15);
|
||||
const std::uint16_t z0 =
|
||||
static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
std::uint16_t z = z0;
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
|
||||
@@ -24,16 +24,73 @@ else:
|
||||
import termios
|
||||
|
||||
PROMPT = b"+"
|
||||
VERSION = 2 # this tool's own version — free to drift from a loader's
|
||||
VERSION = 3 # this tool's own version — free to drift from a loader's
|
||||
# The loader versions this tool speaks. A pureboot version implies its wire
|
||||
# protocol, which carries no number of its own, so this window is where that
|
||||
# map lives: every version so far speaks the same protocol, and one that
|
||||
# changes it becomes the new floor here.
|
||||
OLDEST_LOADER = 1
|
||||
NEWEST_LOADER = 3
|
||||
NEWEST_LOADER = 5
|
||||
SLOT = 512 # the loader slot, on every chip
|
||||
RETRIES = 3 # rewrites of a page that reads back wrong, before the run stops
|
||||
|
||||
# pureboot 5 replaced the four per-memory commands with one unified pair: 'G'
|
||||
# reads and 'g' writes, both taking a selector byte, a 16-bit address and a
|
||||
# count, over the spaces below. The loader carries one transfer loop instead of
|
||||
# four bodies, which is what paid for RAM access and for the calibration fix.
|
||||
UNIFIED_LOADER = 5
|
||||
SP_FLASH, SP_EEPROM, SP_RAM, SP_FUSE, SP_SPM = 0, 1, 2, 3, 4
|
||||
|
||||
# A selector's high nibble is flash's third address byte, so a transfer names a
|
||||
# byte address within a 64 KiB bank and never has to speak word addresses. No
|
||||
# single transfer may cross a bank boundary — the host chunks to keep that true.
|
||||
def selector(space, address):
|
||||
return space | ((address >> 16) << 4)
|
||||
|
||||
|
||||
# The SPM operations pureboot 5 leaves to the host: a write to SP_SPM hands its
|
||||
# data byte to SPMCSR and fires the instruction at the selected flash address.
|
||||
# Every part pureboot targets agrees on these encodings.
|
||||
SPM_ERASE, SPM_WRITE, SPM_RWWSRE = 0x03, 0x05, 0x11
|
||||
|
||||
# Calibration byte for the autobaud loader: 0xC0 is a start bit plus six zero
|
||||
# data bits — one low pulse of seven bit-times, which the loader times into its
|
||||
# per-bit unit. Sent at whatever baud the host chose; the loader locks to it.
|
||||
CALIBRATE = 0xC0
|
||||
|
||||
# The autobaud loader slims its info block to the version and signature; the host
|
||||
# derives the rest of the geometry from the signature. flash, page, eeprom,
|
||||
# patch-vector per distinct signature, over every chip pureboot targets (the
|
||||
# loader computes the same from its chip database at build time). Die revisions
|
||||
# that share a signature share this row, as they share the silicon.
|
||||
AUTOBAUD_GEOMETRY = {
|
||||
# signature : (flash, page, eeprom, patch_vector)
|
||||
(0x1E, 0x90, 0x07): (1024, 32, 64, True), # ATtiny13/13A
|
||||
(0x1E, 0x91, 0x08): (2048, 32, 128, True), # ATtiny25
|
||||
(0x1E, 0x92, 0x06): (4096, 64, 256, True), # ATtiny45
|
||||
(0x1E, 0x93, 0x0B): (8192, 64, 512, True), # ATtiny85
|
||||
(0x1E, 0x92, 0x05): (4096, 64, 256, True), # ATmega48/48A
|
||||
(0x1E, 0x92, 0x0A): (4096, 64, 256, True), # ATmega48P/48PA
|
||||
(0x1E, 0x93, 0x07): (8192, 64, 512, False), # ATmega8/8A
|
||||
(0x1E, 0x93, 0x0A): (8192, 64, 512, False), # ATmega88/88A
|
||||
(0x1E, 0x93, 0x0F): (8192, 64, 512, False), # ATmega88P/88PA
|
||||
(0x1E, 0x94, 0x03): (16384, 128, 512, False), # ATmega16/16A
|
||||
(0x1E, 0x94, 0x06): (16384, 128, 512, False), # ATmega168/168A
|
||||
(0x1E, 0x94, 0x0B): (16384, 128, 512, False), # ATmega168P/168PA
|
||||
(0x1E, 0x94, 0x0A): (16384, 128, 512, False), # ATmega164P/164PA
|
||||
(0x1E, 0x94, 0x0F): (16384, 128, 512, False), # ATmega164A
|
||||
(0x1E, 0x95, 0x02): (32768, 128, 1024, False), # ATmega32/32A
|
||||
(0x1E, 0x95, 0x0F): (32768, 128, 1024, False), # ATmega328P
|
||||
(0x1E, 0x95, 0x14): (32768, 128, 1024, False), # ATmega328
|
||||
(0x1E, 0x95, 0x08): (32768, 128, 1024, False), # ATmega324P
|
||||
(0x1E, 0x95, 0x11): (32768, 128, 1024, False), # ATmega324PA
|
||||
(0x1E, 0x95, 0x15): (32768, 128, 1024, False), # ATmega324A
|
||||
(0x1E, 0x96, 0x09): (65536, 256, 2048, False), # ATmega644/644A
|
||||
(0x1E, 0x96, 0x0A): (65536, 256, 2048, False), # ATmega644P/644PA
|
||||
(0x1E, 0x97, 0x05): (131072, 256, 4096, False),# ATmega1284P
|
||||
(0x1E, 0x97, 0x06): (131072, 256, 4096, False),# ATmega1284
|
||||
}
|
||||
|
||||
VERBOSE = False
|
||||
|
||||
|
||||
@@ -301,6 +358,28 @@ Port = WindowsPort if os.name == "nt" else PosixPort
|
||||
class Info:
|
||||
"""The 12-byte info block."""
|
||||
|
||||
@classmethod
|
||||
def from_slim(cls, raw):
|
||||
"""The autobaud loader's slimmed reply — version and signature only —
|
||||
with the rest of the geometry looked up from the signature (the loader
|
||||
derived it from the same chip facts at build time). Reconstructs the full
|
||||
block so every derived attribute matches the fixed-baud path exactly."""
|
||||
if len(raw) != 4:
|
||||
raise Error(f"bad slim info block: {raw.hex()}")
|
||||
version, signature = raw[0], tuple(raw[1:4])
|
||||
geometry = AUTOBAUD_GEOMETRY.get(signature)
|
||||
if geometry is None:
|
||||
sig = " ".join(f"{b:02x}" for b in signature)
|
||||
raise Error(f"unknown signature {sig} — this tool has no autobaud geometry for it")
|
||||
flash, page, eeprom, patch = geometry
|
||||
base = flash - SLOT
|
||||
word_flash = flash > 0x10000
|
||||
wire_base = base // 2 if word_flash else base
|
||||
flags = (1 if patch else 0) | (2 if word_flash else 0)
|
||||
raw12 = bytes((ord("P"), ord("B"), version, *signature, page & 0xFF,
|
||||
wire_base & 0xFF, wire_base >> 8, eeprom & 0xFF, eeprom >> 8, flags))
|
||||
return cls(raw12)
|
||||
|
||||
def __init__(self, raw):
|
||||
if len(raw) != 12 or raw[0:2] != b"PB":
|
||||
raise Error(f"bad info block: {raw.hex()}")
|
||||
@@ -396,6 +475,37 @@ class Loader:
|
||||
if time.monotonic() > deadline:
|
||||
raise Error("no answer — reset the device within its activation window")
|
||||
|
||||
def connect_autobaud(self, wait):
|
||||
"""The autobaud handshake. Instead of the p+b knock, the host sends the
|
||||
0xC0 calibration pulse — a single seven-bit-time low pulse at the host's
|
||||
chosen baud — which the loader times into its per-bit unit, then a single
|
||||
'p' knock the loader decodes at the rate it just measured. As with
|
||||
connect(), each attempt is the whole handshake, retried until the slim
|
||||
info block comes back or the window closes: a lost pulse or a knock that
|
||||
lands while the loader is mid-frame simply fails to answer, and the
|
||||
loader's measurement loop is back waiting for the next pulse."""
|
||||
deadline = time.monotonic() + wait
|
||||
knocks = 0
|
||||
while True:
|
||||
self.port.flush_input()
|
||||
self.port.write(bytes((CALIBRATE, ord("p"))))
|
||||
knocks += 1
|
||||
if PROMPT in self.port.read_available(0.4):
|
||||
while self.port.read_available(0.3):
|
||||
pass
|
||||
self.port.write(b"b")
|
||||
try:
|
||||
block = self.port.read_exact(4, 2.0)
|
||||
except Error:
|
||||
block = b""
|
||||
if len(block) == 4:
|
||||
self.info = Info.from_slim(block)
|
||||
self._expect_prompt()
|
||||
verbose(f"loader locked on knock {knocks}; slim info read")
|
||||
return self.info
|
||||
if time.monotonic() > deadline:
|
||||
raise Error("no answer — reset the device within its activation window")
|
||||
|
||||
def _expect_prompt(self, timeout=2.0):
|
||||
byte = self.port.read_exact(1, timeout)
|
||||
if byte != PROMPT:
|
||||
@@ -418,7 +528,58 @@ class Loader:
|
||||
count -= chunk
|
||||
return data
|
||||
|
||||
@property
|
||||
def unified(self):
|
||||
"""pureboot 5 and later: one 'G'/'g' pair over selector-named spaces."""
|
||||
return self.info is not None and self.info.version >= UNIFIED_LOADER
|
||||
|
||||
def _read_space(self, space, address, count):
|
||||
"""A run out of any space, chunked to 256 bytes and to bank bounds."""
|
||||
data = b""
|
||||
while count:
|
||||
chunk = min(count, 256, 0x10000 - (address & 0xFFFF))
|
||||
head = bytes((ord("G"), selector(space, address), address & 0xFF,
|
||||
(address >> 8) & 0xFF, chunk & 0xFF))
|
||||
data += self._command(head, chunk, 5.0)
|
||||
address += chunk
|
||||
count -= chunk
|
||||
return data
|
||||
|
||||
def _write_space(self, space, address, data, progress=None):
|
||||
"""A run into any space. Each byte is acked as its write begins — an
|
||||
EEPROM cell and an SPM operation both need that pacing, and the ack is
|
||||
what the loader sends in place of a completion status."""
|
||||
offset = 0
|
||||
while offset < len(data):
|
||||
chunk = data[offset : offset + min(256, 0x10000 - (address & 0xFFFF))]
|
||||
head = bytes((ord("g"), selector(space, address), address & 0xFF,
|
||||
(address >> 8) & 0xFF, len(chunk) & 0xFF))
|
||||
self.port.write(head)
|
||||
for byte in chunk:
|
||||
self.port.write(bytes((byte,)))
|
||||
self._expect_prompt()
|
||||
if progress:
|
||||
progress.step()
|
||||
self._expect_prompt() # the next command prompt
|
||||
address += len(chunk)
|
||||
offset += len(chunk)
|
||||
|
||||
def spm(self, operation, address):
|
||||
"""One SPM operation at a flash address — the erase, write and RWW
|
||||
re-enable that pureboot 4 ran inside 'W' and pureboot 5 leaves here."""
|
||||
self._write_space(SP_SPM, address, bytes((operation,)))
|
||||
|
||||
def read_ram(self, address, count):
|
||||
"""Data space: SRAM, and with it the register file and every I/O
|
||||
register, which share the address space on AVR. New in pureboot 5."""
|
||||
return self._read_space(SP_RAM, address, count)
|
||||
|
||||
def write_ram(self, address, data):
|
||||
self._write_space(SP_RAM, address, data)
|
||||
|
||||
def read_flash(self, address, count):
|
||||
if self.unified:
|
||||
return self._read_space(SP_FLASH, address, count)
|
||||
if not self.info.word_flash:
|
||||
return self._stream_read("R", address, count)
|
||||
# Word-addressed wire: widen to even bounds and never let one read
|
||||
@@ -436,15 +597,32 @@ class Loader:
|
||||
return data[address - start : address - start + count]
|
||||
|
||||
def read_eeprom(self, address, count):
|
||||
if self.unified:
|
||||
return self._read_space(SP_EEPROM, address, count)
|
||||
return self._stream_read("r", address, count)
|
||||
|
||||
def write_page(self, address, data):
|
||||
assert len(data) == self.info.page and address % self.info.page == 0
|
||||
if self.unified:
|
||||
# 'W' fills the page buffer and stops there; the erase and the write
|
||||
# are host-issued SPM operations. Only a chip with a boot section
|
||||
# has RWW to re-enable — on the others bit 4 of SPMCSR means
|
||||
# something else entirely, so it must not be sent.
|
||||
head = bytes((ord("W"), selector(SP_FLASH, address), address & 0xFF, (address >> 8) & 0xFF))
|
||||
self._command(head + data, 0, 2.0)
|
||||
self.spm(SPM_ERASE, address)
|
||||
self.spm(SPM_WRITE, address)
|
||||
if not self.info.patch_vector:
|
||||
self.spm(SPM_RWWSRE, address)
|
||||
return
|
||||
wire = address // (2 if self.info.word_flash else 1)
|
||||
head = bytes((ord("W"), wire & 0xFF, wire >> 8))
|
||||
self._command(head + data, 0, 2.0)
|
||||
|
||||
def write_eeprom(self, address, data, progress=None):
|
||||
if self.unified:
|
||||
self._write_space(SP_EEPROM, address, data, progress)
|
||||
return
|
||||
offset = 0
|
||||
while offset < len(data):
|
||||
chunk = data[offset : offset + 256]
|
||||
@@ -460,6 +638,8 @@ class Loader:
|
||||
offset += len(chunk)
|
||||
|
||||
def read_fuses(self):
|
||||
if self.unified:
|
||||
return self._read_space(SP_FUSE, 0, 4)
|
||||
return self._command(b"F", 4, 2.0)
|
||||
|
||||
def jump(self, word_address):
|
||||
@@ -1016,6 +1196,37 @@ def op_read_eeprom(loader, path):
|
||||
print(f"read EEPROM: {len(data)} B -> {path}")
|
||||
|
||||
|
||||
def _require_unified(loader, what):
|
||||
if not loader.unified:
|
||||
raise Error(f"{what} needs pureboot {UNIFIED_LOADER} or later; this loader is {loader.info.version}")
|
||||
|
||||
|
||||
def _peek_spec(spec):
|
||||
"""ADDR[:N] — addresses and counts in any Python integer base."""
|
||||
address, _, count = spec.partition(":")
|
||||
return int(address, 0), int(count, 0) if count else 1
|
||||
|
||||
|
||||
def op_peek(loader, spec):
|
||||
_require_unified(loader, "--peek")
|
||||
address, count = _peek_spec(spec)
|
||||
data = loader.read_ram(address, count)
|
||||
for offset in range(0, len(data), 16):
|
||||
row = data[offset : offset + 16]
|
||||
text = "".join(chr(b) if 0x20 <= b < 0x7F else "." for b in row)
|
||||
print(f"{address + offset:#06x} {row.hex(' '):<47} {text}")
|
||||
|
||||
|
||||
def op_poke(loader, spec):
|
||||
_require_unified(loader, "--poke")
|
||||
address, _, payload = spec.partition(":")
|
||||
if not payload:
|
||||
raise Error("--poke needs ADDR:HEX, for example 0x200:deadbeef")
|
||||
data = bytes.fromhex(payload.replace(" ", ""))
|
||||
loader.write_ram(int(address, 0), data)
|
||||
print(f"poke: {len(data)} B at {int(address, 0):#06x}")
|
||||
|
||||
|
||||
def op_fuses(loader):
|
||||
low, lock, extended, high = loader.read_fuses()
|
||||
print("fuses:")
|
||||
@@ -1049,6 +1260,9 @@ def main():
|
||||
parser.add_argument("--port", required=True, help="serial device: COM6, /dev/ttyUSB0, or a simavr pty")
|
||||
parser.add_argument("--baud", type=int, default=115200, help="115200 mega, 57600 tinies")
|
||||
parser.add_argument("--wait", type=float, default=30.0, help="seconds to keep knocking")
|
||||
parser.add_argument("--autobaud", action="store_true",
|
||||
help="drive an autobaud loader: send the 0xC0 calibration pulse and a single "
|
||||
"knock, and take geometry from the signature (no clock/baud baked in)")
|
||||
parser.add_argument("--info", action="store_true", help="print the device info block")
|
||||
parser.add_argument("--fuses", action="store_true", help="read the fuse and lock bytes")
|
||||
parser.add_argument("--update-loader", metavar="FILE", help="replace the loader with this pureboot binary")
|
||||
@@ -1064,6 +1278,10 @@ def main():
|
||||
parser.add_argument("--eeprom", metavar="FILE", help="program the EEPROM (bin or ihex)")
|
||||
parser.add_argument("--read-eeprom", metavar="FILE", help="dump the EEPROM")
|
||||
parser.add_argument("--verify-eeprom", metavar="FILE", help="compare EEPROM against an image")
|
||||
parser.add_argument("--peek", metavar="ADDR[:N]", help="read N bytes of data space (SRAM, registers, "
|
||||
"I/O) — pureboot 5 and later")
|
||||
parser.add_argument("--poke", metavar="ADDR:HEX", help="write hex bytes into data space — "
|
||||
"pureboot 5 and later")
|
||||
parser.add_argument("--force", action="store_true", help="override refusable safety checks")
|
||||
parser.add_argument("--stay", action="store_true", help="leave the loader in its session")
|
||||
parser.add_argument("-v", "--verbose", action="store_true",
|
||||
@@ -1086,7 +1304,7 @@ def main():
|
||||
verbose(f"{args.port}: {args.baud} Bd 8N1, DTR/RTS asserted")
|
||||
try:
|
||||
loader = Loader(port)
|
||||
info = loader.connect(args.wait)
|
||||
info = loader.connect_autobaud(args.wait) if args.autobaud else loader.connect(args.wait)
|
||||
if args.info:
|
||||
print("device:")
|
||||
for line in info.lines():
|
||||
@@ -1115,6 +1333,10 @@ def main():
|
||||
op_read_eeprom(loader, args.read_eeprom)
|
||||
if args.verify_eeprom:
|
||||
op_verify_eeprom(loader, args.verify_eeprom)
|
||||
if args.poke:
|
||||
op_poke(loader, args.poke)
|
||||
if args.peek:
|
||||
op_peek(loader, args.peek)
|
||||
if args.stay:
|
||||
print("loader stays in its session (reset to leave)")
|
||||
else:
|
||||
|
||||
396
pureboot/pureboot_autobaud_pure.cpp
Normal file
396
pureboot/pureboot_autobaud_pure.cpp
Normal file
@@ -0,0 +1,396 @@
|
||||
// SUPERSEDED — kept for the record, not built. With the activation hang fixed
|
||||
// (a lone calibration pulse used to wedge the loader, see autobaud.md) this
|
||||
// version is 530 B on the 1284P against a 512 B slot. Its 4 B of margin was
|
||||
// never spare capacity; it was the space the missing fix should have occupied.
|
||||
// pureboot_autobaud_uni.cpp replaces it at 464 B with more features.
|
||||
//
|
||||
// pureboot autobaud — the software-serial variant that measures the host's bit
|
||||
// timing at runtime, so one binary runs at any F_CPU: the image carries no
|
||||
// clock. THE PURE VERSION — no inline assembly, no global register variables,
|
||||
// exactly the constraints the fixed-baud loader keeps.
|
||||
//
|
||||
// Two source files exist for review (autobaud.md):
|
||||
// this one, pure, and pureboot_autobaud_reg.cpp, which keeps the running-slot
|
||||
// write guard at the cost of one global register variable. They differ only in
|
||||
// where the measured unit lives and whether the guard is present.
|
||||
//
|
||||
// What this version trades to fit 512 B in pure C++ (the 1284 at 508), each
|
||||
// licensed by "the host guarantees safety" (README.md) and the owner's approval
|
||||
// to simplify the info block:
|
||||
// - the measured per-bit unit lives in the two general-purpose I/O scratch
|
||||
// registers (GPIOR) where the chip has them, in a static otherwise —
|
||||
// reached through libavr's named register surface, so no asm and no global
|
||||
// register variable; the loader stays pure;
|
||||
// - the info block is slimmed to the version and the signature — the chip's
|
||||
// identity — from which the host derives page size, loader base, EEPROM
|
||||
// size and the addressing flags via its own chip database;
|
||||
// - no running-slot write guard: the host never programs the loader's own
|
||||
// slot, and a broken host bricking the target is the host's bug;
|
||||
// - a single-byte activation knock: the calibration pulse already proves a
|
||||
// host is present.
|
||||
//
|
||||
// Position independence is kept and is in fact total here: control flow is
|
||||
// PC-relative, the wire carries addresses, and with the slimmed info block and
|
||||
// no write guard nothing anchors on the runtime address at all.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <util/delay_basic.h>
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
namespace hw = avr::hw;
|
||||
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Autobaud carries no clock, so PUREBOOT_CLOCK_HZ / PUREBOOT_BAUD are not
|
||||
// consulted; only the software-UART pins are a deployment parameter.
|
||||
#if !defined(PUREBOOT_RX)
|
||||
#define PUREBOOT_RX pb0
|
||||
#endif
|
||||
#if !defined(PUREBOOT_TX)
|
||||
#define PUREBOOT_TX pb1
|
||||
#endif
|
||||
|
||||
// The watchdog reset flag's home: MCUSR, or the classic megas' MCUCSR.
|
||||
consteval std::int16_t wdrf_field()
|
||||
{
|
||||
auto reg = std::string_view{hw::db.regs[static_cast<std::size_t>(avr::power::detail::reset_reg())].name};
|
||||
return hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
// The loader owns the top 512 bytes; a staging copy goes in the slot below.
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = hw::curated::has_boot_section();
|
||||
|
||||
// Past 64 KiB a byte address no longer fits the wire's 16 bits, so flash
|
||||
// addresses there are word addresses ('J' always was one).
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
|
||||
// The loader's one identity number (README.md); the slimmed info block carries
|
||||
// it and the signature, and the host maps a version to its protocol.
|
||||
constexpr std::uint8_t version = 4;
|
||||
|
||||
// The activation window as a fixed poll budget: with no clock, whole seconds
|
||||
// cannot be timed. A __uint24 (AVR's three-byte type) holds it — a fourth byte
|
||||
// would cost two words at each countdown step for range never used.
|
||||
#if !defined(PUREBOOT_AUTOBAUD_POLLS)
|
||||
#define PUREBOOT_AUTOBAUD_POLLS 4000000
|
||||
#endif
|
||||
constexpr __uint24 autobaud_budget = PUREBOOT_AUTOBAUD_POLLS;
|
||||
|
||||
// The measured per-bit delay (in _delay_loop_2 four-cycle iterations). It lives
|
||||
// in the two adjacent general-purpose I/O scratch registers (GPIOR1:GPIOR2)
|
||||
// where the chip has them — in/out reach them in one word where a static's
|
||||
// lds/sts take two, and there is no .bss to clear — and in a plain static
|
||||
// otherwise (the t13, m8 and m16/32 have no GPIOR). Both are pure: the named
|
||||
// register surface, no inline asm, no global register variable.
|
||||
constexpr bool have_gpior = hw::db.reg_index("GPIOR1") >= 0 && hw::db.reg_index("GPIOR2") >= 0;
|
||||
|
||||
std::uint16_t unit_backing;
|
||||
|
||||
template <bool Gpior = have_gpior>
|
||||
[[gnu::always_inline]] inline std::uint16_t get_unit()
|
||||
{
|
||||
if constexpr (Gpior)
|
||||
return static_cast<std::uint16_t>(hw::reg_impl<hw::db.reg_index("GPIOR1")>::read() |
|
||||
(hw::reg_impl<hw::db.reg_index("GPIOR2")>::read() << 8));
|
||||
else
|
||||
return unit_backing;
|
||||
}
|
||||
|
||||
template <bool Gpior = have_gpior>
|
||||
[[gnu::always_inline]] inline void put_unit(std::uint16_t u)
|
||||
{
|
||||
if constexpr (Gpior) {
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR1")>::write(static_cast<std::uint8_t>(u));
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR2")>::write(static_cast<std::uint8_t>(u >> 8));
|
||||
} else
|
||||
unit_backing = u;
|
||||
}
|
||||
|
||||
// The autobaud software link: bit-banged with cycle-counted delays like the
|
||||
// fixed-baud software backend, but the per-bit delay is the measured unit, not
|
||||
// a consteval constant. rx/tx load the unit once into a local, so each bit
|
||||
// spins from a register with no per-bit reload.
|
||||
struct link {
|
||||
using in_t = avr::io::input<avr::PUREBOOT_RX, avr::io::pull::up>;
|
||||
using out_t = avr::io::output<avr::PUREBOOT_TX>;
|
||||
|
||||
static void init()
|
||||
{
|
||||
avr::init<in_t, out_t>();
|
||||
out_t::set(); // idle high
|
||||
}
|
||||
|
||||
// Time the calibration pulse into the unit. The host sends 0xC0 — a start
|
||||
// bit plus six zero data bits are one low pulse of seven bit-times — and the
|
||||
// counted poll loop is seven cycles an iteration, so the count is the pulse
|
||||
// length in cycles ÷ 7 × 7 = one bit period in cycles, and count >> 2 is that
|
||||
// period in _delay_loop_2's four-cycle iterations. Waits for the start edge
|
||||
// under the poll budget; 0 (returned, and stored) means the budget expired.
|
||||
static std::uint16_t measure(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
std::uint16_t count = 0;
|
||||
while (!in_t::read())
|
||||
++count;
|
||||
const std::uint16_t u = static_cast<std::uint16_t>(count >> 2);
|
||||
put_unit(u);
|
||||
return u;
|
||||
}
|
||||
|
||||
// The eight data bits, entered just past the falling edge of the start bit.
|
||||
// Split out of rx() so the activation path can wait for that edge under a
|
||||
// budget while the command loop waits for it indefinitely.
|
||||
static std::uint8_t sample()
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
_delay_loop_2(static_cast<std::uint16_t>(unit + (unit >> 1))); // 1.5 bits to the LSB centre
|
||||
std::uint8_t value = 0;
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
value >>= 1;
|
||||
if (in_t::read())
|
||||
value |= 0x80;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static std::uint8_t rx()
|
||||
{
|
||||
while (in_t::read()) // await the start edge
|
||||
;
|
||||
return sample();
|
||||
}
|
||||
|
||||
// A byte under the poll budget, for the activation knock. An expired budget
|
||||
// returns 0, which is not the knock, so the caller falls back into the
|
||||
// budgeted measure() — and a line that stays idle boots the application
|
||||
// there. Without this the knock's edge wait was unbounded, so a single
|
||||
// spurious calibration pulse (EMI, or a host that opens the port and never
|
||||
// knocks) wedged the loader and the application never ran.
|
||||
static std::uint8_t rx_bounded(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
return sample();
|
||||
}
|
||||
|
||||
static void tx(std::uint8_t value)
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
out_t::clear(); // start bit
|
||||
_delay_loop_2(unit);
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
out_t::write(value & 1);
|
||||
value >>= 1;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
out_t::set(); // stop bit
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
|
||||
static void drain()
|
||||
{
|
||||
// The software transmitter returns only after the stop bit.
|
||||
}
|
||||
};
|
||||
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
{
|
||||
target();
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
[[gnu::noinline, noreturn]] void run_app()
|
||||
{
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
// Inlined: read across a call, the first byte strands in a call-saved register
|
||||
// the caller has to push and pop.
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_near(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(avr::flash_load(reinterpret_cast<const std::uint8_t *>(address++)));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_far(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
std::uint8_t rampz = static_cast<std::uint8_t>(address >> 15);
|
||||
std::uint16_t z = static_cast<std::uint16_t>(address << 1);
|
||||
do {
|
||||
link::tx(avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(rampz) << 16) | z));
|
||||
if (++z == 0)
|
||||
++rampz;
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline void send_flash(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
if constexpr (word_flash)
|
||||
send_flash_far(address, count);
|
||||
else
|
||||
send_flash_near(address, count);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
void send_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(ee::read(address++));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
void store_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do {
|
||||
ee::write<off>(address++, link::rx());
|
||||
tx_ack();
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
// One page into the SPM buffer, then erase and program. No running-slot write
|
||||
// guard: the host guarantees it never targets the loader's own slot (the pure
|
||||
// version's one dropped safety net, licensed — README.md).
|
||||
void program_flash(std::uint16_t wire_address)
|
||||
{
|
||||
spm::flash_address_t address;
|
||||
if constexpr (word_flash) {
|
||||
const std::uint8_t rampz = static_cast<std::uint8_t>(wire_address >> 15);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
std::uint16_t z = z0;
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(rampz) << 16) | z, word_of({low, high}));
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z));
|
||||
address = (static_cast<spm::flash_address_t>(rampz) << 16) | z0;
|
||||
} else {
|
||||
address = static_cast<spm::flash_address_t>(wire_address & ~static_cast<std::uint16_t>(page - 1));
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>(address, word_of({low, high}));
|
||||
address += 2;
|
||||
} while (static_cast<std::uint8_t>(address) & (page - 1));
|
||||
address -= 2; // back inside the page — erase and write ignore the word bits
|
||||
}
|
||||
spm::erase_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
spm::write_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
}
|
||||
|
||||
void send_fuses()
|
||||
{
|
||||
std::uint8_t which = 0;
|
||||
do
|
||||
link::tx(spm::read_fuse<off>(static_cast<spm::fuse>(which)));
|
||||
while (++which != 4);
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
if (hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// Measure the calibration pulse into the unit, then take one 'p' knock. An
|
||||
// expired budget (no host) boots the application; a pulse that decodes to
|
||||
// anything but 'p' re-measures.
|
||||
for (;;) {
|
||||
if (link::measure(autobaud_budget) == 0)
|
||||
run_app();
|
||||
if (link::rx_bounded(autobaud_budget) == 'p')
|
||||
break;
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
ee::wait();
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // chip identity: version then the three signature bytes
|
||||
link::tx(version);
|
||||
link::tx(hw::db.signature[0]);
|
||||
link::tx(hw::db.signature[1]);
|
||||
link::tx(hw::db.signature[2]);
|
||||
break;
|
||||
case 'R': // read flash: addr16, n8 (0 = 256)
|
||||
case 'r': // read EEPROM: addr16, n8
|
||||
case 'w': { // write EEPROM: addr16, n8, then n bytes each acked
|
||||
// One address-and-count path for the three, so the flash streamer
|
||||
// keeps a single call site and inlines into this never-returning
|
||||
// loop — its cursor then lives in the loop's own call-saved
|
||||
// registers instead of being saved and restored around a call
|
||||
// (the call-site-count lesson, autobaud.md).
|
||||
const std::uint16_t address = rx16();
|
||||
const std::uint8_t count = link::rx();
|
||||
if (command == 'r')
|
||||
send_eeprom(address, count);
|
||||
else if (command == 'w')
|
||||
store_eeprom(address, count);
|
||||
else
|
||||
send_flash(address, count);
|
||||
break;
|
||||
}
|
||||
case 'W': // program one flash page: addr16, page bytes
|
||||
program_flash(rx16());
|
||||
break;
|
||||
case 'F': // fuse and lock bytes
|
||||
send_fuses();
|
||||
break;
|
||||
default: // unknown bytes are ignored; the loop re-acks
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace pureboot
|
||||
|
||||
template struct avr::startup::entry<pureboot::run>;
|
||||
371
pureboot/pureboot_autobaud_reg.cpp
Normal file
371
pureboot/pureboot_autobaud_reg.cpp
Normal file
@@ -0,0 +1,371 @@
|
||||
// SUPERSEDED — kept for the record, not built. With the activation hang fixed
|
||||
// (a lone calibration pulse used to wedge the loader, see autobaud.md) this
|
||||
// version is 534 B on the 1284P against a 512 B slot, so the global register
|
||||
// variable it broke purity for buys nothing. pureboot_autobaud_uni.cpp replaces
|
||||
// it at 464 B, strictly pure and with more features.
|
||||
//
|
||||
// pureboot autobaud — the software-serial variant that measures the host's bit
|
||||
// timing at runtime, so one binary runs at any F_CPU: the image carries no
|
||||
// clock. THE REGISTER VERSION — keeps the running-slot write guard, at the cost
|
||||
// of one global register variable (r4) holding the measured unit. That variable
|
||||
// is pureboot's single, deliberate break from its no-global-register-variable
|
||||
// rule, present only in this variant; everything else stays pure C++.
|
||||
//
|
||||
// Two source files exist for review (autobaud.md):
|
||||
// pureboot_autobaud_pure.cpp, fully pure but dropping the write guard, and this
|
||||
// one. They differ only in where the measured unit lives (a call-saved register
|
||||
// here, GPIOR/RAM there) and whether the guard is present.
|
||||
//
|
||||
// The register buys ~32 B over a RAM home — an outlined rx/tx reads it with one
|
||||
// move where a static costs an lds — and that is what lets the write guard stay
|
||||
// while the image still fits 512 B (the 1284 at 512, exactly). The unit is
|
||||
// written once through a noinline setter so the store lands immediately before a
|
||||
// ret: GCC otherwise deletes a global-register store whose only readers are
|
||||
// callees (autobaud.md, upstream bug 6).
|
||||
//
|
||||
// Simplifications shared with the pure version, each licensed: a slimmed info
|
||||
// block (version + signature; the host derives geometry from its chip database)
|
||||
// and a single-byte activation knock (the calibration pulse already proves a
|
||||
// host). Position independence is kept: control flow is PC-relative and the
|
||||
// write guard anchors on the runtime return address, as the fixed-baud loader.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <util/delay_basic.h>
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
namespace hw = avr::hw;
|
||||
|
||||
// The measured per-bit delay (in _delay_loop_2 four-cycle iterations) in a
|
||||
// call-saved register that the serial callees read directly, so no wire path
|
||||
// threads it and rx/tx reach it with a move, not a load. Written only through
|
||||
// set_unit() below.
|
||||
register std::uint16_t g_unit asm("r4");
|
||||
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Autobaud carries no clock; only the software-UART pins are a parameter.
|
||||
#if !defined(PUREBOOT_RX)
|
||||
#define PUREBOOT_RX pb0
|
||||
#endif
|
||||
#if !defined(PUREBOOT_TX)
|
||||
#define PUREBOOT_TX pb1
|
||||
#endif
|
||||
|
||||
consteval std::int16_t wdrf_field()
|
||||
{
|
||||
auto reg = std::string_view{hw::db.regs[static_cast<std::size_t>(avr::power::detail::reset_reg())].name};
|
||||
return hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = hw::curated::has_boot_section();
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
|
||||
constexpr std::uint8_t version = 4;
|
||||
|
||||
#if !defined(PUREBOOT_AUTOBAUD_POLLS)
|
||||
#define PUREBOOT_AUTOBAUD_POLLS 4000000
|
||||
#endif
|
||||
constexpr __uint24 autobaud_budget = PUREBOOT_AUTOBAUD_POLLS;
|
||||
|
||||
// The one store into g_unit, isolated so it lands right before the ret: a
|
||||
// global-register store whose only later readers are callees is dropped
|
||||
// otherwise (autobaud.md, upstream bug 6).
|
||||
[[gnu::noinline]] void set_unit(std::uint16_t v)
|
||||
{
|
||||
g_unit = v;
|
||||
}
|
||||
|
||||
// The autobaud software link: bit-banged with cycle-counted delays, but the
|
||||
// per-bit delay is g_unit, measured from the host's calibration pulse.
|
||||
struct link {
|
||||
using in_t = avr::io::input<avr::PUREBOOT_RX, avr::io::pull::up>;
|
||||
using out_t = avr::io::output<avr::PUREBOOT_TX>;
|
||||
|
||||
static void init()
|
||||
{
|
||||
avr::init<in_t, out_t>();
|
||||
out_t::set(); // idle high
|
||||
}
|
||||
|
||||
// Time the calibration pulse into g_unit. The host sends 0xC0 — a start bit
|
||||
// plus six zero data bits are one low pulse of seven bit-times — and the
|
||||
// counted poll loop is seven cycles an iteration, so count >> 2 is the bit
|
||||
// period in _delay_loop_2's four-cycle iterations. 0 means the budget
|
||||
// expired.
|
||||
static std::uint16_t measure(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
std::uint16_t count = 0;
|
||||
while (!in_t::read())
|
||||
++count;
|
||||
const std::uint16_t u = static_cast<std::uint16_t>(count >> 2);
|
||||
set_unit(u);
|
||||
return u;
|
||||
}
|
||||
|
||||
// The eight data bits, entered just past the falling edge of the start bit.
|
||||
// Split out of rx() so the activation path can wait for that edge under a
|
||||
// budget while the command loop waits for it indefinitely.
|
||||
static std::uint8_t sample()
|
||||
{
|
||||
_delay_loop_2(static_cast<std::uint16_t>(g_unit + (g_unit >> 1))); // 1.5 bits to the LSB centre
|
||||
std::uint8_t value = 0;
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
value >>= 1;
|
||||
if (in_t::read())
|
||||
value |= 0x80;
|
||||
_delay_loop_2(g_unit);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static std::uint8_t rx()
|
||||
{
|
||||
while (in_t::read()) // await the start edge
|
||||
;
|
||||
return sample();
|
||||
}
|
||||
|
||||
// A byte under the poll budget, for the activation knock. An expired budget
|
||||
// returns 0, which is not the knock, so the caller falls back into the
|
||||
// budgeted measure() — and a line that stays idle boots the application
|
||||
// there. Without this the knock's edge wait was unbounded, so a single
|
||||
// spurious calibration pulse (EMI, or a host that opens the port and never
|
||||
// knocks) wedged the loader and the application never ran.
|
||||
static std::uint8_t rx_bounded(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
return sample();
|
||||
}
|
||||
|
||||
static void tx(std::uint8_t value)
|
||||
{
|
||||
out_t::clear(); // start bit
|
||||
_delay_loop_2(g_unit);
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
out_t::write(value & 1);
|
||||
value >>= 1;
|
||||
_delay_loop_2(g_unit);
|
||||
}
|
||||
out_t::set(); // stop bit
|
||||
_delay_loop_2(g_unit);
|
||||
}
|
||||
|
||||
static void drain()
|
||||
{
|
||||
// The software transmitter returns only after the stop bit.
|
||||
}
|
||||
};
|
||||
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
{
|
||||
target();
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
[[gnu::noinline, noreturn]] void run_app()
|
||||
{
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_near(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(avr::flash_load(reinterpret_cast<const std::uint8_t *>(address++)));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_far(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
std::uint8_t rampz = static_cast<std::uint8_t>(address >> 15);
|
||||
std::uint16_t z = static_cast<std::uint16_t>(address << 1);
|
||||
do {
|
||||
link::tx(avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(rampz) << 16) | z));
|
||||
if (++z == 0)
|
||||
++rampz;
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline void send_flash(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
if constexpr (word_flash)
|
||||
send_flash_far(address, count);
|
||||
else
|
||||
send_flash_near(address, count);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
void send_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(ee::read(address++));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
void store_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do {
|
||||
ee::write<off>(address++, link::rx());
|
||||
tx_ack();
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
// One page into the SPM buffer, then erase and program — except the slot this
|
||||
// code is running in (`slot_high`, from run()), which is drained and left
|
||||
// alone. A broken host therefore cannot brick the running loader, and a copy one
|
||||
// slot lower may rewrite the resident one.
|
||||
void program_flash(std::uint16_t wire_address, std::uint8_t slot_high)
|
||||
{
|
||||
spm::flash_address_t address;
|
||||
std::uint8_t page_high;
|
||||
if constexpr (word_flash) {
|
||||
const std::uint8_t rampz = static_cast<std::uint8_t>(wire_address >> 15);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
std::uint16_t z = z0;
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(rampz) << 16) | z, word_of({low, high}));
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z));
|
||||
address = (static_cast<spm::flash_address_t>(rampz) << 16) | z0;
|
||||
page_high = static_cast<std::uint8_t>(wire_address >> 8);
|
||||
} else {
|
||||
address = static_cast<spm::flash_address_t>(wire_address & ~static_cast<std::uint16_t>(page - 1));
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>(address, word_of({low, high}));
|
||||
address += 2;
|
||||
} while (static_cast<std::uint8_t>(address) & (page - 1));
|
||||
address -= 2; // back inside the page — erase and write ignore the word bits
|
||||
page_high = static_cast<std::uint8_t>(address >> 8) & 0xfe;
|
||||
}
|
||||
if (page_high != slot_high) {
|
||||
spm::erase_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
spm::write_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
}
|
||||
if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
}
|
||||
|
||||
void send_fuses()
|
||||
{
|
||||
std::uint8_t which = 0;
|
||||
do
|
||||
link::tx(spm::read_fuse<off>(static_cast<spm::fuse>(which)));
|
||||
while (++which != 4);
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
if (hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// The high byte of the slot this copy runs at, which the write guard
|
||||
// follows: the return address is a word address, so its high byte is the
|
||||
// 256-word slot index, doubled back into byte terms on a byte-addressed
|
||||
// chip. Taken as byteswap's low byte — the builtin already swaps the two
|
||||
// stacked bytes, and the double swap folds away.
|
||||
const std::uint16_t ra_words = reinterpret_cast<std::uint16_t>(__builtin_return_address(0));
|
||||
const std::uint8_t ra_high = static_cast<std::uint8_t>(std::byteswap(ra_words));
|
||||
const std::uint8_t slot_high = word_flash ? ra_high : static_cast<std::uint8_t>(ra_high << 1);
|
||||
|
||||
// Measure the calibration pulse into g_unit, then take one 'p' knock. An
|
||||
// expired budget (no host) boots the application; a pulse that decodes to
|
||||
// anything but 'p' re-measures.
|
||||
for (;;) {
|
||||
if (link::measure(autobaud_budget) == 0)
|
||||
run_app();
|
||||
if (link::rx_bounded(autobaud_budget) == 'p')
|
||||
break;
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
ee::wait();
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // chip identity: version then the three signature bytes
|
||||
link::tx(version);
|
||||
link::tx(hw::db.signature[0]);
|
||||
link::tx(hw::db.signature[1]);
|
||||
link::tx(hw::db.signature[2]);
|
||||
break;
|
||||
case 'R': // read flash: addr16, n8 (0 = 256)
|
||||
case 'r': // read EEPROM: addr16, n8
|
||||
case 'w': { // write EEPROM: addr16, n8, then n bytes each acked
|
||||
// One address-and-count path for the three, so the flash streamer
|
||||
// keeps a single call site and inlines into this never-returning
|
||||
// loop (autobaud.md).
|
||||
const std::uint16_t address = rx16();
|
||||
const std::uint8_t count = link::rx();
|
||||
if (command == 'r')
|
||||
send_eeprom(address, count);
|
||||
else if (command == 'w')
|
||||
store_eeprom(address, count);
|
||||
else
|
||||
send_flash(address, count);
|
||||
break;
|
||||
}
|
||||
case 'W': // program one flash page: addr16, page bytes
|
||||
program_flash(rx16(), slot_high);
|
||||
break;
|
||||
case 'F': // fuse and lock bytes
|
||||
send_fuses();
|
||||
break;
|
||||
default: // unknown bytes are ignored; the loop re-acks
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace pureboot
|
||||
|
||||
template struct avr::startup::entry<pureboot::run>;
|
||||
411
pureboot/pureboot_autobaud_uni.cpp
Normal file
411
pureboot/pureboot_autobaud_uni.cpp
Normal file
@@ -0,0 +1,411 @@
|
||||
// pureboot autobaud, the unified-primitive version — VARIANT A, an explicit
|
||||
// space byte. One read command and one write command carry a space selector, so
|
||||
// flash, EEPROM, RAM and the fuses share a single cursor, a single transfer loop
|
||||
// and a single argument decode instead of one command body each.
|
||||
//
|
||||
// Strictly pure: no inline assembly, no global register variables, and no GPIOR
|
||||
// either — the measured unit lives in a plain static, so the loader claims no
|
||||
// chip resource an application might want. (PUREBOOT_UNIT_GPIOR=1 puts it back
|
||||
// in the I/O scratch registers, kept only as a measurement axis.)
|
||||
//
|
||||
// Against pureboot_autobaud_pure.cpp this version:
|
||||
// - adds RAM read and write, which the loader has never had. Because AVR maps
|
||||
// the register file and the whole I/O space into the data address space,
|
||||
// that one space also gives the host arbitrary peripheral access for free;
|
||||
// - collapses 'R' (read flash), 'r' (read EEPROM), 'w' (write EEPROM) and 'F'
|
||||
// (fuses) — four bodies, four loops — into 'G' and 'P' over four spaces;
|
||||
// - fixes the activation hang: a lone calibration pulse used to leave the
|
||||
// loader blocked forever in the knock's rx(), so a stray edge on an
|
||||
// unattended device wedged it in the loader and the application never ran.
|
||||
//
|
||||
// Position independence is kept and is total: control flow is PC-relative, the
|
||||
// wire carries addresses, and nothing anchors on the runtime address.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <util/delay_basic.h>
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
namespace hw = avr::hw;
|
||||
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Autobaud carries no clock, so PUREBOOT_CLOCK_HZ / PUREBOOT_BAUD are not
|
||||
// consulted; only the software-UART pins are a deployment parameter.
|
||||
#if !defined(PUREBOOT_RX)
|
||||
#define PUREBOOT_RX pb0
|
||||
#endif
|
||||
#if !defined(PUREBOOT_TX)
|
||||
#define PUREBOOT_TX pb1
|
||||
#endif
|
||||
|
||||
// The unit's home. A plain static by default — pureboot claims no GPIOR, so the
|
||||
// application keeps both scratch registers. The GPIOR spelling is retained
|
||||
// behind a macro purely so the two can be measured against each other.
|
||||
#if !defined(PUREBOOT_UNIT_GPIOR)
|
||||
#define PUREBOOT_UNIT_GPIOR 0
|
||||
#endif
|
||||
|
||||
// The watchdog reset flag's home: MCUSR, or the classic megas' MCUCSR.
|
||||
consteval std::int16_t wdrf_field()
|
||||
{
|
||||
auto reg = std::string_view{hw::db.regs[static_cast<std::size_t>(avr::power::detail::reset_reg())].name};
|
||||
return hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
// The loader owns the top 512 bytes; a staging copy goes in the slot below.
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = hw::curated::has_boot_section();
|
||||
|
||||
// Past 64 KiB a byte address no longer fits the wire's 16 bits, so flash
|
||||
// addresses there are word addresses ('J' always was one).
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
|
||||
// The loader's one identity number (README.md); the slimmed info block carries
|
||||
// it and the signature, and the host maps a version to its protocol.
|
||||
constexpr std::uint8_t version = 5;
|
||||
|
||||
// The activation window as a fixed poll budget: with no clock, whole seconds
|
||||
// cannot be timed. A __uint24 (AVR's three-byte type) holds it — a fourth byte
|
||||
// would cost two words at each countdown step for range never used.
|
||||
#if !defined(PUREBOOT_AUTOBAUD_POLLS)
|
||||
#define PUREBOOT_AUTOBAUD_POLLS 4000000
|
||||
#endif
|
||||
constexpr __uint24 autobaud_budget = PUREBOOT_AUTOBAUD_POLLS;
|
||||
|
||||
// The two SPM commands the loader still has to recognise by value, on the chips
|
||||
// where it cannot issue a runtime one generically. Taken from the chip's own
|
||||
// definitions rather than spelled 3 and 5 — though every part pureboot targets
|
||||
// agrees on those, which is what lets the host send the raw SPMCSR byte.
|
||||
constexpr std::uint8_t spm_erase = __BOOT_PAGE_ERASE;
|
||||
constexpr std::uint8_t spm_write = __BOOT_PAGE_WRITE;
|
||||
|
||||
// The spaces a transfer can name. Flash is 0 so it is the cheap default.
|
||||
//
|
||||
// sp_spm is the one that is not memory: a write there hands its data byte to
|
||||
// SPMCSR and fires the instruction at the given flash address, so page erase,
|
||||
// page write and RWW re-enable become host-issued commands instead of a
|
||||
// hardcoded tail inside 'W'. The store side already owns an address, a data
|
||||
// byte and an ack, so the whole sequence costs only the fused out/spm pair. It
|
||||
// also lets the host reach every other SPM operation — lock bits included —
|
||||
// which the loader previously had no way to expose.
|
||||
enum : std::uint8_t { sp_flash = 0, sp_eeprom = 1, sp_ram = 2, sp_fuse = 3, sp_spm = 4 };
|
||||
|
||||
// A transfer's selector byte is `space | bank << 4`: the low nibble names the
|
||||
// space, the high nibble carries flash's third address byte (RAMPZ) on the
|
||||
// chips that have one. Putting the bank here rather than widening the address
|
||||
// keeps the shared cursor sixteen bits for every space — a three-byte cursor
|
||||
// costs its extra increment on EEPROM and RAM reads too, which never need it.
|
||||
// The host must not span a bank boundary in one transfer; it already chunks by
|
||||
// page, so nothing it does today comes close.
|
||||
|
||||
// .noinit, not .bss: the unit is always measured before it is read, so it needs
|
||||
// no zeroing — and a zeroed .bss would drag in __do_clear_bss, 18 bytes of
|
||||
// startup code for a variable that is written before its first use.
|
||||
[[gnu::section(".noinit")]] std::uint16_t unit_backing;
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t get_unit()
|
||||
{
|
||||
#if PUREBOOT_UNIT_GPIOR
|
||||
return static_cast<std::uint16_t>(hw::reg_impl<hw::db.reg_index("GPIOR1")>::read() |
|
||||
(hw::reg_impl<hw::db.reg_index("GPIOR2")>::read() << 8));
|
||||
#else
|
||||
return unit_backing;
|
||||
#endif
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline void put_unit(std::uint16_t u)
|
||||
{
|
||||
#if PUREBOOT_UNIT_GPIOR
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR1")>::write(static_cast<std::uint8_t>(u));
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR2")>::write(static_cast<std::uint8_t>(u >> 8));
|
||||
#else
|
||||
unit_backing = u;
|
||||
#endif
|
||||
}
|
||||
|
||||
// The autobaud software link: bit-banged with cycle-counted delays like the
|
||||
// fixed-baud software backend, but the per-bit delay is the measured unit, not
|
||||
// a consteval constant. rx/tx load the unit once into a local, so each bit
|
||||
// spins from a register with no per-bit reload.
|
||||
struct link {
|
||||
using in_t = avr::io::input<avr::PUREBOOT_RX, avr::io::pull::up>;
|
||||
using out_t = avr::io::output<avr::PUREBOOT_TX>;
|
||||
|
||||
static void init()
|
||||
{
|
||||
avr::init<in_t, out_t>();
|
||||
out_t::set(); // idle high
|
||||
}
|
||||
|
||||
// Time the calibration pulse into the unit. The host sends 0xC0 — a start
|
||||
// bit plus six zero data bits are one low pulse of seven bit-times — and the
|
||||
// counted poll loop is seven cycles an iteration, so the count is the pulse
|
||||
// length in cycles ÷ 7 × 7 = one bit period in cycles, and count >> 2 is that
|
||||
// period in _delay_loop_2's four-cycle iterations. Waits for the start edge
|
||||
// under the poll budget; 0 (returned, and stored) means the budget expired.
|
||||
//
|
||||
// The shape of the counting loop is load-bearing, not incidental: the >> 2
|
||||
// is exact only while the pulse's bit-count equals the loop's cycles per
|
||||
// iteration. Both are 7 here (sbis 1 + rjmp 2 + adiw 2 + rjmp 2). Reshaping
|
||||
// this loop silently changes the lock; test/pbautobaud.py is what pins it.
|
||||
static std::uint16_t measure(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
std::uint16_t count = 0;
|
||||
while (!in_t::read())
|
||||
++count;
|
||||
const std::uint16_t u = static_cast<std::uint16_t>(count >> 2);
|
||||
put_unit(u);
|
||||
return u;
|
||||
}
|
||||
|
||||
// The eight data bits, entered just past the falling edge of the start bit.
|
||||
// Split out of rx() so the activation path can wait for that edge under a
|
||||
// budget while the command loop waits for it indefinitely.
|
||||
static std::uint8_t sample()
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
_delay_loop_2(static_cast<std::uint16_t>(unit + (unit >> 1))); // 1.5 bits to the LSB centre
|
||||
std::uint8_t value = 0;
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
value >>= 1;
|
||||
if (in_t::read())
|
||||
value |= 0x80;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static std::uint8_t rx()
|
||||
{
|
||||
while (in_t::read()) // await the start edge
|
||||
;
|
||||
return sample();
|
||||
}
|
||||
|
||||
// A byte under the poll budget, for the activation knock. An expired budget
|
||||
// returns 0, which is not the knock, so the caller falls back into the
|
||||
// budgeted measure() — and a line that stays idle boots the application
|
||||
// there. That is the whole hang fix: no wait during activation is unbounded,
|
||||
// so a stray calibration pulse can no longer wedge the loader.
|
||||
static std::uint8_t rx_bounded(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
return sample();
|
||||
}
|
||||
|
||||
static void tx(std::uint8_t value)
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
out_t::clear(); // start bit
|
||||
_delay_loop_2(unit);
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
out_t::write(value & 1);
|
||||
value >>= 1;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
out_t::set(); // stop bit
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
|
||||
static void drain()
|
||||
{
|
||||
// The software transmitter returns only after the stop bit.
|
||||
}
|
||||
};
|
||||
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
{
|
||||
target();
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
[[gnu::noinline, noreturn]] void run_app()
|
||||
{
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
// Inlined: read across a call, the first byte strands in a call-saved register
|
||||
// the caller has to push and pop.
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
// One byte out of any space. The four accessors share the cursor, the loop and
|
||||
// the call site — the whole point of the unified commands — so each costs only
|
||||
// its own instruction rather than a body, a loop and a dispatch arm.
|
||||
[[gnu::always_inline]] inline std::uint8_t load(std::uint8_t space, std::uint8_t bank, std::uint16_t at)
|
||||
{
|
||||
if (space == sp_eeprom)
|
||||
return ee::read(at);
|
||||
if (space == sp_ram)
|
||||
return *reinterpret_cast<volatile std::uint8_t *>(at);
|
||||
if (space == sp_fuse)
|
||||
return spm::read_fuse<off>(static_cast<spm::fuse>(at));
|
||||
if constexpr (word_flash)
|
||||
return avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(bank) << 16) | at);
|
||||
else
|
||||
return avr::flash_load(reinterpret_cast<const std::uint8_t *>(at));
|
||||
}
|
||||
|
||||
// One byte into a writable space. Flash is not one of them — it arrives a page
|
||||
// at a time through 'W' — and the fuses are not writable at all.
|
||||
[[gnu::always_inline]] inline void store(std::uint8_t space, [[maybe_unused]] std::uint8_t bank, std::uint16_t at,
|
||||
std::uint8_t value)
|
||||
{
|
||||
if (space == sp_ram) {
|
||||
*reinterpret_cast<volatile std::uint8_t *>(at) = value;
|
||||
return;
|
||||
}
|
||||
if (space == sp_spm) {
|
||||
// The fused store-and-fire: SPMCSR takes the data byte and the SPM
|
||||
// issues against Z in the same unscheduled pair the hardware's
|
||||
// four-cycle window demands, which is exactly why this is one
|
||||
// primitive and not a poke of SPMCSR followed by a poke of anything
|
||||
// else. A host cannot hit that window across a serial link.
|
||||
// The preprocessor rather than `if constexpr` only because
|
||||
// spm::detail::page_command does not exist at all where there is no
|
||||
// RAMPZ, and a discarded constexpr branch outside a template is still
|
||||
// name-checked. A generic spm::command() in libavr would let the
|
||||
// runtime command through on every chip and retire the dispatch below.
|
||||
#if defined(RAMPZ)
|
||||
spm::detail::page_command(value, (static_cast<spm::flash_address_t>(bank) << 16) | at);
|
||||
#else
|
||||
if (value == spm_erase)
|
||||
spm::erase_page<off>(at);
|
||||
else if (value == spm_write)
|
||||
spm::write_page<off>(at);
|
||||
else if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
#endif
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
return;
|
||||
}
|
||||
ee::write<off>(at, value);
|
||||
}
|
||||
|
||||
// One page into the SPM buffer, and only that: the erase, the write and the RWW
|
||||
// re-enable that used to follow are now three host-issued writes to sp_spm,
|
||||
// which reach the same fused out/spm pair through the store path's own address
|
||||
// and data. No running-slot write guard: the host guarantees it never targets
|
||||
// the loader's own slot (licensed — README.md).
|
||||
void program_flash([[maybe_unused]] std::uint8_t bank, std::uint16_t at)
|
||||
{
|
||||
std::uint16_t z = at & ~static_cast<std::uint16_t>(page - 1);
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
#if defined(RAMPZ)
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(bank) << 16) | z, word_of({low, high}));
|
||||
#else
|
||||
spm::fill<off>(z, word_of({low, high}));
|
||||
#endif
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z) & (page - 1));
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
if (hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// Measure the calibration pulse into the unit, then take one 'p' knock —
|
||||
// both under the poll budget. An expired budget (no host) boots the
|
||||
// application; anything but 'p', including the knock timing out, re-measures
|
||||
// and so returns to the budgeted wait that boots it.
|
||||
for (;;) {
|
||||
if (link::measure(autobaud_budget) == 0)
|
||||
run_app();
|
||||
if (link::rx_bounded(autobaud_budget) == 'p')
|
||||
break;
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
ee::wait();
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // chip identity: version then the three signature bytes
|
||||
link::tx(version);
|
||||
link::tx(hw::db.signature[0]);
|
||||
link::tx(hw::db.signature[1]);
|
||||
link::tx(hw::db.signature[2]);
|
||||
break;
|
||||
case 'W': // program one flash page: sel8, addr16, then page bytes
|
||||
case 'G': // read: sel8, addr16, n8 (0 = 256)
|
||||
case 'g': { // write: sel8, addr16, n8, then n bytes each acked
|
||||
// The unified transfer. One decode, one cursor, one loop for every
|
||||
// space and both directions — the four command bodies this replaces
|
||||
// each carried their own copy of all three. Read and write are the
|
||||
// same letter in the two cases, so the direction is bit 5 of the
|
||||
// command and the loop tests it with a one-word skip. 'W' joins the
|
||||
// same selector-and-address decode rather than keeping a word
|
||||
// address of its own, which makes flash addressing uniform across
|
||||
// every command that names it and costs nothing to share.
|
||||
const std::uint8_t sel = link::rx();
|
||||
const std::uint8_t space = sel & 0x0f;
|
||||
const std::uint8_t bank = static_cast<std::uint8_t>(sel >> 4);
|
||||
std::uint16_t at = rx16();
|
||||
if (command == 'W') {
|
||||
program_flash(bank, at);
|
||||
break;
|
||||
}
|
||||
std::uint8_t count = link::rx();
|
||||
do {
|
||||
if (command & 0x20) {
|
||||
store(space, bank, at, link::rx());
|
||||
tx_ack();
|
||||
} else
|
||||
link::tx(load(space, bank, at));
|
||||
++at;
|
||||
} while (--count);
|
||||
break;
|
||||
}
|
||||
default: // unknown bytes are ignored; the loop re-acks
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace pureboot
|
||||
|
||||
template struct avr::startup::entry<pureboot::run>;
|
||||
154
test/pbautobaud.py
Normal file
154
test/pbautobaud.py
Normal file
@@ -0,0 +1,154 @@
|
||||
#!/usr/bin/env python3
|
||||
"""End-to-end autobaud test: drive an autobaud loader in simavr through the
|
||||
calibration handshake and a flash + EEPROM + fuse round-trip, cross-checked
|
||||
against the simulator's ground-truth memory — then repeat at a second F_CPU with
|
||||
the *same* loader binary, which is the property autobaud exists for: one
|
||||
clock-agnostic image that locks onto whatever rate the host sends.
|
||||
|
||||
Usage: pbautobaud.py <device_bin> <loader_elf> <mcu> <base_hex> <page>
|
||||
<app_bin> <app_hz> <app_baud> <tool_py> <workdir>
|
||||
|
||||
The loader is a software-serial build on PB0/PB1 (pureboot_add_autobaud's
|
||||
default), so the runner drives it over the GPIO⇄pty bridge (-l sw:B0,B1). The
|
||||
app fixture is built for (app_hz, app_baud); the hand-over is checked at that
|
||||
point, and a second point at half the clock proves the lock is measured, not
|
||||
baked in.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
def fail(message):
|
||||
print(f"FAIL: {message}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main():
|
||||
(device_bin, elf, mcu, base_hex, page, app_bin, app_hz, app_baud, tool, workdir) = sys.argv[1:]
|
||||
base, page, app_hz, app_baud = int(base_hex, 0), int(page), int(app_hz), int(app_baud)
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(tool)))
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import pbsim
|
||||
import pureboot as pb
|
||||
|
||||
os.makedirs(workdir, exist_ok=True)
|
||||
ee_image = bytes(range(0xA0, 0xB0))
|
||||
ee_path = os.path.join(workdir, "ee.bin")
|
||||
open(ee_path, "wb").write(ee_image)
|
||||
|
||||
# The geometry the surgery planner needs, from the chip class the runner is
|
||||
# told — the same derivation pbtest.py makes: the boot-sectioned megas need
|
||||
# no vector surgery, the tinies and the boot-section-less m48s do, and the
|
||||
# large chips speak word addresses.
|
||||
mega = mcu.startswith("atmega")
|
||||
patch = not mega or mcu.startswith("atmega48")
|
||||
word_flash = base + pb.SLOT > 0x10000
|
||||
wire_base = base // 2 if word_flash else base
|
||||
flags = (1 if patch else 0) | (2 if word_flash else 0)
|
||||
ground_truth = pb.Info(bytes([ord("P"), ord("B"), pb.NEWEST_LOADER, 0, 0, 0, page & 0xFF,
|
||||
wire_base & 0xFF, wire_base >> 8, 0, 0, flags]))
|
||||
|
||||
def round_trip(hz, baud, label, hand_over):
|
||||
"""One clock point: reset, calibrate + knock, program, verify against the
|
||||
simulator's own flash, and (at the app's point) hand over to the fixture."""
|
||||
dump = os.path.join(workdir, f"flash_{label}.bin")
|
||||
device = pbsim.Device(device_bin, elf, mcu, str(hz), base_hex, page, baud, dump, link="sw:B0,B1")
|
||||
try:
|
||||
# The host tool, in autobaud mode, sends the 0xC0 calibration pulse
|
||||
# and a single knock at `baud`; the loader locks to it.
|
||||
out = pbsim.run_tool(tool, device.pty, baud, "--autobaud", "--info", "--fuses",
|
||||
"--flash", app_bin, "--eeprom", ee_path, "--stay")
|
||||
for needed in ("version", "signature", "fuses", "verify:", "stays"):
|
||||
if needed not in out:
|
||||
fail(f"{label}: session output lacks {needed!r}\n{out}")
|
||||
# Read both memories back over the locked link and check them.
|
||||
read_flash = os.path.join(workdir, f"rf_{label}.bin")
|
||||
read_eeprom = os.path.join(workdir, f"re_{label}.bin")
|
||||
out = pbsim.run_tool(tool, device.pty, baud, "--autobaud", "--verify-flash", app_bin,
|
||||
"--verify-eeprom", ee_path, "--read-flash", read_flash,
|
||||
"--read-eeprom", read_eeprom, "--stay")
|
||||
if out.count("verify:") != 2:
|
||||
fail(f"{label}: did not verify both memories\n{out}")
|
||||
if open(read_eeprom, "rb").read()[: len(ee_image)] != ee_image:
|
||||
fail(f"{label}: EEPROM read-back mismatch")
|
||||
|
||||
if hand_over:
|
||||
# Regression: a calibration pulse with no knock behind it must
|
||||
# not wedge the loader. The knock's edge wait used to be
|
||||
# unbudgeted, so one stray low pulse — EMI, or a host that opens
|
||||
# the port and never knocks — held the loader forever and the
|
||||
# application never ran. The whole activation is bounded now, so
|
||||
# the window closes and the app boots; the banner is the proof.
|
||||
# (The pause lets the loader reach its measurement loop, so the
|
||||
# pulse is genuinely seen and the test cannot pass vacuously.)
|
||||
device.reset()
|
||||
port = pb.Port(device.pty, baud)
|
||||
try:
|
||||
time.sleep(0.2)
|
||||
port.write(bytes((pb.CALIBRATE,)))
|
||||
# Accumulate rather than match exactly: the reset leaves the
|
||||
# idle line a framing artefact ahead of the banner, which is
|
||||
# noise here — the question is only whether the app ran.
|
||||
seen = b""
|
||||
deadline = time.monotonic() + 180.0
|
||||
while b"APP" not in seen and time.monotonic() < deadline:
|
||||
seen += port.read_available(1.0)
|
||||
if b"APP" not in seen:
|
||||
fail(f"{label}: lone calibration pulse wedged the loader — app never bannered, saw {seen!r}")
|
||||
print(f" {label}: lone calibration pulse does not wedge the loader")
|
||||
finally:
|
||||
port.close()
|
||||
|
||||
device.reset()
|
||||
port = pb.Port(device.pty, baud)
|
||||
try:
|
||||
loader = pb.Loader(port)
|
||||
live = loader.connect_autobaud(15)
|
||||
if not pb.OLDEST_LOADER <= live.version <= pb.NEWEST_LOADER:
|
||||
fail(f"{label}: loader reports pureboot {live.version}")
|
||||
if loader.unified:
|
||||
# pureboot 5's data space. 0x0200 is clear of the
|
||||
# loader's own .noinit unit at the bottom of SRAM and of
|
||||
# the stack at the top. Reading it back over the same
|
||||
# locked link proves both directions of the new space.
|
||||
probe = bytes(range(0x30, 0x40))
|
||||
loader.write_ram(0x0200, probe)
|
||||
if loader.read_ram(0x0200, len(probe)) != probe:
|
||||
fail(f"{label}: RAM round-trip mismatch")
|
||||
# The register file and the I/O space share the data
|
||||
# address space on AVR, so the same command reaches a
|
||||
# peripheral register. SPMCSR reads back as idle here.
|
||||
verbose_ram = loader.read_ram(0x0200, 4)
|
||||
print(f" {label}: RAM read/write ok ({verbose_ram.hex()})")
|
||||
loader.run_application()
|
||||
banner = port.read_exact(3, 5.0)
|
||||
if banner != b"APP":
|
||||
fail(f"{label}: application banner was {banner!r}")
|
||||
finally:
|
||||
port.close()
|
||||
finally:
|
||||
device.stop()
|
||||
|
||||
# Ground truth (read after the runner exits and writes its dump): what
|
||||
# the tool programmed must be what the simulator actually holds.
|
||||
pages = pb.plan_flash(open(app_bin, "rb").read(), ground_truth)
|
||||
flash_true = open(dump, "rb").read()
|
||||
for address, data in pages.items():
|
||||
if flash_true[address : address + page] != data:
|
||||
fail(f"{label}: simulator flash differs from the programmed image at {address:#06x}")
|
||||
print(f" {label}: locked at {hz} Hz / {baud} Bd, flash+EEPROM verified"
|
||||
+ (", hand-over ok" if hand_over else ""))
|
||||
|
||||
# The app fixture is built for one clock; the hand-over banners there. A
|
||||
# second point at double that clock, same loader binary, proves the lock is
|
||||
# measured, not baked in — the whole point of autobaud. (Doubling keeps the
|
||||
# bit period healthy; halving would drop it below the software UART's floor.)
|
||||
round_trip(app_hz, app_baud, "clock-a", hand_over=True)
|
||||
round_trip(app_hz * 2, app_baud, "clock-b", hand_over=False)
|
||||
print("pbautobaud: calibration lock and flash/EEPROM/fuse round-trip pass at both clocks")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -100,11 +100,14 @@ def main():
|
||||
try:
|
||||
loader = pb.Loader(port)
|
||||
live = loader.connect(15)
|
||||
# The loader built from this tree and the tool beside it must
|
||||
# agree on where the version numbering stands: a bump the tool
|
||||
# was never told about is a loader it would refuse to speak to.
|
||||
if live.version != pb.NEWEST_LOADER:
|
||||
fail(f"loader reports pureboot {live.version}, the tool's newest is {pb.NEWEST_LOADER}")
|
||||
# The loader built from this tree must report a version the tool
|
||||
# beside it speaks — a bump the tool was never told about is a
|
||||
# loader it would refuse to talk to. Not equality with the newest:
|
||||
# the tool now spans two loader generations, the fixed-baud one
|
||||
# here and the unified autobaud loader that follows it.
|
||||
if not pb.OLDEST_LOADER <= live.version <= pb.NEWEST_LOADER:
|
||||
fail(f"loader reports pureboot {live.version}, the tool speaks "
|
||||
f"{pb.OLDEST_LOADER}..{pb.NEWEST_LOADER}")
|
||||
|
||||
# A W addressed inside a page rather than at its base must still
|
||||
# consume exactly one page and prompt. The loader's own slot is
|
||||
|
||||
Reference in New Issue
Block a user