Compare commits
12 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a4da885e36 | |||
| 335e494a31 | |||
| 799709efcf | |||
| 7ae80087b3 | |||
| b477f53ca5 | |||
| 84d3f679c2 | |||
| b5020a1e20 | |||
| 1dbf0089d6 | |||
| 39dfe40dbf | |||
| 8470ce3da0 | |||
| d634741015 | |||
| ab842c8d99 |
134
CMakeLists.txt
134
CMakeLists.txt
@@ -234,12 +234,13 @@ if(PROJECT_IS_TOP_LEVEL)
|
||||
endif()
|
||||
|
||||
# The size matrix: every configuration axis that could move the image
|
||||
# size — the serial backend (different code), the clock and its ladder
|
||||
# baud (different constants and divisor shapes), the USART instance
|
||||
# (different register class) — each combination must still fit the
|
||||
# chip's slot budget. Pins are size-neutral (port and bit are immediate
|
||||
# operands) and the timeout is a constant, so neither adds an axis. The
|
||||
# stock build is one point of this matrix and already has its test.
|
||||
# size — the serial backend (different code), the USART instance
|
||||
# (different registers), the clock (different constants), and the baud
|
||||
# through the shapes its bit timing takes — each combination must still
|
||||
# fit the chip's slot budget. Pins are size-neutral (port and bit are
|
||||
# immediate operands) and the timeout is a constant, so neither adds an
|
||||
# axis. The stock build is one point of this matrix and already has its
|
||||
# test.
|
||||
function(pureboot_size_variant name)
|
||||
pureboot_add_loader(${name} ${ARGN})
|
||||
add_test(NAME ${name}.size
|
||||
@@ -247,23 +248,100 @@ if(PROJECT_IS_TOP_LEVEL)
|
||||
-DLIMIT=${PUREBOOT_LIMIT} -P ${CMAKE_CURRENT_SOURCE_DIR}/test/check_size.cmake)
|
||||
endfunction()
|
||||
|
||||
# The autobaud loader (pureboot/autobaud.md): one clock-agnostic image per
|
||||
# chip, no clock × baud axis, size-tested against the same per-chip budget on
|
||||
# every chip.
|
||||
#
|
||||
# Only the unified version is built. Its two predecessors —
|
||||
# pureboot_autobaud_pure.cpp at 508 B and pureboot_autobaud_reg.cpp at 512 —
|
||||
# had 4 B and 0 B of margin on the 1284P, and the fix for the activation hang
|
||||
# costs ~22, which puts them at 530 and 534. Neither can ship, so the choice
|
||||
# the branch existed to offer is settled by measurement rather than taste.
|
||||
# The sources stay for the record; autobaud.md carries the numbers.
|
||||
function(pureboot_autobaud_variant name source)
|
||||
pureboot_add_autobaud(${name} ${source})
|
||||
add_test(NAME ${name}.size
|
||||
COMMAND ${CMAKE_COMMAND} -DSIZE_TOOL=${CMAKE_SIZE} -DELF=$<TARGET_FILE:${name}>
|
||||
-DLIMIT=${PUREBOOT_LIMIT} -P ${CMAKE_CURRENT_SOURCE_DIR}/test/check_size.cmake)
|
||||
endfunction()
|
||||
pureboot_autobaud_variant(pureboot_autobaud_uni pureboot_autobaud_uni.cpp)
|
||||
|
||||
# One point of the exhaustive matrix, named from its resolved parameters
|
||||
# so the enumeration cannot collide with itself. Unreachable rates drop
|
||||
# out here rather than aborting the configure.
|
||||
function(pureboot_matrix_point hz baud link)
|
||||
if(link STREQUAL "software")
|
||||
pureboot_baud_feasible(${hz} ${baud} 1 _ok)
|
||||
set(_args SERIAL software)
|
||||
else()
|
||||
pureboot_baud_feasible(${hz} ${baud} 0 _ok)
|
||||
set(_args USART ${link})
|
||||
endif()
|
||||
if(_ok)
|
||||
pureboot_size_variant(pbm_${hz}_${baud}_${link} CLOCK ${hz} BAUD ${baud} ${_args})
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Clock points: the shipped-fuse floor (CKDIV8), the calibrated RC, and
|
||||
# the crystal the stock build assumes (the tiny13's ladder is its own RC
|
||||
# menu — it has no crystal option).
|
||||
if(LIBAVR_MCU MATCHES "^attiny13")
|
||||
set(_matrix_clocks 1200000 4800000 9600000)
|
||||
set(_full_clocks 128000 600000 1200000 4800000 9600000)
|
||||
else()
|
||||
set(_matrix_clocks 1000000 8000000 16000000)
|
||||
set(_full_clocks 128000 1000000 1843200 2000000 3686400 4000000 7372800 8000000
|
||||
11059200 12000000 14745600 16000000 18432000 20000000)
|
||||
endif()
|
||||
|
||||
# The exhaustive cross product: every clock a deployment plausibly runs
|
||||
# — the internal oscillators, the shipped CKDIV8 floor, the plain
|
||||
# crystals and the UART crystals — against every rate, against every
|
||||
# backend. Beyond the ladder the list carries the slow rates a
|
||||
# sub-megahertz oscillator is left with, which no ladder rate reaches
|
||||
# (16000 Bd is the only rate the 128 kHz oscillator holds exactly); at
|
||||
# the fast clocks those same rates also select the software UART's
|
||||
# 16-bit _delay_loop_2 bit spin (two words more setup at each of its five
|
||||
# sites), the largest image the space produces and a shape the ladder
|
||||
# default — always the *fastest* rate a clock reaches — never picks.
|
||||
#
|
||||
# Bounded to one chip per size-bearing class: flash addressing (the
|
||||
# word-addressed 1284), hand-over shape (the patched vector on the tinies
|
||||
# and m48s), page size, and USART inventory. Everything else in the image
|
||||
# is chip-independent code, so a further chip buys builds and no
|
||||
# coverage; every chip outside the set carries the compact matrix.
|
||||
get_property(_full_bauds GLOBAL PROPERTY PUREBOOT_BAUD_LADDER)
|
||||
list(APPEND _full_bauds 16000 4800 2400 1200)
|
||||
set(_matrix_spot attiny13a attiny85 atmega48pa atmega8a atmega168pa
|
||||
atmega328p atmega164a atmega644a atmega1284p)
|
||||
if(DEFINED ENV{PUREBOOT_FULL_MATRIX} AND LIBAVR_MCU IN_LIST _matrix_spot)
|
||||
foreach(_matrix_hz IN LISTS _full_clocks)
|
||||
foreach(_matrix_baud IN LISTS _full_bauds)
|
||||
pureboot_matrix_point(${_matrix_hz} ${_matrix_baud} software)
|
||||
if(PUREBOOT_HAS_USART)
|
||||
pureboot_matrix_point(${_matrix_hz} ${_matrix_baud} 0)
|
||||
endif()
|
||||
if(PUREBOOT_HAS_USART1)
|
||||
pureboot_matrix_point(${_matrix_hz} ${_matrix_baud} 1)
|
||||
endif()
|
||||
endforeach()
|
||||
endforeach()
|
||||
else()
|
||||
foreach(_matrix_hz IN LISTS _matrix_clocks)
|
||||
math(EXPR _matrix_khz "${_matrix_hz} / 1000")
|
||||
if(PUREBOOT_HAS_USART OR NOT _matrix_hz EQUAL _pb_stock_hz)
|
||||
pureboot_size_variant(pureboot_sw_${_matrix_khz}k CLOCK ${_matrix_hz} SERIAL software)
|
||||
endif()
|
||||
if(PUREBOOT_HAS_USART AND NOT _matrix_hz EQUAL _pb_stock_hz)
|
||||
pureboot_size_variant(pureboot_hw_${_matrix_khz}k CLOCK ${_matrix_hz} SERIAL hardware)
|
||||
endif()
|
||||
if(PUREBOOT_HAS_USART1 AND NOT _matrix_hz EQUAL _pb_stock_hz)
|
||||
pureboot_size_variant(pureboot_usart1_${_matrix_khz}k CLOCK ${_matrix_hz} USART 1)
|
||||
endif()
|
||||
endforeach()
|
||||
list(GET _matrix_clocks -1 _matrix_top_hz)
|
||||
pureboot_size_variant(pureboot_sw_wide CLOCK ${_matrix_top_hz} BAUD 9600 SERIAL software)
|
||||
endif()
|
||||
foreach(_matrix_hz IN LISTS _matrix_clocks)
|
||||
math(EXPR _matrix_khz "${_matrix_hz} / 1000")
|
||||
if(PUREBOOT_HAS_USART OR NOT _matrix_hz EQUAL _pb_stock_hz)
|
||||
pureboot_size_variant(pureboot_sw_${_matrix_khz}k CLOCK ${_matrix_hz} SERIAL software)
|
||||
endif()
|
||||
if(PUREBOOT_HAS_USART AND NOT _matrix_hz EQUAL _pb_stock_hz)
|
||||
pureboot_size_variant(pureboot_hw_${_matrix_khz}k CLOCK ${_matrix_hz} SERIAL hardware)
|
||||
endif()
|
||||
endforeach()
|
||||
if(PUREBOOT_HAS_USART1)
|
||||
pureboot_size_variant(pureboot_usart1 USART 1)
|
||||
endif()
|
||||
@@ -318,4 +396,30 @@ if(PROJECT_IS_TOP_LEVEL)
|
||||
${CMAKE_BINARY_DIR}/pbusart1-work usart1)
|
||||
set_tests_properties(pureboot.usart1 PROPERTIES TIMEOUT 180)
|
||||
endif()
|
||||
|
||||
# The autobaud variants driven end to end over the software-UART bridge (both
|
||||
# under review — pureboot/autobaud.md): the host sends the 0xC0 calibration
|
||||
# pulse, the loader times it, locks, and programs. Run on the near-flash 328P
|
||||
# and the word-addressed 1284P — the two flash-addressing classes — and each
|
||||
# at two clocks with the one binary, which is the clock-agnostic property
|
||||
# autobaud exists for (test/pbautobaud.py). The fixture application banners
|
||||
# over the same software link at the first clock's rate.
|
||||
if(LIBAVR_MCU MATCHES "^atmega(328p|1284p)$" AND DEFINED PB_DEVICE)
|
||||
add_executable(pbapp_autobaud test/pbapp.cpp)
|
||||
target_link_libraries(pbapp_autobaud PRIVATE libavr)
|
||||
target_compile_definitions(pbapp_autobaud PRIVATE PUREBOOT_CLOCK_HZ=1000000
|
||||
PUREBOOT_BAUD=9600 PUREBOOT_SOFT_SERIAL PUREBOOT_TX=pb1)
|
||||
add_custom_command(TARGET pbapp_autobaud POST_BUILD
|
||||
COMMAND ${CMAKE_OBJCOPY} -O binary
|
||||
$<TARGET_FILE:pbapp_autobaud> $<TARGET_FILE:pbapp_autobaud>.bin)
|
||||
foreach(_variant uni)
|
||||
add_test(NAME pureboot.autobaud_${_variant}
|
||||
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/test/pbautobaud.py
|
||||
${PB_DEVICE} $<TARGET_FILE:pureboot_autobaud_${_variant}> ${PUREBOOT_SIM_MCU}
|
||||
${PUREBOOT_BASE_HEX} ${PUREBOOT_PAGE} $<TARGET_FILE:pbapp_autobaud>.bin
|
||||
1000000 9600 ${CMAKE_CURRENT_SOURCE_DIR}/pureboot/pureboot.py
|
||||
${CMAKE_BINARY_DIR}/pbautobaud-${_variant}-work)
|
||||
set_tests_properties(pureboot.autobaud_${_variant} PROPERTIES TIMEOUT 240)
|
||||
endforeach()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
2
libavr
2
libavr
Submodule libavr updated: e81dad0131...edc77ca43f
@@ -1,27 +1,16 @@
|
||||
# pureboot as a consumable CMake unit: the per-chip geometry, the default
|
||||
# baud ladder, and pureboot_add_loader() — the one way a loader target is
|
||||
# created, both by this port's own build and by a downstream project. A
|
||||
# downstream project brings its usual libavr setup (the `libavr` target and
|
||||
# pureboot as a consumable CMake unit: the per-chip geometry, the default baud
|
||||
# ladder, and pureboot_add_loader() — the one way a loader target is created.
|
||||
# A downstream project brings its usual libavr setup (the `libavr` target and
|
||||
# the LIBAVR_MCU toolchain preset), adds this directory, and states its
|
||||
# deployment:
|
||||
# deployment; every argument is optional (README.md):
|
||||
#
|
||||
# add_subdirectory(bootloader/pureboot)
|
||||
# pureboot_add_loader(myboot CLOCK 1000000 SERIAL software TX pb1 RX pb5)
|
||||
#
|
||||
# Every argument is optional — CLOCK defaults to the family assumption
|
||||
# below, BAUD to the fastest standard rate the clock reaches within 2.5 %
|
||||
# (the ladder), SERIAL to the chip's hardware USART where it has one
|
||||
# (`hardware`/`software` force a backend, USART 1 picks the second
|
||||
# instance), RX/TX to pb0/pb1 for the software UART, TIMEOUT to 8 s.
|
||||
# Infeasible picks fail the build by name: libavr's baud-error and
|
||||
# software-UART cycle-floor static asserts re-check whatever is passed.
|
||||
|
||||
# Per-family geometry: flash/page/EEPROM sizes and the linker wrap the PC
|
||||
# modulo needs, the loader slot (each chip's smallest boot sector — 1 KiB on
|
||||
# the word-addressed 1284s), and the deployment defaults (crystal assumption
|
||||
# on the megas, calibrated RC on the tinies). The USART flags mirror the
|
||||
# hardware inventory the loader's own static asserts check (the plain 644 is
|
||||
# the x4 family's one single-USART die, Atmel-2593).
|
||||
# Per-family geometry, deployment defaults, and the linker wrap the PC modulo
|
||||
# needs. The slot is 512 bytes on every chip. The USART flags mirror the
|
||||
# hardware inventory the loader's own static asserts check — the plain 644 is
|
||||
# the x4 family's one single-USART die (Atmel-2593).
|
||||
set(_pb_has_usart 1)
|
||||
set(_pb_has_usart1 0)
|
||||
if(LIBAVR_MCU MATCHES "^attiny13a?$")
|
||||
@@ -91,10 +80,9 @@ elseif(LIBAVR_MCU MATCHES "^atmega324(a|p|pa)$")
|
||||
set(_pb_eeprom 1024)
|
||||
set(_pb_has_usart1 1)
|
||||
elseif(LIBAVR_MCU MATCHES "^atmega644(a|p|pa)?$")
|
||||
# 64 KiB is exactly the 16-bit byte space: plain LPM reaches everything,
|
||||
# and the smallest boot section (1 KiB) holds the loader and its staging
|
||||
# slot together (see README.md). The plain 644 is the family's one
|
||||
# single-USART die.
|
||||
# 64 KiB is exactly the 16-bit byte space, so plain LPM still reaches
|
||||
# everything and the wire stays byte-addressed. The plain 644 is the
|
||||
# family's one single-USART die.
|
||||
set(_pb_flash 65536)
|
||||
set(_pb_wrap -Wl,--pmem-wrap-around=64k)
|
||||
set(_pb_page 256)
|
||||
@@ -104,33 +92,25 @@ elseif(LIBAVR_MCU MATCHES "^atmega644(a|p|pa)?$")
|
||||
set(_pb_has_usart1 1)
|
||||
endif()
|
||||
elseif(LIBAVR_MCU MATCHES "^atmega1284p?$")
|
||||
# 128 KiB: wire flash addresses are word addresses, reads go through
|
||||
# ELPM, and the PC's modulo wrap exceeds what --pmem-wrap-around models.
|
||||
# The slot is 1 KiB — this chip's own smallest boot sector; the far
|
||||
# machinery cannot fit 512 B (see README.md).
|
||||
# 128 KiB: wire addresses are words, reads go through ELPM, and the PC's
|
||||
# modulo wrap exceeds what --pmem-wrap-around models.
|
||||
set(_pb_flash 131072)
|
||||
set(_pb_wrap "")
|
||||
set(_pb_page 256)
|
||||
set(_pb_hz 16000000)
|
||||
set(_pb_eeprom 4096)
|
||||
set(_pb_slot 1024)
|
||||
set(_pb_limit 1024)
|
||||
set(_pb_has_usart1 1)
|
||||
else()
|
||||
message(FATAL_ERROR "pureboot: no geometry for ${LIBAVR_MCU}")
|
||||
endif()
|
||||
if(NOT DEFINED _pb_slot)
|
||||
set(_pb_slot 512)
|
||||
endif()
|
||||
set(_pb_slot 512)
|
||||
math(EXPR _pb_base "${_pb_flash} - ${_pb_slot}")
|
||||
math(EXPR _pb_base_hex "${_pb_base}" OUTPUT_FORMAT HEXADECIMAL)
|
||||
# Patched-vector chips hand over through the trampoline word below the slot,
|
||||
# which is also the slot's own last word — their budget is slot − 2.
|
||||
if(LIBAVR_MCU MATCHES "^atmega" AND NOT LIBAVR_MCU MATCHES "^atmega48")
|
||||
set(_pb_app 0)
|
||||
if(NOT DEFINED _pb_limit)
|
||||
set(_pb_limit ${_pb_slot})
|
||||
endif()
|
||||
set(_pb_limit ${_pb_slot})
|
||||
else()
|
||||
math(EXPR _pb_app "${_pb_base} - 2")
|
||||
math(EXPR _pb_limit "${_pb_slot} - 2")
|
||||
@@ -166,48 +146,61 @@ set(PUREBOOT_HAS_USART ${_pb_has_usart} PARENT_SCOPE)
|
||||
set(PUREBOOT_HAS_USART1 ${_pb_has_usart1} PARENT_SCOPE)
|
||||
set(PUREBOOT_SIM_MCU ${_pb_sim_mcu} PARENT_SCOPE)
|
||||
|
||||
# The fastest standard rate the clock reaches within 2.5 % — the same
|
||||
# best-of-U2X-and-plain divisor search libavr's solve_baud runs, so a
|
||||
# default never trips the compile-time error it is checked against. A
|
||||
# software build additionally requires the polled receiver's 100-cycles-a-bit
|
||||
# floor (its own static assert): at low clocks the U2X divisor still reaches
|
||||
# rates the bit-banged sampler cannot, so the backend gates the ladder.
|
||||
function(pureboot_default_baud clock software outvar)
|
||||
foreach(baud 115200 57600 38400 19200 9600)
|
||||
math(EXPR _cycles "${clock} / ${baud}")
|
||||
if(software AND _cycles LESS 100)
|
||||
# The rates a default may pick, fastest first.
|
||||
set_property(GLOBAL PROPERTY PUREBOOT_BAUD_LADDER 115200 57600 38400 19200 9600)
|
||||
|
||||
# Whether <baud> is reachable from <clock> within 2.5 %, by the same
|
||||
# best-of-U2X-and-plain divisor search libavr's solve_baud runs, so a build
|
||||
# never trips the compile-time error it is checked against. A software build
|
||||
# also needs the polled receiver's 100-cycles-a-bit floor: at low clocks the
|
||||
# U2X divisor reaches rates the bit-banged sampler cannot.
|
||||
function(pureboot_baud_feasible clock baud software outvar)
|
||||
set(${outvar} 0 PARENT_SCOPE)
|
||||
math(EXPR _cycles "${clock} / ${baud}")
|
||||
if(software AND _cycles LESS 100)
|
||||
return()
|
||||
endif()
|
||||
foreach(divisor 8 16)
|
||||
math(EXPR _step "${divisor} * ${baud}")
|
||||
math(EXPR _n "(${clock} + ${_step} / 2) / ${_step}")
|
||||
if(_n LESS 1 OR _n GREATER 4096)
|
||||
continue()
|
||||
endif()
|
||||
foreach(divisor 8 16)
|
||||
math(EXPR _step "${divisor} * ${baud}")
|
||||
math(EXPR _n "(${clock} + ${_step} / 2) / ${_step}")
|
||||
if(_n LESS 1 OR _n GREATER 4096)
|
||||
continue()
|
||||
endif()
|
||||
math(EXPR _actual "${clock} / (${divisor} * ${_n})")
|
||||
math(EXPR _delta "${_actual} - ${baud}")
|
||||
if(_delta LESS 0)
|
||||
math(EXPR _delta "-(${_delta})")
|
||||
endif()
|
||||
math(EXPR _error_bp "${_delta} * 10000 / ${baud}")
|
||||
if(_error_bp LESS_EQUAL 250)
|
||||
set(${outvar} ${baud} PARENT_SCOPE)
|
||||
return()
|
||||
endif()
|
||||
endforeach()
|
||||
math(EXPR _actual "${clock} / (${divisor} * ${_n})")
|
||||
math(EXPR _delta "${_actual} - ${baud}")
|
||||
if(_delta LESS 0)
|
||||
math(EXPR _delta "-(${_delta})")
|
||||
endif()
|
||||
math(EXPR _error_bp "${_delta} * 10000 / ${baud}")
|
||||
if(_error_bp LESS_EQUAL 250)
|
||||
set(${outvar} 1 PARENT_SCOPE)
|
||||
return()
|
||||
endif()
|
||||
endforeach()
|
||||
message(FATAL_ERROR "pureboot: no standard baud rate fits a ${clock} Hz clock within 2.5 %")
|
||||
endfunction()
|
||||
|
||||
# The fastest ladder rate the clock reaches.
|
||||
function(pureboot_default_baud clock software outvar)
|
||||
get_property(_ladder GLOBAL PROPERTY PUREBOOT_BAUD_LADDER)
|
||||
foreach(baud ${_ladder})
|
||||
pureboot_baud_feasible(${clock} ${baud} ${software} _ok)
|
||||
if(_ok)
|
||||
set(${outvar} ${baud} PARENT_SCOPE)
|
||||
return()
|
||||
endif()
|
||||
endforeach()
|
||||
message(FATAL_ERROR "pureboot: no standard baud rate fits a ${clock} Hz clock within 2.5 % "
|
||||
"— pass BAUD <rate> to deploy a non-standard one")
|
||||
endfunction()
|
||||
|
||||
# pureboot_add_loader(<name> [CLOCK <hz>] [BAUD <bd>]
|
||||
# [SERIAL auto|hardware|software] [USART <n>]
|
||||
# [RX <pin>] [TX <pin>] [TIMEOUT <s>])
|
||||
#
|
||||
# Creates the loader target plus its flashable images (<name>.hex for a
|
||||
# programmer, <name>.bin for --update-loader) and stamps the resolved
|
||||
# deployment on the target: the PUREBOOT_HZ, PUREBOOT_BAUD and PUREBOOT_LINK
|
||||
# properties (the link as usart0/usart1/sw:<RX>,<TX> — what a test harness
|
||||
# needs to speak to the build).
|
||||
# The loader target plus its flashable images (<name>.hex for a programmer,
|
||||
# <name>.bin for --update-loader). The resolved deployment is stamped on the
|
||||
# target as PUREBOOT_HZ / PUREBOOT_BAUD / PUREBOOT_LINK (the link spelled
|
||||
# usart0, usart1 or sw:<RX>,<TX>) — what a test harness speaks to it with.
|
||||
function(pureboot_add_loader name)
|
||||
cmake_parse_arguments(PB "" "CLOCK;BAUD;SERIAL;USART;RX;TX;TIMEOUT" "" ${ARGN})
|
||||
if(PB_UNPARSED_ARGUMENTS)
|
||||
@@ -272,8 +265,7 @@ function(pureboot_add_loader name)
|
||||
endif()
|
||||
endforeach()
|
||||
set(_serial_defines PUREBOOT_SOFT_SERIAL PUREBOOT_RX=${PB_RX} PUREBOOT_TX=${PB_TX})
|
||||
# The link spec a test harness drives a GPIO bridge with: sw:<RX>,<TX>
|
||||
# as the port letter and bit, the loader's own pin naming upcased.
|
||||
# sw:<RX>,<TX> as port letter and bit, upcased.
|
||||
string(SUBSTRING ${PB_RX} 1 2 _rx_pin)
|
||||
string(SUBSTRING ${PB_TX} 1 2 _tx_pin)
|
||||
string(TOUPPER "sw:${_rx_pin},${_tx_pin}" _link)
|
||||
@@ -294,21 +286,22 @@ function(pureboot_add_loader name)
|
||||
add_executable(${name} ${CMAKE_CURRENT_FUNCTION_LIST_DIR}/pureboot.cpp)
|
||||
target_link_libraries(${name} PRIVATE libavr)
|
||||
target_compile_definitions(${name} PRIVATE ${_defines})
|
||||
# Codegen shaping for the loader TU only, worth ~40 B on every chip and
|
||||
# what carries the far-flash 1284 build under 512. At -Os GCC otherwise
|
||||
# rewrites the byte-stream loops' counters into end-pointer forms that
|
||||
# cost registers (-fno-ivopts, -fno-split-wide-types), leaves register
|
||||
# pressure on the table with the default allocator
|
||||
# (-fira-algorithm=priority), and spends bytes on rewrites a
|
||||
# straight-line loader gains nothing from.
|
||||
# Codegen shaping for the loader TU only, worth 14–36 B depending on the
|
||||
# chip. At -Os GCC otherwise rewrites the byte-stream loops' counters into
|
||||
# end-pointer forms that cost registers (-fno-ivopts,
|
||||
# -fno-split-wide-types), leaves register pressure on the table with the
|
||||
# default allocator (-fira-algorithm=priority), and keeps loop-invariant
|
||||
# immediates and expression temporaries in registers
|
||||
# (-fno-move-loop-invariants, -fno-tree-ter) — but every loop body here
|
||||
# contains a call, so a register held across it costs more than the
|
||||
# load-immediate it saves.
|
||||
target_compile_options(${name} PRIVATE
|
||||
-fno-ivopts -fira-algorithm=priority -fno-expensive-optimizations -fno-split-wide-types)
|
||||
-fno-ivopts -fira-algorithm=priority -fno-move-loop-invariants -fno-tree-ter -fno-split-wide-types)
|
||||
target_link_options(${name} PRIVATE -nostartfiles -Wl,--section-start=.text=${_base_hex}
|
||||
-Wl,--defsym=pureboot_app=${_app} ${_wrap})
|
||||
add_custom_command(TARGET ${name} POST_BUILD COMMAND ${CMAKE_SIZE} $<TARGET_FILE:${name}>)
|
||||
# The ELF is a container (symbols, section headers), never flashed; the
|
||||
# flashable forms sit beside it: .hex for a programmer, .bin (the slot's
|
||||
# bare bytes) for the host tool's raw path and --update-loader.
|
||||
# The ELF is a container, never flashed: .hex for a programmer, .bin (the
|
||||
# slot's bare bytes) for --update-loader.
|
||||
add_custom_command(TARGET ${name} POST_BUILD
|
||||
COMMAND ${CMAKE_OBJCOPY} -O ihex -R .eeprom
|
||||
$<TARGET_FILE:${name}> $<TARGET_FILE:${name}>.hex
|
||||
@@ -317,3 +310,38 @@ function(pureboot_add_loader name)
|
||||
set_target_properties(${name} PROPERTIES PUREBOOT_HZ ${PB_CLOCK} PUREBOOT_BAUD ${PB_BAUD}
|
||||
PUREBOOT_LINK ${_link})
|
||||
endfunction()
|
||||
|
||||
# pureboot_add_autobaud(<name> <source> [RX <pin>] [TX <pin>])
|
||||
#
|
||||
# An autobaud software-serial loader from <source> (pureboot_autobaud_*.cpp).
|
||||
# Autobaud measures the host's bit timing at runtime, so the image carries no
|
||||
# clock and no baud — one binary per chip runs at any F_CPU. Same per-chip
|
||||
# geometry, link and codegen flags as pureboot_add_loader(); only the clock and
|
||||
# baud axes fall away. Two source files are under review (autobaud.md):
|
||||
# pureboot_autobaud_pure.cpp and pureboot_autobaud_reg.cpp.
|
||||
function(pureboot_add_autobaud name source)
|
||||
cmake_parse_arguments(PB "" "RX;TX" "" ${ARGN})
|
||||
if(NOT PB_RX)
|
||||
set(PB_RX pb0)
|
||||
endif()
|
||||
if(NOT PB_TX)
|
||||
set(PB_TX pb1)
|
||||
endif()
|
||||
foreach(_pin ${PB_RX} ${PB_TX})
|
||||
if(NOT _pin MATCHES "^p[a-h][0-7]$")
|
||||
message(FATAL_ERROR "pureboot_add_autobaud(${name}): pin '${_pin}' is not of the form pb1")
|
||||
endif()
|
||||
endforeach()
|
||||
get_property(_base_hex GLOBAL PROPERTY PUREBOOT_BASE_HEX)
|
||||
get_property(_app GLOBAL PROPERTY PUREBOOT_APP)
|
||||
get_property(_wrap GLOBAL PROPERTY PUREBOOT_WRAP)
|
||||
|
||||
add_executable(${name} ${CMAKE_CURRENT_FUNCTION_LIST_DIR}/${source})
|
||||
target_link_libraries(${name} PRIVATE libavr)
|
||||
target_compile_definitions(${name} PRIVATE PUREBOOT_RX=${PB_RX} PUREBOOT_TX=${PB_TX})
|
||||
target_compile_options(${name} PRIVATE
|
||||
-fno-ivopts -fira-algorithm=priority -fno-move-loop-invariants -fno-tree-ter -fno-split-wide-types)
|
||||
target_link_options(${name} PRIVATE -nostartfiles -Wl,--section-start=.text=${_base_hex}
|
||||
-Wl,--defsym=pureboot_app=${_app} ${_wrap})
|
||||
add_custom_command(TARGET ${name} POST_BUILD COMMAND ${CMAKE_SIZE} $<TARGET_FILE:${name}>)
|
||||
endfunction()
|
||||
|
||||
@@ -2,46 +2,61 @@
|
||||
|
||||
A serial bootloader on [libavr](https://git.blackmark.me/avr/libavr), pure by
|
||||
constraint: one C++ source, no inline assembly, no global register variables
|
||||
(attributes and compiler flags allowed), built for **every chip libavr
|
||||
targets — all 37 — in 512 bytes each**: 434 B on the tiny13s, 438–442 B on
|
||||
the tiny25/45/85, 412–452 B across the megas, and 506 B on the
|
||||
ATmega1284/1284P, whose far-flash machinery (ELPM reads, RAMPZ page commands,
|
||||
word-addressed wire) is the heaviest. Those are the stock deployments;
|
||||
choosing the software UART where the chip has a USART costs 8–46 B more (a
|
||||
bit-bang against a peripheral), which every chip still absorbs inside its
|
||||
slot — on the 1284s that means their 1 KiB boot sector, where the
|
||||
software-serial image lands at 546 B. Bringing the 1284's default build
|
||||
under 512 at all is what the loop-placement attributes on the byte streamers
|
||||
(`pureboot.cpp`) and the codegen flags on the loader TU (`CMakeLists.txt`)
|
||||
are for; measured against each chip's own budget the tightest is the
|
||||
ATmega328P, 50 B spare. Clock, baud, serial backend and
|
||||
pins are per-build configuration (below); the size matrix in the test suite
|
||||
holds every combination inside its slot. The device speaks primitives; every
|
||||
composite — verify, erase, reset-vector surgery, updating the loader itself —
|
||||
lives in the host tool (`pureboot.py`).
|
||||
|
||||
The 1284s still *deploy* in a 1 KiB slot, their smallest boot sector being
|
||||
512 words; at 506 B the image would also fit the 644's
|
||||
two-512-byte-slots-per-boot-sector geometry.
|
||||
(attributes and compiler flags allowed), **512 bytes on every chip libavr
|
||||
targets — all 37**. The device speaks primitives; every composite — verify,
|
||||
erase, reset-vector surgery, updating the loader itself — lives in the host
|
||||
tool (`pureboot.py`).
|
||||
|
||||
The image is **position-independent**: control flow is PC-relative, the
|
||||
read/write paths take wire addresses, the write guard protects the slot the
|
||||
code is *running* in (from the runtime return address), the info block is
|
||||
addressed from that same anchor, and the application jump is an indirect
|
||||
call to an absolute entry. The identical binary therefore runs from any
|
||||
slot with every command intact — which makes pureboot **its own staging
|
||||
loader**: the host installs the same binary one slot below the resident,
|
||||
jumps into it, and lets it rewrite the resident. The slot is 512 bytes
|
||||
(1 KiB on the word-addressed large chips, matching their boot-sector
|
||||
minimum); on the tinies the budget is 510, not 512: a slot's last word
|
||||
belongs to the host-managed trampoline (below).
|
||||
addressed from that same anchor, and the application jump is an indirect call
|
||||
to an absolute entry. The identical binary therefore runs from any slot with
|
||||
every command intact, which makes pureboot **its own staging loader**: the
|
||||
host installs the same binary one slot below the resident, jumps into it, and
|
||||
lets it rewrite the resident.
|
||||
|
||||
## Chips
|
||||
|
||||
Sizes are the default configuration: the hardware USART0 at 115200 8N1 on a
|
||||
16 MHz crystal, or the software UART on RX = PB0 / TX = PB1 at 57600 8N1 on
|
||||
the tinies' RC oscillator (9.6 MHz on the t13s, 8 MHz above). Every axis moves
|
||||
per build — see *Configuration*; the largest image any of them produces is a
|
||||
software UART at a slow baud, which on the 1284s is 494 B, the tightest fit in
|
||||
the whole matrix at 18 B spare.
|
||||
|
||||
| Chip | Flash | Loader at | Link | Size |
|
||||
|---|---|---|---|---|
|
||||
| ATtiny13, ATtiny13A † | 1 KiB | 0x0200 | software | 416 B |
|
||||
| ATtiny25 † | 2 KiB | 0x0600 | software | 420 B |
|
||||
| ATtiny45 † | 4 KiB | 0x0e00 | software | 424 B |
|
||||
| ATtiny85 † | 8 KiB | 0x1e00 | software | 424 B |
|
||||
| ATmega8, 8A | 8 KiB | 0x1e00 | USART0 | 396 B |
|
||||
| ATmega16, 16A | 16 KiB | 0x3e00 | USART0 | 400 B |
|
||||
| ATmega32, 32A | 32 KiB | 0x7e00 | USART0 | 400 B |
|
||||
| ATmega48, 48A, 48P, 48PA † | 4 KiB | 0x0e00 | USART0 | 414 B |
|
||||
| ATmega88, 88A, 88P, 88PA | 8 KiB | 0x1e00 | USART0 | 434 B |
|
||||
| ATmega168, 168A, 168P, 168PA | 16 KiB | 0x3e00 | USART0 | 438 B |
|
||||
| ATmega328, 328P | 32 KiB | 0x7e00 | USART0 | 438 B |
|
||||
| ATmega164A, 164P, 164PA | 16 KiB | 0x3e00 | USART0 | 438 B |
|
||||
| ATmega324A, 324P, 324PA | 32 KiB | 0x7e00 | USART0 | 438 B |
|
||||
| ATmega644, 644A, 644P, 644PA | 64 KiB | 0xfe00 | USART0 | 432 B |
|
||||
| ATmega1284, 1284P | 128 KiB | 0x1fe00 | USART0 | 478 B |
|
||||
|
||||
† No hardware boot section: the host patches the reset vector, and the budget
|
||||
is 510 bytes, since the slot's last word is the trampoline.
|
||||
|
||||
The 1284s are the heaviest because they alone carry the far-flash machinery —
|
||||
ELPM reads, RAMPZ page commands, a word-addressed wire.
|
||||
|
||||
The software UART enables the RX pull-up; TX idles high. All multi-byte wire
|
||||
quantities are little-endian.
|
||||
|
||||
## Configuration
|
||||
|
||||
Every deployment axis is a build parameter, resolved by the CMake function
|
||||
`pureboot_add_loader()` (in `pureboot/CMakeLists.txt`) — the one way a
|
||||
loader target is created, by this repo's own build and by a downstream
|
||||
project alike:
|
||||
Every deployment axis is a build parameter of `pureboot_add_loader()` (in
|
||||
`pureboot/CMakeLists.txt`) — the one way a loader target is created, by this
|
||||
repo's build and by a downstream project alike:
|
||||
|
||||
| Argument | Meaning | Default |
|
||||
|---|---|---|
|
||||
@@ -55,15 +70,14 @@ project alike:
|
||||
The default baud is the fastest of 115200/57600/38400/19200/9600 the clock
|
||||
reaches within 2.5 % — the same U2X-included divisor search libavr's baud
|
||||
solver runs — and on a software build additionally within the polled
|
||||
receiver's 100-cycles-a-bit floor. 16 MHz lands 115200, 8 MHz 57600,
|
||||
1 MHz 9600. Whatever is picked or overridden is re-checked in the compile:
|
||||
an infeasible clock/baud/backend combination, or a USART the chip does not
|
||||
have, fails with a named static assert.
|
||||
receiver's 100-cycles-a-bit floor. Whatever is picked or overridden is
|
||||
re-checked in the compile: an infeasible combination, or a USART the chip does
|
||||
not have, fails with a named static assert.
|
||||
|
||||
A downstream project brings its usual libavr setup (the `libavr` target,
|
||||
the chip via the `LIBAVR_MCU` toolchain preset), consumes this directory,
|
||||
and states its deployment — for example an ATmega328P on its shipped
|
||||
1 MHz fuses with the software UART on hand-picked pins:
|
||||
A downstream project brings its usual libavr setup (the `libavr` target, the
|
||||
chip via the `LIBAVR_MCU` toolchain preset), consumes this directory, and
|
||||
states its deployment — an ATmega328P on its shipped 1 MHz fuses with the
|
||||
software UART on hand-picked pins, say:
|
||||
|
||||
```cmake
|
||||
FetchContent_Declare(bootloader GIT_REPOSITORY git@git.blackmark.me:avr/bootloader.git GIT_TAG main)
|
||||
@@ -74,63 +88,60 @@ pureboot_add_loader(myboot CLOCK 1000000 SERIAL software TX pb1 RX pb5)
|
||||
```
|
||||
|
||||
The function emits the ELF plus `myboot.hex` (the programmer artifact) and
|
||||
`myboot.bin` (the self-update image), prints the size, and stamps the
|
||||
resolved deployment on the target as the `PUREBOOT_HZ`, `PUREBOOT_BAUD`
|
||||
and `PUREBOOT_LINK` properties — what a flashing script or test harness
|
||||
needs to speak to the build. This exact example deployment runs the full
|
||||
protocol suite in CI (`pureboot.custom`).
|
||||
|
||||
## Link
|
||||
|
||||
The stock builds assume the family's natural deployment; any axis moves
|
||||
per build (above).
|
||||
|
||||
| Chip | Serial | Baud | Clock assumed |
|
||||
|---|---|---|---|
|
||||
| every ATmega | the hardware USART (USART0), RXD/TXD per pinout | 115200 8N1 | 16 MHz crystal |
|
||||
| ATtiny25/45/85 | software UART, RX = PB0, TX = PB1 | 57600 8N1 | 8 MHz internal RC |
|
||||
| ATtiny13/13A | software UART, RX = PB0, TX = PB1 | 57600 8N1 | 9.6 MHz internal RC |
|
||||
|
||||
The software-UART RX pin has its pull-up enabled; TX idles high. All
|
||||
multi-byte quantities on the wire are little-endian.
|
||||
`myboot.bin` (the self-update image), prints the size, and stamps the resolved
|
||||
deployment on the target as the `PUREBOOT_HZ`, `PUREBOOT_BAUD` and
|
||||
`PUREBOOT_LINK` properties — what a flashing script or test harness needs to
|
||||
speak to the build. This exact deployment runs the full protocol suite in CI
|
||||
(`pureboot.custom`).
|
||||
|
||||
## Activation
|
||||
|
||||
Reset enters the loader (BOOTRST on the boot-sectioned megas; the patched
|
||||
reset vector on the tinies and the boot-section-less m48s) — except a
|
||||
watchdog reset, which hands straight to the application (the application
|
||||
owns its watchdog; it must clear WDRF itself, which also releases the
|
||||
WDRF-forced WDE).
|
||||
Reset enters the loader (BOOTRST on the boot-sectioned megas, the patched
|
||||
reset vector elsewhere) — except a watchdog reset, which hands straight to the
|
||||
application with no activation window, since the application owns its watchdog.
|
||||
This is deliberate: it lets an application reboot itself instantly rather than
|
||||
sit through the window. The application must clear WDRF itself (libavr's
|
||||
`watchdog::disable()` does). **Gotcha:** WDRF is sticky (cleared only by
|
||||
software, not by a later reset), so an application that watchdog-resets and
|
||||
never clears it diverts *every* subsequent reset — external ones included —
|
||||
past the window too, and the loader becomes reachable only through an external
|
||||
programmer until the flag is cleared. A serial recovery path therefore assumes
|
||||
the application clears WDRF on its own reset path.
|
||||
|
||||
The host then has one activation window per awaited byte to knock: `p` then
|
||||
`b`. Each awaited byte gets a fresh window; any other byte is discarded and
|
||||
awaited again (line noise cannot lock the loader, only delay it). A window
|
||||
expiring with an idle line boots the application.
|
||||
The host then knocks `p` then `b`, each awaited byte under a fresh activation
|
||||
window; any other byte is discarded and awaited again, so line noise can delay
|
||||
the loader but never lock it. A window expiring on an idle line boots the
|
||||
application.
|
||||
|
||||
The window length is a compile-time constant — 8 s by default, another
|
||||
value via `pureboot_add_loader(... TIMEOUT <s>)` (the stock target keeps
|
||||
the `PUREBOOT_TIMEOUT` cache variable) — so the whole EEPROM belongs to
|
||||
the application; pureboot never uses it for its own state. Re-timing a
|
||||
deployed loader is a self-update with a re-timed build (below).
|
||||
The window is a compile-time constant (`TIMEOUT`, 8 s by default), so the whole
|
||||
EEPROM belongs to the application — pureboot keeps no state of its own.
|
||||
Re-timing a deployed loader is a self-update with a re-timed build.
|
||||
|
||||
## Session
|
||||
|
||||
After the knock the loader stays in its command loop until `J` jumps away or
|
||||
the chip resets. Before reading each command it waits for any pending EEPROM
|
||||
write to finish and sends the prompt `+` (0x2b) — the prompt is therefore
|
||||
also the completion ack of the previous command. A session is: await `+`,
|
||||
send a command, read its reply, repeat.
|
||||
write and sends the prompt `+` (0x2b), which is therefore also the previous
|
||||
command's completion ack. A session is: await `+`, send a command, read its
|
||||
reply, repeat.
|
||||
|
||||
On chips whose flash exceeds 64 KiB (the 1284s — info-block flag bit 1) the
|
||||
`R`/`W` flash addresses are **word** addresses; everywhere else they are byte
|
||||
addresses (the 644s' 64 KiB is exactly the 16-bit byte space and stays
|
||||
byte-addressed). EEPROM addresses are always bytes, counts always bytes.
|
||||
addresses (the 644s' 64 KiB is exactly the 16-bit byte space). EEPROM
|
||||
addresses and all counts are bytes.
|
||||
|
||||
The loader trusts the host to keep addresses in range: it does not bound them
|
||||
against the info block. **Gotcha:** a `w` (or `r`) that runs past `E2END` wraps
|
||||
— EEAR is only as wide as the array, so an address past the end truncates onto
|
||||
low EEPROM and the write silently overwrites it. Keeping writes within the
|
||||
advertised sizes is the host's job (the shipped tool does); the flash budget
|
||||
is better spent on features than on re-checking a bound the host already holds.
|
||||
|
||||
| Cmd | Arguments | Reply |
|
||||
|---|---|---|
|
||||
| `b` | — | the 12-byte info block |
|
||||
| `R` | addr16, n8 | n flash bytes (n = 0 means 256) |
|
||||
| `W` | addr16, then one page of data | — (completion = next prompt) |
|
||||
| `W` | addr16 (any address in the page), then one page of data | — (completion = next prompt) |
|
||||
| `r` | addr16, n8 | n EEPROM bytes (n = 0 means 256) |
|
||||
| `w` | addr16, n8, then n data bytes | `+` per byte, sent once its write has begun |
|
||||
| `F` | — | 4 bytes: low fuse, lock, extended fuse, high fuse |
|
||||
@@ -138,85 +149,74 @@ byte-addressed). EEPROM addresses are always bytes, counts always bytes.
|
||||
| other | — | ignored; the loop re-prompts (send a junk byte, await `+`, to resync) |
|
||||
|
||||
`W` streams exactly one SPM page (size from the info block) into the buffer,
|
||||
then erases and programs; the address must be page-aligned. Pages inside the
|
||||
512-byte slot the loader is *running* in are drained but never programmed — a
|
||||
broken host cannot brick the running copy, and a staged copy may rewrite the
|
||||
resident slot.
|
||||
then erases and programs — except pages inside the 512-byte slot
|
||||
the loader is *running* in, which are drained and left alone, so a broken host
|
||||
cannot brick the running copy and a staged copy may rewrite the resident.
|
||||
|
||||
The loader never clears the SPM buffer before a fill, so **one `W` may
|
||||
program the wrong bytes, and the host is what fixes it**. The buffer is
|
||||
write-once per word until cleared, and two things leave words in it: a
|
||||
refused page (drained, never programmed) and — where SPM runs from anywhere,
|
||||
the tinies and the m48s — an application that self-programmed before
|
||||
entering. The next `W` takes those stale words, and clears them: a page write
|
||||
auto-erases the buffer (§26.2.1; §19.2 on the tinies), so repeating it
|
||||
programs correctly. The host therefore verifies every page it writes and
|
||||
rewrites what comes back wrong (three retries, then it stops); a host that
|
||||
programs without reading back cannot trust the first `W` after either event.
|
||||
The loader never clears the SPM buffer before a fill, so **one `W` may program
|
||||
the wrong bytes, and the host is what fixes it**. The buffer is write-once per
|
||||
word until cleared, and two things leave words in it: a refused page, and —
|
||||
where SPM runs from anywhere, the tinies and the m48s — an application that
|
||||
self-programmed before entering. The next `W` takes those stale words and
|
||||
clears them, since a page write auto-erases the buffer (§26.2.1; §19.2 on the
|
||||
tinies), so repeating it programs correctly. The host therefore verifies every
|
||||
page it writes and rewrites what comes back wrong (three retries, then it
|
||||
stops).
|
||||
|
||||
`w` is host-paced: send the next byte only after the previous
|
||||
byte's `+`. `F` returns the bytes in the hardware's Z order; on a chip
|
||||
without an extended fuse byte (the ATtiny13A) that slot carries no meaning.
|
||||
Fuse *writing* does not exist: SPM reaches flash (and, on the mega, lock
|
||||
bits) only — fuse bytes are external-programming territory by hardware.
|
||||
`w` is host-paced: send the next byte only after the previous byte's `+`. `F`
|
||||
returns the bytes in the hardware's Z order; on a chip without an extended
|
||||
fuse byte that slot carries no meaning. Fuse *writing* does not exist — SPM
|
||||
reaches flash and boot lock bits only.
|
||||
|
||||
`J` is the one control-transfer primitive: the host uses it to run the
|
||||
application (word 0 on the mega, the trampoline word on the tinies — both
|
||||
known from the info block) and to move between loader copies during a
|
||||
self-update. A jump to a loader slot's base re-enters that copy's own
|
||||
startup; it must then be knocked afresh.
|
||||
`J` is the one control-transfer primitive: it runs the application (word 0 or
|
||||
the trampoline word, both known from the info block) and moves between loader
|
||||
copies during a self-update. A jump to a slot's base re-enters that copy's own
|
||||
startup, which must then be knocked afresh.
|
||||
|
||||
The info block (`b`):
|
||||
|
||||
| Offset | Content |
|
||||
|---|---|
|
||||
| 0–2 | `'P'`, `'B'`, protocol version (1) |
|
||||
| 0–2 | `'P'`, `'B'`, pureboot version (3) |
|
||||
| 3–5 | device signature |
|
||||
| 6 | SPM page size in bytes (0 means 256) |
|
||||
| 7–8 | loader base — application flash ends here (a word address when bit 1 is set) |
|
||||
| 9–10 | EEPROM size |
|
||||
| 11 | bit 0: host must patch the reset vector (no hardware boot section); bit 1: flash wire addresses are word addresses |
|
||||
|
||||
Composites are the host's job: verify = read back and compare, erase =
|
||||
write `0xff` (per page for flash, per byte for EEPROM).
|
||||
## Version
|
||||
|
||||
The info block's third byte is the **pureboot version** — the loader's one
|
||||
identity number, and the only way to tell what a deployed loader is. Nothing
|
||||
else is numbered: the wire protocol has no version, a pureboot version implies
|
||||
it, and the host tool holds that map. The tool states the window of loader
|
||||
versions it speaks (`OLDEST_LOADER`/`NEWEST_LOADER` in `pureboot.py`), and a
|
||||
version that changes the protocol becomes the new floor there. None has so
|
||||
far: 1 through 3 speak the identical session. A loader newer than the tool is
|
||||
refused by name rather than decoded on the assumption that nothing moved.
|
||||
|
||||
The tool carries its own version, free to drift; `--version` prints it and the
|
||||
window.
|
||||
|
||||
## Deployment
|
||||
|
||||
The build leaves three artifacts per chip. The ELF is a container for the
|
||||
tests and objcopy — never flashed. The **.hex is the programmer artifact**:
|
||||
it carries its own addresses and lands the loader in its top slot,
|
||||
touching nothing else. The **.bin is the self-update image** — the slot's
|
||||
bare bytes with no addressing, which a programmer would put at address 0.
|
||||
On a boot-sectioned mega a copy at 0 is dead weight (SPM only executes
|
||||
from the boot section, so it cannot even heal itself — reflash the .hex);
|
||||
on the patched-vector chips it *runs* (the image is position-independent
|
||||
and reset enters word 0), reports its canonical geometry, and the ordinary
|
||||
`--update-loader` flow re-homes a build into the top slot from any
|
||||
position — the staging install and the word-0 redirect execute from
|
||||
copies outside page 0's slot, and a copy sitting in the staging slot
|
||||
itself is recognized as the installed staging copy and left in place (it
|
||||
streams the new resident like any staged copy, so an older build installs
|
||||
a newer one). `pureboot.rehome` is the acceptance test for both
|
||||
positions. Flashing the application afterwards overwrites the stale copy,
|
||||
vector surgery included.
|
||||
tests and objcopy, never flashed. The **.hex is the programmer artifact**: it
|
||||
carries its own addresses and lands the loader in its top slot, touching
|
||||
nothing else. The **.bin is the self-update image** — the slot's bare bytes.
|
||||
|
||||
**Boot-sectioned megas**: program the loader at `flash − slot` with an
|
||||
external programmer. Every such mega has a BOOTSZ step whose boot section
|
||||
is exactly the loader slot — 512 B, the second-smallest step on the 8 KiB
|
||||
and 16 KiB chips (m8, m88, m16, m168, m164), the smallest on the 32 KiB
|
||||
ones (m32, m328, m324); on the 1284s that step is the smallest, 512 words,
|
||||
which is why their slot is 1 KiB — so the ATmega328P profiles below apply
|
||||
to every one of them with its own addresses and slot size; the per-chip
|
||||
BOOTSZ ladders live in the host tool (`BOOT_FUSE`). The 1284s' numbers:
|
||||
standalone = BOOTSZ 512 words (reset at the loader base 0x1fc00);
|
||||
self-update = 1024 words, covering both 1 KiB slots, the loader-first
|
||||
reset landing at 0x1f800 — the staging slot, walked across when erased.
|
||||
**Boot-sectioned megas**: program the loader at `flash − 512` with an external
|
||||
programmer. Every such mega has a BOOTSZ step whose boot section is exactly
|
||||
the 512-byte slot — the second-smallest step on the 8 KiB and 16 KiB chips,
|
||||
the smallest on the 32 KiB ones — so the ATmega328P profiles below apply to
|
||||
every one of them with its own addresses; the per-chip BOOTSZ ladders live in
|
||||
the host tool (`BOOT_FUSE`).
|
||||
|
||||
The **644s** are the geometry's sweet spot: their smallest boot section
|
||||
(512 words = 1 KiB) is exactly *two* 512-byte slots, so the resident and
|
||||
its staging slot both live inside the minimum section — self-update needs
|
||||
no fuse step up, and the standalone profile does not exist (reset lands at
|
||||
0xfc00, one erased slot below the loader: the loader-first walk built in).
|
||||
The **644s and 1284s** are the geometry's sweet spot: their smallest boot
|
||||
section (512 words = 1 KiB) is exactly *two* slots, so the resident and its
|
||||
staging slot both live inside the minimum section. Self-update needs no fuse
|
||||
step up, and the standalone profile does not exist — reset lands one erased
|
||||
slot below the loader (0xfc00 / 0x1fc00) and walks up into it.
|
||||
|
||||
ATmega328P profiles (addresses for its 32 KiB):
|
||||
|
||||
@@ -226,154 +226,142 @@ ATmega328P profiles (addresses for its 32 KiB):
|
||||
| 512 words (1 KB) | unprogrammed | *Self-update, app-first*: reset always boots the application, which owns all 31.5 KB and must offer its own jump to 0x7e00 to reach the loader (a virgin chip reaches it by reset across erased flash). Updates are power-fail-safe except mid-rewrite of the resident slot itself (no reset path leads to the staging copy then). |
|
||||
| 512 words (1 KB) | programmed | *Self-update, loader-first*: reset lands at 0x7c00 — the staging slot, normally erased, so execution walks up into the loader; during an update it is the staging copy itself, so a mid-rewrite power loss recovers by reset. The loss windows move to the staging install/retire page writes instead (page-write scale). The host keeps `[0x7c00, 0x7e00)` clear of application data (`--force` overrides). |
|
||||
|
||||
Applications are flashed unmodified — word 0 stays the application's own
|
||||
Applications are flashed unmodified here — word 0 stays the application's own
|
||||
reset vector, and the hand-over jumps to 0.
|
||||
|
||||
**Patched-vector chips — the tinies and the m48s** (no boot section; the
|
||||
m48s' SPM runs from the entire flash, Atmel-8271 §26): program the loader
|
||||
at `flash − 512`; erased flash below it walks up into the loader, so a
|
||||
virgin chip activates. When flashing an application the host performs
|
||||
reset-vector surgery: word 0 is rewritten to `rjmp` to the loader base, and
|
||||
the application's own entry is re-encoded as a trampoline `rjmp` in the
|
||||
word just below the loader (`base − 2`, where the hand-over jumps). Every
|
||||
other vector stays the application's. The patched page 0 and the trampoline
|
||||
page are written *first*, so from the first write on an interrupted flash
|
||||
still resets into the loader; an erase runs top-down for the same reason.
|
||||
The m48s speak this profile over their hardware USART — no fuse preflight,
|
||||
BOOTRST does not exist there.
|
||||
**Patched-vector chips — the tinies and the m48s** (no boot section; the m48s'
|
||||
SPM runs from the entire flash, Atmel-8271 §26): program the loader at
|
||||
`flash − 512`; erased flash below it walks up into the loader, so a virgin
|
||||
chip activates. Flashing an application then takes reset-vector surgery: word
|
||||
0 becomes an `rjmp` to the loader base, and the application's own entry is
|
||||
re-encoded as a trampoline `rjmp` in the word just below the loader
|
||||
(`base − 2`, where the hand-over jumps). Every other vector stays the
|
||||
application's. The patched page 0 and the trampoline page are written *first*
|
||||
and an erase runs top-down, so from the first write on an interruption still
|
||||
resets into the loader.
|
||||
|
||||
A .bin programmed at address 0 by mistake is dead weight on a boot-sectioned
|
||||
mega (SPM only executes from the boot section — reflash the .hex), but *runs*
|
||||
on a patched-vector chip, and the ordinary `--update-loader` flow re-homes it
|
||||
into the top slot from there (`pureboot.rehome`).
|
||||
|
||||
## Updating the loader
|
||||
|
||||
`pureboot.py --update-loader new_pureboot.bin` replaces the resident loader
|
||||
with any pureboot build — a re-timed window, a newer protocol — using the
|
||||
with any pureboot build — a re-timed window, a newer version — using the
|
||||
loader itself as its own staging loader. The image is the loader's own 512
|
||||
bytes as a raw binary, or the Intel HEX the build emits beside it, which
|
||||
links the loader at its base inside an otherwise blank flash image:
|
||||
bytes as a raw binary, or the Intel HEX the build emits beside it.
|
||||
|
||||
The preflight refuses an image built for another chip: the info block
|
||||
embedded in every pureboot binary (signature, page size, loader base,
|
||||
EEPROM size, flags) must match the device's own, and the error names both.
|
||||
Die revisions share their base signature and geometry, so their images are
|
||||
interchangeable — as the silicon is. `loader_image()` also accepts a
|
||||
padded image (a raw .bin padded from 0, or a whole-flash read-back with
|
||||
the loader resident) and peels it to the slot content by the embedded base.
|
||||
The preflight refuses an image built for another chip: the info block embedded
|
||||
in every pureboot binary (signature, page size, loader base, EEPROM size,
|
||||
flags) must match the device's own, and the error names both. Die revisions
|
||||
share their base signature and geometry, so their images are interchangeable —
|
||||
as the silicon is.
|
||||
|
||||
1. The staging slot `[base−slot, base)` is saved to a host-side state file
|
||||
(on the 1 KB tiny13s that is the whole application, vectors included).
|
||||
2. The resident installs the identical update image there. On the
|
||||
patched-vector chips the host composes the slot's last word — the same
|
||||
address as the resident's trampoline — as a jump to the resident base,
|
||||
so even an abandoned staging copy times out into a loader, never into
|
||||
garbage. A loader already sitting whole in the staging slot (its info
|
||||
block in place, the slot unchanged since the update began) is left as
|
||||
the staging copy instead — rewriting it would only meet its own
|
||||
running-slot guard.
|
||||
3. `J` enters the staging copy, which rewrites the resident slot. On the
|
||||
patched-vector chips whose staging slot sits away from page 0 the host
|
||||
first re-aims word 0 at the staging copy, so a power loss mid-rewrite
|
||||
still resets into a loader; on the tiny13s the staging slot carries the
|
||||
reset vector itself.
|
||||
1. The staging slot `[base−512, base)` is saved to a host-side state file (on
|
||||
the 1 KB tiny13s that is the whole application, vectors included).
|
||||
2. The resident installs the update image there. On the patched-vector chips
|
||||
the host composes the slot's last word as a jump to the resident base, so
|
||||
even an abandoned staging copy times out into a loader. A loader already
|
||||
sitting whole in the staging slot is left as the staging copy instead —
|
||||
rewriting it would only meet its own running-slot guard.
|
||||
3. `J` enters the staging copy, which rewrites the resident slot. Where a
|
||||
patched reset vector routes through the resident, the host first re-aims
|
||||
word 0 at the staging copy, so a power loss mid-rewrite still resets into a
|
||||
loader; on the tiny13s the staging slot carries the reset vector itself.
|
||||
4. `J` enters the new resident, which restores the staging slot's saved
|
||||
content (word 0 and the trampoline with it) and the state file is
|
||||
discarded.
|
||||
content, and the state file is discarded.
|
||||
|
||||
Every phase is idempotent and keyed off the actual flash state: re-running
|
||||
the same command after any interruption resumes and completes. The state
|
||||
file carries the only bytes not recoverable from the device; if it is lost
|
||||
mid-update the update still completes, and the staging region is restored by
|
||||
reflashing the application. A boot-sectioned mega needs its fuses for the
|
||||
preflight (BOOTSZ gate, profile notes) — read from the device, or supplied
|
||||
with `--assume-fuses` where reading is impossible (simulators); the
|
||||
patched-vector chips need none.
|
||||
Every phase is idempotent and keyed off the actual flash state, so re-running
|
||||
the same command after any interruption resumes and completes. The state file
|
||||
carries the only bytes not recoverable from the device; losing it mid-update
|
||||
still completes the update, and the staging region comes back by reflashing
|
||||
the application. A boot-sectioned mega needs its fuses for the preflight — read
|
||||
from the device, or supplied with `--assume-fuses` where reading is impossible
|
||||
(simulators).
|
||||
|
||||
## Host tool
|
||||
|
||||
`pureboot.py` — Python 3, standard library only. The port layer is the one
|
||||
platform-specific part: termios drives any tty on POSIX (a USB adapter as
|
||||
well as a simavr pty), the Win32 serial API through `ctypes` drives a COM
|
||||
port on Windows (`--port COM6`; the `\\.\` form for two-digit ports is
|
||||
supplied by the tool). Opening the port asserts DTR and RTS on both, so a
|
||||
board that wires DTR to reset gets its reset pulse and opens the activation
|
||||
window by itself.
|
||||
platform-specific part: termios drives any tty on POSIX (a USB adapter as well
|
||||
as a simavr pty), the Win32 serial API through `ctypes` drives a COM port on
|
||||
Windows (`--port COM6`; the `\\.\` form for two-digit ports is supplied by the
|
||||
tool). Opening the port asserts DTR and RTS on both, so a board that wires DTR
|
||||
to reset gets its reset pulse and opens the activation window by itself.
|
||||
|
||||
pureboot.py --port /dev/ttyUSB0 --baud 57600 \
|
||||
--info --fuses --flash app.hex
|
||||
|
||||
Operations run in a fixed order within one session: info, fuses, loader
|
||||
update, flash (erase / program / read / verify), EEPROM (erase / program /
|
||||
read / verify) — then the loader hands over to the application; `--stay`
|
||||
keeps the session alive instead, and a later invocation reconnects into it
|
||||
(the knock converges there too). `--flash` and `--eeprom` verify by
|
||||
read-back unless `--no-verify`, and a flash page that reads back wrong is
|
||||
rewritten up to three times before the run stops — the loader leaves one
|
||||
recoverable way for a page to land wrong (see `W` above), and rewriting is
|
||||
what clears it. `--verify-flash` only reports. Images are raw binary, or
|
||||
Intel HEX by extension. `--force` overrides the refusable safety checks (today: flashing
|
||||
application data into a mega's reset walk region).
|
||||
update, flash (erase / program / read / verify), EEPROM (the same) — then the
|
||||
loader hands over to the application. `--stay` keeps the session alive
|
||||
instead, and a later invocation reconnects into it. `--flash` and `--eeprom`
|
||||
verify by read-back unless `--no-verify`, and a flash page that reads back
|
||||
wrong is rewritten up to three times before the run stops (see `W` above).
|
||||
`--verify-flash` only reports. Images are raw binary, or Intel HEX by
|
||||
extension. `--force` overrides the refusable safety checks — today, flashing
|
||||
application data into a mega's reset walk region.
|
||||
|
||||
Readouts come one fact per line: `--info` prints the decoded info block
|
||||
field by field, `--fuses` each fuse byte on its own line — plus, on a
|
||||
boot-sectioned mega, the decoded meaning (where the BOOTSZ section starts,
|
||||
what BOOTRST does to reset). Transfers that take wire time — programming,
|
||||
reading, erasing, verifying, the update phases — draw a transient progress
|
||||
bar on stderr when it is a tty; logs and pipes see only the summary lines.
|
||||
`-v`/`--verbose` adds the decisions as they happen: knock counts, the
|
||||
programming plan (vector-surgery targets, skipped blank pages), update
|
||||
state handling and per-phase page counts.
|
||||
Readouts come one fact per line: `--info` decodes the info block field by
|
||||
field, `--fuses` each fuse byte plus, on a boot-sectioned mega, its decoded
|
||||
meaning. Transfers that take wire time draw a transient progress bar on stderr
|
||||
when it is a tty. `-v`/`--verbose` adds the decisions as they happen: knock
|
||||
counts, the programming plan, update state handling and per-phase page counts.
|
||||
|
||||
## Tests
|
||||
|
||||
`tools/check.sh` runs every chip's workflow (`tools/check.sh --full` adds
|
||||
the reflect-mode builds of libavr's spot set; `tools/make_presets.py`
|
||||
regenerates the presets). Per chip preset, `ctest` runs:
|
||||
`tools/check.sh` runs every chip's workflow (`--full` adds the reflect-mode
|
||||
builds of libavr's spot set; `tools/make_presets.py` regenerates the presets).
|
||||
Per chip preset, `ctest` runs:
|
||||
|
||||
- `pureboot.size` — the 510-byte (tinies) / 512-byte (mega) budget;
|
||||
- `pureboot_*.size` — the size matrix: the serial backends × the clock
|
||||
ladder (1/8/16 MHz; the t13s' own RC menu), plus the USART1 build on the
|
||||
x4 chips — every configuration axis that could move the image, each
|
||||
variant against the same slot budget (pins are immediate operands and the
|
||||
timeout is a constant: size-neutral);
|
||||
- `pureboot.custom` (328P) — the configured-deployment acceptance test: the
|
||||
1 MHz software-serial TX=PB1/RX=PB5 build from the configuration example
|
||||
drives the full protocol suite through the runner's GPIO bridge, fixture
|
||||
application included;
|
||||
- `pureboot.usart1` (644A) — the same protocol suite over the second
|
||||
hardware USART: instance selection is compile-checked everywhere, but
|
||||
only a live session proves the loader polls the USART it claims;
|
||||
- `pureboot.pi` — the position-independence lint: no absolute `jmp`/`call`
|
||||
in the image, the info block within its first 256 bytes;
|
||||
- `pureboot.planner` — the host tool's pure logic: programming orders and
|
||||
their recovery properties, the surgery, the staging composition, the
|
||||
boot-fuse decode, the update preflight's error/warning matrix over
|
||||
synthetic fuse bytes, and the repairing verify against a fake device — one
|
||||
bad write repaired in a single rewrite, a page that never comes good
|
||||
stopping after exactly three;
|
||||
- `pureboot.size` — the 510-byte (patched-vector) / 512-byte budget;
|
||||
- `pureboot_*.size` — the size matrix: the serial backends × the clock ladder
|
||||
(1/8/16 MHz; the t13s' own RC menu), the USART1 instance across that same
|
||||
ladder on the x4 chips, and `pureboot_sw_wide`, the slowest ladder rate at
|
||||
the fastest clock — where a software UART's per-bit spin outgrows its
|
||||
one-register delay loop and takes the 16-bit one. That is the largest image
|
||||
the configuration space produces, and a shape the ladder default (always the
|
||||
*fastest* rate a clock reaches) never picks. Pins are immediate operands and
|
||||
the timeout is a constant: neither is an axis;
|
||||
- `pbm_*.size` — under `--full`, the exhaustive cross product replacing that
|
||||
compact matrix: every plausible oscillator (the internal ones, the CKDIV8
|
||||
floor, the plain and the UART crystals) × every rate reachable from it ×
|
||||
every backend, unreachable combinations dropping out rather than aborting
|
||||
the configure. Bounded to one chip per size-bearing class — flash
|
||||
addressing, hand-over shape, page size, USART inventory — since everything
|
||||
else in the image is chip-independent code;
|
||||
- `pureboot.pi` — the position-independence lint: no absolute `jmp`/`call`, the
|
||||
info block within the image's first 256 bytes;
|
||||
- `pureboot.planner` — the host tool's pure logic: programming orders and their
|
||||
recovery properties, the surgery, the staging composition, the boot-fuse
|
||||
decode, the update preflight over synthetic fuse bytes, and the repairing
|
||||
verify against a fake device;
|
||||
- `pureboot.protocol` — end to end against a simavr device
|
||||
(`test/pureboot_device.c` — a hardware USART as a pty, or a cycle-timed
|
||||
GPIO⇄pty bridge for a software-UART build, selected with `-l` to match
|
||||
the loader's link; plus the SPM/NVM module simavr's tiny cores lack)
|
||||
driven by the real host tool through
|
||||
knock-from-reset, program + verify of both memories, session reconnect, an
|
||||
external reset through the patched vector, and the hand-over to a fixture
|
||||
application whose banner proves the launch — cross-checked against the
|
||||
simulator's ground-truth memory dumps and an independent decode of the
|
||||
surgery's rjmp words;
|
||||
- `pureboot.reloc` — the identical image installed one slot below the
|
||||
resident serves the complete command set from there (the
|
||||
position-independence acceptance test);
|
||||
- `pureboot.dirty` (328P) — entering the loader from a running application
|
||||
with no reset between, over an SPM page buffer the fixture deliberately
|
||||
dirtied: the case the loader declines to guard against. A bare verify must
|
||||
see the corruption, the repairing verify must fix it in one rewrite, and a
|
||||
plain verify afterwards must pass. On the boot-sectioned megas hardware
|
||||
forbids the state outright (SPM runs only from the boot section, and reset
|
||||
erases the buffer), but simavr dispatches SPM from anywhere — which is what
|
||||
makes the path constructible at all;
|
||||
- `pureboot.update` — the full `--update-loader` flow to a re-timed build,
|
||||
then every power-fail phase: the device is killed mid-write, restarted
|
||||
from its flash dump, and a re-run must complete the update with the
|
||||
application intact throughout.
|
||||
(`test/pureboot_device.c`: a hardware USART as a pty, or a cycle-timed
|
||||
GPIO⇄pty bridge for a software-UART build, plus the SPM/NVM module simavr's
|
||||
tiny cores lack) driven by the real host tool through knock-from-reset,
|
||||
program + verify of both memories, session reconnect, an external reset
|
||||
through the patched vector, and the hand-over to a fixture application whose
|
||||
banner proves the launch — cross-checked against the simulator's
|
||||
ground-truth memory dumps and an independent decode of the surgery;
|
||||
- `pureboot.reloc` — the identical image one slot below the resident serves the
|
||||
complete command set from there;
|
||||
- `pureboot.rehome` (t85) — a loader programmed at address 0 or in the staging
|
||||
slot re-homes into the top slot through the ordinary update flow;
|
||||
- `pureboot.custom` (328P) — the configuration example's 1 MHz software-serial
|
||||
build driving the full protocol suite, proving the plumbing produces a
|
||||
working loader and not just one that fits;
|
||||
- `pureboot.usart1` (644A) — the same suite over the second hardware USART:
|
||||
instance selection is compile-checked everywhere, but only a live session
|
||||
proves the loader polls the USART it claims;
|
||||
- `pureboot.dirty` (328P) — entering the loader from a running application over
|
||||
an SPM buffer it deliberately dirtied, the case the loader declines to guard:
|
||||
a bare verify must see the corruption and the repairing verify must fix it in
|
||||
one rewrite. Hardware forbids the state here, but simavr dispatches SPM from
|
||||
anywhere, which is what makes the path constructible;
|
||||
- `pureboot.update` — the full `--update-loader` flow, then every power-fail
|
||||
phase: the device is killed mid-write, restarted from its flash dump, and a
|
||||
re-run must complete the update with the application intact.
|
||||
|
||||
`size`, `pi`, and `planner` are host logic and run anywhere; the
|
||||
simulator-driven targets need simavr and a pty, so they are POSIX-only —
|
||||
on Windows the tool is exercised against real hardware.
|
||||
`size`, `pi` and `planner` are host logic and run anywhere; the
|
||||
simulator-driven targets need simavr and a pty, so they are POSIX-only.
|
||||
|
||||
251
pureboot/autobaud.md
Normal file
251
pureboot/autobaud.md
Normal file
@@ -0,0 +1,251 @@
|
||||
# pureboot autobaud — findings, and how the version question settled itself
|
||||
|
||||
The autobaud loader measures the host's bit timing **at runtime** from a
|
||||
calibration pulse, so the image carries no clock: one clock-agnostic binary per
|
||||
chip runs at any F_CPU and locks onto whatever baud the host sends. It exists
|
||||
for the software-serial deployments — the RC-oscillator parts (the tinies,
|
||||
internal-oscillator megas) whose exact clock is uncertain and drifts, so today
|
||||
each needs a per-clock build. Autobaud erases that axis. That is the win —
|
||||
deployment, not bytes.
|
||||
|
||||
This document records how the fit was established, why the two versions that
|
||||
were under review are both dead, and what the loader looks like now.
|
||||
|
||||
## The decision, settled by measurement
|
||||
|
||||
Two autobaud loaders were built for review, differing in one tradeoff — the
|
||||
running-slot write guard against strict purity. `pureboot_autobaud_pure.cpp`
|
||||
landed at **508 B** on the 1284P (4 B spare) and `pureboot_autobaud_reg.cpp` at
|
||||
**512** (zero spare).
|
||||
|
||||
Then hardware testing found a defect that neither could absorb.
|
||||
|
||||
**A single spurious calibration pulse wedged the loader.** `run()` budgeted only
|
||||
the start-edge wait inside `measure()`; the `rx()` that read the knock behind it
|
||||
was unbudgeted and blocked forever. One stray low pulse on an unattended device
|
||||
— EMI, or a host that opens the port and never knocks — held the loader in its
|
||||
activation loop and the application never ran. On a field device that is a hang,
|
||||
not a hiccup, and it is exactly the deployment autobaud is for.
|
||||
|
||||
The fix is to bound the whole activation: an expired knock budget returns a byte
|
||||
that cannot be the knock, so control falls back into the budgeted `measure()`,
|
||||
and a line that stays idle boots the application there. It costs about 22 bytes.
|
||||
|
||||
| 1284P, with the activation fix | size | 512 B budget |
|
||||
|---|---|---|
|
||||
| `pureboot_autobaud_pure.cpp` | 530 | **over by 18** |
|
||||
| `pureboot_autobaud_reg.cpp` | 534 | **over by 22** |
|
||||
| `pureboot_autobaud_uni.cpp` | **464** | **48 B spare** |
|
||||
|
||||
Both candidates were unshippable, and the margin they were competing over was
|
||||
never real — it was the space the missing fix should have occupied. So the
|
||||
choice is not between them. It is the third loader below, which fits with room
|
||||
to spare *and* carries features neither had. The two sources stay in the tree
|
||||
for the record; only the unified one is built.
|
||||
|
||||
## The unified loader
|
||||
|
||||
The insight that paid was the one that had already paid once: **merging command
|
||||
bodies removes cost that moving them around only redistributes.** Folding `R`,
|
||||
`r` and `w` into a single address-and-count path had been worth 14 B earlier.
|
||||
Pushed further — one read command and one write command over *named spaces* —
|
||||
it is worth far more, because four transfer loops collapse into one.
|
||||
|
||||
`pureboot_autobaud_uni.cpp` is pureboot 5. It is strictly pure: no inline
|
||||
assembly, no global register variable, and **no GPIOR either** — the measured
|
||||
unit lives in a plain static, so the loader claims no chip resource an
|
||||
application might want, and the GPIOR-versus-static question disappears along
|
||||
with the chips that have no GPIOR.
|
||||
|
||||
### The protocol
|
||||
|
||||
| command | arguments | |
|
||||
|---|---|---|
|
||||
| `b` | — | version, then the three signature bytes |
|
||||
| `J` | addr16 | ack, then jump (word address) |
|
||||
| `W` | sel8, addr16, page bytes | fill the flash page buffer |
|
||||
| `G` | sel8, addr16, n8 | read n bytes (0 means 256) |
|
||||
| `g` | sel8, addr16, n8, then n bytes | write, each byte acked |
|
||||
|
||||
`sel` is `space | bank << 4`. The low nibble names the space; the high nibble is
|
||||
flash's third address byte, so every transfer speaks a **byte** address inside a
|
||||
64 KiB bank and no command has to carry word addresses. The host must not span a
|
||||
bank boundary in one transfer — it already chunks by page, so nothing it does
|
||||
comes close.
|
||||
|
||||
| space | | |
|
||||
|---|---|---|
|
||||
| 0 | flash | `lpm`/`elpm` |
|
||||
| 1 | EEPROM | |
|
||||
| 2 | data | SRAM — and with it the register file and every I/O register, which share the data address space on AVR |
|
||||
| 3 | fuse and lock | |
|
||||
| 4 | SPM | write-only: the data byte goes to SPMCSR and fires the instruction at the selected address |
|
||||
|
||||
Three things follow from that table that the loader never had:
|
||||
|
||||
- **RAM read and write**, the missing feature. In a per-command design it would
|
||||
have cost a fresh dispatch arm and a fresh loop, ~30 B, on a loader with 4 B
|
||||
spare. As one more space on a shared loop it is a single `ld`/`st`. It also
|
||||
hands the host arbitrary **I/O register access** for free, because AVR maps
|
||||
the peripherals into the same address space.
|
||||
- **Host-driven SPM.** `W` used to end with a hardcoded erase, write and RWW
|
||||
re-enable — 42 B. Those are now three writes to the SPM space, reusing the
|
||||
store path's own address, data byte and ack. The host pays three extra
|
||||
round-trips per page (18 wire bytes against 256 of data) and gains the ability
|
||||
to issue *any* SPM operation, lock bits included.
|
||||
- **`W` on the same footing as everything else.** It takes the same selector and
|
||||
the same byte address instead of a word address of its own, which made flash
|
||||
addressing uniform across the protocol *and* was 20 B cheaper than keeping its
|
||||
private convention.
|
||||
|
||||
Read and write are `G` and `g` — the same letter, one bit apart — so the
|
||||
transfer loop picks its direction with a one-word skip rather than a compare.
|
||||
|
||||
### Why the SPM space has to be one primitive
|
||||
|
||||
It is tempting to go further and expose a generic "poke this I/O register", from
|
||||
which the host could drive SPM itself. The hardware forbids it: `out SPMCSR, x`
|
||||
and the `spm` that follows must issue within four cycles, and EEPROM's
|
||||
EEMPE→EEPE window is the same shape. A host cannot hit a four-cycle window
|
||||
across a serial link. **Atomicity is the floor, and the atomic unit must be
|
||||
resident.** That is the real limit on how low-level a bootloader's primitives
|
||||
can go — not the byte count.
|
||||
|
||||
## Where the bytes went
|
||||
|
||||
The starting point was the 508 B pure build, disassembled and attributed:
|
||||
|
||||
| phase | bytes |
|
||||
|---|---|
|
||||
| `link::rx` + `link::tx` (bit-banged UART) | 102 |
|
||||
| autobaud measure + knock | 62 |
|
||||
| reset vector, pin init, WDRF check, jump, ack, EEPROM wait | 50 |
|
||||
| command loop head + dispatch tree | 42 |
|
||||
| `W` program flash page | 98 |
|
||||
| `R` / `r` / `w` / `F` bodies, plus the shared address decode | 122 |
|
||||
| `b` info, `J` jump | 32 |
|
||||
|
||||
**208 B — 41% — is physical layer and activation**, which no protocol change can
|
||||
touch. The command bodies were the entire addressable surface, and they were
|
||||
four copies of one idea.
|
||||
|
||||
The route from a first attempt to the final loader, all on the 1284P:
|
||||
|
||||
| step | size |
|
||||
|---|---|
|
||||
| unified `G`/`P`, `load`/`store` outlined, `__uint24` cursor | 600 |
|
||||
| …`load`/`store` inlined; unit in `.noinit`, not `.bss` | 534 |
|
||||
| …16-bit cursor with the bank in the selector; direction as a command bit | 510 |
|
||||
| …erase/write/RWW moved out to the SPM space | 484 |
|
||||
| …`W` sharing the selector-and-address decode | **464** |
|
||||
|
||||
Three of those steps are worth keeping as lessons:
|
||||
|
||||
- **Outlining `load`/`store` cost more than the four bodies they replaced.** As
|
||||
functions they were 110 B against the 106 B of inline bodies — the AVR ABI's
|
||||
argument marshalling plus prologue ate the entire saving. Inlined into the one
|
||||
shared loop they cost only their own instructions. The call-site lesson cuts
|
||||
both ways: *merging* call sites pays, *creating* one does not.
|
||||
- **A `.bss` static drags in `__do_clear_bss`** — 18 B of startup code to zero a
|
||||
variable that is always measured before it is read. `.noinit` is correct here
|
||||
and free.
|
||||
- **A three-byte cursor taxes every space.** Widening the shared cursor so flash
|
||||
could reach past 64 KiB put an extra increment on EEPROM and RAM reads that
|
||||
never need it. Moving the bank into the selector byte kept the cursor at
|
||||
sixteen bits and cost nothing on the wire.
|
||||
|
||||
### What did not work
|
||||
|
||||
- **Encoding the space in the command byte** (so dispatch becomes masking rather
|
||||
than a compare tree) cannot carry the bank's four bits alongside a space. The
|
||||
cheap half of the idea survived as the direction bit; the rest lost to the
|
||||
selector byte, which is also more extensible.
|
||||
- **A generic primitive interpreter** — a loader with no logic at all, driven
|
||||
entirely by the host — is not reachable on AVR. Harvard architecture means the
|
||||
program counter cannot fetch from data space, so the classic "upload a flash
|
||||
algorithm into RAM and jump to it" bootstrap is impossible, and on every
|
||||
boot-sectioned part SPM only takes effect from the boot section anyway. What
|
||||
remains is a fixed primitive set: still a protocol, still logic, only at a
|
||||
different granularity. debugWIRE reaches that design point only because its
|
||||
interpreter is *in silicon*; it costs the loader nothing because it is not in
|
||||
the loader.
|
||||
- Below 512 B the saved bytes are largely unspendable on the boot-sectioned
|
||||
chips: the 328P's smallest boot section is exactly 512 B, and the 1284P's is
|
||||
1024 B, of which pureboot already occupies only the top half. The margin
|
||||
matters as headroom for correctness fixes — as this defect showed — not as
|
||||
flash returned to the application. On the patch-vector parts, which have no
|
||||
boot section, it *is* returned: on the ATtiny13 the loader is 43% of a 1 KiB
|
||||
part, and every byte is real.
|
||||
|
||||
## The codegen coupling, still load-bearing
|
||||
|
||||
`count >> 2` is exact only because the calibration pulse's bit-count (7, from
|
||||
the 0xC0 byte) equals the poll loop's cycles per iteration (7 — `sbis` 1,
|
||||
`rjmp` 2, `adiw` 2, `rjmp` 2). The loop shape survived every restructuring here,
|
||||
verified in the disassembly, but a toolchain bump that reshapes it would break
|
||||
the lock silently. `test/pbautobaud.py` is what pins it: a wrong unit fails the
|
||||
flash verify.
|
||||
|
||||
## Sizes — every chip
|
||||
|
||||
Budget 510 B on the patch-vector parts, 512 elsewhere. The 1284P is no longer
|
||||
the tight one: the bank nibble made far flash *cheaper* than the near-flash
|
||||
arithmetic it replaced.
|
||||
|
||||
| size | chips | budget | spare |
|
||||
|---|---|---|---|
|
||||
| 444 | ATtiny13, 13A | 510 | 66 |
|
||||
| 448 | ATmega48, 48A, 48P, 48PA; ATtiny25 | 510 | 62 |
|
||||
| 452 | ATtiny45, 85 | 510 | 58 |
|
||||
| 460 | ATmega644, 644A, 644P, 644PA | 512 | 52 |
|
||||
| 464 | **ATmega1284, 1284P**; ATmega8, 8A, 88, 88A, 88P, 88PA | 512 | 48 |
|
||||
| 466 | ATmega16, 16A, 32, 32A; 164A/P/PA, 168/A/P/PA, 324A/P/PA, 328, 328P | 512 | 46 |
|
||||
|
||||
All 37 chips build and size-test green, plus the 12-preset reflect spot set
|
||||
(guidance rule 4 — the reflect matrix is never run in full), which matches its
|
||||
generated counterpart byte for byte on every chip in the set. Worst case across
|
||||
the whole set is **466 B, 46 under budget**.
|
||||
|
||||
## Host tool and simulation
|
||||
|
||||
- **`pureboot.py`** speaks both generations. `Info.version >= 5` selects the
|
||||
unified path; everything below it keeps the four-command protocol, so the
|
||||
fixed-baud loader is untouched. `--autobaud` sends the 0xC0 pulse and one
|
||||
knock, then derives full geometry from the signature. New: `--peek ADDR[:N]`
|
||||
and `--poke ADDR:HEX` reach the data space.
|
||||
- **`test/pbautobaud.py`** drives the loader over the GPIO⇄pty software-UART
|
||||
bridge through the calibration handshake, a flash + EEPROM + fuse round-trip
|
||||
cross-checked against the simulator's own memory, a RAM read/write round-trip,
|
||||
and a hand-over to the fixture application — then repeats at double the F_CPU
|
||||
with the same binary, which is the clock-agnostic property autobaud exists
|
||||
for. Run on the near-flash 328P and the word-addressed 1284P.
|
||||
- It also **pins the activation hang**: the test sends a lone calibration pulse
|
||||
with no knock behind it and requires the application to boot. Against the
|
||||
unfixed loader that assertion never returns.
|
||||
|
||||
## What remains
|
||||
|
||||
- **Real-hardware acceptance.** A cycle-exact simulator cannot produce what
|
||||
autobaud exists for: a real RC oscillator at ±10% with drift and jitter.
|
||||
simavr proves the arithmetic and the fit at exact clocks; only silicon proves
|
||||
the feature. Drive an internal-oscillator ATtiny at a fixed host baud and
|
||||
confirm lock plus a full flash and verify.
|
||||
- **A generic `spm::command()` in libavr.** The SPM space issues a runtime
|
||||
command through `spm::detail::page_command` where RAMPZ exists, and falls back
|
||||
to a dispatch over the known operations where it does not — the one
|
||||
preprocessor branch in the file. A two-line library addition would make it
|
||||
uniform and save a few bytes on the 36 non-RAMPZ chips, none of which are
|
||||
tight.
|
||||
- **Retire or revive the two dead variants.** They are kept only as the record
|
||||
of the measurement; nothing builds them.
|
||||
|
||||
## Files
|
||||
|
||||
- `pureboot_autobaud_uni.cpp` — the loader. pureboot 5.
|
||||
- `pureboot_autobaud_pure.cpp`, `pureboot_autobaud_reg.cpp` — superseded, not
|
||||
built; 530 and 534 B on the 1284P once the activation hang is fixed.
|
||||
- `pureboot.py` — `--autobaud`, the unified transfer path, `--peek`/`--poke`.
|
||||
- `test/pbautobaud.py` — the end-to-end sim test and the hang regression.
|
||||
- `local/scratch/autobaud/floor_1284.S` (libavr checkout) — the hand-asm floor
|
||||
probe at 506 B, off-tree and gitignored; a size reference only. The unified
|
||||
loader is 42 B under it, with features the probe never had.
|
||||
@@ -1,28 +1,14 @@
|
||||
// pureboot — a serial bootloader on libavr, pure by constraint: one C++
|
||||
// source with no inline assembly and no global register variables, built for
|
||||
// every chip libavr targets, 512 bytes on each. The device speaks primitives
|
||||
// — read/program flash, read/write EEPROM, fuse bytes, an info block, a jump
|
||||
// — and everything composite (verify, erase, reset-vector surgery, updating
|
||||
// the loader itself) lives in the host tool. Protocol reference: README.md
|
||||
// next to this file.
|
||||
// pureboot — a serial bootloader on libavr: one C++ source, no inline
|
||||
// assembly, no global register variables, 512 bytes on every chip libavr
|
||||
// targets. The device speaks primitives; every composite (verify, erase,
|
||||
// reset-vector surgery, self-update) lives in the host tool. Protocol,
|
||||
// deployment and configuration: README.md next to this file.
|
||||
//
|
||||
// The image is position-independent: control flow is PC-relative, the write
|
||||
// and read paths take wire addresses, the write guard refuses the 512-byte
|
||||
// slot the code is *running* in (taken from the runtime return address), the
|
||||
// info block is read relative to that same anchor, and the application jump
|
||||
// is an indirect call to an absolute entry. The identical binary therefore
|
||||
// runs from any 512-byte slot with every command intact: flashed one slot
|
||||
// below the resident loader it becomes the staging loader that rewrites the
|
||||
// resident — how pureboot updates itself, host-driven, with no other
|
||||
// firmware involved.
|
||||
//
|
||||
// Entry: reset lands in avr::startup::entry below (BOOTRST on the
|
||||
// boot-sectioned megas; the patched reset vector — or erased flash walking
|
||||
// up into the loader — on the tinies and the boot-section-less m48s). A
|
||||
// watchdog reset hands straight to the application. Otherwise the
|
||||
// host has one activation window per awaited knock byte ("pb"); an idle line
|
||||
// boots the application. A session then stays in the command loop until 'J'
|
||||
// jumps away or the chip resets.
|
||||
// The image is position-independent — PC-relative control flow, wire
|
||||
// addresses in, the write guard and the info block both anchored on the
|
||||
// runtime return address — so the identical binary runs from any slot. That
|
||||
// is what makes a copy one slot below able to rewrite the resident one, and
|
||||
// every change here has to keep it (test/check_pi.py).
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
@@ -33,17 +19,14 @@ namespace ee = avr::eeprom;
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled — interrupts stay off, every guard folds to nothing.
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Per-deployment personality, passed in by the build — pureboot_add_loader()
|
||||
// (the CMake function next to this file) resolves the defaults: the clock the
|
||||
// board actually runs, the wire baud, the serial backend and its pins. The
|
||||
// device signature needs no configuring — it comes from the chip database
|
||||
// (avr::hw::db.signature), the only universal source, since the tiny13A
|
||||
// cannot even read its signature row from code.
|
||||
// Deployment parameters come from the build (pureboot_add_loader()). The
|
||||
// signature is not one of them: the chip database is the only universal
|
||||
// source — a tiny13A cannot read its own signature row from code.
|
||||
#if !defined(PUREBOOT_CLOCK_HZ) || !defined(PUREBOOT_BAUD)
|
||||
#error \
|
||||
"PUREBOOT_CLOCK_HZ and PUREBOOT_BAUD select this build's clock and baud — create loader targets with pureboot_add_loader() (README.md)"
|
||||
@@ -59,70 +42,62 @@ consteval std::int16_t wdrf_field()
|
||||
return avr::hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
// Geometry: the resident loader owns the top slot of flash — 512 bytes,
|
||||
// except on the >64 KiB chips whose own smallest boot sector is 1 KiB (the
|
||||
// 1284s): there the slot is 1 KiB, matching the hardware boundary the
|
||||
// 512-byte figure comes from everywhere else. The word below the slot is
|
||||
// the trampoline (the application's relocated reset vector) on chips
|
||||
// without a hardware boot section — the tinies and the m48s, whose SPM
|
||||
// runs from anywhere (Atmel-8271 §26). A boot section also means the CPU
|
||||
// runs on while the RWW section programs; everywhere else it halts through
|
||||
// the operation.
|
||||
constexpr std::uint16_t slot_bytes = spm::flash_bytes > 65536 ? 1024 : 512;
|
||||
// The loader owns the top 512 bytes; a staging copy goes in the slot below.
|
||||
// Chips without a hardware boot section — the tinies and the m48s, whose SPM
|
||||
// runs from anywhere (Atmel-8271 §26) — keep the application's relocated
|
||||
// reset vector in the word under the slot.
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = avr::hw::curated::has_boot_section();
|
||||
|
||||
// Past 64 KiB a byte address no longer fits the wire's 16 bits, so on the
|
||||
// large chips every flash address on the wire — and all slot arithmetic —
|
||||
// is a word address instead ('J' always was one). A slot spans the same
|
||||
// wire-high-byte pair in either unit (512 B = 2 x 256 bytes, 1 KiB =
|
||||
// 2 x 256 words), so the slot index is the high byte with its low bit
|
||||
// dropped everywhere.
|
||||
// Past 64 KiB a byte address no longer fits the wire's 16 bits, so flash
|
||||
// addresses there are word addresses ('J' always was one). A slot is 256 of
|
||||
// those — one value of a wire address's high byte, where 512 bytes span two.
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
constexpr std::uint16_t wire_base =
|
||||
word_flash ? static_cast<std::uint16_t>(base / 2) : static_cast<std::uint16_t>(base);
|
||||
constexpr std::uint16_t wire_page_mask = word_flash ? (page / 2 - 1) : (page - 1);
|
||||
|
||||
// The activation window, in seconds, is a compile-time constant (the build
|
||||
// may override it): the whole EEPROM belongs to the application, and
|
||||
// re-timing the loader is a bootloader self-update with a re-timed binary.
|
||||
// A compile-time window, so the whole EEPROM belongs to the application;
|
||||
// re-timing a deployed loader is a self-update with a re-timed build.
|
||||
#if !defined(PUREBOOT_TIMEOUT)
|
||||
#define PUREBOOT_TIMEOUT 8
|
||||
#endif
|
||||
constexpr std::uint8_t timeout_seconds = PUREBOOT_TIMEOUT;
|
||||
|
||||
// The 12-byte info block the host reads with the 'b' command, flash-resident
|
||||
// through flash_table (there is no crt to copy a .data image, and its storage
|
||||
// carries the word alignment 'b' needs to halve the address on the large
|
||||
// chips). The page byte is the wire count convention: 0 means 256.
|
||||
// The loader's one identity number. The protocol carries none of its own —
|
||||
// a version implies it, and the host tool holds that map (README.md).
|
||||
constexpr std::uint8_t version = 4;
|
||||
|
||||
// The 'b' reply, byte for byte (layout: README.md). Flash-resident because
|
||||
// no crt copies a .data image — and flash_table's storage carries the word
|
||||
// alignment 'b' needs to halve the address on the large chips.
|
||||
// One wire byte per line: this is the reply's layout, not a list.
|
||||
// clang-format off
|
||||
inline constexpr avr::flash_table<std::array<std::uint8_t, 12>{
|
||||
'P',
|
||||
'B',
|
||||
1, // magic, protocol version
|
||||
version,
|
||||
avr::hw::db.signature[0],
|
||||
avr::hw::db.signature[1],
|
||||
avr::hw::db.signature[2],
|
||||
static_cast<std::uint8_t>(page),
|
||||
static_cast<std::uint8_t>(page), // 0 means 256
|
||||
wire_base & 0xff,
|
||||
wire_base >> 8, // app flash ends here; resident loader base (a word address on large chips)
|
||||
wire_base >> 8,
|
||||
avr::hw::db.mem.eeprom_size & 0xff,
|
||||
avr::hw::db.mem.eeprom_size >> 8,
|
||||
// bit 0: host must patch the reset vector (no hardware boot section);
|
||||
// bit 1: flash wire addresses are word addresses
|
||||
static_cast<std::uint8_t>((boot_section ? 0 : 1) | (word_flash ? 2 : 0)),
|
||||
static_cast<std::uint8_t>((boot_section ? 0 : 1) | (word_flash ? 2 : 0)), // patch-vector, word-addressed
|
||||
}>
|
||||
info_data;
|
||||
// clang-format on
|
||||
|
||||
// The serial link. PUREBOOT_USART forces a hardware USART instance,
|
||||
// PUREBOOT_SOFT_SERIAL the polled software UART (no vector — the table
|
||||
// belongs to the application) on PUREBOOT_RX/PUREBOOT_TX; with neither, the
|
||||
// chip's first USART where it has one and the software UART elsewhere. Both
|
||||
// are class templates on the clock so only the selected backend is ever
|
||||
// instantiated. pending() is the cheap line test the activation window
|
||||
// polls; rx() then picks the byte up; drain() holds until the last
|
||||
// transmitted frame is fully on the wire (the jump hand-over must not let
|
||||
// the target's re-init clip the ack).
|
||||
// The serial link, per the build's PUREBOOT_USART / PUREBOOT_SOFT_SERIAL,
|
||||
// defaulting to the chip's USART0 where it has one. The software receiver is
|
||||
// the polled one: the vector table belongs to the application. Templates on
|
||||
// the clock, so only the selected backend instantiates. pending() is the
|
||||
// cheap line test the activation window polls; drain() holds until the last
|
||||
// frame is off the wire, so a hand-over cannot let the target's re-init clip
|
||||
// the ack.
|
||||
#if defined(PUREBOOT_SOFT_SERIAL) && defined(PUREBOOT_USART)
|
||||
#error "PUREBOOT_SOFT_SERIAL and PUREBOOT_USART select opposing serial backends"
|
||||
#endif
|
||||
@@ -217,12 +192,10 @@ using link =
|
||||
std::conditional_t<avr::uart::has_usart<usart_digit>(), hardware_link<dev::clock>, software_link<dev::clock>>;
|
||||
#endif
|
||||
|
||||
// The application's entry, an absolute address the linker pins (--defsym in
|
||||
// CMakeLists.txt): 0x0000 on the mega (word 0 stays the application's own
|
||||
// vector — BOOTRST re-vectors a reset into the loader in hardware) and the
|
||||
// trampoline word at base - 2 on the tinies. Reaching it must not depend on
|
||||
// where this copy runs, so the jump goes through a pointer: [[gnu::noipa]]
|
||||
// keeps the constant from folding back into a PC-relative call.
|
||||
// The application's entry, pinned by the linker (--defsym): word 0 on a
|
||||
// boot-sectioned mega, the trampoline at base − 2 elsewhere. Reaching it must
|
||||
// not depend on where this copy runs, so the jump goes through a pointer, and
|
||||
// [[gnu::noipa]] keeps the constant from folding back into a relative call.
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
@@ -236,10 +209,8 @@ extern "C" [[noreturn]] void pureboot_app();
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
// One activation window is a single 32-bit poll countdown. The divisor is
|
||||
// the backend's counted poll-loop cycles (its own comment reads them off the
|
||||
// compiled loop); whole-second precision is all the window promises, so the
|
||||
// nearest cycle count is plenty.
|
||||
// The window as one 32-bit countdown, divided by the backend's counted
|
||||
// poll-loop cycles. Whole seconds is all it promises.
|
||||
consteval std::uint32_t window_polls()
|
||||
{
|
||||
return timeout_seconds * static_cast<std::uint32_t>(dev::clock.hz / link::poll_cycles);
|
||||
@@ -255,8 +226,8 @@ bool pending_before_deadline()
|
||||
return false;
|
||||
}
|
||||
|
||||
// A knock byte under the activation deadline: an idle line means no host is
|
||||
// there, and the application runs.
|
||||
// A knock byte under the deadline: an idle window means no host, so the
|
||||
// application runs.
|
||||
std::uint8_t rx_deadline()
|
||||
{
|
||||
if (!pending_before_deadline())
|
||||
@@ -264,22 +235,26 @@ std::uint8_t rx_deadline()
|
||||
return link::rx();
|
||||
}
|
||||
|
||||
// Inlined into its call sites: reading two bytes across a call otherwise
|
||||
// strands the first in a call-saved register the caller must push/pop; folded
|
||||
// into the (noreturn) command loop that cost disappears.
|
||||
// Inlined: read across a call, the first byte strands in a call-saved
|
||||
// register the caller has to push and pop.
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
// The streamers take the count in the wire's 8-bit form: 0 means 256.
|
||||
//
|
||||
// Two functions, because they want opposite placement and placement is an
|
||||
// attribute: the byte-addressed loop is small enough to inline into both
|
||||
// callers, the word-addressed one stays out of line but flattened — a call to
|
||||
// the transmit inside it would strand the 24-bit cursor in callee-saved
|
||||
// registers. `word_flash` picks at the call site.
|
||||
// The wire's byte pair as the word it is — AVR is little-endian too, so the
|
||||
// cast is the identity a shift-and-or spelling makes the compiler rediscover.
|
||||
// Callers read into named variables first: the wire order is a sequence of
|
||||
// reads, not an argument order.
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
// Counts arrive in the wire's 8-bit form: 0 means 256. Both streamers fold
|
||||
// into the one command that reads flash, which is what lets the far one's
|
||||
// 24-bit cursor sit in the command loop's own call-saved registers.
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_near(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
@@ -287,17 +262,16 @@ std::uint8_t rx_deadline()
|
||||
while (--count);
|
||||
}
|
||||
|
||||
// The 24-bit cursor as the machine holds it: the RAMPZ byte and a 16-bit Z,
|
||||
// carried explicitly (the reassembled 32-bit address folds away inside the
|
||||
// inlined far load).
|
||||
[[maybe_unused, gnu::flatten, gnu::noinline]] void send_flash_far(std::uint16_t address, std::uint8_t count)
|
||||
// The 24-bit cursor as the machine holds it — the RAMPZ byte and a 16-bit Z,
|
||||
// carried apart; the reassembled address folds away inside the far load.
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_far(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
std::uint8_t rampz = static_cast<std::uint8_t>(address >> 15);
|
||||
std::uint16_t z = static_cast<std::uint16_t>(address << 1);
|
||||
do {
|
||||
link::tx(avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(rampz) << 16) | z));
|
||||
// The protocol never reads across 64 KiB, but carrying the wrap is
|
||||
// smaller than the flat 32-bit cursor GCC builds without it.
|
||||
// Carrying the wrap is smaller than the flat 32-bit cursor GCC
|
||||
// builds without it.
|
||||
if (++z == 0)
|
||||
++rampz;
|
||||
} while (--count);
|
||||
@@ -311,6 +285,13 @@ std::uint8_t rx_deadline()
|
||||
send_flash_near(address, count);
|
||||
}
|
||||
|
||||
// Out of line: three sites send it, and a call is shorter than three
|
||||
// load-immediates.
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
void send_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
@@ -318,74 +299,63 @@ void send_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
while (--count);
|
||||
}
|
||||
|
||||
// EEPROM write, host-paced: each ack goes out once the byte's write has
|
||||
// begun, so the next byte arrives while it completes and the following
|
||||
// write's own ready-wait sees an idle line. Nothing is ever missed, on
|
||||
// either serial backend, without a buffer.
|
||||
// Host-paced: the ack goes out once the write has begun, so the next byte
|
||||
// arrives while it completes and nothing is missed without a buffer.
|
||||
void store_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do {
|
||||
ee::write<off>(address++, link::rx());
|
||||
link::tx(ack);
|
||||
tx_ack();
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
// One flash page: stream the bytes into the SPM buffer as little-endian
|
||||
// words, then erase and program — except the 512-byte slot this code runs
|
||||
// in, which is drained but never programmed, so a copy can never erase
|
||||
// itself. `slot_high` is the high byte of that running slot's base (run()
|
||||
// derives it); a broken host thus cannot brick the running loader, and a
|
||||
// copy flashed one slot lower may rewrite the slot above it — how pureboot
|
||||
// updates itself.
|
||||
// One page into the SPM buffer, then erase and program — except the slot
|
||||
// this code is running in (`slot_high`, from run()), which is drained and
|
||||
// left alone. A broken host therefore cannot brick the running loader, and a
|
||||
// copy one slot lower may rewrite the resident one.
|
||||
//
|
||||
// Nothing discards the buffer first: it is write-once per word (§26.2.1), so
|
||||
// filling over a refused page or an application's leavings programs stale
|
||||
// words — but a page write auto-erases it (§26.2.1; §19.2 on the tinies), so
|
||||
// that write clears the condition and the host's read-back rewrites the page.
|
||||
void program_flash(std::uint16_t wire_address, std::uint8_t slot_high)
|
||||
{
|
||||
// No discard before the fill: the buffer is write-once per word
|
||||
// (§26.2.1), so filling over one a refused page or an application left
|
||||
// dirty programs stale words — but a page write auto-erases the buffer
|
||||
// (§26.2.1; §19.2 on the tinies), so that write clears the condition and
|
||||
// the host's read-back rewrites the page.
|
||||
|
||||
// One induction either way. On the byte-addressed chips the wire address
|
||||
// itself walks the page (aligned, so the offset bits wrap to zero); on
|
||||
// the word-addressed large chips the wire word address becomes a 32-bit
|
||||
// byte cursor once, and their 256-byte page makes its low byte the whole
|
||||
// in-page offset. The slot index is one high byte of the wire address —
|
||||
// two values on byte-addressed chips (the & ~1), bits 16:9 re-packed on
|
||||
// the large ones.
|
||||
// The address names a page, so its in-page bits are dropped and the walk
|
||||
// starts at the page base — one induction either way: a byte-addressed
|
||||
// wire address walks the page itself (the offset bits wrap back to zero),
|
||||
// while a word one becomes a byte cursor once. The slot index is the wire
|
||||
// address's high byte — on byte-addressed chips the byte address's, with
|
||||
// the low bit dropped, since a slot is two of those.
|
||||
spm::flash_address_t address;
|
||||
std::uint8_t page_high;
|
||||
if constexpr (word_flash) {
|
||||
// Pages are aligned, so one page never crosses a 64 KiB boundary:
|
||||
// RAMPZ is a per-page constant and the fill cursor is a 16-bit Z
|
||||
// whose low byte is the whole in-page offset (256-byte pages). The
|
||||
// slot index is simply the wire word address's high byte.
|
||||
// A page is aligned, so it never crosses 64 KiB: RAMPZ is a per-page
|
||||
// constant and the 16-bit Z's low byte is the whole in-page offset.
|
||||
const std::uint8_t rampz = static_cast<std::uint8_t>(wire_address >> 15);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
std::uint16_t z = z0;
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(rampz) << 16) | z,
|
||||
static_cast<std::uint16_t>(low | (high << 8)));
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(rampz) << 16) | z, word_of({low, high}));
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z));
|
||||
address = (static_cast<spm::flash_address_t>(rampz) << 16) | z0;
|
||||
page_high = static_cast<std::uint8_t>(wire_address >> 8) & 0xfe;
|
||||
page_high = static_cast<std::uint8_t>(wire_address >> 8);
|
||||
} else {
|
||||
address = static_cast<spm::flash_address_t>(wire_address);
|
||||
address = static_cast<spm::flash_address_t>(wire_address & ~static_cast<std::uint16_t>(page - 1));
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>(address, static_cast<std::uint16_t>(low | (high << 8)));
|
||||
spm::fill<off>(address, word_of({low, high}));
|
||||
address += 2;
|
||||
} while (static_cast<std::uint8_t>(address) & (page - 1));
|
||||
address -= 2; // back inside the page — erase and write ignore the word bits
|
||||
page_high = static_cast<std::uint8_t>(address >> 8) & 0xfe;
|
||||
}
|
||||
if (page_high != slot_high) {
|
||||
// The tinies and the m48s halt the CPU through the erase and the
|
||||
// write, so only the boot-sectioned megas — running on while their
|
||||
// RWW section programs — wait.
|
||||
// Only a boot-sectioned mega runs on while its RWW section programs;
|
||||
// everywhere else the CPU halts through erase and write.
|
||||
spm::erase_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
@@ -393,89 +363,92 @@ void program_flash(std::uint16_t wire_address, std::uint8_t slot_high)
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
}
|
||||
// The megas program with their RWW section disabled; reads need it back
|
||||
// on. The same store discards the buffer (§26.2.2), so they never meet
|
||||
// the stale-word case above. boot_section implies an RWW section.
|
||||
// Programming leaves the RWW section disabled; reads need it back on. The
|
||||
// same store discards the buffer (§26.2.2), so a boot-sectioned mega never
|
||||
// meets the stale-word case above.
|
||||
if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
}
|
||||
|
||||
// The four fuse/lock bytes in the hardware's own Z order: low, lock,
|
||||
// extended, high. Writing fuses is not a thing self-programming can do on
|
||||
// AVR — SPM reaches flash (and boot lock bits) only.
|
||||
// The four fuse and lock bytes in the hardware's own Z order: low, lock,
|
||||
// extended, high.
|
||||
void send_fuses()
|
||||
{
|
||||
std::uint8_t which = 0;
|
||||
do
|
||||
link::tx(spm::read_fuse<off>(static_cast<spm::fuse>(which)));
|
||||
while (++which & 3);
|
||||
while (++which != 4);
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
// A watchdog reset belongs to the application (whose watchdog stays
|
||||
// forced on until it clears WDRF) — no activation window in its way.
|
||||
// The flag register is MCUSR, or the classic megas' MCUCSR.
|
||||
// A watchdog reset belongs to the application, whose watchdog stays forced
|
||||
// on until it clears WDRF — no activation window in its way.
|
||||
if (avr::hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// The high byte of the 512-byte-aligned base this copy runs at: the
|
||||
// return address is a word address, whose high byte is the 256-word slot
|
||||
// index — on byte-addressed chips doubled back into byte terms.
|
||||
// program_flash refuses this one slot and the info block is addressed
|
||||
// from it, so both follow wherever the code was flashed. The high byte is
|
||||
// spelled as byteswap's low byte: the builtin's value is itself built by
|
||||
// swapping the two stacked bytes, and the double swap folds to the single
|
||||
// byte pick a hand assembler writes — `>> 8` leaves the swap materialized.
|
||||
// The high byte of the slot this copy runs at, which the write guard and
|
||||
// the info block both follow: the return address is a word address, so its
|
||||
// high byte is the 256-word slot index, doubled back into byte terms where
|
||||
// the wire counts bytes. Taken as byteswap's low byte — the builtin already
|
||||
// swaps the two stacked bytes, and the double swap folds away, where `>> 8`
|
||||
// would leave the swap materialized.
|
||||
const std::uint16_t ra_words = reinterpret_cast<std::uint16_t>(__builtin_return_address(0));
|
||||
const std::uint8_t ra_high = static_cast<std::uint8_t>(std::byteswap(ra_words));
|
||||
const std::uint8_t slot_high = word_flash ? ra_high & 0xfe : static_cast<std::uint8_t>(ra_high << 1);
|
||||
const std::uint8_t slot_high = word_flash ? ra_high : static_cast<std::uint8_t>(ra_high << 1);
|
||||
|
||||
// The knock: 'p' then 'b', each under a fresh window; any other byte is
|
||||
// line noise and waits again. Falling out of a window runs the app.
|
||||
// 'p' then 'b', each under a fresh window; anything else is line noise.
|
||||
while (rx_deadline() != 'p' || rx_deadline() != 'b') {
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
// No prompt while an EEPROM write runs: a pending write blocks SPM
|
||||
// and fuse reads (§26.2.1), and the ack tells the host all is done.
|
||||
// No prompt while an EEPROM write runs: it blocks SPM and fuse reads
|
||||
// (§26.2.1), and the prompt is the previous command's completion ack.
|
||||
ee::wait();
|
||||
link::tx(ack);
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'b': { // info block, read relative to the running slot
|
||||
// The block sits in the image's first 256 bytes (the build lint
|
||||
// asserts it), and slots are 512-aligned — so the low byte of its
|
||||
// link address (in wire units: bytes, or words on the large
|
||||
// chips) is its offset in any slot, and the high byte of its
|
||||
// runtime address is the running slot's. Composed from the two
|
||||
// bytes — the high half is runtime data, so no absolute address
|
||||
// is ever materialized.
|
||||
const auto link_low = reinterpret_cast<std::uint16_t>(info_data.storage.data());
|
||||
const std::uint8_t low =
|
||||
word_flash ? static_cast<std::uint8_t>(link_low >> 1) : static_cast<std::uint8_t>(link_low);
|
||||
send_flash(static_cast<std::uint16_t>(low | (slot_high << 8)), static_cast<std::uint8_t>(info_data.size()));
|
||||
break;
|
||||
}
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
link::tx(ack);
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // info block, read relative to the running slot
|
||||
case 'R': // read flash: addr16, n8 (0 = 256)
|
||||
case 'r': // read EEPROM: addr16, n8
|
||||
case 'w': { // write EEPROM: addr16, n8, then n bytes each acked
|
||||
std::uint16_t address = rx16();
|
||||
std::uint8_t count = link::rx();
|
||||
if (command == 'R')
|
||||
send_flash(address, count);
|
||||
else if (command == 'r')
|
||||
// One address-and-count path for all four: 'b' is a flash read
|
||||
// whose arguments the loader already knows, so it joins the
|
||||
// wire-argument three rather than streaming from a call site of its
|
||||
// own. That leaves one flash streamer in the image, and lets its
|
||||
// cursor live in this never-returning loop's own call-saved
|
||||
// registers instead of being saved and restored around a call.
|
||||
std::uint16_t address;
|
||||
std::uint8_t count;
|
||||
if (command == 'b') {
|
||||
// The block sits in the image's first 256 bytes (check_pi.py
|
||||
// asserts it) and slots are 512-aligned, so the low byte of its
|
||||
// link address is its offset in any slot — halved where wire
|
||||
// units are words. The high byte is runtime data, so no
|
||||
// absolute address is ever materialized.
|
||||
const auto link_byte =
|
||||
static_cast<std::uint8_t>(reinterpret_cast<std::uint16_t>(info_data.storage.data()));
|
||||
const std::uint8_t low = word_flash ? static_cast<std::uint8_t>(link_byte >> 1) : link_byte;
|
||||
address = static_cast<std::uint16_t>(low | (slot_high << 8));
|
||||
count = static_cast<std::uint8_t>(info_data.size());
|
||||
} else {
|
||||
address = rx16();
|
||||
count = link::rx();
|
||||
}
|
||||
if (command == 'r')
|
||||
send_eeprom(address, count);
|
||||
else
|
||||
else if (command == 'w')
|
||||
store_eeprom(address, count);
|
||||
else
|
||||
send_flash(address, count);
|
||||
break;
|
||||
}
|
||||
case 'W': // program one flash page: addr16, page bytes
|
||||
|
||||
@@ -1,24 +1,13 @@
|
||||
#!/usr/bin/env python3
|
||||
"""pureboot host tool — the smart half of the pureboot protocol (README.md).
|
||||
"""pureboot host tool — the smart half of the protocol (README.md).
|
||||
|
||||
The device exposes primitives; this tool composes them: image loading (raw
|
||||
binary or Intel HEX), flash programming with read-back verification, erase as
|
||||
writing 0xff, EEPROM programming, fuse and info readout, the hand-over jump,
|
||||
and — on chips without a hardware boot section — the reset-vector surgery
|
||||
that re-homes the application's entry through the trampoline word below the
|
||||
loader. Page 0 and the trampoline are written first, so every interruption
|
||||
point of a flash leaves the chip reset-recoverable into the loader.
|
||||
The device exposes primitives; everything composite is here: HEX/raw images,
|
||||
programming with repairing read-back verification, the reset-vector surgery
|
||||
the boot-section-less chips need, and the self-update that stages the loader
|
||||
one slot lower and lets it rewrite the resident.
|
||||
|
||||
It also updates the loader itself (--update-loader): pureboot's image is
|
||||
position-independent, so the tool installs the identical binary one 512-byte
|
||||
slot below the resident loader, jumps into that staging copy, lets it rewrite
|
||||
the resident slot, and restores what the staging slot held — resumable at
|
||||
every phase from the flash state plus a host-side state file carrying the
|
||||
saved bytes.
|
||||
|
||||
Python standard library only; the serial port is driven with termios on POSIX
|
||||
and the Win32 serial API (through ctypes) on Windows, so any tty or COM port
|
||||
works — a USB adapter as well as a simavr pty.
|
||||
Standard library only. The port is termios on POSIX and the Win32 serial API
|
||||
through ctypes on Windows, so any tty or COM port works.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
@@ -35,16 +24,77 @@ else:
|
||||
import termios
|
||||
|
||||
PROMPT = b"+"
|
||||
PROTOCOL_VERSION = 1
|
||||
SLOT = 512 # the loader slot on byte-addressed chips; word-addressed ones (>64 KiB) use 1 KiB — their own smallest boot sector
|
||||
VERSION = 3 # this tool's own version — free to drift from a loader's
|
||||
# The loader versions this tool speaks. A pureboot version implies its wire
|
||||
# protocol, which carries no number of its own, so this window is where that
|
||||
# map lives: every version so far speaks the same protocol, and one that
|
||||
# changes it becomes the new floor here.
|
||||
OLDEST_LOADER = 1
|
||||
NEWEST_LOADER = 5
|
||||
SLOT = 512 # the loader slot, on every chip
|
||||
RETRIES = 3 # rewrites of a page that reads back wrong, before the run stops
|
||||
|
||||
# pureboot 5 replaced the four per-memory commands with one unified pair: 'G'
|
||||
# reads and 'g' writes, both taking a selector byte, a 16-bit address and a
|
||||
# count, over the spaces below. The loader carries one transfer loop instead of
|
||||
# four bodies, which is what paid for RAM access and for the calibration fix.
|
||||
UNIFIED_LOADER = 5
|
||||
SP_FLASH, SP_EEPROM, SP_RAM, SP_FUSE, SP_SPM = 0, 1, 2, 3, 4
|
||||
|
||||
# A selector's high nibble is flash's third address byte, so a transfer names a
|
||||
# byte address within a 64 KiB bank and never has to speak word addresses. No
|
||||
# single transfer may cross a bank boundary — the host chunks to keep that true.
|
||||
def selector(space, address):
|
||||
return space | ((address >> 16) << 4)
|
||||
|
||||
|
||||
# The SPM operations pureboot 5 leaves to the host: a write to SP_SPM hands its
|
||||
# data byte to SPMCSR and fires the instruction at the selected flash address.
|
||||
# Every part pureboot targets agrees on these encodings.
|
||||
SPM_ERASE, SPM_WRITE, SPM_RWWSRE = 0x03, 0x05, 0x11
|
||||
|
||||
# Calibration byte for the autobaud loader: 0xC0 is a start bit plus six zero
|
||||
# data bits — one low pulse of seven bit-times, which the loader times into its
|
||||
# per-bit unit. Sent at whatever baud the host chose; the loader locks to it.
|
||||
CALIBRATE = 0xC0
|
||||
|
||||
# The autobaud loader slims its info block to the version and signature; the host
|
||||
# derives the rest of the geometry from the signature. flash, page, eeprom,
|
||||
# patch-vector per distinct signature, over every chip pureboot targets (the
|
||||
# loader computes the same from its chip database at build time). Die revisions
|
||||
# that share a signature share this row, as they share the silicon.
|
||||
AUTOBAUD_GEOMETRY = {
|
||||
# signature : (flash, page, eeprom, patch_vector)
|
||||
(0x1E, 0x90, 0x07): (1024, 32, 64, True), # ATtiny13/13A
|
||||
(0x1E, 0x91, 0x08): (2048, 32, 128, True), # ATtiny25
|
||||
(0x1E, 0x92, 0x06): (4096, 64, 256, True), # ATtiny45
|
||||
(0x1E, 0x93, 0x0B): (8192, 64, 512, True), # ATtiny85
|
||||
(0x1E, 0x92, 0x05): (4096, 64, 256, True), # ATmega48/48A
|
||||
(0x1E, 0x92, 0x0A): (4096, 64, 256, True), # ATmega48P/48PA
|
||||
(0x1E, 0x93, 0x07): (8192, 64, 512, False), # ATmega8/8A
|
||||
(0x1E, 0x93, 0x0A): (8192, 64, 512, False), # ATmega88/88A
|
||||
(0x1E, 0x93, 0x0F): (8192, 64, 512, False), # ATmega88P/88PA
|
||||
(0x1E, 0x94, 0x03): (16384, 128, 512, False), # ATmega16/16A
|
||||
(0x1E, 0x94, 0x06): (16384, 128, 512, False), # ATmega168/168A
|
||||
(0x1E, 0x94, 0x0B): (16384, 128, 512, False), # ATmega168P/168PA
|
||||
(0x1E, 0x94, 0x0A): (16384, 128, 512, False), # ATmega164P/164PA
|
||||
(0x1E, 0x94, 0x0F): (16384, 128, 512, False), # ATmega164A
|
||||
(0x1E, 0x95, 0x02): (32768, 128, 1024, False), # ATmega32/32A
|
||||
(0x1E, 0x95, 0x0F): (32768, 128, 1024, False), # ATmega328P
|
||||
(0x1E, 0x95, 0x14): (32768, 128, 1024, False), # ATmega328
|
||||
(0x1E, 0x95, 0x08): (32768, 128, 1024, False), # ATmega324P
|
||||
(0x1E, 0x95, 0x11): (32768, 128, 1024, False), # ATmega324PA
|
||||
(0x1E, 0x95, 0x15): (32768, 128, 1024, False), # ATmega324A
|
||||
(0x1E, 0x96, 0x09): (65536, 256, 2048, False), # ATmega644/644A
|
||||
(0x1E, 0x96, 0x0A): (65536, 256, 2048, False), # ATmega644P/644PA
|
||||
(0x1E, 0x97, 0x05): (131072, 256, 4096, False),# ATmega1284P
|
||||
(0x1E, 0x97, 0x06): (131072, 256, 4096, False),# ATmega1284
|
||||
}
|
||||
|
||||
VERBOSE = False
|
||||
|
||||
|
||||
def verbose(message):
|
||||
"""Detail printed only under --verbose: decisions and derived facts, not
|
||||
per-byte chatter — the progress bar carries the bulk transfers."""
|
||||
if VERBOSE:
|
||||
print(f" {message}")
|
||||
|
||||
@@ -54,11 +104,9 @@ class Error(Exception):
|
||||
|
||||
|
||||
class Progress:
|
||||
"""A transient in-place bar on stderr for the operations that take wire
|
||||
time. Drawn only when stderr is a tty — logs, pipes and the test harness
|
||||
see nothing — and erased once done; the summary line each operation
|
||||
prints afterwards is the persistent record. A zero total (or no label)
|
||||
disables it, so callers can pass one through unconditionally."""
|
||||
"""A transient bar on stderr, drawn only for a tty and erased when done —
|
||||
logs and pipes see only the summary line each operation prints. No label
|
||||
or a zero total disables it, so callers can pass one unconditionally."""
|
||||
|
||||
def __init__(self, label, total, unit="pages"):
|
||||
self.label, self.total, self.unit = label, total, unit
|
||||
@@ -310,27 +358,51 @@ Port = WindowsPort if os.name == "nt" else PosixPort
|
||||
class Info:
|
||||
"""The 12-byte info block."""
|
||||
|
||||
@classmethod
|
||||
def from_slim(cls, raw):
|
||||
"""The autobaud loader's slimmed reply — version and signature only —
|
||||
with the rest of the geometry looked up from the signature (the loader
|
||||
derived it from the same chip facts at build time). Reconstructs the full
|
||||
block so every derived attribute matches the fixed-baud path exactly."""
|
||||
if len(raw) != 4:
|
||||
raise Error(f"bad slim info block: {raw.hex()}")
|
||||
version, signature = raw[0], tuple(raw[1:4])
|
||||
geometry = AUTOBAUD_GEOMETRY.get(signature)
|
||||
if geometry is None:
|
||||
sig = " ".join(f"{b:02x}" for b in signature)
|
||||
raise Error(f"unknown signature {sig} — this tool has no autobaud geometry for it")
|
||||
flash, page, eeprom, patch = geometry
|
||||
base = flash - SLOT
|
||||
word_flash = flash > 0x10000
|
||||
wire_base = base // 2 if word_flash else base
|
||||
flags = (1 if patch else 0) | (2 if word_flash else 0)
|
||||
raw12 = bytes((ord("P"), ord("B"), version, *signature, page & 0xFF,
|
||||
wire_base & 0xFF, wire_base >> 8, eeprom & 0xFF, eeprom >> 8, flags))
|
||||
return cls(raw12)
|
||||
|
||||
def __init__(self, raw):
|
||||
if len(raw) != 12 or raw[0:2] != b"PB":
|
||||
raise Error(f"bad info block: {raw.hex()}")
|
||||
if raw[2] != PROTOCOL_VERSION:
|
||||
raise Error(f"protocol version {raw[2]}, tool speaks {PROTOCOL_VERSION}")
|
||||
self.version = raw[2]
|
||||
if not OLDEST_LOADER <= self.version <= NEWEST_LOADER:
|
||||
raise Error(
|
||||
f"pureboot {self.version}: this tool (version {VERSION}) speaks pureboot "
|
||||
f"{OLDEST_LOADER}..{NEWEST_LOADER} — a newer loader needs a newer tool"
|
||||
)
|
||||
self.raw = bytes(raw)
|
||||
self.signature = raw[3:6]
|
||||
self.page = raw[6] or 256 # the wire count convention: 0 means 256
|
||||
self.patch_vector = bool(raw[11] & 1)
|
||||
# Large chips speak word addresses for flash (bit 1); the host keeps
|
||||
# every address in bytes and converts at the wire.
|
||||
# Bit 1: flash addresses are words on the wire. Every address here
|
||||
# stays a byte address and converts at the wire.
|
||||
self.word_flash = bool(raw[11] & 2)
|
||||
scale = 2 if self.word_flash else 1
|
||||
self.base = (raw[7] | (raw[8] << 8)) * scale
|
||||
self.eeprom_size = raw[9] | (raw[10] << 8)
|
||||
self.slot = 1024 if self.word_flash else SLOT
|
||||
self.flash_size = self.base + self.slot
|
||||
self.stage = self.base - self.slot # where a staging copy of the loader goes
|
||||
# The hand-over target, as the word address 'J' takes: the trampoline
|
||||
# below the loader (tinies), or word 0 (mega — the application's own
|
||||
# reset vector; BOOTRST re-vectors a reset into the loader instead).
|
||||
self.flash_size = self.base + SLOT
|
||||
self.stage = self.base - SLOT # where a staging copy of the loader goes
|
||||
# The hand-over target as 'J' takes it: the trampoline below the
|
||||
# loader, or word 0 where BOOTRST re-vectors reset in hardware.
|
||||
self.app_entry_word = (self.base - 2) // 2 if self.patch_vector else 0
|
||||
|
||||
def describe(self):
|
||||
@@ -343,17 +415,18 @@ class Info:
|
||||
)
|
||||
|
||||
def lines(self):
|
||||
"""The info block as one fact per line — what --info prints."""
|
||||
"""One fact per line — what --info prints."""
|
||||
if self.patch_vector:
|
||||
hand_over = f"host-patched reset vector, trampoline at {self.base - 2:#06x}"
|
||||
else:
|
||||
hand_over = "hardware boot section, jump to word 0"
|
||||
return (
|
||||
f"version pureboot {self.version}",
|
||||
f"signature {' '.join(f'{b:02x}' for b in self.signature)}",
|
||||
f"flash {self.flash_size} B, {self.page} B pages"
|
||||
+ (", word-addressed wire" if self.word_flash else ""),
|
||||
f"application 0x0000..{self.base - 1:#06x} ({self.base} B)",
|
||||
f"loader {self.base:#06x} ({self.slot} B slot)",
|
||||
f"loader {self.base:#06x} ({SLOT} B slot)",
|
||||
f"staging {self.stage:#06x}",
|
||||
f"EEPROM {self.eeprom_size} B",
|
||||
f"hand-over {hand_over}",
|
||||
@@ -361,36 +434,77 @@ class Info:
|
||||
|
||||
|
||||
class Loader:
|
||||
"""A pureboot session. Between commands the loader has prompted `+` and
|
||||
awaits a command byte; every method restores that invariant — except
|
||||
jump(), after which the target must be knocked afresh."""
|
||||
"""A session. Between commands the loader has prompted and awaits a
|
||||
command byte; every method restores that, except jump() — after which the
|
||||
target must be knocked afresh."""
|
||||
|
||||
def __init__(self, port):
|
||||
self.port = port
|
||||
self.info = None
|
||||
|
||||
def connect(self, wait):
|
||||
"""Knock until the activation window answers, then read the info
|
||||
block. Also converges when the loader already sits in its command
|
||||
loop: the knock bytes are ignored-or-executed there, and the drain
|
||||
absorbs whatever they produced."""
|
||||
self.port.flush_input()
|
||||
"""Knock until the info block comes back. The block is what proves the
|
||||
loader is listening — a prompt byte alone does not, since one left over
|
||||
from a previous session can still be in the pipeline while the port
|
||||
opening resets the device into a fresh activation window, where a
|
||||
command without its knock is discarded. Each attempt is therefore the
|
||||
whole handshake, retried until it produces the block or the window
|
||||
closes. Also converges into a live session: the knock bytes are ignored
|
||||
there and the drain absorbs whatever they produced."""
|
||||
deadline = time.monotonic() + wait
|
||||
knocks = 0
|
||||
while True:
|
||||
self.port.flush_input()
|
||||
self.port.write(b"pb")
|
||||
knocks += 1
|
||||
if PROMPT in self.port.read_available(0.4):
|
||||
break
|
||||
while self.port.read_available(0.3):
|
||||
pass
|
||||
self.port.write(b"b")
|
||||
try:
|
||||
block = self.port.read_exact(12, 2.0)
|
||||
except Error:
|
||||
block = b""
|
||||
# A version the tool cannot speak is the loader's own answer,
|
||||
# not a failed knock: Info reports it rather than retrying.
|
||||
if block[0:2] == b"PB":
|
||||
self.info = Info(block)
|
||||
self._expect_prompt()
|
||||
verbose(f"loader answered knock {knocks}; info block read")
|
||||
return self.info
|
||||
if time.monotonic() > deadline:
|
||||
raise Error("no answer — reset the device within its activation window")
|
||||
|
||||
def connect_autobaud(self, wait):
|
||||
"""The autobaud handshake. Instead of the p+b knock, the host sends the
|
||||
0xC0 calibration pulse — a single seven-bit-time low pulse at the host's
|
||||
chosen baud — which the loader times into its per-bit unit, then a single
|
||||
'p' knock the loader decodes at the rate it just measured. As with
|
||||
connect(), each attempt is the whole handshake, retried until the slim
|
||||
info block comes back or the window closes: a lost pulse or a knock that
|
||||
lands while the loader is mid-frame simply fails to answer, and the
|
||||
loader's measurement loop is back waiting for the next pulse."""
|
||||
deadline = time.monotonic() + wait
|
||||
knocks = 0
|
||||
while True:
|
||||
self.port.flush_input()
|
||||
self.port.write(bytes((CALIBRATE, ord("p"))))
|
||||
knocks += 1
|
||||
if PROMPT in self.port.read_available(0.4):
|
||||
while self.port.read_available(0.3):
|
||||
pass
|
||||
self.port.write(b"b")
|
||||
try:
|
||||
block = self.port.read_exact(4, 2.0)
|
||||
except Error:
|
||||
block = b""
|
||||
if len(block) == 4:
|
||||
self.info = Info.from_slim(block)
|
||||
self._expect_prompt()
|
||||
verbose(f"loader locked on knock {knocks}; slim info read")
|
||||
return self.info
|
||||
if time.monotonic() > deadline:
|
||||
raise Error("no answer — reset the device within its activation window")
|
||||
while self.port.read_available(0.3):
|
||||
pass
|
||||
self.port.write(b"b")
|
||||
self.info = Info(self.port.read_exact(12, 2.0))
|
||||
self._expect_prompt()
|
||||
verbose(f"loader answered knock {knocks}; info block read")
|
||||
return self.info
|
||||
|
||||
def _expect_prompt(self, timeout=2.0):
|
||||
byte = self.port.read_exact(1, timeout)
|
||||
@@ -414,7 +528,58 @@ class Loader:
|
||||
count -= chunk
|
||||
return data
|
||||
|
||||
@property
|
||||
def unified(self):
|
||||
"""pureboot 5 and later: one 'G'/'g' pair over selector-named spaces."""
|
||||
return self.info is not None and self.info.version >= UNIFIED_LOADER
|
||||
|
||||
def _read_space(self, space, address, count):
|
||||
"""A run out of any space, chunked to 256 bytes and to bank bounds."""
|
||||
data = b""
|
||||
while count:
|
||||
chunk = min(count, 256, 0x10000 - (address & 0xFFFF))
|
||||
head = bytes((ord("G"), selector(space, address), address & 0xFF,
|
||||
(address >> 8) & 0xFF, chunk & 0xFF))
|
||||
data += self._command(head, chunk, 5.0)
|
||||
address += chunk
|
||||
count -= chunk
|
||||
return data
|
||||
|
||||
def _write_space(self, space, address, data, progress=None):
|
||||
"""A run into any space. Each byte is acked as its write begins — an
|
||||
EEPROM cell and an SPM operation both need that pacing, and the ack is
|
||||
what the loader sends in place of a completion status."""
|
||||
offset = 0
|
||||
while offset < len(data):
|
||||
chunk = data[offset : offset + min(256, 0x10000 - (address & 0xFFFF))]
|
||||
head = bytes((ord("g"), selector(space, address), address & 0xFF,
|
||||
(address >> 8) & 0xFF, len(chunk) & 0xFF))
|
||||
self.port.write(head)
|
||||
for byte in chunk:
|
||||
self.port.write(bytes((byte,)))
|
||||
self._expect_prompt()
|
||||
if progress:
|
||||
progress.step()
|
||||
self._expect_prompt() # the next command prompt
|
||||
address += len(chunk)
|
||||
offset += len(chunk)
|
||||
|
||||
def spm(self, operation, address):
|
||||
"""One SPM operation at a flash address — the erase, write and RWW
|
||||
re-enable that pureboot 4 ran inside 'W' and pureboot 5 leaves here."""
|
||||
self._write_space(SP_SPM, address, bytes((operation,)))
|
||||
|
||||
def read_ram(self, address, count):
|
||||
"""Data space: SRAM, and with it the register file and every I/O
|
||||
register, which share the address space on AVR. New in pureboot 5."""
|
||||
return self._read_space(SP_RAM, address, count)
|
||||
|
||||
def write_ram(self, address, data):
|
||||
self._write_space(SP_RAM, address, data)
|
||||
|
||||
def read_flash(self, address, count):
|
||||
if self.unified:
|
||||
return self._read_space(SP_FLASH, address, count)
|
||||
if not self.info.word_flash:
|
||||
return self._stream_read("R", address, count)
|
||||
# Word-addressed wire: widen to even bounds and never let one read
|
||||
@@ -432,15 +597,32 @@ class Loader:
|
||||
return data[address - start : address - start + count]
|
||||
|
||||
def read_eeprom(self, address, count):
|
||||
if self.unified:
|
||||
return self._read_space(SP_EEPROM, address, count)
|
||||
return self._stream_read("r", address, count)
|
||||
|
||||
def write_page(self, address, data):
|
||||
assert len(data) == self.info.page and address % self.info.page == 0
|
||||
if self.unified:
|
||||
# 'W' fills the page buffer and stops there; the erase and the write
|
||||
# are host-issued SPM operations. Only a chip with a boot section
|
||||
# has RWW to re-enable — on the others bit 4 of SPMCSR means
|
||||
# something else entirely, so it must not be sent.
|
||||
head = bytes((ord("W"), selector(SP_FLASH, address), address & 0xFF, (address >> 8) & 0xFF))
|
||||
self._command(head + data, 0, 2.0)
|
||||
self.spm(SPM_ERASE, address)
|
||||
self.spm(SPM_WRITE, address)
|
||||
if not self.info.patch_vector:
|
||||
self.spm(SPM_RWWSRE, address)
|
||||
return
|
||||
wire = address // (2 if self.info.word_flash else 1)
|
||||
head = bytes((ord("W"), wire & 0xFF, wire >> 8))
|
||||
self._command(head + data, 0, 2.0)
|
||||
|
||||
def write_eeprom(self, address, data, progress=None):
|
||||
if self.unified:
|
||||
self._write_space(SP_EEPROM, address, data, progress)
|
||||
return
|
||||
offset = 0
|
||||
while offset < len(data):
|
||||
chunk = data[offset : offset + 256]
|
||||
@@ -456,19 +638,18 @@ class Loader:
|
||||
offset += len(chunk)
|
||||
|
||||
def read_fuses(self):
|
||||
if self.unified:
|
||||
return self._read_space(SP_FUSE, 0, 4)
|
||||
return self._command(b"F", 4, 2.0)
|
||||
|
||||
def jump(self, word_address):
|
||||
"""'J': the device acks, then execution continues at the word
|
||||
address — a loader slot's base (whose copy must then be knocked
|
||||
afresh) or the application entry."""
|
||||
"""The device acks, then execution continues at the word address."""
|
||||
self.port.write(bytes((ord("J"), word_address & 0xFF, word_address >> 8)))
|
||||
self._expect_prompt()
|
||||
|
||||
def enter_copy(self, byte_address, wait):
|
||||
"""Jump into the loader copy at `byte_address` and knock it. Ending
|
||||
up in the copy addressed is guaranteed by construction: a jump to a
|
||||
slot base lands in that slot's entry stub."""
|
||||
"""Jump into the loader copy at `byte_address` and knock it — a slot
|
||||
base is that copy's entry stub, so it can only land there."""
|
||||
self.jump(byte_address // 2)
|
||||
return self.connect(wait)
|
||||
|
||||
@@ -564,15 +745,14 @@ def plan_flash(image, info):
|
||||
|
||||
|
||||
def covered(pages, info, skip_blank):
|
||||
"""Pages in programming order; optionally dropping all-0xff pages (sound
|
||||
only over erased flash) — never a load-bearing one.
|
||||
"""Pages in programming order, optionally dropping all-0xff ones (sound
|
||||
only over erased flash, and never a load-bearing page).
|
||||
|
||||
With a patched vector (tinies), the patched page 0 goes first and the
|
||||
trampoline page second: from the first write on, a reset lands in the
|
||||
loader and the loader's own fall-through lands on the application entry,
|
||||
so every interruption point of the flash is recoverable. With a hardware
|
||||
boot section a reset re-vectors to the loader regardless; ascending
|
||||
order, page 0 last, maximizes what an interrupted image retains."""
|
||||
A patched vector puts page 0 first and the trampoline page second, so from
|
||||
the first write on a reset lands in the loader and its fall-through on the
|
||||
application entry — every interruption point recoverable. A hardware boot
|
||||
section re-vectors reset regardless; page 0 goes last there, which
|
||||
maximizes what an interrupted image retains."""
|
||||
trampoline_page = info.base - info.page if info.patch_vector else None
|
||||
first = [0, trampoline_page] if info.patch_vector else []
|
||||
rest = [a for a in sorted(pages) if a not in first]
|
||||
@@ -638,18 +818,21 @@ def mega_boot(info, fuse_bytes):
|
||||
|
||||
|
||||
def image_info(image):
|
||||
"""The info block embedded in a pureboot binary, or None."""
|
||||
at = image.find(b"PB" + bytes((PROTOCOL_VERSION,)))
|
||||
return Info(image[at : at + 12]) if 0 <= at <= len(image) - 12 else None
|
||||
"""The info block embedded in a pureboot binary, or None. Searched once
|
||||
per known version, so the magic stays three selective bytes rather than
|
||||
two that code could carry by chance."""
|
||||
for version in range(OLDEST_LOADER, NEWEST_LOADER + 1):
|
||||
at = image.find(b"PB" + bytes((version,)))
|
||||
if 0 <= at <= len(image) - 12:
|
||||
return Info(image[at : at + 12])
|
||||
return None
|
||||
|
||||
|
||||
def loader_image(path):
|
||||
"""A loader update image, as the slot's own content. A raw binary is that
|
||||
already; an Intel HEX links the loader at its base inside an otherwise
|
||||
blank flash image, and load_image() anchors every image at zero, so the
|
||||
blank below the base is dropped here. The base comes from the image's own
|
||||
info block rather than the device's, so an image built for somewhere else
|
||||
survives intact and the preflight can say so."""
|
||||
"""An update image as the slot's own content: a raw binary already is,
|
||||
while a HEX carries the blank below the loader's base, which is peeled off
|
||||
here. The base comes from the image's own block, not the device's, so a
|
||||
foreign image survives intact for the preflight to reject by name."""
|
||||
image = load_image(path)
|
||||
embedded = image_info(image)
|
||||
if embedded and len(image) > embedded.base:
|
||||
@@ -658,18 +841,17 @@ def loader_image(path):
|
||||
|
||||
|
||||
def staging_content(image, info):
|
||||
"""The 512-byte staging-slot content: the image, padding, and — on
|
||||
chips whose hand-over jumps through the word below the resident loader —
|
||||
that word, which for a staging copy is the slot's own last word: an rjmp
|
||||
to the resident base. The staging copy's fall-through and 'J'-free exit
|
||||
both land in a loader instead of garbage."""
|
||||
slot = info.slot
|
||||
if len(image) > (slot - 2 if info.patch_vector else slot):
|
||||
raise Error(f"loader image is {len(image)} B, the slot holds {slot - 2 if info.patch_vector else slot}")
|
||||
content = bytearray(image) + bytearray([0xFF] * (slot - len(image)))
|
||||
"""The staging slot's content: the image, padding, and — where the
|
||||
hand-over jumps through the word below the resident — that word, which for
|
||||
a staging copy is its own last one. Composed as an rjmp to the resident,
|
||||
so an abandoned staging copy still falls through into a loader."""
|
||||
budget = SLOT - 2 if info.patch_vector else SLOT
|
||||
if len(image) > budget:
|
||||
raise Error(f"loader image is {len(image)} B, the slot holds {budget}")
|
||||
content = bytearray(image) + bytearray([0xFF] * (SLOT - len(image)))
|
||||
if info.patch_vector:
|
||||
through = rjmp_to((info.base - 2) // 2, info.base // 2, info.flash_size // 2)
|
||||
content[slot - 2], content[slot - 1] = through & 0xFF, through >> 8
|
||||
content[SLOT - 2], content[SLOT - 1] = through & 0xFF, through >> 8
|
||||
return bytes(content)
|
||||
|
||||
|
||||
@@ -677,7 +859,10 @@ def update_preflight(image, info, fuse_bytes):
|
||||
"""Errors and warnings before any flash is touched. Returns warnings."""
|
||||
embedded = image_info(image)
|
||||
if embedded is None:
|
||||
raise Error("no pureboot info block in the update image — not a pureboot binary?")
|
||||
raise Error(
|
||||
"no pureboot info block in the update image — not a pureboot binary, "
|
||||
f"or a version this tool ({VERSION}) does not know"
|
||||
)
|
||||
if embedded.raw[3:] != info.raw[3:]:
|
||||
raise Error(
|
||||
f"update image is for another target: it declares "
|
||||
@@ -692,7 +877,7 @@ def update_preflight(image, info, fuse_bytes):
|
||||
raise Error(
|
||||
f"cannot self-update: the staging slot {info.stage:#06x} lies below the "
|
||||
f"boot section ({bls_start:#06x}) where SPM is disabled "
|
||||
f"— a boot section of at least two slots ({2 * info.slot} B, BOOTSZ) is "
|
||||
f"— a boot section of at least two slots ({2 * SLOT} B, BOOTSZ) is "
|
||||
f"required, and only an external programmer can change fuses"
|
||||
)
|
||||
if not bootrst:
|
||||
@@ -734,7 +919,7 @@ class UpdateState:
|
||||
self.data = {
|
||||
"signature": info.signature.hex(),
|
||||
"base": info.base,
|
||||
"staging": loader.read_flash(info.stage, info.slot).hex(),
|
||||
"staging": loader.read_flash(info.stage, SLOT).hex(),
|
||||
"page0": loader.read_flash(0, info.page).hex() if info.patch_vector else "",
|
||||
}
|
||||
with open(self.path, "w") as f:
|
||||
@@ -753,9 +938,8 @@ class UpdateState:
|
||||
|
||||
|
||||
def write_differing(loader, base, content, order=None, label=None):
|
||||
"""Program the pages of `content` at `base` that differ from flash —
|
||||
idempotent, so a resumed phase redoes only what an interruption left.
|
||||
A label puts the compare-and-program loop on the progress bar."""
|
||||
"""Program the pages of `content` at `base` that differ from flash, so a
|
||||
resumed phase redoes only what an interruption left."""
|
||||
page = loader.info.page
|
||||
offsets = list(order) if order is not None else list(range(0, len(content), page))
|
||||
written = 0
|
||||
@@ -768,9 +952,8 @@ def write_differing(loader, base, content, order=None, label=None):
|
||||
bar.step()
|
||||
if label:
|
||||
verbose(f"{label}: {written} of {len(offsets)} pages differed")
|
||||
# Page-wise read-back with the same bounded repair as verify_pages: this
|
||||
# is the loader-update path, where a page left wrong is a half-written
|
||||
# loader slot.
|
||||
# The same bounded repair as verify_pages: here a page left wrong is a
|
||||
# half-written loader slot.
|
||||
for retry in range(RETRIES + 1):
|
||||
bad = [
|
||||
offset
|
||||
@@ -792,8 +975,8 @@ def write_differing(loader, base, content, order=None, label=None):
|
||||
|
||||
|
||||
def patch_word0(loader, page0, target_base):
|
||||
"""Rewrite page 0 with its word 0 re-aimed at `target_base` — the
|
||||
resume insurance around rewriting a loader slot the reset path uses."""
|
||||
"""Re-aim word 0 at `target_base` — the resume insurance around
|
||||
rewriting a loader slot the reset path goes through."""
|
||||
info = loader.info
|
||||
patched = bytearray(page0)
|
||||
word = rjmp_to(0, target_base // 2, info.flash_size // 2)
|
||||
@@ -803,16 +986,17 @@ def patch_word0(loader, page0, target_base):
|
||||
|
||||
|
||||
def op_update_loader(loader, wait, path, state_path, fuse_bytes):
|
||||
"""Replace the resident loader with `path`, using the loader itself as
|
||||
its own staging loader. Every phase is idempotent and keyed off the
|
||||
actual flash state, so a re-run after any interruption resumes; the
|
||||
state file carries the bytes the staging slot held."""
|
||||
"""Replace the resident loader with `path`, using the loader as its own
|
||||
staging loader. Every phase is idempotent and keyed off the flash state,
|
||||
so a re-run resumes; the state file carries what the staging slot held."""
|
||||
info = loader.info
|
||||
image = loader_image(path)
|
||||
for warning in update_preflight(image, info, fuse_bytes):
|
||||
print(f"note: {warning}")
|
||||
update = image_info(image) # the preflight proved it is there
|
||||
verbose(f"installing pureboot {update.version} over pureboot {info.version}")
|
||||
staged = staging_content(image, info)
|
||||
resident = bytes(image) + bytes([0xFF] * (info.slot - len(image)))
|
||||
resident = bytes(image) + bytes([0xFF] * (SLOT - len(image)))
|
||||
page = info.page
|
||||
|
||||
state = UpdateState(state_path)
|
||||
@@ -822,38 +1006,29 @@ def op_update_loader(loader, wait, path, state_path, fuse_bytes):
|
||||
verbose(f"saving the staging slot to {state_path}")
|
||||
state.load_or_save(loader)
|
||||
|
||||
# Install the staging copy — unless a loader already sits whole in the
|
||||
# staging slot (a build programmed there by hand): that copy IS the
|
||||
# installed staging copy, and rewriting it would only trip its own
|
||||
# running-slot guard on the composed through-word. Any pureboot with
|
||||
# the device's own info block serves — the staged copy just streams
|
||||
# pages, so an older build installs a newer resident all the same. Two
|
||||
# checks make "already a loader" mean a *complete* one: the block must
|
||||
# sit where every image carries it (within the slot's first 256 bytes
|
||||
# — the build's position lint), matching the device's block byte for
|
||||
# byte, and the slot must be unchanged since this update began (the
|
||||
# state file's snapshot) — a resumed, half-written install differs
|
||||
# from its snapshot and takes the install path below, which completes
|
||||
# it page by page.
|
||||
current = loader.read_flash(info.stage, info.slot)
|
||||
# A loader already sitting whole in the staging slot IS the staging copy:
|
||||
# rewriting it would only meet its own running-slot guard. Any pureboot
|
||||
# with the device's info block serves, since a staged copy only streams
|
||||
# pages. "Whole" needs both checks — the block where every image carries
|
||||
# it and matching byte for byte, and the slot unchanged since this update
|
||||
# began, so a half-written install takes the path below instead.
|
||||
current = loader.read_flash(info.stage, SLOT)
|
||||
staged_loader = image_info(current[:268])
|
||||
if staged_loader is not None and staged_loader.raw == info.raw and current == state.staging:
|
||||
print("staging slot already holds a loader — left in place")
|
||||
else:
|
||||
# On a chip whose staging slot starts at address 0 (the 1 KB
|
||||
# tiny13s), its first page carries the reset vector: written last,
|
||||
# so any earlier interruption still resets into the old resident,
|
||||
# and from then on resets enter the staging copy.
|
||||
order = list(range(0, info.slot, page))
|
||||
# Where the staging slot starts at address 0 (the 1 KB tiny13s) its
|
||||
# first page carries the reset vector, so it goes last: until then a
|
||||
# reset still reaches the old resident.
|
||||
order = list(range(0, SLOT, page))
|
||||
if info.stage == 0:
|
||||
order = order[1:] + [0]
|
||||
if write_differing(loader, info.stage, staged, order, label="staging copy"):
|
||||
print(f"staging copy installed at {info.stage:#06x}")
|
||||
|
||||
# Enter it and let it rewrite the resident slot. Where a patched reset
|
||||
# vector routes through the resident (a tiny with the staging slot away
|
||||
# from page 0), word 0 is re-aimed at the staging copy around the
|
||||
# rewrite, so a power failure mid-rewrite still resets into a loader.
|
||||
# Enter it and let it rewrite the resident. Where a patched reset vector
|
||||
# routes through the resident, word 0 is re-aimed at the staging copy for
|
||||
# the rewrite, so a power loss mid-rewrite still resets into a loader.
|
||||
verbose(f"entering the staging copy at {info.stage:#06x}")
|
||||
loader.enter_copy(info.stage, wait)
|
||||
redirect = info.patch_vector and info.stage != 0
|
||||
@@ -871,20 +1046,19 @@ def op_update_loader(loader, wait, path, state_path, fuse_bytes):
|
||||
if redirect:
|
||||
verbose("word 0 restored")
|
||||
write_differing(loader, 0, state.page0)
|
||||
order = list(range(0, info.slot, page))
|
||||
order = list(range(0, SLOT, page))
|
||||
if info.stage == 0:
|
||||
order = [0] + order[1:]
|
||||
write_differing(loader, info.stage, state.staging, order, label="staging restore")
|
||||
|
||||
state.discard()
|
||||
print(f"loader updated: {len(image)} B at {info.base:#06x}, staging region restored")
|
||||
print(f"loader updated: pureboot {update.version}, {len(image)} B at {info.base:#06x}, staging region restored")
|
||||
|
||||
|
||||
def check_walk_region(pages, info, fuse_bytes, force):
|
||||
"""With BOOTRST programmed but targeting below the loader, reset reaches
|
||||
the loader only by walking across erased flash from the boot-section
|
||||
start; application data in that span would divert reset into itself.
|
||||
Only checkable when the fuses are known (--fuses or --assume-fuses)."""
|
||||
"""BOOTRST programmed below the loader means reset reaches it only by
|
||||
walking across erased flash; application data in that span would divert
|
||||
reset into itself. Needs the fuses (--fuses or --assume-fuses)."""
|
||||
if info.patch_vector or fuse_bytes is None:
|
||||
return
|
||||
bootrst, bls_start = mega_boot(info, fuse_bytes)
|
||||
@@ -903,10 +1077,9 @@ def check_walk_region(pages, info, fuse_bytes, force):
|
||||
|
||||
|
||||
def op_erase_flash(loader):
|
||||
"""0xff over the whole application area. Descending on a patched-vector
|
||||
chip: page 0 — the patched reset vector — goes last, so an interrupted
|
||||
erase still resets into the loader, and once it is gone the whole area
|
||||
is erased and the reset walk reaches the loader anyway."""
|
||||
"""0xff over the application area, descending where the reset vector is
|
||||
patched: page 0 goes last, so an interrupted erase still resets into the
|
||||
loader — and once it is gone, the erased walk reaches it anyway."""
|
||||
blank = bytes([0xFF] * loader.info.page)
|
||||
addresses = range(0, loader.info.base, loader.info.page)
|
||||
with Progress("erase", len(addresses)) as bar:
|
||||
@@ -942,11 +1115,10 @@ def op_flash(loader, path, erase, verify, fuse_bytes=None, force=False):
|
||||
|
||||
|
||||
def verify_pages(loader, pages, repair=False):
|
||||
"""Read every page back and compare. With `repair`, a mismatched page is
|
||||
rewritten and re-read, up to RETRIES times before it is raised: a page
|
||||
filled over a dirty SPM buffer takes stale words, and the write that took
|
||||
them cleared the buffer, so one rewrite settles it. Anything still wrong
|
||||
after three is not that, and stops the run."""
|
||||
"""Read every page back and compare. With `repair`, a mismatch is
|
||||
rewritten and re-read up to RETRIES times first: a page filled over a
|
||||
dirty SPM buffer takes stale words, and the write that took them cleared
|
||||
the buffer, so one rewrite settles it. Anything still wrong is not that."""
|
||||
repaired = 0
|
||||
with Progress("verify", len(pages)) as bar:
|
||||
for address in sorted(pages):
|
||||
@@ -1024,6 +1196,37 @@ def op_read_eeprom(loader, path):
|
||||
print(f"read EEPROM: {len(data)} B -> {path}")
|
||||
|
||||
|
||||
def _require_unified(loader, what):
|
||||
if not loader.unified:
|
||||
raise Error(f"{what} needs pureboot {UNIFIED_LOADER} or later; this loader is {loader.info.version}")
|
||||
|
||||
|
||||
def _peek_spec(spec):
|
||||
"""ADDR[:N] — addresses and counts in any Python integer base."""
|
||||
address, _, count = spec.partition(":")
|
||||
return int(address, 0), int(count, 0) if count else 1
|
||||
|
||||
|
||||
def op_peek(loader, spec):
|
||||
_require_unified(loader, "--peek")
|
||||
address, count = _peek_spec(spec)
|
||||
data = loader.read_ram(address, count)
|
||||
for offset in range(0, len(data), 16):
|
||||
row = data[offset : offset + 16]
|
||||
text = "".join(chr(b) if 0x20 <= b < 0x7F else "." for b in row)
|
||||
print(f"{address + offset:#06x} {row.hex(' '):<47} {text}")
|
||||
|
||||
|
||||
def op_poke(loader, spec):
|
||||
_require_unified(loader, "--poke")
|
||||
address, _, payload = spec.partition(":")
|
||||
if not payload:
|
||||
raise Error("--poke needs ADDR:HEX, for example 0x200:deadbeef")
|
||||
data = bytes.fromhex(payload.replace(" ", ""))
|
||||
loader.write_ram(int(address, 0), data)
|
||||
print(f"poke: {len(data)} B at {int(address, 0):#06x}")
|
||||
|
||||
|
||||
def op_fuses(loader):
|
||||
low, lock, extended, high = loader.read_fuses()
|
||||
print("fuses:")
|
||||
@@ -1052,9 +1255,14 @@ def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="pureboot host tool", epilog="operations run in the order listed above"
|
||||
)
|
||||
parser.add_argument("--version", action="version", version=f"%(prog)s {VERSION} "
|
||||
f"(speaks pureboot {OLDEST_LOADER}..{NEWEST_LOADER})")
|
||||
parser.add_argument("--port", required=True, help="serial device: COM6, /dev/ttyUSB0, or a simavr pty")
|
||||
parser.add_argument("--baud", type=int, default=115200, help="115200 mega, 57600 tinies")
|
||||
parser.add_argument("--wait", type=float, default=30.0, help="seconds to keep knocking")
|
||||
parser.add_argument("--autobaud", action="store_true",
|
||||
help="drive an autobaud loader: send the 0xC0 calibration pulse and a single "
|
||||
"knock, and take geometry from the signature (no clock/baud baked in)")
|
||||
parser.add_argument("--info", action="store_true", help="print the device info block")
|
||||
parser.add_argument("--fuses", action="store_true", help="read the fuse and lock bytes")
|
||||
parser.add_argument("--update-loader", metavar="FILE", help="replace the loader with this pureboot binary")
|
||||
@@ -1070,6 +1278,10 @@ def main():
|
||||
parser.add_argument("--eeprom", metavar="FILE", help="program the EEPROM (bin or ihex)")
|
||||
parser.add_argument("--read-eeprom", metavar="FILE", help="dump the EEPROM")
|
||||
parser.add_argument("--verify-eeprom", metavar="FILE", help="compare EEPROM against an image")
|
||||
parser.add_argument("--peek", metavar="ADDR[:N]", help="read N bytes of data space (SRAM, registers, "
|
||||
"I/O) — pureboot 5 and later")
|
||||
parser.add_argument("--poke", metavar="ADDR:HEX", help="write hex bytes into data space — "
|
||||
"pureboot 5 and later")
|
||||
parser.add_argument("--force", action="store_true", help="override refusable safety checks")
|
||||
parser.add_argument("--stay", action="store_true", help="leave the loader in its session")
|
||||
parser.add_argument("-v", "--verbose", action="store_true",
|
||||
@@ -1092,7 +1304,7 @@ def main():
|
||||
verbose(f"{args.port}: {args.baud} Bd 8N1, DTR/RTS asserted")
|
||||
try:
|
||||
loader = Loader(port)
|
||||
info = loader.connect(args.wait)
|
||||
info = loader.connect_autobaud(args.wait) if args.autobaud else loader.connect(args.wait)
|
||||
if args.info:
|
||||
print("device:")
|
||||
for line in info.lines():
|
||||
@@ -1121,6 +1333,10 @@ def main():
|
||||
op_read_eeprom(loader, args.read_eeprom)
|
||||
if args.verify_eeprom:
|
||||
op_verify_eeprom(loader, args.verify_eeprom)
|
||||
if args.poke:
|
||||
op_poke(loader, args.poke)
|
||||
if args.peek:
|
||||
op_peek(loader, args.peek)
|
||||
if args.stay:
|
||||
print("loader stays in its session (reset to leave)")
|
||||
else:
|
||||
|
||||
396
pureboot/pureboot_autobaud_pure.cpp
Normal file
396
pureboot/pureboot_autobaud_pure.cpp
Normal file
@@ -0,0 +1,396 @@
|
||||
// SUPERSEDED — kept for the record, not built. With the activation hang fixed
|
||||
// (a lone calibration pulse used to wedge the loader, see autobaud.md) this
|
||||
// version is 530 B on the 1284P against a 512 B slot. Its 4 B of margin was
|
||||
// never spare capacity; it was the space the missing fix should have occupied.
|
||||
// pureboot_autobaud_uni.cpp replaces it at 464 B with more features.
|
||||
//
|
||||
// pureboot autobaud — the software-serial variant that measures the host's bit
|
||||
// timing at runtime, so one binary runs at any F_CPU: the image carries no
|
||||
// clock. THE PURE VERSION — no inline assembly, no global register variables,
|
||||
// exactly the constraints the fixed-baud loader keeps.
|
||||
//
|
||||
// Two source files exist for review (autobaud.md):
|
||||
// this one, pure, and pureboot_autobaud_reg.cpp, which keeps the running-slot
|
||||
// write guard at the cost of one global register variable. They differ only in
|
||||
// where the measured unit lives and whether the guard is present.
|
||||
//
|
||||
// What this version trades to fit 512 B in pure C++ (the 1284 at 508), each
|
||||
// licensed by "the host guarantees safety" (README.md) and the owner's approval
|
||||
// to simplify the info block:
|
||||
// - the measured per-bit unit lives in the two general-purpose I/O scratch
|
||||
// registers (GPIOR) where the chip has them, in a static otherwise —
|
||||
// reached through libavr's named register surface, so no asm and no global
|
||||
// register variable; the loader stays pure;
|
||||
// - the info block is slimmed to the version and the signature — the chip's
|
||||
// identity — from which the host derives page size, loader base, EEPROM
|
||||
// size and the addressing flags via its own chip database;
|
||||
// - no running-slot write guard: the host never programs the loader's own
|
||||
// slot, and a broken host bricking the target is the host's bug;
|
||||
// - a single-byte activation knock: the calibration pulse already proves a
|
||||
// host is present.
|
||||
//
|
||||
// Position independence is kept and is in fact total here: control flow is
|
||||
// PC-relative, the wire carries addresses, and with the slimmed info block and
|
||||
// no write guard nothing anchors on the runtime address at all.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <util/delay_basic.h>
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
namespace hw = avr::hw;
|
||||
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Autobaud carries no clock, so PUREBOOT_CLOCK_HZ / PUREBOOT_BAUD are not
|
||||
// consulted; only the software-UART pins are a deployment parameter.
|
||||
#if !defined(PUREBOOT_RX)
|
||||
#define PUREBOOT_RX pb0
|
||||
#endif
|
||||
#if !defined(PUREBOOT_TX)
|
||||
#define PUREBOOT_TX pb1
|
||||
#endif
|
||||
|
||||
// The watchdog reset flag's home: MCUSR, or the classic megas' MCUCSR.
|
||||
consteval std::int16_t wdrf_field()
|
||||
{
|
||||
auto reg = std::string_view{hw::db.regs[static_cast<std::size_t>(avr::power::detail::reset_reg())].name};
|
||||
return hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
// The loader owns the top 512 bytes; a staging copy goes in the slot below.
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = hw::curated::has_boot_section();
|
||||
|
||||
// Past 64 KiB a byte address no longer fits the wire's 16 bits, so flash
|
||||
// addresses there are word addresses ('J' always was one).
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
|
||||
// The loader's one identity number (README.md); the slimmed info block carries
|
||||
// it and the signature, and the host maps a version to its protocol.
|
||||
constexpr std::uint8_t version = 4;
|
||||
|
||||
// The activation window as a fixed poll budget: with no clock, whole seconds
|
||||
// cannot be timed. A __uint24 (AVR's three-byte type) holds it — a fourth byte
|
||||
// would cost two words at each countdown step for range never used.
|
||||
#if !defined(PUREBOOT_AUTOBAUD_POLLS)
|
||||
#define PUREBOOT_AUTOBAUD_POLLS 4000000
|
||||
#endif
|
||||
constexpr __uint24 autobaud_budget = PUREBOOT_AUTOBAUD_POLLS;
|
||||
|
||||
// The measured per-bit delay (in _delay_loop_2 four-cycle iterations). It lives
|
||||
// in the two adjacent general-purpose I/O scratch registers (GPIOR1:GPIOR2)
|
||||
// where the chip has them — in/out reach them in one word where a static's
|
||||
// lds/sts take two, and there is no .bss to clear — and in a plain static
|
||||
// otherwise (the t13, m8 and m16/32 have no GPIOR). Both are pure: the named
|
||||
// register surface, no inline asm, no global register variable.
|
||||
constexpr bool have_gpior = hw::db.reg_index("GPIOR1") >= 0 && hw::db.reg_index("GPIOR2") >= 0;
|
||||
|
||||
std::uint16_t unit_backing;
|
||||
|
||||
template <bool Gpior = have_gpior>
|
||||
[[gnu::always_inline]] inline std::uint16_t get_unit()
|
||||
{
|
||||
if constexpr (Gpior)
|
||||
return static_cast<std::uint16_t>(hw::reg_impl<hw::db.reg_index("GPIOR1")>::read() |
|
||||
(hw::reg_impl<hw::db.reg_index("GPIOR2")>::read() << 8));
|
||||
else
|
||||
return unit_backing;
|
||||
}
|
||||
|
||||
template <bool Gpior = have_gpior>
|
||||
[[gnu::always_inline]] inline void put_unit(std::uint16_t u)
|
||||
{
|
||||
if constexpr (Gpior) {
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR1")>::write(static_cast<std::uint8_t>(u));
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR2")>::write(static_cast<std::uint8_t>(u >> 8));
|
||||
} else
|
||||
unit_backing = u;
|
||||
}
|
||||
|
||||
// The autobaud software link: bit-banged with cycle-counted delays like the
|
||||
// fixed-baud software backend, but the per-bit delay is the measured unit, not
|
||||
// a consteval constant. rx/tx load the unit once into a local, so each bit
|
||||
// spins from a register with no per-bit reload.
|
||||
struct link {
|
||||
using in_t = avr::io::input<avr::PUREBOOT_RX, avr::io::pull::up>;
|
||||
using out_t = avr::io::output<avr::PUREBOOT_TX>;
|
||||
|
||||
static void init()
|
||||
{
|
||||
avr::init<in_t, out_t>();
|
||||
out_t::set(); // idle high
|
||||
}
|
||||
|
||||
// Time the calibration pulse into the unit. The host sends 0xC0 — a start
|
||||
// bit plus six zero data bits are one low pulse of seven bit-times — and the
|
||||
// counted poll loop is seven cycles an iteration, so the count is the pulse
|
||||
// length in cycles ÷ 7 × 7 = one bit period in cycles, and count >> 2 is that
|
||||
// period in _delay_loop_2's four-cycle iterations. Waits for the start edge
|
||||
// under the poll budget; 0 (returned, and stored) means the budget expired.
|
||||
static std::uint16_t measure(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
std::uint16_t count = 0;
|
||||
while (!in_t::read())
|
||||
++count;
|
||||
const std::uint16_t u = static_cast<std::uint16_t>(count >> 2);
|
||||
put_unit(u);
|
||||
return u;
|
||||
}
|
||||
|
||||
// The eight data bits, entered just past the falling edge of the start bit.
|
||||
// Split out of rx() so the activation path can wait for that edge under a
|
||||
// budget while the command loop waits for it indefinitely.
|
||||
static std::uint8_t sample()
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
_delay_loop_2(static_cast<std::uint16_t>(unit + (unit >> 1))); // 1.5 bits to the LSB centre
|
||||
std::uint8_t value = 0;
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
value >>= 1;
|
||||
if (in_t::read())
|
||||
value |= 0x80;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static std::uint8_t rx()
|
||||
{
|
||||
while (in_t::read()) // await the start edge
|
||||
;
|
||||
return sample();
|
||||
}
|
||||
|
||||
// A byte under the poll budget, for the activation knock. An expired budget
|
||||
// returns 0, which is not the knock, so the caller falls back into the
|
||||
// budgeted measure() — and a line that stays idle boots the application
|
||||
// there. Without this the knock's edge wait was unbounded, so a single
|
||||
// spurious calibration pulse (EMI, or a host that opens the port and never
|
||||
// knocks) wedged the loader and the application never ran.
|
||||
static std::uint8_t rx_bounded(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
return sample();
|
||||
}
|
||||
|
||||
static void tx(std::uint8_t value)
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
out_t::clear(); // start bit
|
||||
_delay_loop_2(unit);
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
out_t::write(value & 1);
|
||||
value >>= 1;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
out_t::set(); // stop bit
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
|
||||
static void drain()
|
||||
{
|
||||
// The software transmitter returns only after the stop bit.
|
||||
}
|
||||
};
|
||||
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
{
|
||||
target();
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
[[gnu::noinline, noreturn]] void run_app()
|
||||
{
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
// Inlined: read across a call, the first byte strands in a call-saved register
|
||||
// the caller has to push and pop.
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_near(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(avr::flash_load(reinterpret_cast<const std::uint8_t *>(address++)));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_far(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
std::uint8_t rampz = static_cast<std::uint8_t>(address >> 15);
|
||||
std::uint16_t z = static_cast<std::uint16_t>(address << 1);
|
||||
do {
|
||||
link::tx(avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(rampz) << 16) | z));
|
||||
if (++z == 0)
|
||||
++rampz;
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline void send_flash(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
if constexpr (word_flash)
|
||||
send_flash_far(address, count);
|
||||
else
|
||||
send_flash_near(address, count);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
void send_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(ee::read(address++));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
void store_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do {
|
||||
ee::write<off>(address++, link::rx());
|
||||
tx_ack();
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
// One page into the SPM buffer, then erase and program. No running-slot write
|
||||
// guard: the host guarantees it never targets the loader's own slot (the pure
|
||||
// version's one dropped safety net, licensed — README.md).
|
||||
void program_flash(std::uint16_t wire_address)
|
||||
{
|
||||
spm::flash_address_t address;
|
||||
if constexpr (word_flash) {
|
||||
const std::uint8_t rampz = static_cast<std::uint8_t>(wire_address >> 15);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
std::uint16_t z = z0;
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(rampz) << 16) | z, word_of({low, high}));
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z));
|
||||
address = (static_cast<spm::flash_address_t>(rampz) << 16) | z0;
|
||||
} else {
|
||||
address = static_cast<spm::flash_address_t>(wire_address & ~static_cast<std::uint16_t>(page - 1));
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>(address, word_of({low, high}));
|
||||
address += 2;
|
||||
} while (static_cast<std::uint8_t>(address) & (page - 1));
|
||||
address -= 2; // back inside the page — erase and write ignore the word bits
|
||||
}
|
||||
spm::erase_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
spm::write_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
}
|
||||
|
||||
void send_fuses()
|
||||
{
|
||||
std::uint8_t which = 0;
|
||||
do
|
||||
link::tx(spm::read_fuse<off>(static_cast<spm::fuse>(which)));
|
||||
while (++which != 4);
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
if (hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// Measure the calibration pulse into the unit, then take one 'p' knock. An
|
||||
// expired budget (no host) boots the application; a pulse that decodes to
|
||||
// anything but 'p' re-measures.
|
||||
for (;;) {
|
||||
if (link::measure(autobaud_budget) == 0)
|
||||
run_app();
|
||||
if (link::rx_bounded(autobaud_budget) == 'p')
|
||||
break;
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
ee::wait();
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // chip identity: version then the three signature bytes
|
||||
link::tx(version);
|
||||
link::tx(hw::db.signature[0]);
|
||||
link::tx(hw::db.signature[1]);
|
||||
link::tx(hw::db.signature[2]);
|
||||
break;
|
||||
case 'R': // read flash: addr16, n8 (0 = 256)
|
||||
case 'r': // read EEPROM: addr16, n8
|
||||
case 'w': { // write EEPROM: addr16, n8, then n bytes each acked
|
||||
// One address-and-count path for the three, so the flash streamer
|
||||
// keeps a single call site and inlines into this never-returning
|
||||
// loop — its cursor then lives in the loop's own call-saved
|
||||
// registers instead of being saved and restored around a call
|
||||
// (the call-site-count lesson, autobaud.md).
|
||||
const std::uint16_t address = rx16();
|
||||
const std::uint8_t count = link::rx();
|
||||
if (command == 'r')
|
||||
send_eeprom(address, count);
|
||||
else if (command == 'w')
|
||||
store_eeprom(address, count);
|
||||
else
|
||||
send_flash(address, count);
|
||||
break;
|
||||
}
|
||||
case 'W': // program one flash page: addr16, page bytes
|
||||
program_flash(rx16());
|
||||
break;
|
||||
case 'F': // fuse and lock bytes
|
||||
send_fuses();
|
||||
break;
|
||||
default: // unknown bytes are ignored; the loop re-acks
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace pureboot
|
||||
|
||||
template struct avr::startup::entry<pureboot::run>;
|
||||
371
pureboot/pureboot_autobaud_reg.cpp
Normal file
371
pureboot/pureboot_autobaud_reg.cpp
Normal file
@@ -0,0 +1,371 @@
|
||||
// SUPERSEDED — kept for the record, not built. With the activation hang fixed
|
||||
// (a lone calibration pulse used to wedge the loader, see autobaud.md) this
|
||||
// version is 534 B on the 1284P against a 512 B slot, so the global register
|
||||
// variable it broke purity for buys nothing. pureboot_autobaud_uni.cpp replaces
|
||||
// it at 464 B, strictly pure and with more features.
|
||||
//
|
||||
// pureboot autobaud — the software-serial variant that measures the host's bit
|
||||
// timing at runtime, so one binary runs at any F_CPU: the image carries no
|
||||
// clock. THE REGISTER VERSION — keeps the running-slot write guard, at the cost
|
||||
// of one global register variable (r4) holding the measured unit. That variable
|
||||
// is pureboot's single, deliberate break from its no-global-register-variable
|
||||
// rule, present only in this variant; everything else stays pure C++.
|
||||
//
|
||||
// Two source files exist for review (autobaud.md):
|
||||
// pureboot_autobaud_pure.cpp, fully pure but dropping the write guard, and this
|
||||
// one. They differ only in where the measured unit lives (a call-saved register
|
||||
// here, GPIOR/RAM there) and whether the guard is present.
|
||||
//
|
||||
// The register buys ~32 B over a RAM home — an outlined rx/tx reads it with one
|
||||
// move where a static costs an lds — and that is what lets the write guard stay
|
||||
// while the image still fits 512 B (the 1284 at 512, exactly). The unit is
|
||||
// written once through a noinline setter so the store lands immediately before a
|
||||
// ret: GCC otherwise deletes a global-register store whose only readers are
|
||||
// callees (autobaud.md, upstream bug 6).
|
||||
//
|
||||
// Simplifications shared with the pure version, each licensed: a slimmed info
|
||||
// block (version + signature; the host derives geometry from its chip database)
|
||||
// and a single-byte activation knock (the calibration pulse already proves a
|
||||
// host). Position independence is kept: control flow is PC-relative and the
|
||||
// write guard anchors on the runtime return address, as the fixed-baud loader.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <util/delay_basic.h>
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
namespace hw = avr::hw;
|
||||
|
||||
// The measured per-bit delay (in _delay_loop_2 four-cycle iterations) in a
|
||||
// call-saved register that the serial callees read directly, so no wire path
|
||||
// threads it and rx/tx reach it with a move, not a load. Written only through
|
||||
// set_unit() below.
|
||||
register std::uint16_t g_unit asm("r4");
|
||||
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Autobaud carries no clock; only the software-UART pins are a parameter.
|
||||
#if !defined(PUREBOOT_RX)
|
||||
#define PUREBOOT_RX pb0
|
||||
#endif
|
||||
#if !defined(PUREBOOT_TX)
|
||||
#define PUREBOOT_TX pb1
|
||||
#endif
|
||||
|
||||
consteval std::int16_t wdrf_field()
|
||||
{
|
||||
auto reg = std::string_view{hw::db.regs[static_cast<std::size_t>(avr::power::detail::reset_reg())].name};
|
||||
return hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = hw::curated::has_boot_section();
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
|
||||
constexpr std::uint8_t version = 4;
|
||||
|
||||
#if !defined(PUREBOOT_AUTOBAUD_POLLS)
|
||||
#define PUREBOOT_AUTOBAUD_POLLS 4000000
|
||||
#endif
|
||||
constexpr __uint24 autobaud_budget = PUREBOOT_AUTOBAUD_POLLS;
|
||||
|
||||
// The one store into g_unit, isolated so it lands right before the ret: a
|
||||
// global-register store whose only later readers are callees is dropped
|
||||
// otherwise (autobaud.md, upstream bug 6).
|
||||
[[gnu::noinline]] void set_unit(std::uint16_t v)
|
||||
{
|
||||
g_unit = v;
|
||||
}
|
||||
|
||||
// The autobaud software link: bit-banged with cycle-counted delays, but the
|
||||
// per-bit delay is g_unit, measured from the host's calibration pulse.
|
||||
struct link {
|
||||
using in_t = avr::io::input<avr::PUREBOOT_RX, avr::io::pull::up>;
|
||||
using out_t = avr::io::output<avr::PUREBOOT_TX>;
|
||||
|
||||
static void init()
|
||||
{
|
||||
avr::init<in_t, out_t>();
|
||||
out_t::set(); // idle high
|
||||
}
|
||||
|
||||
// Time the calibration pulse into g_unit. The host sends 0xC0 — a start bit
|
||||
// plus six zero data bits are one low pulse of seven bit-times — and the
|
||||
// counted poll loop is seven cycles an iteration, so count >> 2 is the bit
|
||||
// period in _delay_loop_2's four-cycle iterations. 0 means the budget
|
||||
// expired.
|
||||
static std::uint16_t measure(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
std::uint16_t count = 0;
|
||||
while (!in_t::read())
|
||||
++count;
|
||||
const std::uint16_t u = static_cast<std::uint16_t>(count >> 2);
|
||||
set_unit(u);
|
||||
return u;
|
||||
}
|
||||
|
||||
// The eight data bits, entered just past the falling edge of the start bit.
|
||||
// Split out of rx() so the activation path can wait for that edge under a
|
||||
// budget while the command loop waits for it indefinitely.
|
||||
static std::uint8_t sample()
|
||||
{
|
||||
_delay_loop_2(static_cast<std::uint16_t>(g_unit + (g_unit >> 1))); // 1.5 bits to the LSB centre
|
||||
std::uint8_t value = 0;
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
value >>= 1;
|
||||
if (in_t::read())
|
||||
value |= 0x80;
|
||||
_delay_loop_2(g_unit);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static std::uint8_t rx()
|
||||
{
|
||||
while (in_t::read()) // await the start edge
|
||||
;
|
||||
return sample();
|
||||
}
|
||||
|
||||
// A byte under the poll budget, for the activation knock. An expired budget
|
||||
// returns 0, which is not the knock, so the caller falls back into the
|
||||
// budgeted measure() — and a line that stays idle boots the application
|
||||
// there. Without this the knock's edge wait was unbounded, so a single
|
||||
// spurious calibration pulse (EMI, or a host that opens the port and never
|
||||
// knocks) wedged the loader and the application never ran.
|
||||
static std::uint8_t rx_bounded(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
return sample();
|
||||
}
|
||||
|
||||
static void tx(std::uint8_t value)
|
||||
{
|
||||
out_t::clear(); // start bit
|
||||
_delay_loop_2(g_unit);
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
out_t::write(value & 1);
|
||||
value >>= 1;
|
||||
_delay_loop_2(g_unit);
|
||||
}
|
||||
out_t::set(); // stop bit
|
||||
_delay_loop_2(g_unit);
|
||||
}
|
||||
|
||||
static void drain()
|
||||
{
|
||||
// The software transmitter returns only after the stop bit.
|
||||
}
|
||||
};
|
||||
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
{
|
||||
target();
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
[[gnu::noinline, noreturn]] void run_app()
|
||||
{
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_near(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(avr::flash_load(reinterpret_cast<const std::uint8_t *>(address++)));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
[[maybe_unused, gnu::always_inline]] inline void send_flash_far(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
std::uint8_t rampz = static_cast<std::uint8_t>(address >> 15);
|
||||
std::uint16_t z = static_cast<std::uint16_t>(address << 1);
|
||||
do {
|
||||
link::tx(avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(rampz) << 16) | z));
|
||||
if (++z == 0)
|
||||
++rampz;
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline void send_flash(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
if constexpr (word_flash)
|
||||
send_flash_far(address, count);
|
||||
else
|
||||
send_flash_near(address, count);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
void send_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do
|
||||
link::tx(ee::read(address++));
|
||||
while (--count);
|
||||
}
|
||||
|
||||
void store_eeprom(std::uint16_t address, std::uint8_t count)
|
||||
{
|
||||
do {
|
||||
ee::write<off>(address++, link::rx());
|
||||
tx_ack();
|
||||
} while (--count);
|
||||
}
|
||||
|
||||
// One page into the SPM buffer, then erase and program — except the slot this
|
||||
// code is running in (`slot_high`, from run()), which is drained and left
|
||||
// alone. A broken host therefore cannot brick the running loader, and a copy one
|
||||
// slot lower may rewrite the resident one.
|
||||
void program_flash(std::uint16_t wire_address, std::uint8_t slot_high)
|
||||
{
|
||||
spm::flash_address_t address;
|
||||
std::uint8_t page_high;
|
||||
if constexpr (word_flash) {
|
||||
const std::uint8_t rampz = static_cast<std::uint8_t>(wire_address >> 15);
|
||||
const std::uint16_t z0 = static_cast<std::uint16_t>(wire_address << 1) & ~static_cast<std::uint16_t>(page - 1);
|
||||
std::uint16_t z = z0;
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(rampz) << 16) | z, word_of({low, high}));
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z));
|
||||
address = (static_cast<spm::flash_address_t>(rampz) << 16) | z0;
|
||||
page_high = static_cast<std::uint8_t>(wire_address >> 8);
|
||||
} else {
|
||||
address = static_cast<spm::flash_address_t>(wire_address & ~static_cast<std::uint16_t>(page - 1));
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
spm::fill<off>(address, word_of({low, high}));
|
||||
address += 2;
|
||||
} while (static_cast<std::uint8_t>(address) & (page - 1));
|
||||
address -= 2; // back inside the page — erase and write ignore the word bits
|
||||
page_high = static_cast<std::uint8_t>(address >> 8) & 0xfe;
|
||||
}
|
||||
if (page_high != slot_high) {
|
||||
spm::erase_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
spm::write_page<off>(address);
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
}
|
||||
if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
}
|
||||
|
||||
void send_fuses()
|
||||
{
|
||||
std::uint8_t which = 0;
|
||||
do
|
||||
link::tx(spm::read_fuse<off>(static_cast<spm::fuse>(which)));
|
||||
while (++which != 4);
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
if (hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// The high byte of the slot this copy runs at, which the write guard
|
||||
// follows: the return address is a word address, so its high byte is the
|
||||
// 256-word slot index, doubled back into byte terms on a byte-addressed
|
||||
// chip. Taken as byteswap's low byte — the builtin already swaps the two
|
||||
// stacked bytes, and the double swap folds away.
|
||||
const std::uint16_t ra_words = reinterpret_cast<std::uint16_t>(__builtin_return_address(0));
|
||||
const std::uint8_t ra_high = static_cast<std::uint8_t>(std::byteswap(ra_words));
|
||||
const std::uint8_t slot_high = word_flash ? ra_high : static_cast<std::uint8_t>(ra_high << 1);
|
||||
|
||||
// Measure the calibration pulse into g_unit, then take one 'p' knock. An
|
||||
// expired budget (no host) boots the application; a pulse that decodes to
|
||||
// anything but 'p' re-measures.
|
||||
for (;;) {
|
||||
if (link::measure(autobaud_budget) == 0)
|
||||
run_app();
|
||||
if (link::rx_bounded(autobaud_budget) == 'p')
|
||||
break;
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
ee::wait();
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // chip identity: version then the three signature bytes
|
||||
link::tx(version);
|
||||
link::tx(hw::db.signature[0]);
|
||||
link::tx(hw::db.signature[1]);
|
||||
link::tx(hw::db.signature[2]);
|
||||
break;
|
||||
case 'R': // read flash: addr16, n8 (0 = 256)
|
||||
case 'r': // read EEPROM: addr16, n8
|
||||
case 'w': { // write EEPROM: addr16, n8, then n bytes each acked
|
||||
// One address-and-count path for the three, so the flash streamer
|
||||
// keeps a single call site and inlines into this never-returning
|
||||
// loop (autobaud.md).
|
||||
const std::uint16_t address = rx16();
|
||||
const std::uint8_t count = link::rx();
|
||||
if (command == 'r')
|
||||
send_eeprom(address, count);
|
||||
else if (command == 'w')
|
||||
store_eeprom(address, count);
|
||||
else
|
||||
send_flash(address, count);
|
||||
break;
|
||||
}
|
||||
case 'W': // program one flash page: addr16, page bytes
|
||||
program_flash(rx16(), slot_high);
|
||||
break;
|
||||
case 'F': // fuse and lock bytes
|
||||
send_fuses();
|
||||
break;
|
||||
default: // unknown bytes are ignored; the loop re-acks
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace pureboot
|
||||
|
||||
template struct avr::startup::entry<pureboot::run>;
|
||||
411
pureboot/pureboot_autobaud_uni.cpp
Normal file
411
pureboot/pureboot_autobaud_uni.cpp
Normal file
@@ -0,0 +1,411 @@
|
||||
// pureboot autobaud, the unified-primitive version — VARIANT A, an explicit
|
||||
// space byte. One read command and one write command carry a space selector, so
|
||||
// flash, EEPROM, RAM and the fuses share a single cursor, a single transfer loop
|
||||
// and a single argument decode instead of one command body each.
|
||||
//
|
||||
// Strictly pure: no inline assembly, no global register variables, and no GPIOR
|
||||
// either — the measured unit lives in a plain static, so the loader claims no
|
||||
// chip resource an application might want. (PUREBOOT_UNIT_GPIOR=1 puts it back
|
||||
// in the I/O scratch registers, kept only as a measurement axis.)
|
||||
//
|
||||
// Against pureboot_autobaud_pure.cpp this version:
|
||||
// - adds RAM read and write, which the loader has never had. Because AVR maps
|
||||
// the register file and the whole I/O space into the data address space,
|
||||
// that one space also gives the host arbitrary peripheral access for free;
|
||||
// - collapses 'R' (read flash), 'r' (read EEPROM), 'w' (write EEPROM) and 'F'
|
||||
// (fuses) — four bodies, four loops — into 'G' and 'P' over four spaces;
|
||||
// - fixes the activation hang: a lone calibration pulse used to leave the
|
||||
// loader blocked forever in the knock's rx(), so a stray edge on an
|
||||
// unattended device wedged it in the loader and the application never ran.
|
||||
//
|
||||
// Position independence is kept and is total: control flow is PC-relative, the
|
||||
// wire carries addresses, and nothing anchors on the runtime address.
|
||||
|
||||
#include <libavr/libavr.hpp>
|
||||
|
||||
#include <util/delay_basic.h>
|
||||
|
||||
using namespace avr::literals;
|
||||
namespace spm = avr::spm;
|
||||
namespace ee = avr::eeprom;
|
||||
namespace hw = avr::hw;
|
||||
|
||||
namespace pureboot {
|
||||
namespace {
|
||||
|
||||
// Purely polled: every interrupt guard folds to nothing.
|
||||
constexpr auto off = avr::irq::guard_policy::unused;
|
||||
|
||||
constexpr std::uint8_t ack = '+';
|
||||
|
||||
// Autobaud carries no clock, so PUREBOOT_CLOCK_HZ / PUREBOOT_BAUD are not
|
||||
// consulted; only the software-UART pins are a deployment parameter.
|
||||
#if !defined(PUREBOOT_RX)
|
||||
#define PUREBOOT_RX pb0
|
||||
#endif
|
||||
#if !defined(PUREBOOT_TX)
|
||||
#define PUREBOOT_TX pb1
|
||||
#endif
|
||||
|
||||
// The unit's home. A plain static by default — pureboot claims no GPIOR, so the
|
||||
// application keeps both scratch registers. The GPIOR spelling is retained
|
||||
// behind a macro purely so the two can be measured against each other.
|
||||
#if !defined(PUREBOOT_UNIT_GPIOR)
|
||||
#define PUREBOOT_UNIT_GPIOR 0
|
||||
#endif
|
||||
|
||||
// The watchdog reset flag's home: MCUSR, or the classic megas' MCUCSR.
|
||||
consteval std::int16_t wdrf_field()
|
||||
{
|
||||
auto reg = std::string_view{hw::db.regs[static_cast<std::size_t>(avr::power::detail::reset_reg())].name};
|
||||
return hw::db.field_index(reg, "WDRF");
|
||||
}
|
||||
|
||||
// The loader owns the top 512 bytes; a staging copy goes in the slot below.
|
||||
constexpr std::uint16_t slot_bytes = 512;
|
||||
constexpr std::uint32_t base = spm::flash_bytes - slot_bytes;
|
||||
constexpr std::uint16_t page = spm::page_bytes;
|
||||
constexpr bool boot_section = hw::curated::has_boot_section();
|
||||
|
||||
// Past 64 KiB a byte address no longer fits the wire's 16 bits, so flash
|
||||
// addresses there are word addresses ('J' always was one).
|
||||
constexpr bool word_flash = spm::flash_bytes > 65536;
|
||||
|
||||
// The loader's one identity number (README.md); the slimmed info block carries
|
||||
// it and the signature, and the host maps a version to its protocol.
|
||||
constexpr std::uint8_t version = 5;
|
||||
|
||||
// The activation window as a fixed poll budget: with no clock, whole seconds
|
||||
// cannot be timed. A __uint24 (AVR's three-byte type) holds it — a fourth byte
|
||||
// would cost two words at each countdown step for range never used.
|
||||
#if !defined(PUREBOOT_AUTOBAUD_POLLS)
|
||||
#define PUREBOOT_AUTOBAUD_POLLS 4000000
|
||||
#endif
|
||||
constexpr __uint24 autobaud_budget = PUREBOOT_AUTOBAUD_POLLS;
|
||||
|
||||
// The two SPM commands the loader still has to recognise by value, on the chips
|
||||
// where it cannot issue a runtime one generically. Taken from the chip's own
|
||||
// definitions rather than spelled 3 and 5 — though every part pureboot targets
|
||||
// agrees on those, which is what lets the host send the raw SPMCSR byte.
|
||||
constexpr std::uint8_t spm_erase = __BOOT_PAGE_ERASE;
|
||||
constexpr std::uint8_t spm_write = __BOOT_PAGE_WRITE;
|
||||
|
||||
// The spaces a transfer can name. Flash is 0 so it is the cheap default.
|
||||
//
|
||||
// sp_spm is the one that is not memory: a write there hands its data byte to
|
||||
// SPMCSR and fires the instruction at the given flash address, so page erase,
|
||||
// page write and RWW re-enable become host-issued commands instead of a
|
||||
// hardcoded tail inside 'W'. The store side already owns an address, a data
|
||||
// byte and an ack, so the whole sequence costs only the fused out/spm pair. It
|
||||
// also lets the host reach every other SPM operation — lock bits included —
|
||||
// which the loader previously had no way to expose.
|
||||
enum : std::uint8_t { sp_flash = 0, sp_eeprom = 1, sp_ram = 2, sp_fuse = 3, sp_spm = 4 };
|
||||
|
||||
// A transfer's selector byte is `space | bank << 4`: the low nibble names the
|
||||
// space, the high nibble carries flash's third address byte (RAMPZ) on the
|
||||
// chips that have one. Putting the bank here rather than widening the address
|
||||
// keeps the shared cursor sixteen bits for every space — a three-byte cursor
|
||||
// costs its extra increment on EEPROM and RAM reads too, which never need it.
|
||||
// The host must not span a bank boundary in one transfer; it already chunks by
|
||||
// page, so nothing it does today comes close.
|
||||
|
||||
// .noinit, not .bss: the unit is always measured before it is read, so it needs
|
||||
// no zeroing — and a zeroed .bss would drag in __do_clear_bss, 18 bytes of
|
||||
// startup code for a variable that is written before its first use.
|
||||
[[gnu::section(".noinit")]] std::uint16_t unit_backing;
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t get_unit()
|
||||
{
|
||||
#if PUREBOOT_UNIT_GPIOR
|
||||
return static_cast<std::uint16_t>(hw::reg_impl<hw::db.reg_index("GPIOR1")>::read() |
|
||||
(hw::reg_impl<hw::db.reg_index("GPIOR2")>::read() << 8));
|
||||
#else
|
||||
return unit_backing;
|
||||
#endif
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline void put_unit(std::uint16_t u)
|
||||
{
|
||||
#if PUREBOOT_UNIT_GPIOR
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR1")>::write(static_cast<std::uint8_t>(u));
|
||||
hw::reg_impl<hw::db.reg_index("GPIOR2")>::write(static_cast<std::uint8_t>(u >> 8));
|
||||
#else
|
||||
unit_backing = u;
|
||||
#endif
|
||||
}
|
||||
|
||||
// The autobaud software link: bit-banged with cycle-counted delays like the
|
||||
// fixed-baud software backend, but the per-bit delay is the measured unit, not
|
||||
// a consteval constant. rx/tx load the unit once into a local, so each bit
|
||||
// spins from a register with no per-bit reload.
|
||||
struct link {
|
||||
using in_t = avr::io::input<avr::PUREBOOT_RX, avr::io::pull::up>;
|
||||
using out_t = avr::io::output<avr::PUREBOOT_TX>;
|
||||
|
||||
static void init()
|
||||
{
|
||||
avr::init<in_t, out_t>();
|
||||
out_t::set(); // idle high
|
||||
}
|
||||
|
||||
// Time the calibration pulse into the unit. The host sends 0xC0 — a start
|
||||
// bit plus six zero data bits are one low pulse of seven bit-times — and the
|
||||
// counted poll loop is seven cycles an iteration, so the count is the pulse
|
||||
// length in cycles ÷ 7 × 7 = one bit period in cycles, and count >> 2 is that
|
||||
// period in _delay_loop_2's four-cycle iterations. Waits for the start edge
|
||||
// under the poll budget; 0 (returned, and stored) means the budget expired.
|
||||
//
|
||||
// The shape of the counting loop is load-bearing, not incidental: the >> 2
|
||||
// is exact only while the pulse's bit-count equals the loop's cycles per
|
||||
// iteration. Both are 7 here (sbis 1 + rjmp 2 + adiw 2 + rjmp 2). Reshaping
|
||||
// this loop silently changes the lock; test/pbautobaud.py is what pins it.
|
||||
static std::uint16_t measure(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
std::uint16_t count = 0;
|
||||
while (!in_t::read())
|
||||
++count;
|
||||
const std::uint16_t u = static_cast<std::uint16_t>(count >> 2);
|
||||
put_unit(u);
|
||||
return u;
|
||||
}
|
||||
|
||||
// The eight data bits, entered just past the falling edge of the start bit.
|
||||
// Split out of rx() so the activation path can wait for that edge under a
|
||||
// budget while the command loop waits for it indefinitely.
|
||||
static std::uint8_t sample()
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
_delay_loop_2(static_cast<std::uint16_t>(unit + (unit >> 1))); // 1.5 bits to the LSB centre
|
||||
std::uint8_t value = 0;
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
value >>= 1;
|
||||
if (in_t::read())
|
||||
value |= 0x80;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
static std::uint8_t rx()
|
||||
{
|
||||
while (in_t::read()) // await the start edge
|
||||
;
|
||||
return sample();
|
||||
}
|
||||
|
||||
// A byte under the poll budget, for the activation knock. An expired budget
|
||||
// returns 0, which is not the knock, so the caller falls back into the
|
||||
// budgeted measure() — and a line that stays idle boots the application
|
||||
// there. That is the whole hang fix: no wait during activation is unbounded,
|
||||
// so a stray calibration pulse can no longer wedge the loader.
|
||||
static std::uint8_t rx_bounded(__uint24 budget)
|
||||
{
|
||||
while (in_t::read())
|
||||
if (!--budget)
|
||||
return 0;
|
||||
return sample();
|
||||
}
|
||||
|
||||
static void tx(std::uint8_t value)
|
||||
{
|
||||
const std::uint16_t unit = get_unit();
|
||||
out_t::clear(); // start bit
|
||||
_delay_loop_2(unit);
|
||||
for (std::uint8_t bit = 0; bit < 8; ++bit) {
|
||||
out_t::write(value & 1);
|
||||
value >>= 1;
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
out_t::set(); // stop bit
|
||||
_delay_loop_2(unit);
|
||||
}
|
||||
|
||||
static void drain()
|
||||
{
|
||||
// The software transmitter returns only after the stop bit.
|
||||
}
|
||||
};
|
||||
|
||||
extern "C" [[noreturn]] void pureboot_app();
|
||||
|
||||
[[gnu::noipa, noreturn]] void jump(void (*target)())
|
||||
{
|
||||
target();
|
||||
__builtin_unreachable();
|
||||
}
|
||||
|
||||
[[gnu::noinline, noreturn]] void run_app()
|
||||
{
|
||||
jump(pureboot_app);
|
||||
}
|
||||
|
||||
// Inlined: read across a call, the first byte strands in a call-saved register
|
||||
// the caller has to push and pop.
|
||||
[[gnu::always_inline]] inline std::uint16_t rx16()
|
||||
{
|
||||
std::uint16_t low = link::rx();
|
||||
return static_cast<std::uint16_t>(low | (link::rx() << 8));
|
||||
}
|
||||
|
||||
[[gnu::always_inline]] inline std::uint16_t word_of(std::array<std::uint8_t, 2> pair)
|
||||
{
|
||||
return std::bit_cast<std::uint16_t>(pair);
|
||||
}
|
||||
|
||||
[[gnu::noinline]] void tx_ack()
|
||||
{
|
||||
link::tx(ack);
|
||||
}
|
||||
|
||||
// One byte out of any space. The four accessors share the cursor, the loop and
|
||||
// the call site — the whole point of the unified commands — so each costs only
|
||||
// its own instruction rather than a body, a loop and a dispatch arm.
|
||||
[[gnu::always_inline]] inline std::uint8_t load(std::uint8_t space, std::uint8_t bank, std::uint16_t at)
|
||||
{
|
||||
if (space == sp_eeprom)
|
||||
return ee::read(at);
|
||||
if (space == sp_ram)
|
||||
return *reinterpret_cast<volatile std::uint8_t *>(at);
|
||||
if (space == sp_fuse)
|
||||
return spm::read_fuse<off>(static_cast<spm::fuse>(at));
|
||||
if constexpr (word_flash)
|
||||
return avr::flash_load_far<std::uint8_t>((static_cast<std::uint32_t>(bank) << 16) | at);
|
||||
else
|
||||
return avr::flash_load(reinterpret_cast<const std::uint8_t *>(at));
|
||||
}
|
||||
|
||||
// One byte into a writable space. Flash is not one of them — it arrives a page
|
||||
// at a time through 'W' — and the fuses are not writable at all.
|
||||
[[gnu::always_inline]] inline void store(std::uint8_t space, [[maybe_unused]] std::uint8_t bank, std::uint16_t at,
|
||||
std::uint8_t value)
|
||||
{
|
||||
if (space == sp_ram) {
|
||||
*reinterpret_cast<volatile std::uint8_t *>(at) = value;
|
||||
return;
|
||||
}
|
||||
if (space == sp_spm) {
|
||||
// The fused store-and-fire: SPMCSR takes the data byte and the SPM
|
||||
// issues against Z in the same unscheduled pair the hardware's
|
||||
// four-cycle window demands, which is exactly why this is one
|
||||
// primitive and not a poke of SPMCSR followed by a poke of anything
|
||||
// else. A host cannot hit that window across a serial link.
|
||||
// The preprocessor rather than `if constexpr` only because
|
||||
// spm::detail::page_command does not exist at all where there is no
|
||||
// RAMPZ, and a discarded constexpr branch outside a template is still
|
||||
// name-checked. A generic spm::command() in libavr would let the
|
||||
// runtime command through on every chip and retire the dispatch below.
|
||||
#if defined(RAMPZ)
|
||||
spm::detail::page_command(value, (static_cast<spm::flash_address_t>(bank) << 16) | at);
|
||||
#else
|
||||
if (value == spm_erase)
|
||||
spm::erase_page<off>(at);
|
||||
else if (value == spm_write)
|
||||
spm::write_page<off>(at);
|
||||
else if constexpr (boot_section)
|
||||
spm::rww_enable<off>();
|
||||
#endif
|
||||
if constexpr (boot_section)
|
||||
spm::wait();
|
||||
return;
|
||||
}
|
||||
ee::write<off>(at, value);
|
||||
}
|
||||
|
||||
// One page into the SPM buffer, and only that: the erase, the write and the RWW
|
||||
// re-enable that used to follow are now three host-issued writes to sp_spm,
|
||||
// which reach the same fused out/spm pair through the store path's own address
|
||||
// and data. No running-slot write guard: the host guarantees it never targets
|
||||
// the loader's own slot (licensed — README.md).
|
||||
void program_flash([[maybe_unused]] std::uint8_t bank, std::uint16_t at)
|
||||
{
|
||||
std::uint16_t z = at & ~static_cast<std::uint16_t>(page - 1);
|
||||
do {
|
||||
std::uint8_t low = link::rx();
|
||||
std::uint8_t high = link::rx();
|
||||
#if defined(RAMPZ)
|
||||
spm::fill<off>((static_cast<spm::flash_address_t>(bank) << 16) | z, word_of({low, high}));
|
||||
#else
|
||||
spm::fill<off>(z, word_of({low, high}));
|
||||
#endif
|
||||
z += 2;
|
||||
} while (static_cast<std::uint8_t>(z) & (page - 1));
|
||||
}
|
||||
|
||||
[[noreturn]] void run()
|
||||
{
|
||||
if (hw::field_impl<wdrf_field()>::test())
|
||||
run_app();
|
||||
|
||||
link::init();
|
||||
|
||||
// Measure the calibration pulse into the unit, then take one 'p' knock —
|
||||
// both under the poll budget. An expired budget (no host) boots the
|
||||
// application; anything but 'p', including the knock timing out, re-measures
|
||||
// and so returns to the budgeted wait that boots it.
|
||||
for (;;) {
|
||||
if (link::measure(autobaud_budget) == 0)
|
||||
run_app();
|
||||
if (link::rx_bounded(autobaud_budget) == 'p')
|
||||
break;
|
||||
}
|
||||
|
||||
for (;;) {
|
||||
ee::wait();
|
||||
tx_ack();
|
||||
const std::uint8_t command = link::rx();
|
||||
switch (command) {
|
||||
case 'J': { // jump to a wire word address: hand-over and staging transfer
|
||||
auto target = reinterpret_cast<void (*)()>(rx16());
|
||||
tx_ack();
|
||||
link::drain();
|
||||
jump(target);
|
||||
}
|
||||
case 'b': // chip identity: version then the three signature bytes
|
||||
link::tx(version);
|
||||
link::tx(hw::db.signature[0]);
|
||||
link::tx(hw::db.signature[1]);
|
||||
link::tx(hw::db.signature[2]);
|
||||
break;
|
||||
case 'W': // program one flash page: sel8, addr16, then page bytes
|
||||
case 'G': // read: sel8, addr16, n8 (0 = 256)
|
||||
case 'g': { // write: sel8, addr16, n8, then n bytes each acked
|
||||
// The unified transfer. One decode, one cursor, one loop for every
|
||||
// space and both directions — the four command bodies this replaces
|
||||
// each carried their own copy of all three. Read and write are the
|
||||
// same letter in the two cases, so the direction is bit 5 of the
|
||||
// command and the loop tests it with a one-word skip. 'W' joins the
|
||||
// same selector-and-address decode rather than keeping a word
|
||||
// address of its own, which makes flash addressing uniform across
|
||||
// every command that names it and costs nothing to share.
|
||||
const std::uint8_t sel = link::rx();
|
||||
const std::uint8_t space = sel & 0x0f;
|
||||
const std::uint8_t bank = static_cast<std::uint8_t>(sel >> 4);
|
||||
std::uint16_t at = rx16();
|
||||
if (command == 'W') {
|
||||
program_flash(bank, at);
|
||||
break;
|
||||
}
|
||||
std::uint8_t count = link::rx();
|
||||
do {
|
||||
if (command & 0x20) {
|
||||
store(space, bank, at, link::rx());
|
||||
tx_ack();
|
||||
} else
|
||||
link::tx(load(space, bank, at));
|
||||
++at;
|
||||
} while (--count);
|
||||
break;
|
||||
}
|
||||
default: // unknown bytes are ignored; the loop re-acks
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
} // namespace pureboot
|
||||
|
||||
template struct avr::startup::entry<pureboot::run>;
|
||||
@@ -1,17 +1,11 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Position-independence lint for the pureboot image.
|
||||
"""Position-independence lint: the two link-time facts that let the identical
|
||||
image run from any slot, asserted from the built ELF.
|
||||
|
||||
The self-staging design lets the identical binary run from any 512-byte
|
||||
slot, which holds only if nothing in the image addresses itself absolutely.
|
||||
Two link-time facts guarantee it, both asserted here from the built ELF:
|
||||
|
||||
1. No absolute jmp/call opcodes — all control flow is PC-relative
|
||||
(rjmp/rcall/ijmp/icall). -mrelax normally guarantees this; a code
|
||||
change that grows a branch out of relaxation range would break it
|
||||
silently.
|
||||
2. The info block sits within the image's first 256 bytes: the 'b'
|
||||
command rebuilds its address as (running slot high byte : low byte of
|
||||
the link address), which needs the offset to fit that low byte.
|
||||
1. No absolute jmp/call — -mrelax normally guarantees it, but a branch that
|
||||
grows out of relaxation range would break it silently.
|
||||
2. The info block within the image's first 256 bytes: 'b' rebuilds its
|
||||
address as (running slot high byte : link address low byte).
|
||||
|
||||
Usage: check_pi.py <objdump> <nm> <elf> <text_start_hex>
|
||||
"""
|
||||
|
||||
@@ -66,9 +66,10 @@ struct link {
|
||||
}
|
||||
[[noreturn]] static void idle()
|
||||
{
|
||||
// 'L' hands back to the loader at the top slot — 512 bytes, or the
|
||||
// 1 KiB the >64 KiB chips use.
|
||||
constexpr std::uint32_t slot = avr::hw::db.mem.flash_size > 65536 ? 1024 : 512;
|
||||
// 'L' hands back to the loader in the top slot — 512 bytes on every
|
||||
// chip. The jump takes a word address, which is what makes the
|
||||
// >64 KiB chips' entry reachable through a 16-bit pointer at all.
|
||||
constexpr std::uint32_t slot = 512;
|
||||
for (;;) {
|
||||
auto command = tx_t::read_blocking();
|
||||
if (command == 'L')
|
||||
|
||||
154
test/pbautobaud.py
Normal file
154
test/pbautobaud.py
Normal file
@@ -0,0 +1,154 @@
|
||||
#!/usr/bin/env python3
|
||||
"""End-to-end autobaud test: drive an autobaud loader in simavr through the
|
||||
calibration handshake and a flash + EEPROM + fuse round-trip, cross-checked
|
||||
against the simulator's ground-truth memory — then repeat at a second F_CPU with
|
||||
the *same* loader binary, which is the property autobaud exists for: one
|
||||
clock-agnostic image that locks onto whatever rate the host sends.
|
||||
|
||||
Usage: pbautobaud.py <device_bin> <loader_elf> <mcu> <base_hex> <page>
|
||||
<app_bin> <app_hz> <app_baud> <tool_py> <workdir>
|
||||
|
||||
The loader is a software-serial build on PB0/PB1 (pureboot_add_autobaud's
|
||||
default), so the runner drives it over the GPIO⇄pty bridge (-l sw:B0,B1). The
|
||||
app fixture is built for (app_hz, app_baud); the hand-over is checked at that
|
||||
point, and a second point at half the clock proves the lock is measured, not
|
||||
baked in.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
def fail(message):
|
||||
print(f"FAIL: {message}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main():
|
||||
(device_bin, elf, mcu, base_hex, page, app_bin, app_hz, app_baud, tool, workdir) = sys.argv[1:]
|
||||
base, page, app_hz, app_baud = int(base_hex, 0), int(page), int(app_hz), int(app_baud)
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(tool)))
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import pbsim
|
||||
import pureboot as pb
|
||||
|
||||
os.makedirs(workdir, exist_ok=True)
|
||||
ee_image = bytes(range(0xA0, 0xB0))
|
||||
ee_path = os.path.join(workdir, "ee.bin")
|
||||
open(ee_path, "wb").write(ee_image)
|
||||
|
||||
# The geometry the surgery planner needs, from the chip class the runner is
|
||||
# told — the same derivation pbtest.py makes: the boot-sectioned megas need
|
||||
# no vector surgery, the tinies and the boot-section-less m48s do, and the
|
||||
# large chips speak word addresses.
|
||||
mega = mcu.startswith("atmega")
|
||||
patch = not mega or mcu.startswith("atmega48")
|
||||
word_flash = base + pb.SLOT > 0x10000
|
||||
wire_base = base // 2 if word_flash else base
|
||||
flags = (1 if patch else 0) | (2 if word_flash else 0)
|
||||
ground_truth = pb.Info(bytes([ord("P"), ord("B"), pb.NEWEST_LOADER, 0, 0, 0, page & 0xFF,
|
||||
wire_base & 0xFF, wire_base >> 8, 0, 0, flags]))
|
||||
|
||||
def round_trip(hz, baud, label, hand_over):
|
||||
"""One clock point: reset, calibrate + knock, program, verify against the
|
||||
simulator's own flash, and (at the app's point) hand over to the fixture."""
|
||||
dump = os.path.join(workdir, f"flash_{label}.bin")
|
||||
device = pbsim.Device(device_bin, elf, mcu, str(hz), base_hex, page, baud, dump, link="sw:B0,B1")
|
||||
try:
|
||||
# The host tool, in autobaud mode, sends the 0xC0 calibration pulse
|
||||
# and a single knock at `baud`; the loader locks to it.
|
||||
out = pbsim.run_tool(tool, device.pty, baud, "--autobaud", "--info", "--fuses",
|
||||
"--flash", app_bin, "--eeprom", ee_path, "--stay")
|
||||
for needed in ("version", "signature", "fuses", "verify:", "stays"):
|
||||
if needed not in out:
|
||||
fail(f"{label}: session output lacks {needed!r}\n{out}")
|
||||
# Read both memories back over the locked link and check them.
|
||||
read_flash = os.path.join(workdir, f"rf_{label}.bin")
|
||||
read_eeprom = os.path.join(workdir, f"re_{label}.bin")
|
||||
out = pbsim.run_tool(tool, device.pty, baud, "--autobaud", "--verify-flash", app_bin,
|
||||
"--verify-eeprom", ee_path, "--read-flash", read_flash,
|
||||
"--read-eeprom", read_eeprom, "--stay")
|
||||
if out.count("verify:") != 2:
|
||||
fail(f"{label}: did not verify both memories\n{out}")
|
||||
if open(read_eeprom, "rb").read()[: len(ee_image)] != ee_image:
|
||||
fail(f"{label}: EEPROM read-back mismatch")
|
||||
|
||||
if hand_over:
|
||||
# Regression: a calibration pulse with no knock behind it must
|
||||
# not wedge the loader. The knock's edge wait used to be
|
||||
# unbudgeted, so one stray low pulse — EMI, or a host that opens
|
||||
# the port and never knocks — held the loader forever and the
|
||||
# application never ran. The whole activation is bounded now, so
|
||||
# the window closes and the app boots; the banner is the proof.
|
||||
# (The pause lets the loader reach its measurement loop, so the
|
||||
# pulse is genuinely seen and the test cannot pass vacuously.)
|
||||
device.reset()
|
||||
port = pb.Port(device.pty, baud)
|
||||
try:
|
||||
time.sleep(0.2)
|
||||
port.write(bytes((pb.CALIBRATE,)))
|
||||
# Accumulate rather than match exactly: the reset leaves the
|
||||
# idle line a framing artefact ahead of the banner, which is
|
||||
# noise here — the question is only whether the app ran.
|
||||
seen = b""
|
||||
deadline = time.monotonic() + 180.0
|
||||
while b"APP" not in seen and time.monotonic() < deadline:
|
||||
seen += port.read_available(1.0)
|
||||
if b"APP" not in seen:
|
||||
fail(f"{label}: lone calibration pulse wedged the loader — app never bannered, saw {seen!r}")
|
||||
print(f" {label}: lone calibration pulse does not wedge the loader")
|
||||
finally:
|
||||
port.close()
|
||||
|
||||
device.reset()
|
||||
port = pb.Port(device.pty, baud)
|
||||
try:
|
||||
loader = pb.Loader(port)
|
||||
live = loader.connect_autobaud(15)
|
||||
if not pb.OLDEST_LOADER <= live.version <= pb.NEWEST_LOADER:
|
||||
fail(f"{label}: loader reports pureboot {live.version}")
|
||||
if loader.unified:
|
||||
# pureboot 5's data space. 0x0200 is clear of the
|
||||
# loader's own .noinit unit at the bottom of SRAM and of
|
||||
# the stack at the top. Reading it back over the same
|
||||
# locked link proves both directions of the new space.
|
||||
probe = bytes(range(0x30, 0x40))
|
||||
loader.write_ram(0x0200, probe)
|
||||
if loader.read_ram(0x0200, len(probe)) != probe:
|
||||
fail(f"{label}: RAM round-trip mismatch")
|
||||
# The register file and the I/O space share the data
|
||||
# address space on AVR, so the same command reaches a
|
||||
# peripheral register. SPMCSR reads back as idle here.
|
||||
verbose_ram = loader.read_ram(0x0200, 4)
|
||||
print(f" {label}: RAM read/write ok ({verbose_ram.hex()})")
|
||||
loader.run_application()
|
||||
banner = port.read_exact(3, 5.0)
|
||||
if banner != b"APP":
|
||||
fail(f"{label}: application banner was {banner!r}")
|
||||
finally:
|
||||
port.close()
|
||||
finally:
|
||||
device.stop()
|
||||
|
||||
# Ground truth (read after the runner exits and writes its dump): what
|
||||
# the tool programmed must be what the simulator actually holds.
|
||||
pages = pb.plan_flash(open(app_bin, "rb").read(), ground_truth)
|
||||
flash_true = open(dump, "rb").read()
|
||||
for address, data in pages.items():
|
||||
if flash_true[address : address + page] != data:
|
||||
fail(f"{label}: simulator flash differs from the programmed image at {address:#06x}")
|
||||
print(f" {label}: locked at {hz} Hz / {baud} Bd, flash+EEPROM verified"
|
||||
+ (", hand-over ok" if hand_over else ""))
|
||||
|
||||
# The app fixture is built for one clock; the hand-over banners there. A
|
||||
# second point at double that clock, same loader binary, proves the lock is
|
||||
# measured, not baked in — the whole point of autobaud. (Doubling keeps the
|
||||
# bit period healthy; halving would drop it below the software UART's floor.)
|
||||
round_trip(app_hz, app_baud, "clock-a", hand_over=True)
|
||||
round_trip(app_hz * 2, app_baud, "clock-b", hand_over=False)
|
||||
print("pbautobaud: calibration lock and flash/EEPROM/fuse round-trip pass at both clocks")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,15 +1,13 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Dirty-page-buffer acceptance test: the loader carries no buffer discard,
|
||||
so a page filled over words an earlier writer left behind programs those
|
||||
instead. This asserts the whole contract — the corruption is real and a bare
|
||||
verify sees it, the repairing verify fixes it in one rewrite (the write that
|
||||
took the stale words auto-erased the buffer), and it stays fixed.
|
||||
"""Dirty-page-buffer acceptance test: with no discard in the loader, a page
|
||||
filled over words an earlier writer left takes those instead. The whole
|
||||
contract is asserted — a bare verify sees the corruption, the repairing
|
||||
verify fixes it in one rewrite, and it stays fixed.
|
||||
|
||||
The state is reached the way the loader cannot prevent: an application
|
||||
dirties the buffer and jumps in with no reset between. Real boot-sectioned
|
||||
megas forbid that outright — SPM executes only from the boot section
|
||||
(Atmel-8271 §26.2) — but simavr dispatches SPM from anywhere, which is what
|
||||
makes the path constructible at all.
|
||||
The state is reached the one way the loader cannot prevent: an application
|
||||
dirties the buffer and jumps in with no reset between. Boot-sectioned megas
|
||||
forbid that outright (SPM runs only from the boot section, Atmel-8271 §26.2),
|
||||
but simavr dispatches SPM from anywhere, which is what makes it constructible.
|
||||
|
||||
Usage: pbdirty.py <device_bin> <pureboot_elf> <mcu> <hz> <base_hex> <page>
|
||||
<baud> <app_bin> <tool_py> <workdir>
|
||||
|
||||
@@ -1,19 +1,13 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Re-homing acceptance test: a pureboot image programmed somewhere other
|
||||
than its canonical top slot must still be a working loader —
|
||||
position-independent, guarding its accidental slot — and the ordinary
|
||||
"""Re-homing acceptance test: an image programmed somewhere other than its
|
||||
canonical slot must still be a working loader, and the ordinary
|
||||
--update-loader flow must put a build into the top slot from there.
|
||||
|
||||
Two positions are exercised. Address 0 (a raw .bin handed to a programmer,
|
||||
which defaults to offset 0): the staging install and the word-0 redirect
|
||||
both run from copies whose slots are not page 0's, so the running-slot
|
||||
guard never blocks the flow. The staging slot itself: a loader already
|
||||
sitting there IS the installed staging copy — the tool recognizes it by
|
||||
its embedded info block and leaves it in place instead of tripping the
|
||||
copy's own guard on the composed through-word — and that (older) copy
|
||||
streams the new resident like any staged copy. In both cases flashing an
|
||||
application through the healed resident overwrites the stale copy, vector
|
||||
surgery included, and the banner proves the launch.
|
||||
Two positions. Address 0, a raw .bin handed to a programmer: the staging
|
||||
install and the word-0 redirect run from copies outside page 0's slot, so the
|
||||
running-slot guard never blocks them. And the staging slot itself, where a
|
||||
loader already sitting there IS the staging copy — recognized by its embedded
|
||||
block and left in place, then streaming the new resident like any staged copy.
|
||||
|
||||
Usage: pbrehome.py <device_bin> <pureboot_elf> <update_bin> <mcu> <hz>
|
||||
<base_hex> <page> <baud> <app_bin> <tool_py> <workdir>
|
||||
@@ -90,7 +84,7 @@ def main():
|
||||
# The staging slot: erased flash with the loader sitting exactly where
|
||||
# a staging copy would — the tool must leave it in place and let it
|
||||
# stream the (different) update build into the resident slot.
|
||||
stage = base - 512
|
||||
stage = base - pb.SLOT
|
||||
rehome_from(pbsim, pb, device_bin, elf, hex(stage), hex(stage), update_bin, base, page, baud, app_bin, workdir,
|
||||
mcu, hz)
|
||||
print("re-home from the staging slot: converged")
|
||||
|
||||
@@ -1,11 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Position-independence acceptance test: the identical pureboot binary,
|
||||
flashed one slot below the resident loader, must serve the complete command
|
||||
set from there. The resident installs it (through-word composed by the host
|
||||
layer), 'J' transfers control, and every command is exercised against the
|
||||
staged copy — the info block must come back byte-identical, the write guard
|
||||
must protect the staged copy's own slot and permit the resident's, and the
|
||||
staged copy must be able to rewrite the resident slot verbatim.
|
||||
"""Position-independence acceptance test: the identical binary, flashed one
|
||||
slot below the resident, must serve the complete command set from there. The
|
||||
info block must come back byte-identical, the write guard must refuse the
|
||||
staged copy's own slot and permit the resident's, and the staged copy must be
|
||||
able to rewrite the resident verbatim.
|
||||
|
||||
Usage: pbreloc.py <device_bin> <pureboot_elf> <mcu> <hz> <base_hex> <page>
|
||||
<baud> <tool_py> <workdir>
|
||||
@@ -83,7 +81,7 @@ def main():
|
||||
|
||||
# Restore the resident image through the staged copy, then 'J' back
|
||||
# into it and prove it lives.
|
||||
resident = image + b"\xff" * (info.slot - len(image))
|
||||
resident = image + b"\xff" * (pb.SLOT - len(image))
|
||||
pb.write_differing(loader, base, resident)
|
||||
back_info = loader.enter_copy(base, 25)
|
||||
if back_info.raw != resident_info:
|
||||
|
||||
@@ -1,15 +1,13 @@
|
||||
#!/usr/bin/env python3
|
||||
"""End-to-end pureboot protocol test: spawn the simavr device, then drive it
|
||||
with the real host tool (pureboot.py, as a subprocess over the device's pty)
|
||||
through flash + EEPROM + fuse + hand-over scenarios, and cross-check
|
||||
the tool's view against the simulator's ground-truth memory dumps.
|
||||
"""End-to-end protocol test: drive the simavr device with the real host tool
|
||||
over its pty through flash, EEPROM, fuse and hand-over scenarios, and
|
||||
cross-check the tool's view against the simulator's ground-truth dumps.
|
||||
|
||||
Usage: pbtest.py <device_bin> <pureboot_elf> <mcu> <hz> <base_hex> <page>
|
||||
<baud> <eeprom_size> <app_bin> <tool_py> <workdir> [link]
|
||||
|
||||
The optional link is the runner's -l spec (usart1, sw:B5,B1, ...) for a
|
||||
The optional link is the runner's -l spec (usart1, sw:B5,B1, ...), for a
|
||||
loader built off the chip's natural serial default.
|
||||
Exits 0 if every scenario passes.
|
||||
"""
|
||||
|
||||
import os
|
||||
@@ -57,11 +55,11 @@ def main():
|
||||
# the page byte is the wire's 0-means-256.
|
||||
mega = mcu.startswith("atmega")
|
||||
patch = not mega or mcu.startswith("atmega48")
|
||||
word_flash = base + 512 > 0x10000
|
||||
word_flash = base + pb.SLOT > 0x10000
|
||||
wire_base = base // 2 if word_flash else base
|
||||
flags = (1 if patch else 0) | (2 if word_flash else 0)
|
||||
info = pb.Info(
|
||||
bytes([ord("P"), ord("B"), 1, 0, 0, 0, page & 0xFF])
|
||||
bytes([ord("P"), ord("B"), pb.NEWEST_LOADER, 0, 0, 0, page & 0xFF])
|
||||
+ bytes([wire_base & 0xFF, wire_base >> 8, eeprom_size & 0xFF, eeprom_size >> 8])
|
||||
+ bytes([flags])
|
||||
)
|
||||
@@ -71,7 +69,7 @@ def main():
|
||||
# Session 1: knock from reset, identify, program everything, stay.
|
||||
out = pbsim.run_tool(tool, device.pty, baud, "--info", "--fuses", "--flash", app_bin,
|
||||
"--eeprom", ee_path, "--stay")
|
||||
for needed in ("signature", "fuses", "verify:", "stays"):
|
||||
for needed in ("version", "signature", "fuses", "verify:", "stays"):
|
||||
if needed not in out:
|
||||
fail(f"session 1 output lacks {needed!r}")
|
||||
|
||||
@@ -101,7 +99,26 @@ def main():
|
||||
port = pb.Port(device.pty, baud)
|
||||
try:
|
||||
loader = pb.Loader(port)
|
||||
loader.connect(15)
|
||||
live = loader.connect(15)
|
||||
# The loader built from this tree must report a version the tool
|
||||
# beside it speaks — a bump the tool was never told about is a
|
||||
# loader it would refuse to talk to. Not equality with the newest:
|
||||
# the tool now spans two loader generations, the fixed-baud one
|
||||
# here and the unified autobaud loader that follows it.
|
||||
if not pb.OLDEST_LOADER <= live.version <= pb.NEWEST_LOADER:
|
||||
fail(f"loader reports pureboot {live.version}, the tool speaks "
|
||||
f"{pb.OLDEST_LOADER}..{pb.NEWEST_LOADER}")
|
||||
|
||||
# A W addressed inside a page rather than at its base must still
|
||||
# consume exactly one page and prompt. The loader's own slot is
|
||||
# the target — it is drained and never programmed — and the
|
||||
# payload is erased-state bytes, so the probe can disturb neither
|
||||
# the image nor the page buffer it leaves behind.
|
||||
wire = wire_base + 1
|
||||
port.write(bytes((ord("W"), wire & 0xFF, wire >> 8)) + b"\xff" * page)
|
||||
if port.read_exact(1, 5.0) != pb.PROMPT:
|
||||
fail("unaligned W did not return to the prompt")
|
||||
|
||||
loader.run_application()
|
||||
banner = port.read_exact(3, 5.0)
|
||||
if banner != b"APP":
|
||||
@@ -122,7 +139,7 @@ def main():
|
||||
# loader, the trampoline on the application's own entry (patched-vector
|
||||
# chips only — a boot-sectioned mega's word 0 stays the application's).
|
||||
if patch:
|
||||
flash_words = (base + 512) // 2
|
||||
flash_words = (base + pb.SLOT) // 2
|
||||
app = open(app_bin, "rb").read()
|
||||
word0 = flash_true[0] | (flash_true[1] << 8)
|
||||
if rjmp_decode(word0, 0, flash_words) != base // 2:
|
||||
|
||||
@@ -1,17 +1,12 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Self-update end-to-end: an application is flashed, then the loader
|
||||
replaces itself with a re-timed build through the host tool's
|
||||
--update-loader — and the power-fail phases of that update are rehearsed by
|
||||
killing the simulated device mid-write, restarting it from its flash dump,
|
||||
and letting a re-run complete the update.
|
||||
"""Self-update end-to-end: an application is flashed, the loader replaces
|
||||
itself with a re-timed build, and every power-fail phase is rehearsed by
|
||||
killing the device mid-write, restarting it from its flash dump, and letting
|
||||
a re-run complete the update.
|
||||
|
||||
The boot-sectioned megas run the BOOTRST-unprogrammed profile (reset boots
|
||||
the application; the fixture application's 'L' jump is the application-owned
|
||||
loader entry), with --assume-fuses standing in for the fuse read simavr
|
||||
cannot model. The patched-vector chips — the tinies and the m48s — reset
|
||||
into a loader at every phase by construction: the t13a because its staging
|
||||
slot carries the reset vector itself, the others through the word-0 redirect
|
||||
the tool plants around the resident rewrite.
|
||||
The boot-sectioned megas run the BOOTRST-unprogrammed profile — reset boots
|
||||
the application, whose 'L' is the application-owned loader entry — with
|
||||
--assume-fuses standing in for the fuse read simavr cannot model.
|
||||
|
||||
Usage: pbupdate.py <device_bin> <pureboot_elf> <update_elf> <mcu> <hz>
|
||||
<base_hex> <page> <baud> <app_bin> <tool_py> <workdir>
|
||||
@@ -50,7 +45,7 @@ def assumed_fuses(pb, image):
|
||||
image's embedded signature."""
|
||||
info = pb.image_info(image)
|
||||
which, ladder = pb.BOOT_FUSE[bytes(info.signature[1:3])]
|
||||
bits = min((b for b in ladder if ladder[b] * 2 >= 2 * info.slot), key=lambda b: ladder[b])
|
||||
bits = min((b for b in ladder if ladder[b] * 2 >= 2 * pb.SLOT), key=lambda b: ladder[b])
|
||||
fuses = bytearray((0xFF, 0xFF, 0xFF, 0xFF))
|
||||
fuses[which] = 0xF8 | (bits << 1) | 1
|
||||
return bytes(fuses)
|
||||
@@ -93,16 +88,13 @@ def main():
|
||||
# The m48s are megas without a boot section: patched vector, no fuse
|
||||
# preflight, and the same reset-to-0 the tinies get.
|
||||
patch = not mega or mcu.startswith("atmega48")
|
||||
# Word-addressed (>64 KiB) chips use the 1 KiB slot; their loader base
|
||||
# itself sits beyond the 16-bit byte space — the 644's base + slot only
|
||||
# touches the 64 KiB boundary and stays byte-addressed.
|
||||
slot = 1024 if base >= 0x10000 and mega else 512
|
||||
reset_hex = "0" if mega else None # the boot-sectioned mega runs BOOTRST-unprogrammed here
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(tool)))
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import pbsim
|
||||
import pureboot as pb
|
||||
|
||||
slot = pb.SLOT
|
||||
os.makedirs(workdir, exist_ok=True)
|
||||
objcopy = os.environ.get("PB_OBJCOPY", "avr-objcopy")
|
||||
images = {}
|
||||
|
||||
@@ -1,9 +1,8 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Host-tool unit tests — the pure planning and policy logic, no simulator:
|
||||
the flash-programming orders and their recovery properties, the reset-vector
|
||||
surgery, the staging-slot composition, the mega boot-fuse decode, and the
|
||||
update preflight's error/warning matrix (fuse combinations simavr cannot
|
||||
model reach it here as synthetic bytes).
|
||||
"""Host-tool unit tests — the planning and policy logic, no simulator:
|
||||
programming orders and their recovery properties, the reset-vector surgery,
|
||||
the staging composition, the boot-fuse decode, and the update preflight over
|
||||
fuse combinations simavr cannot model.
|
||||
|
||||
Usage: test_planner.py <tool_py>
|
||||
"""
|
||||
@@ -27,14 +26,15 @@ def expect_error(what, fn, *needles):
|
||||
fail(f"{what}: no error raised")
|
||||
|
||||
|
||||
def info_of(pb, base, page, patch, flash, signature=(0x1E, 0x93, 0x0B), word_flash=False):
|
||||
def info_of(pb, base, page, patch, flash, signature=(0x1E, 0x93, 0x0B), word_flash=False, version=None):
|
||||
scale = 2 if word_flash else 1
|
||||
wire_base = base // scale
|
||||
flags = (1 if patch else 0) | (2 if word_flash else 0)
|
||||
raw = bytes((0x50, 0x42, 1, *signature, page & 0xFF, wire_base & 0xFF, wire_base >> 8,
|
||||
0, 2, flags))
|
||||
raw = bytes((0x50, 0x42, pb.NEWEST_LOADER if version is None else version,
|
||||
*signature, page & 0xFF, wire_base & 0xFF, wire_base >> 8, 0, 2, flags))
|
||||
info = pb.Info(raw)
|
||||
assert info.flash_size == flash
|
||||
if info.flash_size != flash:
|
||||
fail(f"info_of({base:#x}) decodes to {info.flash_size:#x} of flash, not {flash:#x}")
|
||||
return info
|
||||
|
||||
|
||||
@@ -55,6 +55,21 @@ def main():
|
||||
tiny = info_of(pb, 0x1E00, 64, True, 0x2000)
|
||||
mega = info_of(pb, 0x7E00, 128, False, 0x8000, signature=(0x1E, 0x95, 0x0F))
|
||||
|
||||
# Versioning: the block's third byte is the loader's version, and the tool
|
||||
# speaks a window of them. Every version in the window decodes, so an older
|
||||
# deployed loader stays usable; one above the window is refused by name,
|
||||
# since which version changed the protocol is knowledge only the tool
|
||||
# holds, and it holds none about a version it has never heard of.
|
||||
for version in range(pb.OLDEST_LOADER, pb.NEWEST_LOADER + 1):
|
||||
if info_of(pb, 0x1E00, 64, True, 0x2000, version=version).version != version:
|
||||
fail(f"pureboot {version} does not decode")
|
||||
expect_error(
|
||||
"unknown loader version",
|
||||
lambda: info_of(pb, 0x1E00, 64, True, 0x2000, version=pb.NEWEST_LOADER + 1),
|
||||
f"pureboot {pb.NEWEST_LOADER + 1}",
|
||||
"newer tool",
|
||||
)
|
||||
|
||||
# mega_boot: BOOTSZ words and the BOOTRST sense per chip — the fuse byte
|
||||
# index (EXTENDED on the x8 line except the m328s' HIGH, HIGH elsewhere)
|
||||
# and the per-family ladders (Atmel-2486/2466/2503/2545/8271/DS40002065/
|
||||
@@ -79,9 +94,7 @@ def main():
|
||||
((0x1E, 0x97, 0x05), 0x20000, 3, {0b11: 0x1FC00, 0b10: 0x1F800, 0b01: 0x1F000, 0b00: 0x1E000}), # 1284P
|
||||
)
|
||||
for signature, flash, which, ladder in cases:
|
||||
# Word-addressed chips carry the 1 KiB slot (their smallest boot sector).
|
||||
slot = 1024 if flash > 0x10000 else 512
|
||||
chip = info_of(pb, flash - slot, 128 if flash < 0x20000 else 0, False, flash,
|
||||
chip = info_of(pb, flash - pb.SLOT, 128 if flash < 0x20000 else 0, False, flash,
|
||||
signature=signature, word_flash=flash > 0x10000)
|
||||
for bits, start in ladder.items():
|
||||
fuses = bytearray((0xFF, 0xFF, 0xFF, 0xFF))
|
||||
@@ -94,10 +107,12 @@ def main():
|
||||
if prog or at != start:
|
||||
fail(f"mega_boot {signature[1]:02x}{signature[2]:02b} unprogrammed: {prog} {at:#07x}")
|
||||
|
||||
# Word-addressed info decode: the 1284P's base/page ride the wire scaled,
|
||||
# and its slot is 1 KiB.
|
||||
big = info_of(pb, 0x1FC00, 0, False, 0x20000, signature=(0x1E, 0x97, 0x05), word_flash=True)
|
||||
if big.page != 256 or big.base != 0x1FC00 or big.stage != 0x1F800 or big.slot != 1024:
|
||||
# Word-addressed info decode: the 1284P's base and page ride the wire
|
||||
# scaled — a 17-bit base halved into the block's two bytes, a 256-byte page
|
||||
# spelled 0 — and its slot is the same 512 bytes as everywhere else, so its
|
||||
# staging slot lands inside the 1 KiB minimum boot section.
|
||||
big = info_of(pb, 0x1FE00, 0, False, 0x20000, signature=(0x1E, 0x97, 0x05), word_flash=True)
|
||||
if big.page != 256 or big.base != 0x1FE00 or big.stage != 0x1FC00:
|
||||
fail(f"word-addressed info decode: page {big.page}, base {big.base:#x}, stage {big.stage:#x}")
|
||||
|
||||
# Surgery: word 0 lands on the loader, the trampoline on the original
|
||||
@@ -154,6 +169,12 @@ def main():
|
||||
fail("image_info misses the embedded block")
|
||||
if pb.image_info(bytes((0xAA,)) * 40) is not None:
|
||||
fail("image_info invents a block")
|
||||
# An older loader's image stays readable, so a deployed build can be
|
||||
# identified and installed like any other.
|
||||
old = info_of(pb, 0x1E00, 64, True, 0x2000, version=pb.OLDEST_LOADER)
|
||||
found_old = pb.image_info(bytes((0xAA,)) * 10 + old.raw)
|
||||
if found_old is None or found_old.version != pb.OLDEST_LOADER:
|
||||
fail("image_info misses an older loader's block")
|
||||
|
||||
# loader_image must peel a padded image down to the slot content: a raw
|
||||
# .bin padded from address 0 (or a whole-flash read-back with the loader
|
||||
@@ -194,6 +215,15 @@ def main():
|
||||
if pb.update_preflight(bytes((0xAA,)) * 8 + tiny.raw, tiny, None) != []:
|
||||
fail("tiny preflight should pass without fuses")
|
||||
|
||||
# The 1284s' smallest boot section (512 words) is exactly the resident
|
||||
# slot plus its staging slot, so self-update is possible at the minimum
|
||||
# BOOTSZ — no fuse step up, the 644's geometry. That holds only while a
|
||||
# slot is 512 B: at 1 KiB the staging slot would fall outside the section
|
||||
# and the preflight would refuse.
|
||||
notes = pb.update_preflight(bytes((0xAA,)) * 8 + big.raw, big, fuses(0xFE))
|
||||
if not any("staging slot" in n for n in notes):
|
||||
fail(f"1284 minimum-BOOTSZ notes: {notes}")
|
||||
|
||||
# The walk-region refusal: BOOTRST aimed below the loader plus app data
|
||||
# in the walk span errors without --force; erased spans and unprogrammed
|
||||
# BOOTRST pass.
|
||||
@@ -250,6 +280,68 @@ def main():
|
||||
if device.writes != pb.RETRIES + 1:
|
||||
fail(f"unrepairable page took {device.writes} writes, expected {pb.RETRIES + 1}")
|
||||
|
||||
# The knock handshake against a device that is not listening yet — the
|
||||
# state a port open leaves behind: it resets the chip into a fresh
|
||||
# activation window while the previous session's prompt is still in
|
||||
# flight, so the first knock is lost and a prompt arrives anyway.
|
||||
class FakePort:
|
||||
"""A loader in its activation window, plus `lost` leading writes the
|
||||
reset swallows and one stale prompt still on the wire."""
|
||||
|
||||
def __init__(self, info_raw, lost=0, stale=b"", active=False):
|
||||
self.info_raw = info_raw
|
||||
self.lost = lost
|
||||
self.inflight = bytearray(stale)
|
||||
self.rx = bytearray()
|
||||
self.active = active
|
||||
self.last = None
|
||||
|
||||
def flush_input(self):
|
||||
self.rx.clear()
|
||||
|
||||
def write(self, data):
|
||||
if self.lost:
|
||||
self.lost -= 1
|
||||
return
|
||||
for byte in bytes(data):
|
||||
if not self.active:
|
||||
self.active = self.last == ord("p") and byte == ord("b")
|
||||
self.last = byte
|
||||
if self.active:
|
||||
self.rx += pb.PROMPT
|
||||
elif byte == ord("b"):
|
||||
self.rx += self.info_raw + pb.PROMPT
|
||||
else:
|
||||
self.rx += pb.PROMPT
|
||||
|
||||
def read_available(self, wait):
|
||||
self.rx = self.inflight + self.rx # the stale prompt lands late
|
||||
self.inflight.clear()
|
||||
out, self.rx = bytes(self.rx), bytearray()
|
||||
return out
|
||||
|
||||
def read_exact(self, count, timeout):
|
||||
if len(self.rx) < count:
|
||||
raise pb.Error(f"timeout: got {len(self.rx)} of {count} bytes")
|
||||
out, self.rx = bytes(self.rx[:count]), self.rx[count:]
|
||||
return out
|
||||
|
||||
raw = info_of(pb, 0x7E00, 128, False, 0x8000).raw
|
||||
for what, port in (
|
||||
("clean window", FakePort(raw)),
|
||||
("stale prompt over a lost knock", FakePort(raw, lost=1, stale=pb.PROMPT)),
|
||||
("live session", FakePort(raw, active=True)),
|
||||
):
|
||||
info = pb.Loader(port).connect(5)
|
||||
if info.raw != raw:
|
||||
fail(f"connect ({what}) returned {info.raw.hex()}")
|
||||
|
||||
# A device that never answers still says so, and a version the tool cannot
|
||||
# speak is reported as such rather than retried into a timeout.
|
||||
expect_error("dead device", lambda: pb.Loader(FakePort(raw, lost=99)).connect(0), "no answer")
|
||||
old = bytes(raw[:2]) + bytes((pb.NEWEST_LOADER + 1,)) + bytes(raw[3:])
|
||||
expect_error("unspeakable version", lambda: pb.Loader(FakePort(old)).connect(5), "needs a newer tool")
|
||||
|
||||
print("test_planner: all planner and policy checks pass")
|
||||
|
||||
|
||||
|
||||
@@ -2,12 +2,14 @@
|
||||
# The port's gate: every chip's generated workflow — build, size matrix, and
|
||||
# the simulator-driven protocol suites. --full adds the reflect-spot builds
|
||||
# (libavr's rule: reflect compiles are bounded to its spot set, never the
|
||||
# full matrix). LIBAVR_ROOT must point at the libavr checkout.
|
||||
# full matrix) and swaps the compact size matrix for the exhaustive
|
||||
# clock × baud × backend cross product. LIBAVR_ROOT must point at the libavr
|
||||
# checkout.
|
||||
set -e
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
full=0
|
||||
[[ "$1" == "--full" ]] && { full=1; shift; }
|
||||
[[ "$1" == "--full" ]] && { full=1; shift; export PUREBOOT_FULL_MATRIX=1; }
|
||||
|
||||
CHIPS=(attiny13 attiny13a attiny25 attiny45 attiny85
|
||||
atmega8 atmega8a atmega16 atmega16a atmega32 atmega32a
|
||||
|
||||
Reference in New Issue
Block a user