diff --git a/AGENTS.md b/AGENTS.md index b9341d8..4f88661 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -459,6 +459,33 @@ Two constraints are not negotiable, both from untimed spin loops in the firmware measuring. Not hypothetical: the first test run decoded 48 registers correctly and still reported RSSI as 0 for precisely this reason. +### Running the tests + + bash tools/run_tests.sh # everything + bash tools/run_tests.sh -q # unit tests only, no emulator, ~15 s + +Use the runner rather than pasting individual commands. It checks the build first and +**stops** on failure, which matters more than it sounds: `ninja` leaves the previous +binary in place when it fails, so tests run happily against code that was never +compiled. That produced two rounds of entirely meaningless results before the habit +stuck. + +It also rebuilds only when `qemu/py32f071.c` differs from the copy in the QEMU tree, so +a plain test run does not pay for a rebuild it does not need. + +The runner checks *itself* first, via `tools/test_run_tests.sh`. Its first version wrote + + if "$@" 2>&1 | sed 's/^/ /'; then + +which tests **sed's** exit status, not the test's — so every test would have counted as +passing whatever broke. Hence `PIPESTATUS[0]`, and a self-check that asserts a failing +test really is counted and named. A runner that cannot fail is worse than none, because +it gets trusted. Test output also goes through `tr -cd` first: gdb-driven tests emit +stray bytes that make the log a "binary file" to grep, which swallows the summary. + +Emulator tests boot their own QEMU on private ports and take 20-30 s each, so they do +not disturb a running `run.sh` or web UI session. + ### Counting distinct frames proves less than it looks Worth knowing before writing any test that watches the screen. diff --git a/README.md b/README.md index eb84340..ae5338e 100644 --- a/README.md +++ b/README.md @@ -86,6 +86,8 @@ keypresses silently stop working. Run the test after touching that code; test_smeter.py the S-meter reads a signal when monitoring test_ptt.py PTT keys the radio and releases cleanly test_scan.py a busy band does not stall a scan + run_tests.sh runs all of the above, build-checked first + test_run_tests.sh that the runner actually notices failures lib_kill_emulator.sh cleanup that only ever kills emulators webui.py web remote control: live LCD plus clickable keypad dn42_firewall.sh restrict the web UI port to DN42 sources @@ -125,6 +127,13 @@ Needs a QEMU 7.2 source tree, `meson`, `ninja`, `libfdt-dev`, `libglib2.0-dev`, Then check the build actually works, which takes about a minute: + bash tools/run_tests.sh # everything, a few minutes + bash tools/run_tests.sh -q # unit tests only, ~15 s, no emulator + +The runner checks the build first and refuses to continue if it fails, because ninja +leaves the previous binary in place and the tests would otherwise pass against code +that was never compiled. Individual tests still run standalone: + python3 tools/keypad_test.py python3 tools/test_flash_persist.py python3 tools/test_freq_entry.py diff --git a/tools/run_tests.sh b/tools/run_tests.sh new file mode 100755 index 0000000..4e50dae --- /dev/null +++ b/tools/run_tests.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Run every test, report what failed, and exit non-zero if anything did. +# +# This exists because the suite had grown to ten separate invocations that had to be +# remembered and pasted in the right order. That is how regressions slip through: it is +# too easy to run the two tests related to what you just changed and miss the one that +# broke. +# +# Emulator tests boot their own QEMU and take 20-30 s each, so a full run is a few +# minutes. Pass -q to run only the fast unit tests, which need no emulator at all and +# finish in about 15 s -- useful while iterating. +# +# The build is checked FIRST and a failure stops everything. ninja leaves the previous +# binary in place when it fails, so the tests would otherwise run happily against a +# stale build and report results for code that was never compiled. That has produced +# two rounds of meaningless output before now. +set -u + +HERE=$(cd "$(dirname "$0")" && pwd) +SIM=$(dirname "$HERE") +QEMU_SRC=${QEMU_SRC:-/root/qemu-build/qemu-7.2+dfsg} + +QUICK=0 +[ "${1:-}" = "-q" ] && QUICK=1 + +pass=0 +fail=0 +failed_names="" + +run() { + local name="$1"; shift + printf '\n=== %s\n' "$name" + # Strip control characters. Some tests shell out to gdb, whose output can carry + # escape sequences and stray bytes; left alone they make the combined log a + # "binary file" as far as grep is concerned, which silently swallows the summary. + # + # PIPESTATUS, not $?, because $? here is sed's status and would report success + # for every failing test. + "$@" 2>&1 | tr -cd '\11\12\15\40-\176' | sed 's/^/ /' + if [ "${PIPESTATUS[0]}" = "0" ]; then + pass=$((pass + 1)) + else + fail=$((fail + 1)) + failed_names="$failed_names $name" + fi +} + +# --- build --------------------------------------------------------------- +# Only when the model source differs from what the QEMU tree holds, so a plain test +# run does not pay for a rebuild it does not need. +if [ -d "$QEMU_SRC/build" ] && \ + ! cmp -s "$SIM/qemu/py32f071.c" "$QEMU_SRC/hw/arm/py32f071.c"; then + echo "=== build (model source changed)" + cp "$SIM/qemu/py32f071.c" "$QEMU_SRC/hw/arm/py32f071.c" + if (cd "$QEMU_SRC/build" && ninja qemu-system-arm 2>&1 | tail -20) \ + | grep -qE 'FAILED|error:'; then + echo " BUILD FAILED -- stopping before any test runs" + echo " (a stale binary would otherwise be tested silently)" + exit 1 + fi + echo " ok" +fi + +# --- unit tests, no emulator -------------------------------------------- +cd "$HERE" +# First, that this script itself reports failures. A runner that silently counts every +# test as passing is worse than no runner, because it gets trusted. +run "runner self-check" bash "$HERE/test_run_tests.sh" +run "unit: model helpers" python3 -m unittest discover -p 'test_uvk5*.py' -q +run "unit: web UI" python3 -m unittest test_webui -q + +if [ "$QUICK" = "1" ]; then + printf '\n%d passed, %d failed (unit tests only)\n' "$pass" "$fail" + [ "$fail" = "0" ] || { echo "failed:$failed_names"; exit 1; } + exit 0 +fi + +# --- emulator tests ----------------------------------------------------- +# Ordered cheapest first, so an obvious breakage surfaces without waiting for the +# whole run. +cd "$SIM" +run "keypad" python3 tools/keypad_test.py +run "BK4819 registers" python3 tools/test_bk4819.py +run "register readback" bash tools/test_bk4819_readback.sh +run "S-meter" python3 tools/test_smeter.py +run "PTT" python3 tools/test_ptt.py +run "scan" python3 tools/test_scan.py +run "serial receive" python3 tools/test_serial_rx.py +run "flash persistence" python3 tools/test_flash_persist.py +run "frequency entry" python3 tools/test_freq_entry.py + +printf '\n%d passed, %d failed\n' "$pass" "$fail" +if [ "$fail" != "0" ]; then + echo "failed:$failed_names" + exit 1 +fi diff --git a/tools/test_run_tests.sh b/tools/test_run_tests.sh new file mode 100755 index 0000000..1c52307 --- /dev/null +++ b/tools/test_run_tests.sh @@ -0,0 +1,89 @@ +#!/usr/bin/env bash +# The test runner must actually notice a failing test. +# +# This is not a hypothetical worry. The first version of run_tests.sh used +# +# if "$@" 2>&1 | sed 's/^/ /'; then +# +# which checks *sed's* exit status, not the test's. sed almost always succeeds, so every +# test would have been counted as passing and the runner would have reported a clean +# suite no matter what broke. A test runner that cannot fail is worse than none, because +# it is trusted. +# +# Checked here: +# 1. a failing command is counted as a failure and named +# 2. a passing command is counted as a pass +# 3. the runner's own exit status is non-zero when something failed +# 4. binary noise in a test's output does not break the accounting +set -u + +HERE=$(cd "$(dirname "$0")" && pwd) +RUNNER="$HERE/run_tests.sh" + +[ -f "$RUNNER" ] || { echo "FAIL run_tests.sh not found"; exit 1; } + +failures=0 + +# Extract the run() helper and exercise it in isolation, so this test does not have to +# boot an emulator to check the accounting logic. +harness=$(mktemp /tmp/run-harness-XXXX.sh) +trap 'rm -f "$harness" /tmp/rt-good /tmp/rt-bad /tmp/rt-noisy' EXIT + +sed -n '/^run() {/,/^}/p' "$RUNNER" > "$harness" +if ! grep -q PIPESTATUS "$harness"; then + echo "FAIL run() does not use PIPESTATUS; it is checking the wrong exit status" + echo " (a piped command's \$? is the last stage, so every test would 'pass')" + exit 1 +fi +echo "PASS run() checks the test's status, not the pipeline's last stage" + +cat >> "$harness" <<'EOF' +pass=0; fail=0; failed_names="" +run "good" /tmp/rt-good +run "bad" /tmp/rt-bad +run "noisy" /tmp/rt-noisy +echo "RESULT pass=$pass fail=$fail failed:$failed_names" +EOF + +printf '#!/bin/sh\necho fine\n' > /tmp/rt-good +printf '#!/bin/sh\necho broken\nexit 1\n' > /tmp/rt-bad +# Emits raw bytes, as gdb-driven tests can. Left unfiltered these make the combined +# output a "binary file" to grep, which silently swallows the summary line. +printf '#!/bin/sh\nprintf "noise\\001\\002\\003\\n"\nexit 0\n' > /tmp/rt-noisy +chmod +x /tmp/rt-good /tmp/rt-bad /tmp/rt-noisy + +out=$(bash "$harness" 2>&1) +result=$(printf '%s\n' "$out" | grep '^RESULT' || true) + +echo " $result" + +case "$result" in + *"pass=2"*) echo "PASS both passing tests counted" ;; + *) echo "FAIL expected pass=2"; failures=$((failures + 1)) ;; +esac + +case "$result" in + *"fail=1"*) echo "PASS the failing test was counted" ;; + *) echo "FAIL expected fail=1; a failing test went unnoticed" + failures=$((failures + 1)) ;; +esac + +case "$result" in + *"failed: bad"*) echo "PASS the failing test was named" ;; + *) echo "FAIL the failing test was not named"; failures=$((failures + 1)) ;; +esac + +# The runner must exit non-zero on failure, or CI and shell && chains ignore it. +if grep -q 'exit 1' "$RUNNER"; then + echo "PASS the runner exits non-zero when tests fail" +else + echo "FAIL the runner never exits non-zero" + failures=$((failures + 1)) +fi + +if [ "$failures" = "0" ]; then + echo + echo "the test runner reports failures correctly" + exit 0 +fi +exit 1