The rig could drive the UI and detect one outcome: the process died. Nothing could ask the UI what page it was on or what a tap would land on, because that channel is a FIFO inside the guest and the initramfs is busybox-only with no sshd. Scenarios therefore asserted nothing and screenshots went unread. run.sh --ctl exposes a second pci-serial port as a unix socket, the same device the RS485 bridge already rides, listed first so it is always ttyS0. It also puts warden.ctl on the kernel command line, and init bridges only when that marker is present: a VM launched with --rs485 alone has a ttyS0 too, and that one is the Modbus wire. The bridge relays one command line in and the FIFO's reply out, then a sentinel so the reader needs no timeout. qmp.py gains the channel verbs (nav, page, stats, hit, assert_page, assert_hit), records every step to results.jsonl as ok/fail/fatal, continues past an assertion mismatch so one run reports every broken expectation, and checks the console after EVERY step for the stage-2 init's EXITED line so a crash is pinned to the step that caused it. assert_hit matches the widget's bounding box: an icon has no usable caption and two list rows share a class, but the geometry the UI itself resolved is exact. The vocabulary is what tools/warden-ctl already speaks over SSH to a real panel, so a script that runs here runs there. Verified end to end on the rig (11/11 verbs round-tripped) and against the bench panel, where the same commands returned byte-identical results. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013aHKWzT5EF86RFKRMtAv9n
175 lines
7.6 KiB
Plaintext
Executable File
175 lines
7.6 KiB
Plaintext
Executable File
#!/bin/busybox sh
|
|
# shellcheck shell=dash
|
|
# Stage-2 init: PID 1 on the disk rootfs (rootfs_a or rootfs_b), reached via
|
|
# switch_root from the initramfs. Brings up the minimum a WardenOS userspace
|
|
# needs (mounts, by-name symlinks, network, serial alias), then starts any
|
|
# payload daemons and holds. This stands in for the device's BusyBox SysV
|
|
# /etc/init.d/S* sequence; it is deliberately tiny, not a model of it.
|
|
|
|
/bin/busybox mount -t devtmpfs devtmpfs /dev 2>/dev/null
|
|
exec </dev/console >/dev/console 2>&1
|
|
/bin/busybox --install -s /bin
|
|
|
|
mount -t proc proc /proc
|
|
mount -t sysfs sysfs /sys
|
|
mount -t tmpfs tmpfs /tmp
|
|
# The UI's Terminal page opens a PTY (posix_openpt), which needs devpts mounted
|
|
# and /dev/ptmx pointing into it. Without this the page can only ever report
|
|
# "no PTY available" -- it renders, so a screenshot scenario looks fine, while
|
|
# the one thing the page does is untestable in the VM.
|
|
mkdir -p /dev/pts
|
|
mount -t devpts devpts /dev/pts 2>/dev/null
|
|
|
|
# shellcheck source=qemu/rootfs/etc/warden-lib.sh disable=SC1091
|
|
. /etc/warden-lib.sh
|
|
|
|
# Fresh devtmpfs: repopulate the by-name contract; same VALIDATED slot rule
|
|
# as stage 1 (shared helper, so the two can never drift).
|
|
warden_populate_by_name
|
|
slot="$(warden_slot)"
|
|
|
|
# The device's matched mounts: persistent state and the slot's oem partition.
|
|
# Fail-fast: a scenario against an image whose userdata cannot mount would
|
|
# otherwise burn its whole deadline before failing generically. warden.shell
|
|
# still gets a shell for post-mortem.
|
|
mount_fatal() {
|
|
if ! mount -t ext4 "/dev/block/by-name/$1" "$2"; then
|
|
echo "WARDEN-QEMU-MOUNT-FAILED $1"
|
|
if grep -qw warden.shell /proc/cmdline; then
|
|
echo "warden.shell: post-mortem shell (exit powers off)"
|
|
setsid cttyhack sh
|
|
fi
|
|
poweroff -f
|
|
fi
|
|
}
|
|
mount_fatal userdata /userdata
|
|
mount_fatal "oem${slot}" /oem
|
|
mkdir -p /userdata/warden
|
|
|
|
# RS485: warden-modbus hardcodes /dev/ttyS4 at compile time; alias it to the
|
|
# VM's pci-serial UART when one is present (needs the virt.fragment kernel).
|
|
[ -c /dev/ttyS0 ] && ln -sf /dev/ttyS0 /dev/ttyS4
|
|
|
|
# Network: slirp user-mode net on eth0 (DHCP, fallback to QEMU's static map).
|
|
# The fallback keys off the interface actually having an address: udhcpc
|
|
# exiting 0 only proves a lease, not that the hook script applied it.
|
|
ip link set lo up
|
|
if [ -e /sys/class/net/eth0 ]; then
|
|
ip link set eth0 up
|
|
udhcpc -i eth0 -n -q -t 5 -T 2 >/dev/null 2>&1 || true
|
|
if ! ip -4 addr show dev eth0 | grep -q 'inet '; then
|
|
ip addr add 10.0.2.15/24 dev eth0 2>/dev/null
|
|
ip route replace default via 10.0.2.2 dev eth0
|
|
echo "nameserver 10.0.2.3" > /etc/resolv.conf
|
|
fi
|
|
fi
|
|
|
|
hostname warden-qemu
|
|
|
|
echo "WARDEN-QEMU-ROOTFS-OK slot=${slot}"
|
|
|
|
# Payload daemons (dropped into /usr/bin by qemu/mkimage.sh from qemu/payload/).
|
|
# WARDEN_FLARE_INSECURE=1: the VM's portal is the desk mock over plain HTTP.
|
|
# This is a dev instrument: a production device build never sets it.
|
|
export WARDEN_FLARE_INSECURE=1
|
|
# No HPMCU on -M virt: the mailbox SRAM (0xff6fff00) is unmapped bus space
|
|
# here, and flared's /dev/mem poke dies with an external abort (SIGBUS). The
|
|
# SCR1 supervisor state machine is modeled in sim/src/hpmcu.rs instead.
|
|
export WARDEN_HPMCU=0
|
|
# Same class: the CRU reset ladder's /dev/mem poke is fatal on virt. A
|
|
# post-apply "reboot" surfaces as a clean flared error; the scenario harness
|
|
# performs the actual reboot into the applied slot.
|
|
export WARDEN_HARD_RESET=0
|
|
# OTA apply is opt-in per boot (run.sh --allow-apply): writing rootfs_b is
|
|
# safe inside disk.img but must never be the default posture.
|
|
if grep -qw warden.fwapply /proc/cmdline; then
|
|
export WARDEN_FW_ALLOW_APPLY=1
|
|
echo "init: OTA APPLY ENABLED (warden.fwapply)"
|
|
fi
|
|
for d in /usr/bin/warden-flared /usr/bin/warden-modbus; do
|
|
if [ -x "$d" ]; then
|
|
name="$(basename "$d")"
|
|
echo "init: starting $name"
|
|
"$d" > "/tmp/${name}.log" 2>&1 &
|
|
fi
|
|
done
|
|
|
|
# The UI (LVGL fbdev+evdev build from flare-edge tools/build-ui-vm.sh) needs
|
|
# virtio-gpu's fbdev: present only with run.sh --display on|headless AND the
|
|
# virt.fragment kernel.
|
|
if [ -x /usr/bin/warden-ui ] && [ -c /dev/fb0 ]; then
|
|
echo "init: starting warden-ui (fbdev)"
|
|
# Announce the exit on the CONSOLE, not just in the log. A UI that dies
|
|
# mid-scenario otherwise looks exactly like a UI that stopped repainting:
|
|
# the framebuffer holds its last frame, screendumps keep working, and the
|
|
# scenario reports a stale picture as the current state. With this, a crash
|
|
# is one grep away for any scenario driving the VM from outside, and the
|
|
# signal or status that caused it is on the line.
|
|
(
|
|
# tee, not a plain redirect: the UI's own log (LV_LOG_USER and friends)
|
|
# is the most useful thing there is when a scenario does not do what it
|
|
# should, and a scenario driving the VM from outside can only see the
|
|
# CONSOLE. The file is kept as well so the exit dump below still works.
|
|
/usr/bin/warden-ui 2>&1 | tee /tmp/warden-ui.log
|
|
rc=$?
|
|
echo "init: warden-ui EXITED rc=$rc"
|
|
# 128+n is a signal death (139 = SIGSEGV); dump the tail so the
|
|
# scenario's console log carries the UI's own last words.
|
|
echo "init: warden-ui log tail:"
|
|
tail -n 20 /tmp/warden-ui.log 2>/dev/null
|
|
) &
|
|
fi
|
|
|
|
# Control bridge: the UI's debug FIFO, reachable from OUTSIDE the VM.
|
|
#
|
|
# warden-ui answers nav/page/stats/hit on /tmp/warden-ui.ctl (warden_debug.c),
|
|
# but that FIFO lives in here and a scenario drives the VM from the host. On
|
|
# real hardware the same channel is reached over SSH (flare-edge
|
|
# tools/warden-ctl); this initramfs is busybox-only and has no sshd, so the
|
|
# equivalent seam is a second 16550 that run.sh --ctl exposes as a unix socket
|
|
# (pci-serial, the same device the RS485 bridge already rides). One command
|
|
# per line in, the FIFO's reply out, and a sentinel line so the reader knows
|
|
# the reply is complete without a timeout. The vocabulary is identical on both
|
|
# sides of that seam, which is what lets one flow script run against the sim
|
|
# and against a panel.
|
|
#
|
|
# run.sh lists the ctl port before any other pci-serial, so it is always the
|
|
# first 8250, and it says so with warden.ctl on the command line. The marker,
|
|
# not the mere presence of a ttyS0, is what arms the bridge: a VM launched
|
|
# with --rs485 alone also has a ttyS0, and that one is the Modbus wire.
|
|
if grep -qw warden.ctl /proc/cmdline && [ -c /dev/ttyS0 ]; then
|
|
ctl=/dev/ttyS0
|
|
echo "init: control bridge on $ctl"
|
|
(
|
|
# Opened ONCE, read-write, on fd 3. Reopening a serial port per line
|
|
# can block on carrier detect; one open at bridge start either works
|
|
# or fails visibly on the console. The tty stays in its default cooked
|
|
# mode: the host discards echoed input, and a whole line arrives per
|
|
# read.
|
|
exec 3<> "$ctl"
|
|
while IFS= read -r cmd <&3; do
|
|
[ -n "$cmd" ] || continue
|
|
if [ -p /tmp/warden-ui.ctl ]; then
|
|
printf '%s\n' "$cmd" > /tmp/warden-ui.ctl
|
|
# The UI polls its FIFO every 100 ms and truncates the reply
|
|
# file on each command, so a short settle then a read is the
|
|
# same protocol warden-ctl uses over SSH.
|
|
sleep 0.3
|
|
cat /tmp/warden-ui.dbg 2>/dev/null >&3
|
|
else
|
|
echo "bridge: warden-ui control FIFO not present" >&3
|
|
fi
|
|
echo "<<END>>" >&3
|
|
done
|
|
) &
|
|
fi
|
|
|
|
if grep -qw warden.shell /proc/cmdline; then
|
|
echo "warden.shell: interactive shell (exit powers off)"
|
|
setsid cttyhack sh
|
|
poweroff -f
|
|
fi
|
|
|
|
# Hold: daemons run, console idles, scenarios drive the VM from outside.
|
|
while :; do sleep 3600; done
|