The control bridge sent a command, slept 0.3 s and read /tmp/warden-ui.dbg, so a page that took longer to build handed the host the previous command's reply as if it were this one (flare-edge #152). It now removes the old reply before sending and waits, up to 5 s, for warden-ui to rename the new one into place, the same recipe tools/warden-ctl and flow-run-hw.sh use. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013aHKWzT5EF86RFKRMtAv9n
230 lines
11 KiB
Plaintext
Executable File
230 lines
11 KiB
Plaintext
Executable File
#!/bin/busybox sh
|
|
# shellcheck shell=dash
|
|
# Stage-2 init: PID 1 on the disk rootfs (rootfs_a or rootfs_b), reached via
|
|
# switch_root from the initramfs. Brings up the minimum a WardenOS userspace
|
|
# needs (mounts, by-name symlinks, network, serial alias), then starts any
|
|
# payload daemons and holds. This stands in for the device's BusyBox SysV
|
|
# /etc/init.d/S* sequence; it is deliberately tiny, not a model of it.
|
|
|
|
/bin/busybox mount -t devtmpfs devtmpfs /dev 2>/dev/null
|
|
exec </dev/console >/dev/console 2>&1
|
|
/bin/busybox --install -s /bin
|
|
|
|
mount -t proc proc /proc
|
|
mount -t sysfs sysfs /sys
|
|
mount -t tmpfs tmpfs /tmp
|
|
# The UI's Terminal page opens a PTY (posix_openpt), which needs devpts mounted
|
|
# and /dev/ptmx pointing into it. Without this the page can only ever report
|
|
# "no PTY available" -- it renders, so a screenshot scenario looks fine, while
|
|
# the one thing the page does is untestable in the VM.
|
|
mkdir -p /dev/pts
|
|
mount -t devpts devpts /dev/pts 2>/dev/null
|
|
|
|
# shellcheck source=qemu/rootfs/etc/warden-lib.sh disable=SC1091
|
|
. /etc/warden-lib.sh
|
|
|
|
# Fresh devtmpfs: repopulate the by-name contract; same VALIDATED slot rule
|
|
# as stage 1 (shared helper, so the two can never drift).
|
|
warden_populate_by_name
|
|
slot="$(warden_slot)"
|
|
|
|
# The device's matched mounts: persistent state and the slot's oem partition.
|
|
# Fail-fast: a scenario against an image whose userdata cannot mount would
|
|
# otherwise burn its whole deadline before failing generically. warden.shell
|
|
# still gets a shell for post-mortem.
|
|
mount_fatal() {
|
|
if ! mount -t ext4 "/dev/block/by-name/$1" "$2"; then
|
|
echo "WARDEN-QEMU-MOUNT-FAILED $1"
|
|
if grep -qw warden.shell /proc/cmdline; then
|
|
echo "warden.shell: post-mortem shell (exit powers off)"
|
|
setsid cttyhack sh
|
|
fi
|
|
poweroff -f
|
|
fi
|
|
}
|
|
mount_fatal userdata /userdata
|
|
mount_fatal "oem${slot}" /oem
|
|
mkdir -p /userdata/warden
|
|
|
|
# RS485: warden-modbus hardcodes /dev/ttyS4 at compile time; alias it to the
|
|
# VM's RS485 pci-serial UART (needs the virt.fragment kernel). run.sh lists
|
|
# the control channel's port first when there is one (warden.ctl on the
|
|
# cmdline), so the RS485 UART is ttyS1 then and ttyS0 otherwise. Aliasing
|
|
# ttyS0 blindly put Modbus polls on the control channel.
|
|
if grep -qw warden.ctl /proc/cmdline; then rs485="/dev/ttyS1"; else rs485="/dev/ttyS0"; fi
|
|
[ -c "$rs485" ] && ln -sf "$rs485" /dev/ttyS4
|
|
|
|
# Network: slirp user-mode net on eth0 (DHCP, fallback to QEMU's static map).
|
|
# The fallback keys off the interface actually having an address: udhcpc
|
|
# exiting 0 only proves a lease, not that the hook script applied it.
|
|
ip link set lo up
|
|
if [ -e /sys/class/net/eth0 ]; then
|
|
ip link set eth0 up
|
|
udhcpc -i eth0 -n -q -t 5 -T 2 >/dev/null 2>&1 || true
|
|
if ! ip -4 addr show dev eth0 | grep -q 'inet '; then
|
|
ip addr add 10.0.2.15/24 dev eth0 2>/dev/null
|
|
ip route replace default via 10.0.2.2 dev eth0
|
|
echo "nameserver 10.0.2.3" > /etc/resolv.conf
|
|
fi
|
|
fi
|
|
|
|
hostname warden-qemu
|
|
|
|
echo "WARDEN-QEMU-ROOTFS-OK slot=${slot}"
|
|
|
|
# Payload daemons (dropped into /usr/bin by qemu/mkimage.sh from qemu/payload/).
|
|
# WARDEN_FLARE_INSECURE=1: the VM's portal is the desk mock over plain HTTP.
|
|
# This is a dev instrument: a production device build never sets it.
|
|
export WARDEN_FLARE_INSECURE=1
|
|
# No HPMCU on -M virt: the mailbox SRAM (0xff6fff00) is unmapped bus space
|
|
# here, and flared's /dev/mem poke dies with an external abort (SIGBUS). The
|
|
# SCR1 supervisor state machine is modeled in sim/src/hpmcu.rs instead.
|
|
export WARDEN_HPMCU=0
|
|
# Same class: the CRU reset ladder's /dev/mem poke is fatal on virt. A
|
|
# post-apply "reboot" surfaces as a clean flared error; the scenario harness
|
|
# performs the actual reboot into the applied slot.
|
|
export WARDEN_HARD_RESET=0
|
|
# OTA apply is opt-in per boot (run.sh --allow-apply): writing rootfs_b is
|
|
# safe inside disk.img but must never be the default posture.
|
|
if grep -qw warden.fwapply /proc/cmdline; then
|
|
export WARDEN_FW_ALLOW_APPLY=1
|
|
echo "init: OTA APPLY ENABLED (warden.fwapply)"
|
|
fi
|
|
# The same daemons the panel's SysV scripts start, in their S-number order
|
|
# (S92 ai, S93 automation, S94 modbus, S95 mikrotik, S96 asic/starlink/
|
|
# stratum, S97 flared), each with no arguments and no environment, exactly
|
|
# as start-stop-daemon runs them there. A daemon that cannot live on -M virt
|
|
# (warden-ai wants the NPU) exits into its log and the UI shows the same
|
|
# "not running" it would show on a panel whose daemon died; that is the
|
|
# panel's behaviour, not a rig substitute for it. warden-watchdog stays out:
|
|
# it and a payload flared do not mix (README).
|
|
for d in /usr/bin/warden-ai /usr/bin/warden-automation /usr/bin/warden-modbus \
|
|
/usr/bin/warden-mikrotik /usr/bin/warden-asic /usr/bin/warden-starlink \
|
|
/usr/bin/warden-stratum /usr/bin/warden-flared; do
|
|
if [ -x "$d" ]; then
|
|
name="$(basename "$d")"
|
|
echo "init: starting $name"
|
|
"$d" > "/tmp/${name}.log" 2>&1 &
|
|
fi
|
|
done
|
|
|
|
# The UI (LVGL fbdev+evdev build from flare-edge tools/build-ui-vm.sh) needs
|
|
# virtio-gpu's fbdev: present only with run.sh --display on|headless AND the
|
|
# virt.fragment kernel.
|
|
if [ -x /usr/bin/warden-ui ] && [ -c /dev/fb0 ]; then
|
|
echo "init: starting warden-ui (fbdev)"
|
|
# Announce the exit on the CONSOLE, not just in the log. A UI that dies
|
|
# mid-scenario otherwise looks exactly like a UI that stopped repainting:
|
|
# the framebuffer holds its last frame, screendumps keep working, and the
|
|
# scenario reports a stale picture as the current state. With this, a crash
|
|
# is one grep away for any scenario driving the VM from outside, and the
|
|
# signal or status that caused it is on the line.
|
|
(
|
|
# tee, not a plain redirect: the UI's own log (LV_LOG_USER and friends)
|
|
# is the most useful thing there is when a scenario does not do what it
|
|
# should, and a scenario driving the VM from outside can only see the
|
|
# CONSOLE. The file is kept as well so the exit dump below still works.
|
|
/usr/bin/warden-ui 2>&1 | tee /tmp/warden-ui.log
|
|
rc=$?
|
|
echo "init: warden-ui EXITED rc=$rc"
|
|
# 128+n is a signal death (139 = SIGSEGV); dump the tail so the
|
|
# scenario's console log carries the UI's own last words.
|
|
echo "init: warden-ui log tail:"
|
|
tail -n 20 /tmp/warden-ui.log 2>/dev/null
|
|
) &
|
|
fi
|
|
|
|
# Control bridge: the UI's debug FIFO, reachable from OUTSIDE the VM.
|
|
#
|
|
# warden-ui answers nav/page/stats/hit on /tmp/warden-ui.ctl (warden_debug.c),
|
|
# but that FIFO lives in here and a scenario drives the VM from the host. On
|
|
# real hardware the same channel is reached over SSH (flare-edge
|
|
# tools/warden-ctl); this initramfs is busybox-only and has no sshd, so the
|
|
# equivalent seam is a second 16550 that run.sh --ctl exposes as a unix socket
|
|
# (pci-serial, the same device the RS485 bridge already rides). One command
|
|
# per line in, the FIFO's reply out, and a sentinel line so the reader knows
|
|
# the reply is complete without a timeout. The FIFO vocabulary itself is
|
|
# identical on both sides of that seam, which is what lets one flow script
|
|
# run against the sim and against a panel.
|
|
#
|
|
# One exception, answered by the bridge itself and never forwarded to the
|
|
# FIFO: `@cat PATH` replies with PATH's contents (or one "bridge: no such
|
|
# file: PATH" line if it is missing), then the same sentinel. This is how the
|
|
# json flow channel reads webstatus.c's /tmp/warden-web-status.json snapshot
|
|
# from OUTSIDE the VM -- on a panel that file is just as reachable over the
|
|
# SSH session tools/warden-ctl already has, so hardware needs no equivalent.
|
|
#
|
|
# run.sh lists the ctl port before any other pci-serial, so it is always the
|
|
# first 8250, and it says so with warden.ctl on the command line. The marker,
|
|
# not the mere presence of a ttyS0, is what arms the bridge: a VM launched
|
|
# with --rs485 alone also has a ttyS0, and that one is the Modbus wire.
|
|
if grep -qw warden.ctl /proc/cmdline && [ -c /dev/ttyS0 ]; then
|
|
ctl=/dev/ttyS0
|
|
echo "init: control bridge on $ctl"
|
|
(
|
|
# Opened ONCE, read-write, on fd 3. Reopening a serial port per line
|
|
# can block on carrier detect; one open at bridge start either works
|
|
# or fails visibly on the console. The tty stays in its default cooked
|
|
# mode: the host discards echoed input, and a whole line arrives per
|
|
# read.
|
|
exec 3<> "$ctl"
|
|
while IFS= read -r cmd <&3; do
|
|
[ -n "$cmd" ] || continue
|
|
case "$cmd" in
|
|
"@cat "*)
|
|
# A bridge-local command, never forwarded to warden-ui's
|
|
# FIFO: `@cat PATH` reads PATH directly off the GUEST's
|
|
# own filesystem and answers with it, which is how the
|
|
# json flow channel gets webstatus.c's snapshot out to
|
|
# the host driving the VM from outside. `-f` so a
|
|
# directory or device node reports as missing rather than
|
|
# cat hanging or erroring oddly.
|
|
path="${cmd#@cat }"
|
|
if [ -f "$path" ]; then
|
|
cat "$path" >&3
|
|
# Force a newline after the file's own bytes: the
|
|
# status json (webstatus.c) is written with NO
|
|
# trailing newline, and without this the sentinel
|
|
# below would land on the SAME line as the content
|
|
# and the reader (qmp.py Ctl.send, line-based) would
|
|
# block forever waiting for a line that never comes.
|
|
echo >&3
|
|
else
|
|
echo "bridge: no such file: $path" >&3
|
|
fi
|
|
;;
|
|
*)
|
|
if [ -p /tmp/warden-ui.ctl ]; then
|
|
# Remove the previous reply BEFORE sending, then wait
|
|
# for the new one to appear (the UI renames it into
|
|
# place whole). A fixed settle used to hand back the
|
|
# previous command's reply whenever a page took longer
|
|
# than 0.3 s to build (flare-edge #152); this is the
|
|
# same recipe tools/warden-ctl uses over SSH.
|
|
rm -f /tmp/warden-ui.dbg
|
|
printf '%s\n' "$cmd" > /tmp/warden-ui.ctl
|
|
n=0
|
|
while [ ! -s /tmp/warden-ui.dbg ] && [ "$n" -lt 100 ]; do
|
|
sleep 0.05
|
|
n=$((n + 1))
|
|
done
|
|
cat /tmp/warden-ui.dbg 2>/dev/null >&3
|
|
else
|
|
echo "bridge: warden-ui control FIFO not present" >&3
|
|
fi
|
|
;;
|
|
esac
|
|
echo "<<END>>" >&3
|
|
done
|
|
) &
|
|
fi
|
|
|
|
if grep -qw warden.shell /proc/cmdline; then
|
|
echo "warden.shell: interactive shell (exit powers off)"
|
|
setsid cttyhack sh
|
|
poweroff -f
|
|
fi
|
|
|
|
# Hold: daemons run, console idles, scenarios drive the VM from outside.
|
|
while :; do sleep 3600; done
|