loop recovery: review-ladder fixes, root never runs a fly-writable binary

B1: fly-loop-reset runs only /opt/fly/sbin/flysim, a root-owned copy
05-deploy installs from the release tarball after checking it against
the tarball's MANIFEST; fixed paths, fly.env parsed as data, env -i.
B2: the wrapper pauses fly-watchdog.timer (and waits out a running
probe) for the reset, and starts flysim and the timer on every exit.
H1: a step that raises or times out is a failed step; the ladder state
is saved before the step runs. H2: one --list call computes the build's
compatibility once; TimeoutStartSec 25 min. H3: level 2+ resets to the
rung below the best or restarts, never lower.
Tests run the wrapper against a fake flysim, systemctl and archives.
This commit is contained in:
acamilo 2026-09-28 21:34:32 +00:00
parent 9c9cec49a9
commit f205d95e5e
6 changed files with 304 additions and 90 deletions

View file

@ -272,6 +272,21 @@ if [ -n "$RELEASE_TARBALL" ]; then
cpu_pin chown -R fly:fly "$release_path"
fi
# A root-owned flysim for fly-loop-reset, which runs as root and so must never execute the
# fly-owned copy above (the fly account could replace it). Taken from the tarball on the host
# and checked against the tarball's own MANIFEST on every deploy, so a release installed by
# an earlier deploy is covered too.
log "05-deploy: installing a root-owned flysim at /opt/fly/sbin/flysim for fly-loop-reset"
root_flysim="$(mktemp)"
tar -xzOf "$RELEASE_TARBALL" ./flysim > "$root_flysim"
want_sha="$(tar -xzOf "$RELEASE_TARBALL" ./MANIFEST | awk '$2 == "flysim" || $2 == "./flysim" {print $1; exit}')"
[ -n "$want_sha" ] && [ "$(sha256sum "$root_flysim" | cut -d' ' -f1)" = "$want_sha" ] \
|| { rm -f "$root_flysim"; die "the tarball's flysim does not match its MANIFEST; not installing /opt/fly/sbin/flysim"; }
ct_exec "$CTID" -- install -d -o root -g root -m 0755 /opt/fly/sbin
ct_push_file "$CTID" "$root_flysim" /opt/fly/sbin/flysim.new 0755
ct_exec "$CTID" -- sh -c 'chown root:root /opt/fly/sbin/flysim.new && mv -f /opt/fly/sbin/flysim.new /opt/fly/sbin/flysim'
rm -f "$root_flysim"
log "05-deploy: overlaying infra-owned stage/serve.{mjs,sh} (docs/design/infra.md's 'own tiny static server' clarification)"
converge_file "$CTID" "$INFRA_DIR/config/serve.mjs" "${release_path}/stage/serve.mjs" 0644 fly:fly >/dev/null
converge_file "$CTID" "$INFRA_DIR/config/serve.sh" "${release_path}/stage/serve.sh" 0755 fly:fly >/dev/null

View file

@ -32,6 +32,9 @@ MAX_RESETS = 2 # milestone resets per DAY
VETO_LIMIT = 3 # model vetoes of one step before it goes ahead anyway
MODEL_BUDGET = 90 # seconds across all models in one decision
VERIFY_TIMEOUT = 240
LIST_TIMEOUT = 120
RESTART_TIMEOUT = 180
RESET_TIMEOUT = 480 # the wrapper waits up to 60 s for the watchdog, then stop, reset, start
# docs/design/ladder.md; mirrors RANK_LADDER in flybrain-gb's pokemon_red/mod.rs (a test pins it).
LADDER = [
@ -140,27 +143,30 @@ def rungs():
def plan(level, rank, best, resets_left):
"""(action, target rung) for a ladder level; a spent budget or no archive falls back to restart.
"""(action, target rung) for a ladder level; a spent budget or no restorable archive is a restart.
Level 1 resets to the current rung's archive, level 2 and beyond to the archive below the
best rung (never lower), so repeated days of a trap cannot walk the run down the ladder.
Level 1 resets to the current rung's archive; level 2 and beyond to the rung below the best,
and never lower, so repeated days of a trap cannot walk the run down the ladder.
"""
if level == 0 or not resets_left or rank is None:
return "restart", None
ceiling = rank if level == 1 else min(rank, max(best, rank) - 1)
for rung in reversed([rung for rung in rungs() if rung <= ceiling]):
if restorable(rung):
return "reset", rung
target = rank if level == 1 else max(best, rank) - 1
if target > rank or target not in restorable_rungs():
return "restart", None
return "reset", target
def restorable(rung):
"""Whether the running build can restore this archive (fly-loop-reset --check)."""
def restorable_rungs():
"""The rungs the running build can restore (fly-loop-reset --list), empty on any failure."""
try:
return subprocess.run(["sudo", "-n", RESET_BIN, "--check", str(rung)], check=False, timeout=60,
stdout=subprocess.DEVNULL).returncode == 0
except (OSError, subprocess.SubprocessError):
return False
done = subprocess.run(["sudo", "-n", RESET_BIN, "--list"], check=False, timeout=LIST_TIMEOUT,
capture_output=True, text=True)
except (OSError, subprocess.SubprocessError) as error:
print("restorable rungs unknown:", type(error).__name__, file=sys.stderr)
return set()
if done.returncode != 0:
return set()
return {int(line) for line in done.stdout.split() if line.isdigit()}
def status():
@ -184,10 +190,14 @@ def verify(target, deadline):
def act(action, target):
if action == "restart":
command = ["sudo", "-n", "systemctl", "restart", "flysim.service"]
command, limit = ["sudo", "-n", "systemctl", "restart", "flysim.service"], RESTART_TIMEOUT
else:
command = ["sudo", "-n", RESET_BIN, str(target)]
done = subprocess.run(command, check=False, timeout=VERIFY_TIMEOUT)
command, limit = ["sudo", "-n", RESET_BIN, str(target)], RESET_TIMEOUT
try:
done = subprocess.run(command, check=False, timeout=limit)
except (OSError, subprocess.SubprocessError) as error:
print(f"{action} did not complete: {type(error).__name__}", file=sys.stderr)
return False
if done.returncode != 0:
return False
return verify(target, time.monotonic() + VERIFY_TIMEOUT)
@ -268,15 +278,18 @@ def run(now=None, sleep=time.sleep, clock=None):
base.update(toRung=target, toLabel=LADDER[target] if 0 <= target < len(LADDER) else None)
notice(base, "countdown", now)
sleep(COUNTDOWN)
notice(base, "acting", clock())
started = int(clock())
# Recorded before acting, so a helper killed mid-step has still climbed and spent the reset.
state.update(level=level + 1, actedAt=started, lastAction=action, vetoes=0)
state.pop("observedAt", None)
if action == "reset":
state["resets"].append(started)
write_json(STATE, state)
notice(base, "acting", started)
ok = act(action, target)
finished = int(clock())
notice(base, "done" if ok else "failed", finished)
state.update(level=level + 1, actedAt=finished, lastAction=action, vetoes=0)
state.pop("observedAt", None)
if action == "reset":
state["resets"].append(finished)
state["actedAt"] = finished
write_json(STATE, state)
log({"at": finished, "event": action, "target": target, "ok": ok, "why": why, "rank": rank,
"level": level, "sequence": report.get("sequence"), "map": report.get("map")})

View file

@ -3,47 +3,80 @@
#
# fly-loop-recover runs as User=fly and reaches this through one NOPASSWD sudoers line
# (config/fly-sudoers). It does what infra/docs/runbook.md "Restart the run from a rung"
# does by hand for a release whose adapter already wrote the archive: stop flysim, promote
# milestone-<N>.checkpoint with fly-reset-to-milestone, start flysim. flysim is started
# again whether or not the reset worked, so the stream never stays down on a failure;
# fly-reset-to-milestone copies both stores aside before it rewrites anything.
# does by hand: stop flysim, promote milestone-<N>.checkpoint with fly-reset-to-milestone,
# start flysim.
#
# fly-reset-to-milestone does not check that the running build can restore the archive, and a
# flysim that refuses every checkpoint refuses to start: a black stream. So an archive is only
# used when its compatibility string equals this build's, or differs only in an adapter id that
# FLY_ACCEPT_ADAPTERS in /etc/fly/fly.env names (05-deploy.sh's adapter_migration_accepted).
# Root never executes anything the fly account can write. The release tree under
# /opt/fly/releases is fly-owned, so the flysim run here is the root-owned copy 05-deploy
# installs at /opt/fly/sbin/flysim straight from the release tarball; without it every rung is
# refused and the ladder falls back to restarts. Paths are fixed (test overrides are honoured
# only when not root), and /etc/fly/fly.env is read as KEY=VALUE data, never sourced.
#
# Usage: fly-loop-reset [--check] <rung> (--check: exit 0 if restorable, 3 if not; touch nothing)
# fly-reset-to-milestone does not check that the build can restore the archive, and a flysim
# that refuses every checkpoint refuses to start: a black stream. So a rung is only used when
# its archive's compatibility string equals this build's, or differs only in an adapter id that
# FLY_ACCEPT_ADAPTERS names (05-deploy.sh's adapter_migration_accepted).
#
# fly-watchdog restarts a flysim whose /healthz fails, which during a reset would start it on a
# half-rewritten store; its timer is stopped for the reset and started again on every exit path.
#
# Usage: fly-loop-reset <rung> reset to that rung
# fly-loop-reset --list print the rungs this build can restore, one per line
set -euo pipefail
: "${FLY_STATE_DIR:=/srv/fly/state}"
: "${FLY_SERVICE:=flysim.service}"
: "${FLY_RESET_BIN:=/opt/fly/bin/fly-reset-to-milestone}"
: "${FLY_ENV_FILE:=/etc/fly/fly.env}"
: "${FLY_BIN:=/opt/fly/current/flysim}"
STATE_DIR=/srv/fly/state
ENV_FILE=/etc/fly/fly.env
FLYSIM=/opt/fly/sbin/flysim
RESET_BIN=/opt/fly/bin/fly-reset-to-milestone
if [ "$(id -u)" -ne 0 ]; then
STATE_DIR="${FLY_LOOP_RESET_TEST_STATE_DIR:-$STATE_DIR}"
ENV_FILE="${FLY_LOOP_RESET_TEST_ENV_FILE:-$ENV_FILE}"
FLYSIM="${FLY_LOOP_RESET_TEST_FLYSIM:-$FLYSIM}"
RESET_BIN="${FLY_LOOP_RESET_TEST_RESET_BIN:-$RESET_BIN}"
fi
SERVICE=flysim.service
WATCHDOG_TIMER=fly-watchdog.timer
WATCHDOG_SERVICE=fly-watchdog.service
readonly STATE_DIR ENV_FILE FLYSIM RESET_BIN SERVICE WATCHDOG_TIMER WATCHDOG_SERVICE
log() { echo "fly-loop-reset: $*" >&2; }
usage() { log "usage: fly-loop-reset <rung> | --list"; exit 2; }
CHECK=0
if [ "${1:-}" = "--check" ]; then
CHECK=1
shift
LIST=0
RANK=""
[ "$#" -eq 1 ] || usage
if [ "$1" = --list ]; then
LIST=1
elif [[ "$1" =~ ^[0-9]{1,2}$ ]]; then
RANK="$1"
else
usage
fi
RANK="${1:-}"
if [ "$#" -ne 1 ] || ! [[ "$RANK" =~ ^[0-9]{1,2}$ ]]; then
log "usage: fly-loop-reset [--check] <rung>"
exit 2
fi
ARCHIVE="${FLY_STATE_DIR}/milestone-${RANK}.checkpoint"
if [ ! -f "$ARCHIVE" ]; then
log "no milestone archive for rung ${RANK}; nothing touched"
exit 1
# FLY_* lines of fly.env as `env` arguments. Values may be quoted the systemd way; nothing
# is evaluated.
env_args=()
accepted=""
if [ -r "$ENV_FILE" ]; then
while IFS= read -r line || [ -n "$line" ]; do
[[ "$line" =~ ^(FLY_[A-Z0-9_]*)=(.*)$ ]] || continue
key="${BASH_REMATCH[1]}"
value="${BASH_REMATCH[2]}"
if [[ "$value" =~ ^\"(.*)\"$ ]] || [[ "$value" =~ ^\'(.*)\'$ ]]; then
value="${BASH_REMATCH[1]}"
fi
env_args+=("$key=$value")
if [ "$key" = FLY_ACCEPT_ADAPTERS ]; then
accepted="$value"
fi
done < "$ENV_FILE"
fi
clean_env() { env -i PATH=/usr/sbin:/usr/bin:/sbin:/bin HOME=/root "${env_args[@]}" "$@"; }
# The same rule as 05-deploy.sh: exactly one '/'-separated field differs, it is the adapter
# (index 1), and the archive's adapter id is listed.
migration_accepted() {
local old="$1" new="$2" accepted="$3" i differing=0 index=-1 entry
local old="$1" new="$2" i differing=0 index=-1 entry
local -a old_parts new_parts
IFS='/' read -r -a old_parts <<< "$old"
IFS='/' read -r -a new_parts <<< "$new"
@ -61,30 +94,63 @@ migration_accepted() {
return 1
}
accepted=""
if [ -r "$FLY_ENV_FILE" ]; then
set -a
# shellcheck disable=SC1090
. "$FLY_ENV_FILE"
set +a
accepted="${FLY_ACCEPT_ADAPTERS:-}"
build_compat=""
if [ -x "$FLYSIM" ] && [ "$(stat -c %u "$FLYSIM")" = "$(id -u)" ]; then
build_compat="$(clean_env timeout 90 "$FLYSIM" --print-compatibility 2>/dev/null | tail -n1 || true)"
fi
archive_compat="$(head -c 262144 "$ARCHIVE" 2>/dev/null | grep -a -o -m1 '"compatibility":"[^"]*"' | head -n1 | cut -d'"' -f4 || true)"
build_compat="$("$FLY_BIN" --print-compatibility 2>/dev/null | tail -n1 || true)"
if [ -z "$archive_compat" ] || [ -z "$build_compat" ]; then
log "cannot read the compatibility of rung ${RANK}'s archive or of this build; nothing touched"
if [ -z "$build_compat" ]; then
log "no compatibility string from $FLYSIM (missing, not owned by $(id -un), or failed); no rung is restorable"
[ "$LIST" -eq 1 ] && exit 0
exit 3
fi
if [ "$archive_compat" != "$build_compat" ] && ! migration_accepted "$archive_compat" "$build_compat" "$accepted"; then
log "rung ${RANK}'s archive ($(echo "$archive_compat" | cut -d/ -f2)) is not restorable by this build ($(echo "$build_compat" | cut -d/ -f2)), FLY_ACCEPT_ADAPTERS='${accepted}'; nothing touched"
exit 3
fi
[ "$CHECK" -eq 0 ] || exit 0
log "stopping ${FLY_SERVICE} to reset to rung ${RANK}"
systemctl stop "$FLY_SERVICE"
restorable() {
local archive="${STATE_DIR}/milestone-$1.checkpoint" compat
[ -f "$archive" ] || return 1
compat="$(head -c 262144 "$archive" 2>/dev/null | grep -a -o -m1 '"compatibility"[[:space:]]*:[[:space:]]*"[^"]*"' | head -n1 | cut -d'"' -f4 || true)"
[ -n "$compat" ] || return 1
[ "$compat" = "$build_compat" ] || migration_accepted "$compat" "$build_compat"
}
if [ "$LIST" -eq 1 ]; then
for archive in "${STATE_DIR}"/milestone-*.checkpoint; do
[ -e "$archive" ] || continue
rung="${archive##*/milestone-}"
rung="${rung%.checkpoint}"
if [[ "$rung" =~ ^[0-9]{1,2}$ ]] && restorable "$rung"; then
echo "$rung"
fi
done | sort -n
exit 0
fi
if ! restorable "$RANK"; then
log "rung ${RANK}'s archive is missing or not restorable by this build (FLY_ACCEPT_ADAPTERS='${accepted}'); nothing touched"
exit 3
fi
# shellcheck disable=SC2317 # invoked by the EXIT trap
finish() {
local status=$?
systemctl start "$SERVICE" || log "starting ${SERVICE} failed"
systemctl start "$WATCHDOG_TIMER" || log "starting ${WATCHDOG_TIMER} failed"
exit "$status"
}
trap finish EXIT
trap 'exit 143' TERM INT HUP
log "pausing ${WATCHDOG_TIMER} and stopping ${SERVICE} to reset to rung ${RANK}"
systemctl stop "$WATCHDOG_TIMER"
for _ in $(seq 1 60); do
systemctl is-active --quiet "$WATCHDOG_SERVICE" || break
sleep 1
done
if systemctl is-active --quiet "$WATCHDOG_SERVICE"; then
log "${WATCHDOG_SERVICE} still running after 60 s; not resetting"
exit 4
fi
systemctl stop "$SERVICE"
status=0
"$FLY_RESET_BIN" "$RANK" || status=$?
[ "$status" -eq 0 ] || log "fly-reset-to-milestone ${RANK} failed (exit ${status}); starting ${FLY_SERVICE} on the state it left"
systemctl start "$FLY_SERVICE"
clean_env FLY_BIN="$FLYSIM" "$RESET_BIN" "$RANK" || status=$?
[ "$status" -eq 0 ] || log "fly-reset-to-milestone ${RANK} failed (exit ${status}); starting ${SERVICE} on the state it left"
exit "$status"

View file

@ -19,21 +19,29 @@ a ladder one step per trap that outlives the previous step:
- **Starting over:** the ladder returns to level 0 when the fly reaches a new best rung, or after
six hours with no suspected report.
A milestone step only uses an archive the running build can restore: `fly-loop-reset --check`
compares the archive's compatibility string with `flysim --print-compatibility` and accepts an
A milestone step only uses an archive the running build can restore. `fly-loop-reset --list`
compares each archive's compatibility string with `flysim --print-compatibility` and accepts an
adapter-only difference that `FLY_ACCEPT_ADAPTERS` in `/etc/fly/fly.env` names (the rule
`05-deploy.sh` applies). A flysim that refuses every checkpoint refuses to start, so an archive
from an older adapter is skipped for the next one down unless the deploy named its adapter; with
none restorable the step is a restart. Keep `FLY_ACCEPT_ADAPTERS` in the release env file so a
deploy does not drop it.
`05-deploy.sh` applies). A flysim that refuses every checkpoint refuses to start, so a step whose
rung is not restorable is a restart instead, never a lower rung. Keep `FLY_ACCEPT_ADAPTERS` in the
release env file so a deploy does not drop it.
A milestone step runs `/opt/fly/bin/fly-loop-reset <rung>` through sudo (the one line in
`config/fly-sudoers`): stop flysim, `fly-reset-to-milestone`, start flysim. flysim is started again
even when the reset fails. The reset copies both stores to `/srv/fly/state.reset-<UTC>` first, as
in the runbook, and needs no deploy when the running release wrote the archive or
`FLY_ACCEPT_ADAPTERS` already names its adapter. After each step the helper waits up to four
minutes for `/status` to report `running` (and the target rank for a reset); otherwise the step is
recorded as failed and the ladder still climbs.
The step runs `/opt/fly/bin/fly-loop-reset <rung>` through sudo (the one line in
`config/fly-sudoers`). As root it:
- runs only the root-owned `/opt/fly/sbin/flysim` that `05-deploy.sh` installs from the release
tarball after checking it against the tarball's MANIFEST, never the fly-owned release tree;
without that copy no rung is restorable;
- reads `fly.env` as `KEY=VALUE` data, never sources it, and ignores the caller's environment;
- stops `fly-watchdog.timer` and waits for a running probe to finish, so the watchdog cannot
start flysim on a half-rewritten store; stops flysim; runs `fly-reset-to-milestone`; and on every
exit path starts flysim and the watchdog timer again.
The reset copies both stores to `/srv/fly/state.reset-<UTC>` first, as in the runbook, and
clears milestone archives above the rung. After each step the helper waits up to four minutes for
`/status` to report `running` (and the target rank for a reset). Otherwise the step is recorded
as failed and the ladder still climbs. The step is recorded before it runs, so a helper killed
mid-step has still climbed and spent the reset.
## On stream

View file

@ -12,6 +12,8 @@ script = repo / "infra/bin/fly-loop-recover"
spec = importlib.util.spec_from_loader("recover", importlib.machinery.SourceFileLoader("recover", str(script)))
recover = importlib.util.module_from_spec(spec)
spec.loader.exec_module(recover)
RESTORABLE_RUNGS = recover.restorable_rungs
WRAPPER = repo / "infra/bin/fly-loop-reset"
T0 = 1_790_000_000
@ -41,7 +43,7 @@ class RecoveryTests(unittest.TestCase):
recover.os.environ.pop(key, None)
self.acts = []
self.unrestorable = set()
restorable = patch.object(recover, "restorable", side_effect=lambda rung: rung not in self.unrestorable)
restorable = patch.object(recover, "restorable_rungs", side_effect=lambda: {1, 9, 10, 11, 12} - self.unrestorable)
restorable.start()
self.addCleanup(restorable.stop)
self.act = patch.object(recover, "act", side_effect=lambda action, target: self.acts.append((action, target)) or True)
@ -93,11 +95,16 @@ class RecoveryTests(unittest.TestCase):
self.confirm(T0, rank=11)
self.assertEqual(self.acts, [("reset", 11)])
def test_an_archive_this_build_cannot_restore_is_skipped(self):
def test_an_unrestorable_rung_below_the_best_means_a_restart_never_a_lower_rung(self):
self.unrestorable = {11}
recover.write_json(recover.STATE, {"level": 2, "bestRank": 12, "resets": [], "actedAt": 0})
self.confirm(T0)
self.assertEqual(self.acts, [("reset", 10)])
self.assertEqual(self.acts, [("restart", None)])
def test_a_fly_already_below_the_rung_under_its_best_is_restarted_not_reset(self):
recover.write_json(recover.STATE, {"level": 2, "bestRank": 12, "resets": [], "actedAt": 0})
self.confirm(T0, rank=10)
self.assertEqual(self.acts, [("restart", None)])
def test_no_restorable_archive_means_a_restart(self):
self.unrestorable = {1, 9, 10, 11, 12}
@ -207,6 +214,32 @@ class RecoveryTests(unittest.TestCase):
self.assertEqual(json.loads(recover.NOTICE.read_text())["phase"], "failed")
self.assertEqual(json.loads(recover.STATE.read_text())["level"], 1)
def test_a_step_that_raises_or_times_out_is_a_failed_step_with_state_saved(self):
self.act.stop()
with patch.object(recover.subprocess, "run", side_effect=recover.subprocess.TimeoutExpired("sudo", 1)):
self.assertIn("FAILED", self.confirm(T0))
self.act.start()
self.assertEqual(json.loads(recover.NOTICE.read_text())["phase"], "failed")
self.assertEqual(json.loads(recover.STATE.read_text())["level"], 1)
def test_state_is_saved_before_the_step_runs(self):
seen = []
self.act.stop()
with patch.object(recover, "act", side_effect=lambda a, t: seen.append(json.loads(recover.STATE.read_text())) or True):
self.confirm(T0)
self.act.start()
self.assertEqual(seen[0]["level"], 1)
self.assertNotIn("observedAt", seen[0])
def test_restorable_rungs_parses_the_list_and_is_empty_on_any_failure(self):
ok = recover.subprocess.CompletedProcess([], 0, stdout="11\n12\n", stderr="")
with patch.object(recover.subprocess, "run", return_value=ok):
self.assertEqual(RESTORABLE_RUNGS(), {11, 12})
with patch.object(recover.subprocess, "run", side_effect=OSError("no sudo")):
self.assertEqual(RESTORABLE_RUNGS(), set())
with patch.object(recover.subprocess, "run", return_value=recover.subprocess.CompletedProcess([], 1, "12", "")):
self.assertEqual(RESTORABLE_RUNGS(), set())
def test_history_records_every_step(self):
self.confirm(T0)
lines = [json.loads(line) for line in recover.HISTORY.read_text().splitlines()]
@ -219,5 +252,83 @@ class RecoveryTests(unittest.TestCase):
self.assertEqual(recover.LADDER, re.findall(r'"([^"]*)"', table))
BUILD = "lif-1ms-f64-v2/pokered-unique8-v7/abc/fly-kc-mbon-rstdp-v2"
@unittest.skipIf(recover.os.geteuid() == 0, "the wrapper ignores test overrides as root")
class WrapperTests(unittest.TestCase):
"""infra/bin/fly-loop-reset against a fake flysim, systemctl, reset tool and archives."""
def setUp(self):
self.temp = tempfile.TemporaryDirectory()
self.addCleanup(self.temp.cleanup)
root = Path(self.temp.name)
self.root = root
self.calls = root / "calls.log"
(root / "state").mkdir()
(root / "bin").mkdir()
self.script(root / "bin/systemctl", f'echo "systemctl $*" >> {self.calls}; [ "$1" != is-active ] || exit 3')
self.script(root / "flysim", f'[ "$1" = --print-compatibility ] && echo "{BUILD}"')
self.script(root / "reset", f'echo "reset $* FLY_BIN=$FLY_BIN" >> {self.calls}')
self.env_file = root / "fly.env"
self.env_file.write_text('FLY_MACRO_MODE=macros\nGAME_TITLE=Pokemon (Red) $(touch pwned)\n')
self.archive(12, BUILD)
self.archive(11, BUILD.replace("-v7", "-v6"))
self.archive(1, BUILD.replace("-v7", "-v5"))
def script(self, path, body):
path.write_text("#!/bin/sh\n" + body + "\n")
path.chmod(0o755)
def archive(self, rung, compat):
(self.root / f"state/milestone-{rung}.checkpoint").write_bytes(
b"FLYSIM01" + json.dumps({"generation": 1, "compatibility": compat}).encode() + b"\x00" * 64)
def run_wrapper(self, *args):
env = {"PATH": f"{self.root}/bin:/usr/bin:/bin", "FLY_LOOP_RESET_TEST_STATE_DIR": str(self.root / "state"),
"FLY_LOOP_RESET_TEST_ENV_FILE": str(self.env_file), "FLY_LOOP_RESET_TEST_FLYSIM": str(self.root / "flysim"),
"FLY_LOOP_RESET_TEST_RESET_BIN": str(self.root / "reset")}
return recover.subprocess.run([str(WRAPPER), *args], env=env, capture_output=True, text=True, timeout=60)
def log(self):
return self.calls.read_text() if self.calls.exists() else ""
def test_list_names_only_rungs_this_build_restores(self):
self.assertEqual(self.run_wrapper("--list").stdout.split(), ["12"])
self.env_file.write_text("FLY_ACCEPT_ADAPTERS=pokered-unique8-v6\n")
self.assertEqual(self.run_wrapper("--list").stdout.split(), ["11", "12"])
self.assertFalse((self.root / "pwned").exists())
def test_reset_pauses_the_watchdog_and_always_starts_flysim_again(self):
done = self.run_wrapper("12")
self.assertEqual(done.returncode, 0, done.stderr)
self.assertEqual([line.split()[:3] for line in self.log().splitlines() if not line.startswith("systemctl is-active")], [
["systemctl", "stop", "fly-watchdog.timer"], ["systemctl", "stop", "flysim.service"],
["reset", "12", f"FLY_BIN={self.root}/flysim"],
["systemctl", "start", "flysim.service"], ["systemctl", "start", "fly-watchdog.timer"]])
def test_a_failed_reset_still_starts_flysim_and_the_watchdog(self):
self.script(self.root / "reset", f'echo "reset $*" >> {self.calls}; exit 7')
self.assertEqual(self.run_wrapper("12").returncode, 7)
self.assertIn("systemctl start flysim.service", self.log())
self.assertIn("systemctl start fly-watchdog.timer", self.log())
def test_an_unrestorable_or_missing_rung_touches_nothing(self):
for rung in ("11", "5"):
self.assertEqual(self.run_wrapper(rung).returncode, 3)
self.assertEqual(self.log(), "")
def test_no_build_compatibility_means_nothing_is_restorable(self):
(self.root / "flysim").unlink()
self.assertEqual(self.run_wrapper("--list").stdout, "")
self.assertEqual(self.run_wrapper("12").returncode, 3)
self.assertEqual(self.log(), "")
def test_bad_arguments_are_refused(self):
for args in ((), ("--check", "12"), ("12", "13"), ("../12",), ("123",), ("-1",)):
self.assertEqual(self.run_wrapper(*args).returncode, 2, args)
self.assertEqual(self.log(), "")
if __name__ == "__main__":
unittest.main()

View file

@ -7,6 +7,7 @@ User=fly
EnvironmentFile=-/etc/fly/loop-recovery.env
# The ladder's state and history outlive a reboot (/run does not).
StateDirectory=fly-loop-recover
# A step is a 60 s on-stream countdown, up to 90 s of model calls and 4 min to verify.
TimeoutStartSec=12min
# A step: the rung list (<=2 min), up to 90 s of model calls, a 60 s on-stream countdown,
# the reset (<=8 min) and 4 min to verify. systemd must never kill the wrapper mid-reset.
TimeoutStartSec=25min
ExecStart=/opt/fly/bin/fly-loop-recover