diff --git a/infra/05-deploy.sh b/infra/05-deploy.sh index 29e6620..061e6dd 100644 --- a/infra/05-deploy.sh +++ b/infra/05-deploy.sh @@ -272,6 +272,21 @@ if [ -n "$RELEASE_TARBALL" ]; then cpu_pin chown -R fly:fly "$release_path" fi + # A root-owned flysim for fly-loop-reset, which runs as root and so must never execute the + # fly-owned copy above (the fly account could replace it). Taken from the tarball on the host + # and checked against the tarball's own MANIFEST on every deploy, so a release installed by + # an earlier deploy is covered too. + log "05-deploy: installing a root-owned flysim at /opt/fly/sbin/flysim for fly-loop-reset" + root_flysim="$(mktemp)" + tar -xzOf "$RELEASE_TARBALL" ./flysim > "$root_flysim" + want_sha="$(tar -xzOf "$RELEASE_TARBALL" ./MANIFEST | awk '$2 == "flysim" || $2 == "./flysim" {print $1; exit}')" + [ -n "$want_sha" ] && [ "$(sha256sum "$root_flysim" | cut -d' ' -f1)" = "$want_sha" ] \ + || { rm -f "$root_flysim"; die "the tarball's flysim does not match its MANIFEST; not installing /opt/fly/sbin/flysim"; } + ct_exec "$CTID" -- install -d -o root -g root -m 0755 /opt/fly/sbin + ct_push_file "$CTID" "$root_flysim" /opt/fly/sbin/flysim.new 0755 + ct_exec "$CTID" -- sh -c 'chown root:root /opt/fly/sbin/flysim.new && mv -f /opt/fly/sbin/flysim.new /opt/fly/sbin/flysim' + rm -f "$root_flysim" + log "05-deploy: overlaying infra-owned stage/serve.{mjs,sh} (docs/design/infra.md's 'own tiny static server' clarification)" converge_file "$CTID" "$INFRA_DIR/config/serve.mjs" "${release_path}/stage/serve.mjs" 0644 fly:fly >/dev/null converge_file "$CTID" "$INFRA_DIR/config/serve.sh" "${release_path}/stage/serve.sh" 0755 fly:fly >/dev/null diff --git a/infra/bin/fly-loop-recover b/infra/bin/fly-loop-recover index 1fdc355..1d46b62 100755 --- a/infra/bin/fly-loop-recover +++ b/infra/bin/fly-loop-recover @@ -32,6 +32,9 @@ MAX_RESETS = 2 # milestone resets per DAY VETO_LIMIT = 3 # model vetoes of one step before it goes ahead anyway MODEL_BUDGET = 90 # seconds across all models in one decision VERIFY_TIMEOUT = 240 +LIST_TIMEOUT = 120 +RESTART_TIMEOUT = 180 +RESET_TIMEOUT = 480 # the wrapper waits up to 60 s for the watchdog, then stop, reset, start # docs/design/ladder.md; mirrors RANK_LADDER in flybrain-gb's pokemon_red/mod.rs (a test pins it). LADDER = [ @@ -140,27 +143,30 @@ def rungs(): def plan(level, rank, best, resets_left): - """(action, target rung) for a ladder level; a spent budget or no archive falls back to restart. + """(action, target rung) for a ladder level; a spent budget or no restorable archive is a restart. - Level 1 resets to the current rung's archive, level 2 and beyond to the archive below the - best rung (never lower), so repeated days of a trap cannot walk the run down the ladder. + Level 1 resets to the current rung's archive; level 2 and beyond to the rung below the best, + and never lower, so repeated days of a trap cannot walk the run down the ladder. """ if level == 0 or not resets_left or rank is None: return "restart", None - ceiling = rank if level == 1 else min(rank, max(best, rank) - 1) - for rung in reversed([rung for rung in rungs() if rung <= ceiling]): - if restorable(rung): - return "reset", rung - return "restart", None + target = rank if level == 1 else max(best, rank) - 1 + if target > rank or target not in restorable_rungs(): + return "restart", None + return "reset", target -def restorable(rung): - """Whether the running build can restore this archive (fly-loop-reset --check).""" +def restorable_rungs(): + """The rungs the running build can restore (fly-loop-reset --list), empty on any failure.""" try: - return subprocess.run(["sudo", "-n", RESET_BIN, "--check", str(rung)], check=False, timeout=60, - stdout=subprocess.DEVNULL).returncode == 0 - except (OSError, subprocess.SubprocessError): - return False + done = subprocess.run(["sudo", "-n", RESET_BIN, "--list"], check=False, timeout=LIST_TIMEOUT, + capture_output=True, text=True) + except (OSError, subprocess.SubprocessError) as error: + print("restorable rungs unknown:", type(error).__name__, file=sys.stderr) + return set() + if done.returncode != 0: + return set() + return {int(line) for line in done.stdout.split() if line.isdigit()} def status(): @@ -184,10 +190,14 @@ def verify(target, deadline): def act(action, target): if action == "restart": - command = ["sudo", "-n", "systemctl", "restart", "flysim.service"] + command, limit = ["sudo", "-n", "systemctl", "restart", "flysim.service"], RESTART_TIMEOUT else: - command = ["sudo", "-n", RESET_BIN, str(target)] - done = subprocess.run(command, check=False, timeout=VERIFY_TIMEOUT) + command, limit = ["sudo", "-n", RESET_BIN, str(target)], RESET_TIMEOUT + try: + done = subprocess.run(command, check=False, timeout=limit) + except (OSError, subprocess.SubprocessError) as error: + print(f"{action} did not complete: {type(error).__name__}", file=sys.stderr) + return False if done.returncode != 0: return False return verify(target, time.monotonic() + VERIFY_TIMEOUT) @@ -268,15 +278,18 @@ def run(now=None, sleep=time.sleep, clock=None): base.update(toRung=target, toLabel=LADDER[target] if 0 <= target < len(LADDER) else None) notice(base, "countdown", now) sleep(COUNTDOWN) - notice(base, "acting", clock()) + started = int(clock()) + # Recorded before acting, so a helper killed mid-step has still climbed and spent the reset. + state.update(level=level + 1, actedAt=started, lastAction=action, vetoes=0) + state.pop("observedAt", None) + if action == "reset": + state["resets"].append(started) + write_json(STATE, state) + notice(base, "acting", started) ok = act(action, target) finished = int(clock()) notice(base, "done" if ok else "failed", finished) - - state.update(level=level + 1, actedAt=finished, lastAction=action, vetoes=0) - state.pop("observedAt", None) - if action == "reset": - state["resets"].append(finished) + state["actedAt"] = finished write_json(STATE, state) log({"at": finished, "event": action, "target": target, "ok": ok, "why": why, "rank": rank, "level": level, "sequence": report.get("sequence"), "map": report.get("map")}) diff --git a/infra/bin/fly-loop-reset b/infra/bin/fly-loop-reset index 99a5e4f..00d2d9c 100755 --- a/infra/bin/fly-loop-reset +++ b/infra/bin/fly-loop-reset @@ -3,47 +3,80 @@ # # fly-loop-recover runs as User=fly and reaches this through one NOPASSWD sudoers line # (config/fly-sudoers). It does what infra/docs/runbook.md "Restart the run from a rung" -# does by hand for a release whose adapter already wrote the archive: stop flysim, promote -# milestone-.checkpoint with fly-reset-to-milestone, start flysim. flysim is started -# again whether or not the reset worked, so the stream never stays down on a failure; -# fly-reset-to-milestone copies both stores aside before it rewrites anything. +# does by hand: stop flysim, promote milestone-.checkpoint with fly-reset-to-milestone, +# start flysim. # -# fly-reset-to-milestone does not check that the running build can restore the archive, and a -# flysim that refuses every checkpoint refuses to start: a black stream. So an archive is only -# used when its compatibility string equals this build's, or differs only in an adapter id that -# FLY_ACCEPT_ADAPTERS in /etc/fly/fly.env names (05-deploy.sh's adapter_migration_accepted). +# Root never executes anything the fly account can write. The release tree under +# /opt/fly/releases is fly-owned, so the flysim run here is the root-owned copy 05-deploy +# installs at /opt/fly/sbin/flysim straight from the release tarball; without it every rung is +# refused and the ladder falls back to restarts. Paths are fixed (test overrides are honoured +# only when not root), and /etc/fly/fly.env is read as KEY=VALUE data, never sourced. # -# Usage: fly-loop-reset [--check] (--check: exit 0 if restorable, 3 if not; touch nothing) +# fly-reset-to-milestone does not check that the build can restore the archive, and a flysim +# that refuses every checkpoint refuses to start: a black stream. So a rung is only used when +# its archive's compatibility string equals this build's, or differs only in an adapter id that +# FLY_ACCEPT_ADAPTERS names (05-deploy.sh's adapter_migration_accepted). +# +# fly-watchdog restarts a flysim whose /healthz fails, which during a reset would start it on a +# half-rewritten store; its timer is stopped for the reset and started again on every exit path. +# +# Usage: fly-loop-reset reset to that rung +# fly-loop-reset --list print the rungs this build can restore, one per line set -euo pipefail -: "${FLY_STATE_DIR:=/srv/fly/state}" -: "${FLY_SERVICE:=flysim.service}" -: "${FLY_RESET_BIN:=/opt/fly/bin/fly-reset-to-milestone}" -: "${FLY_ENV_FILE:=/etc/fly/fly.env}" -: "${FLY_BIN:=/opt/fly/current/flysim}" +STATE_DIR=/srv/fly/state +ENV_FILE=/etc/fly/fly.env +FLYSIM=/opt/fly/sbin/flysim +RESET_BIN=/opt/fly/bin/fly-reset-to-milestone +if [ "$(id -u)" -ne 0 ]; then + STATE_DIR="${FLY_LOOP_RESET_TEST_STATE_DIR:-$STATE_DIR}" + ENV_FILE="${FLY_LOOP_RESET_TEST_ENV_FILE:-$ENV_FILE}" + FLYSIM="${FLY_LOOP_RESET_TEST_FLYSIM:-$FLYSIM}" + RESET_BIN="${FLY_LOOP_RESET_TEST_RESET_BIN:-$RESET_BIN}" +fi +SERVICE=flysim.service +WATCHDOG_TIMER=fly-watchdog.timer +WATCHDOG_SERVICE=fly-watchdog.service +readonly STATE_DIR ENV_FILE FLYSIM RESET_BIN SERVICE WATCHDOG_TIMER WATCHDOG_SERVICE log() { echo "fly-loop-reset: $*" >&2; } +usage() { log "usage: fly-loop-reset | --list"; exit 2; } -CHECK=0 -if [ "${1:-}" = "--check" ]; then - CHECK=1 - shift +LIST=0 +RANK="" +[ "$#" -eq 1 ] || usage +if [ "$1" = --list ]; then + LIST=1 +elif [[ "$1" =~ ^[0-9]{1,2}$ ]]; then + RANK="$1" +else + usage fi -RANK="${1:-}" -if [ "$#" -ne 1 ] || ! [[ "$RANK" =~ ^[0-9]{1,2}$ ]]; then - log "usage: fly-loop-reset [--check] " - exit 2 -fi -ARCHIVE="${FLY_STATE_DIR}/milestone-${RANK}.checkpoint" -if [ ! -f "$ARCHIVE" ]; then - log "no milestone archive for rung ${RANK}; nothing touched" - exit 1 + +# FLY_* lines of fly.env as `env` arguments. Values may be quoted the systemd way; nothing +# is evaluated. +env_args=() +accepted="" +if [ -r "$ENV_FILE" ]; then + while IFS= read -r line || [ -n "$line" ]; do + [[ "$line" =~ ^(FLY_[A-Z0-9_]*)=(.*)$ ]] || continue + key="${BASH_REMATCH[1]}" + value="${BASH_REMATCH[2]}" + if [[ "$value" =~ ^\"(.*)\"$ ]] || [[ "$value" =~ ^\'(.*)\'$ ]]; then + value="${BASH_REMATCH[1]}" + fi + env_args+=("$key=$value") + if [ "$key" = FLY_ACCEPT_ADAPTERS ]; then + accepted="$value" + fi + done < "$ENV_FILE" fi +clean_env() { env -i PATH=/usr/sbin:/usr/bin:/sbin:/bin HOME=/root "${env_args[@]}" "$@"; } # The same rule as 05-deploy.sh: exactly one '/'-separated field differs, it is the adapter # (index 1), and the archive's adapter id is listed. migration_accepted() { - local old="$1" new="$2" accepted="$3" i differing=0 index=-1 entry + local old="$1" new="$2" i differing=0 index=-1 entry local -a old_parts new_parts IFS='/' read -r -a old_parts <<< "$old" IFS='/' read -r -a new_parts <<< "$new" @@ -61,30 +94,63 @@ migration_accepted() { return 1 } -accepted="" -if [ -r "$FLY_ENV_FILE" ]; then - set -a - # shellcheck disable=SC1090 - . "$FLY_ENV_FILE" - set +a - accepted="${FLY_ACCEPT_ADAPTERS:-}" +build_compat="" +if [ -x "$FLYSIM" ] && [ "$(stat -c %u "$FLYSIM")" = "$(id -u)" ]; then + build_compat="$(clean_env timeout 90 "$FLYSIM" --print-compatibility 2>/dev/null | tail -n1 || true)" fi -archive_compat="$(head -c 262144 "$ARCHIVE" 2>/dev/null | grep -a -o -m1 '"compatibility":"[^"]*"' | head -n1 | cut -d'"' -f4 || true)" -build_compat="$("$FLY_BIN" --print-compatibility 2>/dev/null | tail -n1 || true)" -if [ -z "$archive_compat" ] || [ -z "$build_compat" ]; then - log "cannot read the compatibility of rung ${RANK}'s archive or of this build; nothing touched" +if [ -z "$build_compat" ]; then + log "no compatibility string from $FLYSIM (missing, not owned by $(id -un), or failed); no rung is restorable" + [ "$LIST" -eq 1 ] && exit 0 exit 3 fi -if [ "$archive_compat" != "$build_compat" ] && ! migration_accepted "$archive_compat" "$build_compat" "$accepted"; then - log "rung ${RANK}'s archive ($(echo "$archive_compat" | cut -d/ -f2)) is not restorable by this build ($(echo "$build_compat" | cut -d/ -f2)), FLY_ACCEPT_ADAPTERS='${accepted}'; nothing touched" - exit 3 -fi -[ "$CHECK" -eq 0 ] || exit 0 -log "stopping ${FLY_SERVICE} to reset to rung ${RANK}" -systemctl stop "$FLY_SERVICE" +restorable() { + local archive="${STATE_DIR}/milestone-$1.checkpoint" compat + [ -f "$archive" ] || return 1 + compat="$(head -c 262144 "$archive" 2>/dev/null | grep -a -o -m1 '"compatibility"[[:space:]]*:[[:space:]]*"[^"]*"' | head -n1 | cut -d'"' -f4 || true)" + [ -n "$compat" ] || return 1 + [ "$compat" = "$build_compat" ] || migration_accepted "$compat" "$build_compat" +} + +if [ "$LIST" -eq 1 ]; then + for archive in "${STATE_DIR}"/milestone-*.checkpoint; do + [ -e "$archive" ] || continue + rung="${archive##*/milestone-}" + rung="${rung%.checkpoint}" + if [[ "$rung" =~ ^[0-9]{1,2}$ ]] && restorable "$rung"; then + echo "$rung" + fi + done | sort -n + exit 0 +fi + +if ! restorable "$RANK"; then + log "rung ${RANK}'s archive is missing or not restorable by this build (FLY_ACCEPT_ADAPTERS='${accepted}'); nothing touched" + exit 3 +fi + +# shellcheck disable=SC2317 # invoked by the EXIT trap +finish() { + local status=$? + systemctl start "$SERVICE" || log "starting ${SERVICE} failed" + systemctl start "$WATCHDOG_TIMER" || log "starting ${WATCHDOG_TIMER} failed" + exit "$status" +} +trap finish EXIT +trap 'exit 143' TERM INT HUP + +log "pausing ${WATCHDOG_TIMER} and stopping ${SERVICE} to reset to rung ${RANK}" +systemctl stop "$WATCHDOG_TIMER" +for _ in $(seq 1 60); do + systemctl is-active --quiet "$WATCHDOG_SERVICE" || break + sleep 1 +done +if systemctl is-active --quiet "$WATCHDOG_SERVICE"; then + log "${WATCHDOG_SERVICE} still running after 60 s; not resetting" + exit 4 +fi +systemctl stop "$SERVICE" status=0 -"$FLY_RESET_BIN" "$RANK" || status=$? -[ "$status" -eq 0 ] || log "fly-reset-to-milestone ${RANK} failed (exit ${status}); starting ${FLY_SERVICE} on the state it left" -systemctl start "$FLY_SERVICE" +clean_env FLY_BIN="$FLYSIM" "$RESET_BIN" "$RANK" || status=$? +[ "$status" -eq 0 ] || log "fly-reset-to-milestone ${RANK} failed (exit ${status}); starting ${SERVICE} on the state it left" exit "$status" diff --git a/infra/docs/loop-recovery.md b/infra/docs/loop-recovery.md index cff1c24..953353a 100644 --- a/infra/docs/loop-recovery.md +++ b/infra/docs/loop-recovery.md @@ -19,21 +19,29 @@ a ladder one step per trap that outlives the previous step: - **Starting over:** the ladder returns to level 0 when the fly reaches a new best rung, or after six hours with no suspected report. -A milestone step only uses an archive the running build can restore: `fly-loop-reset --check` -compares the archive's compatibility string with `flysim --print-compatibility` and accepts an +A milestone step only uses an archive the running build can restore. `fly-loop-reset --list` +compares each archive's compatibility string with `flysim --print-compatibility` and accepts an adapter-only difference that `FLY_ACCEPT_ADAPTERS` in `/etc/fly/fly.env` names (the rule -`05-deploy.sh` applies). A flysim that refuses every checkpoint refuses to start, so an archive -from an older adapter is skipped for the next one down unless the deploy named its adapter; with -none restorable the step is a restart. Keep `FLY_ACCEPT_ADAPTERS` in the release env file so a -deploy does not drop it. +`05-deploy.sh` applies). A flysim that refuses every checkpoint refuses to start, so a step whose +rung is not restorable is a restart instead, never a lower rung. Keep `FLY_ACCEPT_ADAPTERS` in the +release env file so a deploy does not drop it. -A milestone step runs `/opt/fly/bin/fly-loop-reset ` through sudo (the one line in -`config/fly-sudoers`): stop flysim, `fly-reset-to-milestone`, start flysim. flysim is started again -even when the reset fails. The reset copies both stores to `/srv/fly/state.reset-` first, as -in the runbook, and needs no deploy when the running release wrote the archive or -`FLY_ACCEPT_ADAPTERS` already names its adapter. After each step the helper waits up to four -minutes for `/status` to report `running` (and the target rank for a reset); otherwise the step is -recorded as failed and the ladder still climbs. +The step runs `/opt/fly/bin/fly-loop-reset ` through sudo (the one line in +`config/fly-sudoers`). As root it: + +- runs only the root-owned `/opt/fly/sbin/flysim` that `05-deploy.sh` installs from the release + tarball after checking it against the tarball's MANIFEST, never the fly-owned release tree; + without that copy no rung is restorable; +- reads `fly.env` as `KEY=VALUE` data, never sources it, and ignores the caller's environment; +- stops `fly-watchdog.timer` and waits for a running probe to finish, so the watchdog cannot + start flysim on a half-rewritten store; stops flysim; runs `fly-reset-to-milestone`; and on every + exit path starts flysim and the watchdog timer again. + +The reset copies both stores to `/srv/fly/state.reset-` first, as in the runbook, and +clears milestone archives above the rung. After each step the helper waits up to four minutes for +`/status` to report `running` (and the target rank for a reset). Otherwise the step is recorded +as failed and the ladder still climbs. The step is recorded before it runs, so a helper killed +mid-step has still climbed and spent the reset. ## On stream diff --git a/infra/tests/test_loop_recover.py b/infra/tests/test_loop_recover.py index 264a3a7..405135c 100644 --- a/infra/tests/test_loop_recover.py +++ b/infra/tests/test_loop_recover.py @@ -12,6 +12,8 @@ script = repo / "infra/bin/fly-loop-recover" spec = importlib.util.spec_from_loader("recover", importlib.machinery.SourceFileLoader("recover", str(script))) recover = importlib.util.module_from_spec(spec) spec.loader.exec_module(recover) +RESTORABLE_RUNGS = recover.restorable_rungs +WRAPPER = repo / "infra/bin/fly-loop-reset" T0 = 1_790_000_000 @@ -41,7 +43,7 @@ class RecoveryTests(unittest.TestCase): recover.os.environ.pop(key, None) self.acts = [] self.unrestorable = set() - restorable = patch.object(recover, "restorable", side_effect=lambda rung: rung not in self.unrestorable) + restorable = patch.object(recover, "restorable_rungs", side_effect=lambda: {1, 9, 10, 11, 12} - self.unrestorable) restorable.start() self.addCleanup(restorable.stop) self.act = patch.object(recover, "act", side_effect=lambda action, target: self.acts.append((action, target)) or True) @@ -93,11 +95,16 @@ class RecoveryTests(unittest.TestCase): self.confirm(T0, rank=11) self.assertEqual(self.acts, [("reset", 11)]) - def test_an_archive_this_build_cannot_restore_is_skipped(self): + def test_an_unrestorable_rung_below_the_best_means_a_restart_never_a_lower_rung(self): self.unrestorable = {11} recover.write_json(recover.STATE, {"level": 2, "bestRank": 12, "resets": [], "actedAt": 0}) self.confirm(T0) - self.assertEqual(self.acts, [("reset", 10)]) + self.assertEqual(self.acts, [("restart", None)]) + + def test_a_fly_already_below_the_rung_under_its_best_is_restarted_not_reset(self): + recover.write_json(recover.STATE, {"level": 2, "bestRank": 12, "resets": [], "actedAt": 0}) + self.confirm(T0, rank=10) + self.assertEqual(self.acts, [("restart", None)]) def test_no_restorable_archive_means_a_restart(self): self.unrestorable = {1, 9, 10, 11, 12} @@ -207,6 +214,32 @@ class RecoveryTests(unittest.TestCase): self.assertEqual(json.loads(recover.NOTICE.read_text())["phase"], "failed") self.assertEqual(json.loads(recover.STATE.read_text())["level"], 1) + def test_a_step_that_raises_or_times_out_is_a_failed_step_with_state_saved(self): + self.act.stop() + with patch.object(recover.subprocess, "run", side_effect=recover.subprocess.TimeoutExpired("sudo", 1)): + self.assertIn("FAILED", self.confirm(T0)) + self.act.start() + self.assertEqual(json.loads(recover.NOTICE.read_text())["phase"], "failed") + self.assertEqual(json.loads(recover.STATE.read_text())["level"], 1) + + def test_state_is_saved_before_the_step_runs(self): + seen = [] + self.act.stop() + with patch.object(recover, "act", side_effect=lambda a, t: seen.append(json.loads(recover.STATE.read_text())) or True): + self.confirm(T0) + self.act.start() + self.assertEqual(seen[0]["level"], 1) + self.assertNotIn("observedAt", seen[0]) + + def test_restorable_rungs_parses_the_list_and_is_empty_on_any_failure(self): + ok = recover.subprocess.CompletedProcess([], 0, stdout="11\n12\n", stderr="") + with patch.object(recover.subprocess, "run", return_value=ok): + self.assertEqual(RESTORABLE_RUNGS(), {11, 12}) + with patch.object(recover.subprocess, "run", side_effect=OSError("no sudo")): + self.assertEqual(RESTORABLE_RUNGS(), set()) + with patch.object(recover.subprocess, "run", return_value=recover.subprocess.CompletedProcess([], 1, "12", "")): + self.assertEqual(RESTORABLE_RUNGS(), set()) + def test_history_records_every_step(self): self.confirm(T0) lines = [json.loads(line) for line in recover.HISTORY.read_text().splitlines()] @@ -219,5 +252,83 @@ class RecoveryTests(unittest.TestCase): self.assertEqual(recover.LADDER, re.findall(r'"([^"]*)"', table)) +BUILD = "lif-1ms-f64-v2/pokered-unique8-v7/abc/fly-kc-mbon-rstdp-v2" + + +@unittest.skipIf(recover.os.geteuid() == 0, "the wrapper ignores test overrides as root") +class WrapperTests(unittest.TestCase): + """infra/bin/fly-loop-reset against a fake flysim, systemctl, reset tool and archives.""" + + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + root = Path(self.temp.name) + self.root = root + self.calls = root / "calls.log" + (root / "state").mkdir() + (root / "bin").mkdir() + self.script(root / "bin/systemctl", f'echo "systemctl $*" >> {self.calls}; [ "$1" != is-active ] || exit 3') + self.script(root / "flysim", f'[ "$1" = --print-compatibility ] && echo "{BUILD}"') + self.script(root / "reset", f'echo "reset $* FLY_BIN=$FLY_BIN" >> {self.calls}') + self.env_file = root / "fly.env" + self.env_file.write_text('FLY_MACRO_MODE=macros\nGAME_TITLE=Pokemon (Red) $(touch pwned)\n') + self.archive(12, BUILD) + self.archive(11, BUILD.replace("-v7", "-v6")) + self.archive(1, BUILD.replace("-v7", "-v5")) + + def script(self, path, body): + path.write_text("#!/bin/sh\n" + body + "\n") + path.chmod(0o755) + + def archive(self, rung, compat): + (self.root / f"state/milestone-{rung}.checkpoint").write_bytes( + b"FLYSIM01" + json.dumps({"generation": 1, "compatibility": compat}).encode() + b"\x00" * 64) + + def run_wrapper(self, *args): + env = {"PATH": f"{self.root}/bin:/usr/bin:/bin", "FLY_LOOP_RESET_TEST_STATE_DIR": str(self.root / "state"), + "FLY_LOOP_RESET_TEST_ENV_FILE": str(self.env_file), "FLY_LOOP_RESET_TEST_FLYSIM": str(self.root / "flysim"), + "FLY_LOOP_RESET_TEST_RESET_BIN": str(self.root / "reset")} + return recover.subprocess.run([str(WRAPPER), *args], env=env, capture_output=True, text=True, timeout=60) + + def log(self): + return self.calls.read_text() if self.calls.exists() else "" + + def test_list_names_only_rungs_this_build_restores(self): + self.assertEqual(self.run_wrapper("--list").stdout.split(), ["12"]) + self.env_file.write_text("FLY_ACCEPT_ADAPTERS=pokered-unique8-v6\n") + self.assertEqual(self.run_wrapper("--list").stdout.split(), ["11", "12"]) + self.assertFalse((self.root / "pwned").exists()) + + def test_reset_pauses_the_watchdog_and_always_starts_flysim_again(self): + done = self.run_wrapper("12") + self.assertEqual(done.returncode, 0, done.stderr) + self.assertEqual([line.split()[:3] for line in self.log().splitlines() if not line.startswith("systemctl is-active")], [ + ["systemctl", "stop", "fly-watchdog.timer"], ["systemctl", "stop", "flysim.service"], + ["reset", "12", f"FLY_BIN={self.root}/flysim"], + ["systemctl", "start", "flysim.service"], ["systemctl", "start", "fly-watchdog.timer"]]) + + def test_a_failed_reset_still_starts_flysim_and_the_watchdog(self): + self.script(self.root / "reset", f'echo "reset $*" >> {self.calls}; exit 7') + self.assertEqual(self.run_wrapper("12").returncode, 7) + self.assertIn("systemctl start flysim.service", self.log()) + self.assertIn("systemctl start fly-watchdog.timer", self.log()) + + def test_an_unrestorable_or_missing_rung_touches_nothing(self): + for rung in ("11", "5"): + self.assertEqual(self.run_wrapper(rung).returncode, 3) + self.assertEqual(self.log(), "") + + def test_no_build_compatibility_means_nothing_is_restorable(self): + (self.root / "flysim").unlink() + self.assertEqual(self.run_wrapper("--list").stdout, "") + self.assertEqual(self.run_wrapper("12").returncode, 3) + self.assertEqual(self.log(), "") + + def test_bad_arguments_are_refused(self): + for args in ((), ("--check", "12"), ("12", "13"), ("../12",), ("123",), ("-1",)): + self.assertEqual(self.run_wrapper(*args).returncode, 2, args) + self.assertEqual(self.log(), "") + + if __name__ == "__main__": unittest.main() diff --git a/infra/units/fly-loop-recover.service b/infra/units/fly-loop-recover.service index a2dff4d..dc93b16 100644 --- a/infra/units/fly-loop-recover.service +++ b/infra/units/fly-loop-recover.service @@ -7,6 +7,7 @@ User=fly EnvironmentFile=-/etc/fly/loop-recovery.env # The ladder's state and history outlive a reboot (/run does not). StateDirectory=fly-loop-recover -# A step is a 60 s on-stream countdown, up to 90 s of model calls and 4 min to verify. -TimeoutStartSec=12min +# A step: the rung list (<=2 min), up to 90 s of model calls, a 60 s on-stream countdown, +# the reset (<=8 min) and 4 min to verify. systemd must never kill the wrapper mid-reset. +TimeoutStartSec=25min ExecStart=/opt/fly/bin/fly-loop-recover