From 432991cbc5765b546fc218df518fdca390a2b2e5 Mon Sep 17 00:00:00 2001 From: acamilo Date: Sat, 26 Sep 2026 01:12:46 +0000 Subject: [PATCH] Add conservative automated loop recovery timer --- infra/05-deploy.sh | 2 +- infra/07-enable.sh | 2 +- infra/bin/fly-loop-recover | 66 +++++++++++++++++++++ infra/docs/loop-recovery.md | 29 ++++++++++ infra/tests/test_loop_recover.py | 85 ++++++++++++++++++++++++++++ infra/units/fly-loop-recover.service | 8 +++ infra/units/fly-loop-recover.timer | 11 ++++ 7 files changed, 201 insertions(+), 2 deletions(-) create mode 100755 infra/bin/fly-loop-recover create mode 100644 infra/docs/loop-recovery.md create mode 100644 infra/tests/test_loop_recover.py create mode 100644 infra/units/fly-loop-recover.service create mode 100644 infra/units/fly-loop-recover.timer diff --git a/infra/05-deploy.sh b/infra/05-deploy.sh index c6cd3db..5d79ce4 100644 --- a/infra/05-deploy.sh +++ b/infra/05-deploy.sh @@ -685,7 +685,7 @@ fi # --------------------------------------------------------------------------- log "05-deploy: converging bin/ helpers to /opt/fly/bin" ct_exec "$CTID" -- mkdir -p /opt/fly/bin -for name in fly-watchdog fly-recap fly-retention fly-reset-to-milestone flypush flystage-launch flycast-launch wait-for-x wait-for-stage wait-for-health; do +for name in fly-watchdog fly-loop-recover fly-recap fly-retention fly-reset-to-milestone flypush flystage-launch flycast-launch wait-for-x wait-for-stage wait-for-health; do converge_file "$CTID" "$INFRA_DIR/bin/$name" "/opt/fly/bin/$name" 0755 root:root >/dev/null done diff --git a/infra/07-enable.sh b/infra/07-enable.sh index 03cf184..59b9590 100755 --- a/infra/07-enable.sh +++ b/infra/07-enable.sh @@ -27,7 +27,7 @@ require_pve_host need pct ALWAYS_ON_UNITS="xvfb.service pulse.service mediamtx.service flysim.service flystage-web.service flystage.service flycast.service flybridge.service" -ALWAYS_ON_TIMERS="fly-recap.timer fly-retention.timer fly-watchdog.timer" +ALWAYS_ON_TIMERS="fly-recap.timer fly-retention.timer fly-watchdog.timer fly-loop-recover.timer" log "07-enable: enabling app units (not yet starting)" for u in $ALWAYS_ON_UNITS; do diff --git a/infra/bin/fly-loop-recover b/infra/bin/fly-loop-recover new file mode 100755 index 0000000..404ad0b --- /dev/null +++ b/infra/bin/fly-loop-recover @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Conservative, out-of-process loop recovery; invoked after the watchdog probe.""" +import json +import os +from pathlib import Path +import subprocess +import sys +import time +from urllib.request import Request, urlopen + +REPORT = Path(os.environ.get("FLY_LOOP_REPORT", "/run/fly/wd/loop.json")) +STATE = Path(os.environ.get("FLY_LOOP_RECOVERY_STATE", "/run/fly/wd/recovery.json")) +INTERVAL = int(os.environ.get("WD_LOOP_INTERVAL", "300")) +COOLDOWN = 3600 + + +def verdict(report): + url = os.environ.get("FLY_LOOP_ROUTER_URL") + model = os.environ.get("FLY_LOOP_MODEL") + if not url or not model: + return True + payload = {"model": model, "temperature": 0, "messages": [ + {"role": "system", "content": "Classify whether a suspected repeating game macro loop is truly stuck. Return only JSON {\"stuck\":true|false}. No instructions or commands."}, + {"role": "user", "content": json.dumps({key: report.get(key) for key in ("reason", "sequence", "window", "places", "milestone")})}, + ]} + headers = {"Content-Type": "application/json"} + key = os.environ.get("FLY_LOOP_ROUTER_KEY") + if key: + headers["Authorization"] = "Bearer " + key + request = Request(url.rstrip("/") + "/chat/completions", json.dumps(payload).encode(), headers) + with urlopen(request, timeout=20) as response: + answer = json.load(response) + decision = json.loads(answer["choices"][0]["message"]["content"]) + return type(decision) is dict and type(decision.get("stuck")) is bool and decision["stuck"] + + +def run(now=None): + now = int(time.time() if now is None else now) + report = json.loads(REPORT.read_text()) + previous = json.loads(STATE.read_text()) if STATE.exists() else {} + at = report.get("at") + if (report.get("suspected") != 1 or type(at) is not int + or not 0 <= now - at <= 600 or report.get("action") != "none"): + STATE.write_text(json.dumps({key: previous[key] for key in ("restarted_at",) if key in previous}) + "\n") + return "not a fresh suspected loop" + if "restarted_at" in previous and now - previous["restarted_at"] < COOLDOWN: + return "recovery cooldown" + if previous.get("observed_at") == at: + return "waiting for next probe" + if not previous.get("observed_at") or at - previous["observed_at"] < INTERVAL - 30: + STATE.write_text(json.dumps({"observed_at": at}) + "\n") + return "waiting for second probe" + try: + approved = verdict(report) + except (OSError, ValueError, KeyError, IndexError) as error: + print("router unavailable or invalid:", type(error).__name__, file=sys.stderr) + return "router unavailable" + if not approved: + return "router did not confirm" + subprocess.run(["sudo", "-n", "systemctl", "restart", "flysim.service"], check=True) + STATE.write_text(json.dumps({"observed_at": at, "restarted_at": now}) + "\n") + return "flysim restarted; checkpoint restore keeps the rung" + + +if __name__ == "__main__": + print(run()) diff --git a/infra/docs/loop-recovery.md b/infra/docs/loop-recovery.md new file mode 100644 index 0000000..41c01df --- /dev/null +++ b/infra/docs/loop-recovery.md @@ -0,0 +1,29 @@ +# Automated loop recovery + +`fly-watchdog` remains report-only. The separate `fly-loop-recover.timer` checks its +`/run/fly/wd/loop.json` every five minutes. Two fresh suspected reports from +separate watchdog probes, at least ~5 minutes apart, cause **only** a +`flysim.service` restart. The ordinary checkpoint restore keeps the current +rung and learned brain state, but clears session macro ledgers. This does not +press buttons or promote a milestone archive. A successful restart imposes a +one-hour cooldown, including across intervening clear reports. The timer's +journal records every decision; inspect it with +`journalctl -u fly-loop-recover.service`. Disable the timer to stop automatic +recovery: `systemctl disable --now fly-loop-recover.timer`. + +By default confirmation is deterministic, using watchdog's existing signal. +An optional OpenAI-compatible local router can veto a recovery: provision a +root-managed `/etc/fly/loop-recovery.env` readable by the `fly` account, +containing `FLY_LOOP_ROUTER_URL` (base URL ending in `/v1`) and +`FLY_LOOP_MODEL` (an available free-tier model). A private router can additionally +use `FLY_LOOP_ROUTER_KEY`; provision it outside this public checkout and limit +file permissions to `0640 root:fly`. Never put credentials in the unit or the +repository. When a router is configured, malformed or unavailable responses +prevent recovery; the model may only return `{"stuck": true|false}` and cannot +choose commands, buttons, or checkpoint paths. Confirm the model actually +exists and is reachable from the release container before configuring it. + +This is an unstick mechanism, not a macro bug fix. A recurring trap still needs +the checkpoint-based loop review in `docs/loop-review.md`. The live recovery +must be recorded in the host claim log by the operator reviewing the unit +journal; the unit itself has no host claim-log access. diff --git a/infra/tests/test_loop_recover.py b/infra/tests/test_loop_recover.py new file mode 100644 index 0000000..190b42b --- /dev/null +++ b/infra/tests/test_loop_recover.py @@ -0,0 +1,85 @@ +import importlib.machinery +import importlib.util +import json +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +script = Path(__file__).resolve().parents[1] / "bin/fly-loop-recover" +spec = importlib.util.spec_from_loader("recover", importlib.machinery.SourceFileLoader("recover", str(script))) +recover = importlib.util.module_from_spec(spec) +spec.loader.exec_module(recover) + + +class RecoveryTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + root = Path(self.temp.name) + recover.REPORT = root / "loop.json" + recover.STATE = root / "recovery.json" + self.report = {"suspected": 1, "at": 1000, "action": "none", "reason": "unrewarded"} + self.save() + + def save(self): + recover.REPORT.write_text(json.dumps(self.report)) + + def test_two_probes_and_cooldown(self): + with patch.object(recover, "verdict", return_value=True), patch.object(recover.subprocess, "run") as restart: + self.assertIn("second probe", recover.run(1001)) + self.assertIn("next probe", recover.run(1002)) + self.report["at"] = 1300 + self.save() + self.assertIn("restarted", recover.run(1301)) + restart.assert_called_once_with(["sudo", "-n", "systemctl", "restart", "flysim.service"], check=True) + self.assertIn("cooldown", recover.run(1302)) + + def test_stale_and_clear_never_restart(self): + with patch.object(recover.subprocess, "run") as restart: + self.assertIn("not a fresh", recover.run(1700)) + self.report.update(at=1700, suspected=0) + self.save() + self.assertIn("not a fresh", recover.run(1700)) + restart.assert_not_called() + + def test_router_failure_does_not_act(self): + recover.run(1000) + self.report["at"] = 1300 + self.save() + with patch.object(recover, "verdict", side_effect=ValueError("bad response")), patch.object(recover.subprocess, "run") as restart: + self.assertEqual(recover.run(1300), "router unavailable") + restart.assert_not_called() + + def test_cleared_probe_preserves_cooldown(self): + with patch.object(recover, "verdict", return_value=True), patch.object(recover.subprocess, "run") as restart: + recover.run(1000) + self.report["at"] = 1300 + self.save() + recover.run(1300) + self.report.update(at=1600, suspected=0) + self.save() + recover.run(1600) + self.report.update(at=1900, suspected=1) + self.save() + recover.run(1900) + self.report["at"] = 2200 + self.save() + self.assertIn("cooldown", recover.run(2200)) + restart.assert_called_once() + + def test_router_rejection_does_not_restart(self): + recover.run(1000) + self.report["at"] = 1300 + self.save() + with patch.object(recover, "verdict", return_value=False), patch.object(recover.subprocess, "run") as restart: + self.assertIn("did not confirm", recover.run(1300)) + restart.assert_not_called() + + def test_unconfigured_router_uses_deterministic_confirmation(self): + with patch.dict("os.environ", {}, clear=True): + self.assertTrue(recover.verdict(self.report)) + + +if __name__ == "__main__": + unittest.main() diff --git a/infra/units/fly-loop-recover.service b/infra/units/fly-loop-recover.service new file mode 100644 index 0000000..429fb2b --- /dev/null +++ b/infra/units/fly-loop-recover.service @@ -0,0 +1,8 @@ +[Unit] +Description=Check confirmed macro loops for minimal flysim recovery + +[Service] +Type=oneshot +User=fly +EnvironmentFile=-/etc/fly/loop-recovery.env +ExecStart=/opt/fly/bin/fly-loop-recover diff --git a/infra/units/fly-loop-recover.timer b/infra/units/fly-loop-recover.timer new file mode 100644 index 0000000..59f5ce5 --- /dev/null +++ b/infra/units/fly-loop-recover.timer @@ -0,0 +1,11 @@ +[Unit] +Description=Check for repeated macro loops after watchdog probes + +[Timer] +OnBootSec=7min +OnUnitActiveSec=5min +AccuracySec=10s +Unit=fly-loop-recover.service + +[Install] +WantedBy=timers.target