Add conservative automated loop recovery timer

This commit is contained in:
acamilo 2026-09-26 01:12:46 +00:00
parent 862e343e1c
commit 432991cbc5
7 changed files with 201 additions and 2 deletions

View file

@ -685,7 +685,7 @@ fi
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
log "05-deploy: converging bin/ helpers to /opt/fly/bin" log "05-deploy: converging bin/ helpers to /opt/fly/bin"
ct_exec "$CTID" -- mkdir -p /opt/fly/bin ct_exec "$CTID" -- mkdir -p /opt/fly/bin
for name in fly-watchdog fly-recap fly-retention fly-reset-to-milestone flypush flystage-launch flycast-launch wait-for-x wait-for-stage wait-for-health; do for name in fly-watchdog fly-loop-recover fly-recap fly-retention fly-reset-to-milestone flypush flystage-launch flycast-launch wait-for-x wait-for-stage wait-for-health; do
converge_file "$CTID" "$INFRA_DIR/bin/$name" "/opt/fly/bin/$name" 0755 root:root >/dev/null converge_file "$CTID" "$INFRA_DIR/bin/$name" "/opt/fly/bin/$name" 0755 root:root >/dev/null
done done

View file

@ -27,7 +27,7 @@ require_pve_host
need pct need pct
ALWAYS_ON_UNITS="xvfb.service pulse.service mediamtx.service flysim.service flystage-web.service flystage.service flycast.service flybridge.service" ALWAYS_ON_UNITS="xvfb.service pulse.service mediamtx.service flysim.service flystage-web.service flystage.service flycast.service flybridge.service"
ALWAYS_ON_TIMERS="fly-recap.timer fly-retention.timer fly-watchdog.timer" ALWAYS_ON_TIMERS="fly-recap.timer fly-retention.timer fly-watchdog.timer fly-loop-recover.timer"
log "07-enable: enabling app units (not yet starting)" log "07-enable: enabling app units (not yet starting)"
for u in $ALWAYS_ON_UNITS; do for u in $ALWAYS_ON_UNITS; do

66
infra/bin/fly-loop-recover Executable file
View file

@ -0,0 +1,66 @@
#!/usr/bin/env python3
"""Conservative, out-of-process loop recovery; invoked after the watchdog probe."""
import json
import os
from pathlib import Path
import subprocess
import sys
import time
from urllib.request import Request, urlopen
REPORT = Path(os.environ.get("FLY_LOOP_REPORT", "/run/fly/wd/loop.json"))
STATE = Path(os.environ.get("FLY_LOOP_RECOVERY_STATE", "/run/fly/wd/recovery.json"))
INTERVAL = int(os.environ.get("WD_LOOP_INTERVAL", "300"))
COOLDOWN = 3600
def verdict(report):
url = os.environ.get("FLY_LOOP_ROUTER_URL")
model = os.environ.get("FLY_LOOP_MODEL")
if not url or not model:
return True
payload = {"model": model, "temperature": 0, "messages": [
{"role": "system", "content": "Classify whether a suspected repeating game macro loop is truly stuck. Return only JSON {\"stuck\":true|false}. No instructions or commands."},
{"role": "user", "content": json.dumps({key: report.get(key) for key in ("reason", "sequence", "window", "places", "milestone")})},
]}
headers = {"Content-Type": "application/json"}
key = os.environ.get("FLY_LOOP_ROUTER_KEY")
if key:
headers["Authorization"] = "Bearer " + key
request = Request(url.rstrip("/") + "/chat/completions", json.dumps(payload).encode(), headers)
with urlopen(request, timeout=20) as response:
answer = json.load(response)
decision = json.loads(answer["choices"][0]["message"]["content"])
return type(decision) is dict and type(decision.get("stuck")) is bool and decision["stuck"]
def run(now=None):
now = int(time.time() if now is None else now)
report = json.loads(REPORT.read_text())
previous = json.loads(STATE.read_text()) if STATE.exists() else {}
at = report.get("at")
if (report.get("suspected") != 1 or type(at) is not int
or not 0 <= now - at <= 600 or report.get("action") != "none"):
STATE.write_text(json.dumps({key: previous[key] for key in ("restarted_at",) if key in previous}) + "\n")
return "not a fresh suspected loop"
if "restarted_at" in previous and now - previous["restarted_at"] < COOLDOWN:
return "recovery cooldown"
if previous.get("observed_at") == at:
return "waiting for next probe"
if not previous.get("observed_at") or at - previous["observed_at"] < INTERVAL - 30:
STATE.write_text(json.dumps({"observed_at": at}) + "\n")
return "waiting for second probe"
try:
approved = verdict(report)
except (OSError, ValueError, KeyError, IndexError) as error:
print("router unavailable or invalid:", type(error).__name__, file=sys.stderr)
return "router unavailable"
if not approved:
return "router did not confirm"
subprocess.run(["sudo", "-n", "systemctl", "restart", "flysim.service"], check=True)
STATE.write_text(json.dumps({"observed_at": at, "restarted_at": now}) + "\n")
return "flysim restarted; checkpoint restore keeps the rung"
if __name__ == "__main__":
print(run())

View file

@ -0,0 +1,29 @@
# Automated loop recovery
`fly-watchdog` remains report-only. The separate `fly-loop-recover.timer` checks its
`/run/fly/wd/loop.json` every five minutes. Two fresh suspected reports from
separate watchdog probes, at least ~5 minutes apart, cause **only** a
`flysim.service` restart. The ordinary checkpoint restore keeps the current
rung and learned brain state, but clears session macro ledgers. This does not
press buttons or promote a milestone archive. A successful restart imposes a
one-hour cooldown, including across intervening clear reports. The timer's
journal records every decision; inspect it with
`journalctl -u fly-loop-recover.service`. Disable the timer to stop automatic
recovery: `systemctl disable --now fly-loop-recover.timer`.
By default confirmation is deterministic, using watchdog's existing signal.
An optional OpenAI-compatible local router can veto a recovery: provision a
root-managed `/etc/fly/loop-recovery.env` readable by the `fly` account,
containing `FLY_LOOP_ROUTER_URL` (base URL ending in `/v1`) and
`FLY_LOOP_MODEL` (an available free-tier model). A private router can additionally
use `FLY_LOOP_ROUTER_KEY`; provision it outside this public checkout and limit
file permissions to `0640 root:fly`. Never put credentials in the unit or the
repository. When a router is configured, malformed or unavailable responses
prevent recovery; the model may only return `{"stuck": true|false}` and cannot
choose commands, buttons, or checkpoint paths. Confirm the model actually
exists and is reachable from the release container before configuring it.
This is an unstick mechanism, not a macro bug fix. A recurring trap still needs
the checkpoint-based loop review in `docs/loop-review.md`. The live recovery
must be recorded in the host claim log by the operator reviewing the unit
journal; the unit itself has no host claim-log access.

View file

@ -0,0 +1,85 @@
import importlib.machinery
import importlib.util
import json
from pathlib import Path
import tempfile
import unittest
from unittest.mock import patch
script = Path(__file__).resolve().parents[1] / "bin/fly-loop-recover"
spec = importlib.util.spec_from_loader("recover", importlib.machinery.SourceFileLoader("recover", str(script)))
recover = importlib.util.module_from_spec(spec)
spec.loader.exec_module(recover)
class RecoveryTests(unittest.TestCase):
def setUp(self):
self.temp = tempfile.TemporaryDirectory()
self.addCleanup(self.temp.cleanup)
root = Path(self.temp.name)
recover.REPORT = root / "loop.json"
recover.STATE = root / "recovery.json"
self.report = {"suspected": 1, "at": 1000, "action": "none", "reason": "unrewarded"}
self.save()
def save(self):
recover.REPORT.write_text(json.dumps(self.report))
def test_two_probes_and_cooldown(self):
with patch.object(recover, "verdict", return_value=True), patch.object(recover.subprocess, "run") as restart:
self.assertIn("second probe", recover.run(1001))
self.assertIn("next probe", recover.run(1002))
self.report["at"] = 1300
self.save()
self.assertIn("restarted", recover.run(1301))
restart.assert_called_once_with(["sudo", "-n", "systemctl", "restart", "flysim.service"], check=True)
self.assertIn("cooldown", recover.run(1302))
def test_stale_and_clear_never_restart(self):
with patch.object(recover.subprocess, "run") as restart:
self.assertIn("not a fresh", recover.run(1700))
self.report.update(at=1700, suspected=0)
self.save()
self.assertIn("not a fresh", recover.run(1700))
restart.assert_not_called()
def test_router_failure_does_not_act(self):
recover.run(1000)
self.report["at"] = 1300
self.save()
with patch.object(recover, "verdict", side_effect=ValueError("bad response")), patch.object(recover.subprocess, "run") as restart:
self.assertEqual(recover.run(1300), "router unavailable")
restart.assert_not_called()
def test_cleared_probe_preserves_cooldown(self):
with patch.object(recover, "verdict", return_value=True), patch.object(recover.subprocess, "run") as restart:
recover.run(1000)
self.report["at"] = 1300
self.save()
recover.run(1300)
self.report.update(at=1600, suspected=0)
self.save()
recover.run(1600)
self.report.update(at=1900, suspected=1)
self.save()
recover.run(1900)
self.report["at"] = 2200
self.save()
self.assertIn("cooldown", recover.run(2200))
restart.assert_called_once()
def test_router_rejection_does_not_restart(self):
recover.run(1000)
self.report["at"] = 1300
self.save()
with patch.object(recover, "verdict", return_value=False), patch.object(recover.subprocess, "run") as restart:
self.assertIn("did not confirm", recover.run(1300))
restart.assert_not_called()
def test_unconfigured_router_uses_deterministic_confirmation(self):
with patch.dict("os.environ", {}, clear=True):
self.assertTrue(recover.verdict(self.report))
if __name__ == "__main__":
unittest.main()

View file

@ -0,0 +1,8 @@
[Unit]
Description=Check confirmed macro loops for minimal flysim recovery
[Service]
Type=oneshot
User=fly
EnvironmentFile=-/etc/fly/loop-recovery.env
ExecStart=/opt/fly/bin/fly-loop-recover

View file

@ -0,0 +1,11 @@
[Unit]
Description=Check for repeated macro loops after watchdog probes
[Timer]
OnBootSec=7min
OnUnitActiveSec=5min
AccuracySec=10s
Unit=fly-loop-recover.service
[Install]
WantedBy=timers.target