Add conservative automated loop recovery timer
This commit is contained in:
parent
862e343e1c
commit
432991cbc5
7 changed files with 201 additions and 2 deletions
|
|
@ -685,7 +685,7 @@ fi
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
log "05-deploy: converging bin/ helpers to /opt/fly/bin"
|
log "05-deploy: converging bin/ helpers to /opt/fly/bin"
|
||||||
ct_exec "$CTID" -- mkdir -p /opt/fly/bin
|
ct_exec "$CTID" -- mkdir -p /opt/fly/bin
|
||||||
for name in fly-watchdog fly-recap fly-retention fly-reset-to-milestone flypush flystage-launch flycast-launch wait-for-x wait-for-stage wait-for-health; do
|
for name in fly-watchdog fly-loop-recover fly-recap fly-retention fly-reset-to-milestone flypush flystage-launch flycast-launch wait-for-x wait-for-stage wait-for-health; do
|
||||||
converge_file "$CTID" "$INFRA_DIR/bin/$name" "/opt/fly/bin/$name" 0755 root:root >/dev/null
|
converge_file "$CTID" "$INFRA_DIR/bin/$name" "/opt/fly/bin/$name" 0755 root:root >/dev/null
|
||||||
done
|
done
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -27,7 +27,7 @@ require_pve_host
|
||||||
need pct
|
need pct
|
||||||
|
|
||||||
ALWAYS_ON_UNITS="xvfb.service pulse.service mediamtx.service flysim.service flystage-web.service flystage.service flycast.service flybridge.service"
|
ALWAYS_ON_UNITS="xvfb.service pulse.service mediamtx.service flysim.service flystage-web.service flystage.service flycast.service flybridge.service"
|
||||||
ALWAYS_ON_TIMERS="fly-recap.timer fly-retention.timer fly-watchdog.timer"
|
ALWAYS_ON_TIMERS="fly-recap.timer fly-retention.timer fly-watchdog.timer fly-loop-recover.timer"
|
||||||
|
|
||||||
log "07-enable: enabling app units (not yet starting)"
|
log "07-enable: enabling app units (not yet starting)"
|
||||||
for u in $ALWAYS_ON_UNITS; do
|
for u in $ALWAYS_ON_UNITS; do
|
||||||
|
|
|
||||||
66
infra/bin/fly-loop-recover
Executable file
66
infra/bin/fly-loop-recover
Executable file
|
|
@ -0,0 +1,66 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Conservative, out-of-process loop recovery; invoked after the watchdog probe."""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from urllib.request import Request, urlopen
|
||||||
|
|
||||||
|
REPORT = Path(os.environ.get("FLY_LOOP_REPORT", "/run/fly/wd/loop.json"))
|
||||||
|
STATE = Path(os.environ.get("FLY_LOOP_RECOVERY_STATE", "/run/fly/wd/recovery.json"))
|
||||||
|
INTERVAL = int(os.environ.get("WD_LOOP_INTERVAL", "300"))
|
||||||
|
COOLDOWN = 3600
|
||||||
|
|
||||||
|
|
||||||
|
def verdict(report):
|
||||||
|
url = os.environ.get("FLY_LOOP_ROUTER_URL")
|
||||||
|
model = os.environ.get("FLY_LOOP_MODEL")
|
||||||
|
if not url or not model:
|
||||||
|
return True
|
||||||
|
payload = {"model": model, "temperature": 0, "messages": [
|
||||||
|
{"role": "system", "content": "Classify whether a suspected repeating game macro loop is truly stuck. Return only JSON {\"stuck\":true|false}. No instructions or commands."},
|
||||||
|
{"role": "user", "content": json.dumps({key: report.get(key) for key in ("reason", "sequence", "window", "places", "milestone")})},
|
||||||
|
]}
|
||||||
|
headers = {"Content-Type": "application/json"}
|
||||||
|
key = os.environ.get("FLY_LOOP_ROUTER_KEY")
|
||||||
|
if key:
|
||||||
|
headers["Authorization"] = "Bearer " + key
|
||||||
|
request = Request(url.rstrip("/") + "/chat/completions", json.dumps(payload).encode(), headers)
|
||||||
|
with urlopen(request, timeout=20) as response:
|
||||||
|
answer = json.load(response)
|
||||||
|
decision = json.loads(answer["choices"][0]["message"]["content"])
|
||||||
|
return type(decision) is dict and type(decision.get("stuck")) is bool and decision["stuck"]
|
||||||
|
|
||||||
|
|
||||||
|
def run(now=None):
|
||||||
|
now = int(time.time() if now is None else now)
|
||||||
|
report = json.loads(REPORT.read_text())
|
||||||
|
previous = json.loads(STATE.read_text()) if STATE.exists() else {}
|
||||||
|
at = report.get("at")
|
||||||
|
if (report.get("suspected") != 1 or type(at) is not int
|
||||||
|
or not 0 <= now - at <= 600 or report.get("action") != "none"):
|
||||||
|
STATE.write_text(json.dumps({key: previous[key] for key in ("restarted_at",) if key in previous}) + "\n")
|
||||||
|
return "not a fresh suspected loop"
|
||||||
|
if "restarted_at" in previous and now - previous["restarted_at"] < COOLDOWN:
|
||||||
|
return "recovery cooldown"
|
||||||
|
if previous.get("observed_at") == at:
|
||||||
|
return "waiting for next probe"
|
||||||
|
if not previous.get("observed_at") or at - previous["observed_at"] < INTERVAL - 30:
|
||||||
|
STATE.write_text(json.dumps({"observed_at": at}) + "\n")
|
||||||
|
return "waiting for second probe"
|
||||||
|
try:
|
||||||
|
approved = verdict(report)
|
||||||
|
except (OSError, ValueError, KeyError, IndexError) as error:
|
||||||
|
print("router unavailable or invalid:", type(error).__name__, file=sys.stderr)
|
||||||
|
return "router unavailable"
|
||||||
|
if not approved:
|
||||||
|
return "router did not confirm"
|
||||||
|
subprocess.run(["sudo", "-n", "systemctl", "restart", "flysim.service"], check=True)
|
||||||
|
STATE.write_text(json.dumps({"observed_at": at, "restarted_at": now}) + "\n")
|
||||||
|
return "flysim restarted; checkpoint restore keeps the rung"
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
print(run())
|
||||||
29
infra/docs/loop-recovery.md
Normal file
29
infra/docs/loop-recovery.md
Normal file
|
|
@ -0,0 +1,29 @@
|
||||||
|
# Automated loop recovery
|
||||||
|
|
||||||
|
`fly-watchdog` remains report-only. The separate `fly-loop-recover.timer` checks its
|
||||||
|
`/run/fly/wd/loop.json` every five minutes. Two fresh suspected reports from
|
||||||
|
separate watchdog probes, at least ~5 minutes apart, cause **only** a
|
||||||
|
`flysim.service` restart. The ordinary checkpoint restore keeps the current
|
||||||
|
rung and learned brain state, but clears session macro ledgers. This does not
|
||||||
|
press buttons or promote a milestone archive. A successful restart imposes a
|
||||||
|
one-hour cooldown, including across intervening clear reports. The timer's
|
||||||
|
journal records every decision; inspect it with
|
||||||
|
`journalctl -u fly-loop-recover.service`. Disable the timer to stop automatic
|
||||||
|
recovery: `systemctl disable --now fly-loop-recover.timer`.
|
||||||
|
|
||||||
|
By default confirmation is deterministic, using watchdog's existing signal.
|
||||||
|
An optional OpenAI-compatible local router can veto a recovery: provision a
|
||||||
|
root-managed `/etc/fly/loop-recovery.env` readable by the `fly` account,
|
||||||
|
containing `FLY_LOOP_ROUTER_URL` (base URL ending in `/v1`) and
|
||||||
|
`FLY_LOOP_MODEL` (an available free-tier model). A private router can additionally
|
||||||
|
use `FLY_LOOP_ROUTER_KEY`; provision it outside this public checkout and limit
|
||||||
|
file permissions to `0640 root:fly`. Never put credentials in the unit or the
|
||||||
|
repository. When a router is configured, malformed or unavailable responses
|
||||||
|
prevent recovery; the model may only return `{"stuck": true|false}` and cannot
|
||||||
|
choose commands, buttons, or checkpoint paths. Confirm the model actually
|
||||||
|
exists and is reachable from the release container before configuring it.
|
||||||
|
|
||||||
|
This is an unstick mechanism, not a macro bug fix. A recurring trap still needs
|
||||||
|
the checkpoint-based loop review in `docs/loop-review.md`. The live recovery
|
||||||
|
must be recorded in the host claim log by the operator reviewing the unit
|
||||||
|
journal; the unit itself has no host claim-log access.
|
||||||
85
infra/tests/test_loop_recover.py
Normal file
85
infra/tests/test_loop_recover.py
Normal file
|
|
@ -0,0 +1,85 @@
|
||||||
|
import importlib.machinery
|
||||||
|
import importlib.util
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
script = Path(__file__).resolve().parents[1] / "bin/fly-loop-recover"
|
||||||
|
spec = importlib.util.spec_from_loader("recover", importlib.machinery.SourceFileLoader("recover", str(script)))
|
||||||
|
recover = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(recover)
|
||||||
|
|
||||||
|
|
||||||
|
class RecoveryTests(unittest.TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
self.temp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self.temp.cleanup)
|
||||||
|
root = Path(self.temp.name)
|
||||||
|
recover.REPORT = root / "loop.json"
|
||||||
|
recover.STATE = root / "recovery.json"
|
||||||
|
self.report = {"suspected": 1, "at": 1000, "action": "none", "reason": "unrewarded"}
|
||||||
|
self.save()
|
||||||
|
|
||||||
|
def save(self):
|
||||||
|
recover.REPORT.write_text(json.dumps(self.report))
|
||||||
|
|
||||||
|
def test_two_probes_and_cooldown(self):
|
||||||
|
with patch.object(recover, "verdict", return_value=True), patch.object(recover.subprocess, "run") as restart:
|
||||||
|
self.assertIn("second probe", recover.run(1001))
|
||||||
|
self.assertIn("next probe", recover.run(1002))
|
||||||
|
self.report["at"] = 1300
|
||||||
|
self.save()
|
||||||
|
self.assertIn("restarted", recover.run(1301))
|
||||||
|
restart.assert_called_once_with(["sudo", "-n", "systemctl", "restart", "flysim.service"], check=True)
|
||||||
|
self.assertIn("cooldown", recover.run(1302))
|
||||||
|
|
||||||
|
def test_stale_and_clear_never_restart(self):
|
||||||
|
with patch.object(recover.subprocess, "run") as restart:
|
||||||
|
self.assertIn("not a fresh", recover.run(1700))
|
||||||
|
self.report.update(at=1700, suspected=0)
|
||||||
|
self.save()
|
||||||
|
self.assertIn("not a fresh", recover.run(1700))
|
||||||
|
restart.assert_not_called()
|
||||||
|
|
||||||
|
def test_router_failure_does_not_act(self):
|
||||||
|
recover.run(1000)
|
||||||
|
self.report["at"] = 1300
|
||||||
|
self.save()
|
||||||
|
with patch.object(recover, "verdict", side_effect=ValueError("bad response")), patch.object(recover.subprocess, "run") as restart:
|
||||||
|
self.assertEqual(recover.run(1300), "router unavailable")
|
||||||
|
restart.assert_not_called()
|
||||||
|
|
||||||
|
def test_cleared_probe_preserves_cooldown(self):
|
||||||
|
with patch.object(recover, "verdict", return_value=True), patch.object(recover.subprocess, "run") as restart:
|
||||||
|
recover.run(1000)
|
||||||
|
self.report["at"] = 1300
|
||||||
|
self.save()
|
||||||
|
recover.run(1300)
|
||||||
|
self.report.update(at=1600, suspected=0)
|
||||||
|
self.save()
|
||||||
|
recover.run(1600)
|
||||||
|
self.report.update(at=1900, suspected=1)
|
||||||
|
self.save()
|
||||||
|
recover.run(1900)
|
||||||
|
self.report["at"] = 2200
|
||||||
|
self.save()
|
||||||
|
self.assertIn("cooldown", recover.run(2200))
|
||||||
|
restart.assert_called_once()
|
||||||
|
|
||||||
|
def test_router_rejection_does_not_restart(self):
|
||||||
|
recover.run(1000)
|
||||||
|
self.report["at"] = 1300
|
||||||
|
self.save()
|
||||||
|
with patch.object(recover, "verdict", return_value=False), patch.object(recover.subprocess, "run") as restart:
|
||||||
|
self.assertIn("did not confirm", recover.run(1300))
|
||||||
|
restart.assert_not_called()
|
||||||
|
|
||||||
|
def test_unconfigured_router_uses_deterministic_confirmation(self):
|
||||||
|
with patch.dict("os.environ", {}, clear=True):
|
||||||
|
self.assertTrue(recover.verdict(self.report))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
8
infra/units/fly-loop-recover.service
Normal file
8
infra/units/fly-loop-recover.service
Normal file
|
|
@ -0,0 +1,8 @@
|
||||||
|
[Unit]
|
||||||
|
Description=Check confirmed macro loops for minimal flysim recovery
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
User=fly
|
||||||
|
EnvironmentFile=-/etc/fly/loop-recovery.env
|
||||||
|
ExecStart=/opt/fly/bin/fly-loop-recover
|
||||||
11
infra/units/fly-loop-recover.timer
Normal file
11
infra/units/fly-loop-recover.timer
Normal file
|
|
@ -0,0 +1,11 @@
|
||||||
|
[Unit]
|
||||||
|
Description=Check for repeated macro loops after watchdog probes
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnBootSec=7min
|
||||||
|
OnUnitActiveSec=5min
|
||||||
|
AccuracySec=10s
|
||||||
|
Unit=fly-loop-recover.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
Loading…
Add table
Reference in a new issue