Improve Local Agent update reliability v1.26.24

This commit is contained in:
A R R R Associates
2026-09-30 22:28:39 +05:30
parent 56098aaeb9
commit 24d7a8b1ec
6 changed files with 96 additions and 8 deletions
@@ -23,6 +23,7 @@ STATE_FILE = DATA_DIR / "supervisor_state.json"
OWNER_FILE = DATA_DIR / "supervisor_owner.json"
REQUEST_FILE = UPDATES_DIR / "supervisor_request.json"
UPDATE_PROGRESS_FILE = DATA_DIR / "update_progress.json"
FAILED_UPDATES_FILE = DATA_DIR / "failed_updates.json"
WORKER_LOG = LOG_DIR / "worker-supervisor.log"
SUPERVISOR_LOG = LOG_DIR / "supervisor.log"
DASHBOARD_URL = "http://127.0.0.1:8788"
@@ -110,6 +111,36 @@ def _update_progress(phase: str, percent: int, message: str, *, version: str = "
_write_json_atomic(UPDATE_PROGRESS_FILE, payload)
def _read_failed_updates() -> dict:
try:
if FAILED_UPDATES_FILE.exists():
value = json.loads(FAILED_UPDATES_FILE.read_text(encoding="utf-8"))
return value if isinstance(value, dict) else {}
except Exception:
pass
return {}
def _mark_failed_update(version: str, error: str) -> None:
if not version:
return
payload = _read_failed_updates()
previous = payload.get(version) if isinstance(payload.get(version), dict) else {}
payload[version] = {
"failed_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
"error": str(error)[:2000],
"attempts": int(previous.get("attempts") or 0) + 1,
}
_write_json_atomic(FAILED_UPDATES_FILE, payload)
def _clear_failed_update(version: str) -> None:
payload = _read_failed_updates()
if version in payload:
payload.pop(version, None)
_write_json_atomic(FAILED_UPDATES_FILE, payload)
def _dashboard_ready(timeout: float = 1.5) -> bool:
try:
with urllib.request.urlopen(DASHBOARD_URL + "/api/status", timeout=timeout) as response:
@@ -443,6 +474,7 @@ class Supervisor:
raise RuntimeError(f"Updated worker exited with code {self.worker.returncode}.")
if _dashboard_ready():
_update_progress("complete", 100, f"ERP Local Agent {version} updated successfully.", version=version, status="updated")
_clear_failed_update(version)
_state(
"updated",
f"ERP Local Agent {version} installed and restarted successfully.",
@@ -456,6 +488,7 @@ class Supervisor:
time.sleep(1)
raise RuntimeError("Updated worker did not become healthy within the restart timeout.")
except Exception as exc:
_mark_failed_update(version, str(exc))
_update_progress("rollback", 92, f"Update failed; restoring the previous runtime: {exc}", version=version, status="rollback")
_state("rollback", f"Update {version} failed. Restoring previous worker.", target_version=version)
self.stop_worker()