Repair Local Agent supervisor and add update progress UI

This commit is contained in:
A R R R Associates
2026-09-04 13:33:34 +05:30
parent 4deef57364
commit 7bc97be1b1
8 changed files with 282 additions and 343 deletions
@@ -10,7 +10,6 @@ from pathlib import Path
import shutil
import signal
import subprocess
import sys
import time
import urllib.request
@@ -23,14 +22,14 @@ LOCK_FILE = DATA_DIR / "supervisor.lock"
STATE_FILE = DATA_DIR / "supervisor_state.json"
OWNER_FILE = DATA_DIR / "supervisor_owner.json"
REQUEST_FILE = UPDATES_DIR / "supervisor_request.json"
UPDATE_PROGRESS_FILE = DATA_DIR / "update_progress.json"
WORKER_LOG = LOG_DIR / "worker-supervisor.log"
SUPERVISOR_LOG = LOG_DIR / "supervisor.log"
DASHBOARD_URL = "http://127.0.0.1:8788"
WORKER_MODULE = "erp_local_agent.main"
RESTART_DELAY_SECONDS = 3
HEALTH_TIMEOUT_SECONDS = 60
HEALTH_TIMEOUT_SECONDS = 90
ERROR_ALREADY_EXISTS = 183
SINGLETON_WATCHDOG_SECONDS = 10
def _supervisor_mutex_name() -> str:
@@ -64,28 +63,6 @@ def _close_os_mutex(handle) -> None:
pass
def _kill_duplicate_supervisors() -> None:
if os.name != "nt":
return
root = str(INSTALL_ROOT).replace("'", "''")
script = (
"$root='" + root + "';$self=" + str(os.getpid()) + ";"
"Get-CimInstance Win32_Process -ErrorAction SilentlyContinue | "
"Where-Object { $_.ProcessId -ne $self -and $_.CommandLine -and "
"$_.CommandLine -match 'ERPAgentSupervisor\\.pyw' -and "
"$_.CommandLine -like ('*'+$root+'*') } | "
"ForEach-Object { Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue }"
)
subprocess.run(
["powershell.exe", "-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", script],
cwd=str(INSTALL_ROOT),
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
creationflags=_create_no_window(),
check=False,
)
def _create_no_window() -> int:
return getattr(subprocess, "CREATE_NO_WINDOW", 0)
@@ -100,16 +77,6 @@ def _log(message: str) -> None:
pass
def _pid_alive(pid: int) -> bool:
if pid <= 0:
return False
try:
os.kill(pid, 0)
return True
except OSError:
return False
def _write_json_atomic(path: Path, payload: dict) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(path.suffix + ".tmp")
@@ -129,6 +96,20 @@ def _state(status: str, message: str, **extra) -> None:
_log(f"{status}: {message}")
def _update_progress(phase: str, percent: int, message: str, *, version: str = "", status: str = "installing") -> None:
payload = {
"operation": "install",
"status": status,
"phase": phase,
"percent": max(0, min(100, int(percent))),
"message": message,
"target_version": version,
"updated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
"supervisor_pid": os.getpid(),
}
_write_json_atomic(UPDATE_PROGRESS_FILE, payload)
def _dashboard_ready(timeout: float = 1.5) -> bool:
try:
with urllib.request.urlopen(DASHBOARD_URL + "/api/status", timeout=timeout) as response:
@@ -149,9 +130,7 @@ def _python(console: bool = True) -> Path:
def _terminate_process_tree(process: subprocess.Popen | None) -> None:
if process is None:
return
if process.poll() is not None:
if process is None or process.poll() is not None:
return
try:
subprocess.run(
@@ -167,91 +146,41 @@ def _terminate_process_tree(process: subprocess.Popen | None) -> None:
except Exception:
pass
try:
process.wait(timeout=15)
process.wait(timeout=20)
except Exception:
pass
def _kill_unmanaged_workers() -> None:
script = (
"$root='" + str(INSTALL_ROOT).replace("'", "''") + "';"
"Get-CimInstance Win32_Process -ErrorAction SilentlyContinue | "
"Where-Object { $_.CommandLine -and $_.CommandLine -match 'erp_local_agent\\.main' "
"-and $_.CommandLine -like ('*'+$root+'*') } | "
"ForEach-Object { Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue }"
)
subprocess.run(
["powershell.exe", "-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", script],
cwd=str(INSTALL_ROOT),
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
creationflags=_create_no_window(),
check=False,
)
def _enforce_singleton_processes(*, keep_worker_pid: int | None = None) -> None:
"""Kill stale/duplicate supervisors and workers for this install only.
The Global named mutex is the primary atomic guard. This watchdog is a
recovery layer for legacy processes started before the hardened runtime.
"""
if os.name != "nt":
return
root = str(INSTALL_ROOT.resolve()).replace("'", "''")
keep_worker = int(keep_worker_pid or 0)
script = rf"""
$root='{root}'
$selfPid={os.getpid()}
$keepWorker={keep_worker}
Get-CimInstance Win32_Process -ErrorAction SilentlyContinue | ForEach-Object {{
$cmd=[string]$_.CommandLine
if (-not $cmd -or $cmd -notlike ('*'+$root+'*')) {{ return }}
if (($cmd -match 'ERPAgentSupervisor\.pyw') -and ($_.ProcessId -ne $selfPid)) {{
Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue
return
}}
if (($cmd -match 'erp_local_agent\.main') -and ($_.ProcessId -ne $keepWorker)) {{
Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue
}}
}}
"""
subprocess.run(
["powershell.exe", "-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", script],
cwd=str(INSTALL_ROOT),
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
creationflags=_create_no_window(),
check=False,
)
class Supervisor:
def __init__(self):
self.worker: subprocess.Popen | None = None
self.running = True
self._os_mutex = None
self._worker_log_handle = None
def acquire_lock(self) -> bool:
# The named mutex is atomic at the Windows kernel level and prevents
# the lock-file race that allowed multiple supervisors before 1.22.3.
"""Acquire the only authoritative supervisor singleton guard.
The Global Windows mutex is atomic across SYSTEM and user sessions. PID/owner
files are diagnostics only and must never veto a successfully acquired mutex.
"""
_log("startup: acquiring Global supervisor mutex")
self._os_mutex = _acquire_os_mutex()
if self._os_mutex is None:
_log("startup: another supervisor already owns the Global mutex; exiting")
return False
DATA_DIR.mkdir(parents=True, exist_ok=True)
if LOCK_FILE.exists():
try:
existing = int(LOCK_FILE.read_text(encoding="utf-8").strip() or "0")
except Exception:
existing = 0
if existing and existing != os.getpid() and _pid_alive(existing):
return False
LOCK_FILE.write_text(str(os.getpid()), encoding="utf-8")
_write_json_atomic(OWNER_FILE, {
"supervisor_pid": os.getpid(),
"install_root": str(INSTALL_ROOT),
"started_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
})
_write_json_atomic(
OWNER_FILE,
{
"supervisor_pid": os.getpid(),
"install_root": str(INSTALL_ROOT),
"started_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
"singleton": "global_mutex",
},
)
_log("startup: Global supervisor mutex acquired")
return True
def release_lock(self) -> None:
@@ -273,26 +202,32 @@ class Supervisor:
def start_worker(self) -> None:
if self.worker is not None and self.worker.poll() is None:
return
_kill_unmanaged_workers()
LOG_DIR.mkdir(parents=True, exist_ok=True)
py = _python(console=True)
handle = WORKER_LOG.open("a", encoding="utf-8")
handle.write(f"\n--- worker start {time.strftime('%Y-%m-%d %H:%M:%S')} ---\n")
handle.flush()
self._worker_log_handle = WORKER_LOG.open("a", encoding="utf-8")
self._worker_log_handle.write(f"\n--- worker start {time.strftime('%Y-%m-%d %H:%M:%S')} ---\n")
self._worker_log_handle.flush()
self.worker = subprocess.Popen(
[str(py), "-m", WORKER_MODULE],
cwd=str(INSTALL_ROOT),
stdout=handle,
stderr=handle,
stdout=self._worker_log_handle,
stderr=self._worker_log_handle,
creationflags=_create_no_window(),
)
_state("running", "ERP Local Agent worker started.", worker_pid=self.worker.pid)
def stop_worker(self) -> None:
if self.worker is None:
return
_state("stopping", "Stopping ERP Local Agent worker.")
_terminate_process_tree(self.worker)
self.worker = None
_kill_unmanaged_workers()
if self._worker_log_handle is not None:
try:
self._worker_log_handle.close()
except Exception:
pass
self._worker_log_handle = None
@staticmethod
def _read_request() -> dict | None:
@@ -345,6 +280,8 @@ class Supervisor:
"stop_task_scheduler.bat",
"status_task_scheduler.bat",
"uninstall_task_scheduler.bat",
"ERPAgentSupervisor.pyw",
"ERPAgentDashboard.pyw",
):
src = INSTALL_ROOT / name
if src.exists():
@@ -356,11 +293,9 @@ class Supervisor:
target = INSTALL_ROOT / "erp_local_agent"
new_target = INSTALL_ROOT / "erp_local_agent.__new__"
old_target = INSTALL_ROOT / "erp_local_agent.__old__"
shutil.rmtree(new_target, ignore_errors=True)
shutil.rmtree(old_target, ignore_errors=True)
shutil.copytree(staged / "erp_local_agent", new_target)
if target.exists():
target.rename(old_target)
try:
@@ -371,9 +306,6 @@ class Supervisor:
raise
shutil.rmtree(old_target, ignore_errors=True)
# Root helper files are safe to refresh while the stable supervisor is running.
# Supervisor/dashboard launchers themselves are intentionally excluded here;
# they are updated only by the explicit direct installer.
for name in (
"requirements.txt",
"README_ERP_LOCAL_AGENT.txt",
@@ -387,6 +319,8 @@ class Supervisor:
"stop_task_scheduler.bat",
"status_task_scheduler.bat",
"uninstall_task_scheduler.bat",
"ERPAgentSupervisor.pyw",
"ERPAgentDashboard.pyw",
):
src = staged / name
if src.exists():
@@ -407,7 +341,6 @@ class Supervisor:
)
if result.returncode != 0:
raise RuntimeError("Dependency installation failed:\n" + result.stdout[-4000:])
code = (
"import erp_local_agent;"
"from erp_local_agent import accounting_store,commands,tally,updater;"
@@ -443,16 +376,26 @@ class Supervisor:
staged_text = str(request.get("staged_dir") or "").strip()
if not version or not staged_text:
raise RuntimeError("Supervisor update request is missing version/staged_dir.")
staged = Path(staged_text).resolve()
self._validate_staged(staged)
_state("installing", f"Installing ERP Local Agent {version}.", target_version=version)
self.stop_worker()
backup = self._backup_runtime(version)
_update_progress("preparing", 8, f"Preparing ERP Local Agent {version} update.", version=version)
_state("installing", f"Installing ERP Local Agent {version}.", target_version=version, progress_pct=8)
backup = None
try:
_update_progress("stopping", 18, "Stopping the current Local Agent worker safely.", version=version)
self.stop_worker()
_update_progress("backup", 30, "Creating rollback backup.", version=version)
backup = self._backup_runtime(version)
_update_progress("replacing", 48, "Replacing Local Agent runtime files.", version=version)
self._replace_worker_runtime(staged)
_update_progress("dependencies", 66, "Verifying runtime dependencies and imports.", version=version)
self._install_requirements_and_verify()
_update_progress("starting", 82, "Starting the updated Local Agent.", version=version)
self.start_worker()
deadline = time.time() + HEALTH_TIMEOUT_SECONDS
@@ -460,42 +403,36 @@ class Supervisor:
if self.worker is not None and self.worker.poll() is not None:
raise RuntimeError(f"Updated worker exited with code {self.worker.returncode}.")
if _dashboard_ready():
_update_progress("complete", 100, f"ERP Local Agent {version} updated successfully.", version=version, status="updated")
_state(
"updated",
f"ERP Local Agent {version} installed and restarted successfully.",
target_version=version,
worker_pid=(self.worker.pid if self.worker else None),
backup_path=str(backup),
progress_pct=100,
)
return
_update_progress("health_check", 90, "Waiting for the updated dashboard to become ready.", version=version)
time.sleep(1)
raise RuntimeError("Updated worker did not become healthy within the restart timeout.")
except Exception:
except Exception as exc:
_update_progress("rollback", 92, f"Update failed; restoring the previous runtime: {exc}", version=version, status="rollback")
_state("rollback", f"Update {version} failed. Restoring previous worker.", target_version=version)
self.stop_worker()
self._restore_backup(backup)
if backup is not None:
self._restore_backup(backup)
self.start_worker()
_update_progress("error", 100, f"Update failed and the previous version was restored: {exc}", version=version, status="error")
raise
def run(self) -> int:
if not self.acquire_lock():
return 0
try:
_kill_duplicate_supervisors()
_state("starting", "ERP Local Agent supervisor starting.")
self.start_worker()
last_singleton_watchdog = 0.0
while self.running:
now = time.time()
if now - last_singleton_watchdog >= SINGLETON_WATCHDOG_SECONDS:
_enforce_singleton_processes(
keep_worker_pid=(self.worker.pid if self.worker is not None and self.worker.poll() is None else None)
)
last_singleton_watchdog = now
request = self._read_request()
if request and str(request.get("action") or "") == "install_update":
self._clear_request()
@@ -503,37 +440,34 @@ class Supervisor:
self.apply_update(request)
except Exception as exc:
_state("error", f"Update failed and worker was restored/restarted: {exc}")
if self.worker is None or self.worker.poll() is not None:
code = None if self.worker is None else self.worker.returncode
_state("restarting", f"ERP Local Agent worker stopped (exit={code}); restarting.")
time.sleep(RESTART_DELAY_SECONDS)
self.start_worker()
time.sleep(1)
return 0
except BaseException as exc:
_log(f"fatal: {type(exc).__name__}: {exc}")
try:
_state("fatal", f"Supervisor stopped unexpectedly: {exc}")
except Exception:
pass
raise
finally:
self.stop_worker()
self.release_lock()
_state("stopped", "ERP Local Agent supervisor stopped.")
try:
self.stop_worker()
finally:
self.release_lock()
_state("stopped", "ERP Local Agent supervisor stopped.")
def main() -> int:
parser = argparse.ArgumentParser(description="ERP Local Agent desktop/background supervisor")
parser.add_argument("--background", action="store_true")
parser.add_argument("--replace", action="store_true", help="Explicit controlled replacement of an older supervisor instance")
args = parser.parse_args()
if args.replace and os.name == "nt":
# Explicit restart/update only. Normal duplicate launches never disturb
# the healthy owner; they simply fail the Global mutex and exit.
_kill_duplicate_supervisors()
deadline = time.time() + 15
while time.time() < deadline:
time.sleep(0.25)
break
parser.add_argument("--replace", action="store_true", help="Compatibility flag; the Global mutex remains authoritative")
parser.parse_args()
_log(f"startup: supervisor process entered main pid={os.getpid()} root={INSTALL_ROOT}")
supervisor = Supervisor()
def stop_handler(*_args):
@@ -544,9 +478,14 @@ def main() -> int:
signal.signal(signal.SIGINT, stop_handler)
except Exception:
pass
return supervisor.run()
if __name__ == "__main__":
raise SystemExit(main())
try:
raise SystemExit(main())
except SystemExit:
raise
except BaseException as exc:
_log(f"startup-fatal: {type(exc).__name__}: {exc}")
raise