aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/daemon.py
diff options
context:
space:
mode:
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/daemon.py')
-rw-r--r--packages/meshbay-node/src/meshbay_node/daemon.py47
1 files changed, 45 insertions, 2 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/daemon.py b/packages/meshbay-node/src/meshbay_node/daemon.py
index e1aac41..6fba120 100644
--- a/packages/meshbay-node/src/meshbay_node/daemon.py
+++ b/packages/meshbay-node/src/meshbay_node/daemon.py
@@ -31,6 +31,7 @@ import os
import signal
import socket
import sys
+import threading
import time
from dataclasses import asdict
from pathlib import Path
@@ -81,6 +82,31 @@ if WEBRTC_AVAILABLE:
log = logging.getLogger(__name__)
+# How long a stopped node's process may outlive its _shutdown(), and the most
+# _shutdown() itself may take. Python's own exit waits for every worker thread,
+# and one still walking or hashing a large tree kept the process alive for as
+# long as that took, with its control API already closed: the desktop app saw
+# no node and said "Stopped" about a process that was still there.
+EXIT_GRACE_SECS = 3.0
+SHUTDOWN_DEADLINE_SECS = 30.0
+
+
+def exit_after(seconds: float, why: str, code: int = 0) -> threading.Timer:
+ """End this process in `seconds`, whatever is still running in it."""
+ def _exit() -> None:
+ busy = sorted(t.name for t in threading.enumerate()
+ if t is not threading.current_thread() and not t.daemon
+ and t is not threading.main_thread())
+ log.warning("%s -- ending the process now (still busy: %s)",
+ why, ", ".join(busy) or "nothing")
+ logging.shutdown()
+ os._exit(code)
+ timer = threading.Timer(seconds, _exit)
+ timer.daemon = True
+ timer.start()
+ return timer
+
+
# ── Hub WS sender bridge ─────────────────────────────────────────────────────
class _WsSender:
@@ -101,6 +127,9 @@ def _root_shape(roots) -> set[tuple]:
class NodeDaemon(EnrichmentMixin):
+ # Set by main(), never by a test that runs a daemon in its own process.
+ exit_process_when_stopped = False
+
def __init__(self, config: Config, config_path: Path = DEFAULT_CONFIG_PATH):
self._config = config
self._config_path = config_path
@@ -1564,6 +1593,9 @@ class NodeDaemon(EnrichmentMixin):
async def _shutdown(self) -> None:
log.info("Shutting down...")
self._state["status"] = "stopping"
+ if self.exit_process_when_stopped:
+ exit_after(SHUTDOWN_DEADLINE_SECS,
+ f"shutdown took more than {SHUTDOWN_DEADLINE_SECS:.0f}s", code=1)
for handle in self._pending_broadcasts.values():
handle.cancel()
@@ -1624,6 +1656,8 @@ class NodeDaemon(EnrichmentMixin):
token_file.unlink(missing_ok=True)
log.info("Node stopped")
+ if self.exit_process_when_stopped:
+ exit_after(EXIT_GRACE_SECS, "the node has stopped")
# ── Entry point ───────────────────────────────────────────────────────────────
@@ -1668,7 +1702,15 @@ def main() -> None:
# Printed for whoever ran it, and logged too: a daemon started by the
# service task has no console, and these are why it would refuse to start.
- cfg = load_config(args.config or DEFAULT_CONFIG_PATH)
+ config_path = args.config or DEFAULT_CONFIG_PATH
+ try:
+ cfg = load_config(config_path)
+ except Exception as e:
+ # Uncaught, this went to a stderr that Task Scheduler discards: the
+ # node died with "Logging to ..." as its last word.
+ print(f"Error: cannot read {config_path}: {e}")
+ log.error("cannot read %s: %s", config_path, e)
+ sys.exit(1)
if not cfg.hub.username:
print("Error: hub.username not set in config. Run: meshbay-node init")
log.error("hub.username not set in config. Run: meshbay-node init")
@@ -1681,7 +1723,8 @@ def main() -> None:
log.error("%s", e)
sys.exit(1)
- daemon = NodeDaemon(cfg, Path(args.config or DEFAULT_CONFIG_PATH))
+ daemon = NodeDaemon(cfg, Path(config_path))
+ daemon.exit_process_when_stopped = True
try:
asyncio.run(daemon.run())
except ControlPortTaken as e: