fix(cli): warn about stale dashboard processes after hermes update

The dashboard is a long-lived server process users start and forget.
When hermes update replaces files on disk, the running process holds
the old Python backend in memory while the JS bundle gets updated,
producing a silent frontend/backend mismatch (e.g. v0.11.0 changed
the session token header -- old backends reject every API call).

Scan for running dashboard processes after a successful update (both
git and ZIP paths) and print a warning with their PIDs and restart
instructions. Mirrors the existing pattern for gateway processes.

Fixes #16872
This commit is contained in:
Societus
2026-04-28 00:40:20 -07:00
committed by Teknium
parent 6b6fc28e85
commit 66b1142384
2 changed files with 185 additions and 0 deletions

View File

@@ -5213,6 +5213,80 @@ def _build_web_ui(web_dir: Path, *, fatal: bool = False) -> bool:
return True
def _warn_stale_dashboard_processes() -> None:
"""Warn about running dashboard processes that still hold pre-update code.
``hermes dashboard`` is a long-lived server process commonly started and
forgotten. When ``hermes update`` replaces files on disk, the running
process keeps the old Python backend in memory while the JS bundle on
disk is updated, causing a silent frontend/backend mismatch (e.g. new
auth headers the old backend doesn't recognise → every API call 401s).
Unlike the gateway, the dashboard has no service manager (systemd /
launchd), so we can only warn — we don't auto-kill user-managed
background processes.
"""
patterns = [
"hermes dashboard",
"hermes_cli.main dashboard",
"hermes_cli/main.py dashboard",
]
self_pid = os.getpid()
dashboard_pids: list[int] = []
try:
if sys.platform == "win32":
result = subprocess.run(
["wmic", "process", "get", "ProcessId,CommandLine",
"/FORMAT:LIST"],
capture_output=True, text=True, timeout=10,
)
if result.returncode != 0:
return
current_cmd = ""
for line in result.stdout.split("\n"):
line = line.strip()
if line.startswith("CommandLine="):
current_cmd = line[len("CommandLine="):]
elif line.startswith("ProcessId="):
pid_str = line[len("ProcessId="):]
if (any(p in current_cmd for p in patterns)
and int(pid_str) != self_pid):
try:
dashboard_pids.append(int(pid_str))
except ValueError:
pass
else:
# Linux / macOS: scan /proc or fall back to ps.
result = subprocess.run(
["pgrep", "-f", "hermes.*dashboard"],
capture_output=True, text=True, timeout=5,
)
if result.returncode == 0:
for line in result.stdout.strip().splitlines():
try:
pid = int(line.strip())
if pid != self_pid:
dashboard_pids.append(pid)
except ValueError:
pass
except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
return
if not dashboard_pids:
return
print()
print(f"{len(dashboard_pids)} dashboard process(es) still running "
f"with the previous version:")
for pid in dashboard_pids:
print(f" PID {pid}")
print(" The running backend may not match the updated frontend,")
print(" causing silent auth failures or empty data.")
print(" Restart them to pick up the changes:")
print(" kill <pid> && hermes dashboard --port <port> ...")
def _update_via_zip(args):
"""Update Hermes Agent by downloading a ZIP archive.
@@ -5347,6 +5421,7 @@ def _update_via_zip(args):
print()
print("✓ Update complete!")
_warn_stale_dashboard_processes()
def _stash_local_changes_if_needed(git_cmd: list[str], cwd: Path) -> Optional[str]:
@@ -7163,6 +7238,10 @@ def _cmd_update_impl(args, gateway_mode: bool):
except Exception as e:
logger.debug("Legacy unit check during update failed: %s", e)
# Warn about stale dashboard processes — the dashboard has no
# service manager, so we can only tell the user to restart them.
_warn_stale_dashboard_processes()
print()
print("Tip: You can now select a provider and model:")
print(" hermes model # Select provider and model")

View File

@@ -0,0 +1,106 @@
"""Tests for _warn_stale_dashboard_processes — stale dashboard detection.
Ensures ``hermes update`` warns the user when dashboard processes from a
previous version are still running after files on disk have been replaced.
See #16872.
"""
from __future__ import annotations
import os
import sys
from unittest.mock import patch, MagicMock
import pytest
from hermes_cli.main import _warn_stale_dashboard_processes
class TestWarnStaleDashboardProcesses:
"""Unit tests for the stale dashboard process warning."""
def test_no_warning_when_no_dashboard_running(self, capsys):
"""pgrep finds nothing — no warning should be printed."""
with patch("subprocess.run") as mock_run:
mock_run.return_value = MagicMock(
returncode=1, stdout="", stderr=""
)
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert "dashboard process" not in output
def test_warning_printed_for_running_dashboard(self, capsys):
"""pgrep finds a dashboard PID — warning with PID should appear."""
with patch("subprocess.run") as mock_run:
mock_run.return_value = MagicMock(
returncode=0, stdout="12345\n", stderr=""
)
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert "1 dashboard process" in output
assert "PID 12345" in output
assert "kill <pid>" in output
def test_multiple_dashboard_pids(self, capsys):
"""Multiple dashboard processes — all PIDs listed."""
with patch("subprocess.run") as mock_run:
mock_run.return_value = MagicMock(
returncode=0, stdout="12345\n12346\n12347\n", stderr=""
)
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert "3 dashboard process" in output
assert "PID 12345" in output
assert "PID 12346" in output
assert "PID 12347" in output
def test_self_pid_excluded(self, capsys):
"""The current process PID should not be reported."""
with patch("subprocess.run") as mock_run:
# Return the current process PID
mock_run.return_value = MagicMock(
returncode=0,
stdout=f"{os.getpid()}\n12345\n",
stderr="",
)
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert str(os.getpid()) not in output
assert "PID 12345" in output
def test_pgrep_not_found_silently_ignored(self, capsys):
"""If pgrep is missing (FileNotFoundError), no crash, no warning."""
with patch("subprocess.run", side_effect=FileNotFoundError):
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert output == ""
def test_pgrep_timeout_silently_ignored(self, capsys):
"""If pgrep times out, no crash, no warning."""
import subprocess as sp
with patch("subprocess.run", side_effect=sp.TimeoutExpired("pgrep", 5)):
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert output == ""
def test_empty_pgrep_output_no_warning(self, capsys):
"""pgrep returns 0 but empty stdout — no warning."""
with patch("subprocess.run") as mock_run:
mock_run.return_value = MagicMock(
returncode=0, stdout="\n", stderr=""
)
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert "dashboard process" not in output
def test_invalid_pid_lines_skipped(self, capsys):
"""Non-numeric lines from pgrep should be skipped gracefully."""
with patch("subprocess.run") as mock_run:
mock_run.return_value = MagicMock(
returncode=0, stdout="notapid\n12345\nalso_bad\n", stderr=""
)
_warn_stale_dashboard_processes()
output = capsys.readouterr().out
assert "PID 12345" in output
assert "1 dashboard process" in output