From bb68dee386f5aab00bead2ee0a5995910ebab8dd Mon Sep 17 00:00:00 2001 From: tomfuertes Date: Fri, 1 May 2026 13:22:51 -0500 Subject: [PATCH 1/2] slim-shim: replace /setup with Caddy + native dashboard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Drop the 1500-line Python admin server and the hand-rolled Alpine.js SPA. Replace with: - Caddy at $PORT terminating basic_auth → reverse_proxy to the native hermes dashboard on 127.0.0.1:9119. - start.sh as a tiny supervisor: runs hermes dashboard in background, caddy in background, hermes gateway in foreground (under tini). - `hermes gateway run --replace` supersedes the manual stale-PID cleanup the old setup needed; it's also why the in-memory gateway state machine in server.py existed in the first place. Operator surface area moves to the native dashboard plus the upstream CLI: `hermes pairing`, `hermes config`, `hermes status` via railway ssh. The /setup curated channel UI is gone — channels are configured in the native dashboard's config tab. -2682 net lines. --- Caddyfile.tmpl | 17 + Dockerfile | 52 +- requirements.txt | 10 - server.py | 1190 ---------------------------------- start.sh | 40 +- templates/index.html | 1469 ------------------------------------------ 6 files changed, 48 insertions(+), 2730 deletions(-) create mode 100644 Caddyfile.tmpl delete mode 100644 requirements.txt delete mode 100644 server.py delete mode 100644 templates/index.html diff --git a/Caddyfile.tmpl b/Caddyfile.tmpl new file mode 100644 index 0000000..42d78ea --- /dev/null +++ b/Caddyfile.tmpl @@ -0,0 +1,17 @@ +{ + admin off + auto_https off +} + +:{$PORT} { + handle /health { + respond "ok" 200 + } + + handle { + basic_auth { + {$ADMIN_USERNAME} {$ADMIN_PASSWORD_HASH} + } + reverse_proxy 127.0.0.1:9119 + } +} diff --git a/Dockerfile b/Dockerfile index 85586e8..3e53e3d 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,37 +1,17 @@ +FROM caddy:2-alpine AS caddy-bin + FROM ghcr.io/astral-sh/uv:python3.12-bookworm-slim -# Which hermes-agent revision to install. Accepts any git ref the upstream -# repo publishes — a release tag (recommended for reproducibility) or a -# branch name (`main`) for bleeding edge. -# -# To bump: check https://github.com/NousResearch/hermes-agent/releases for the -# newest tag (format `vYYYY.M.D`, e.g. `v2026.4.23`) and update the default -# below. Use `main` only if you accept that every rebuild can pull arbitrary -# new upstream commits. ARG HERMES_REF=v2026.4.30 -# tini = tiny init that we run as PID 1. Without it, hermes's grandchild -# processes (MCP stdio servers, git, bun, browser daemons spawned by tools) -# reparent to PID 1 when their parents exit and pile up as zombies. After -# weeks of uptime that exhausts the kernel's PID table → "fork: cannot -# allocate memory" and the container dies. tini reaps zombies in the -# background and forwards SIGTERM/SIGINT to our entrypoint so Railway's -# stop signal still triggers our graceful shutdown. Standard container init -# (same as Docker's `--init` flag and Kubernetes' pause container). -# -# Node.js is required only at build time to compile the Hermes React dashboard. -# We strip the source + apt lists afterwards to keep the image lean. RUN apt-get update && \ apt-get install -y --no-install-recommends curl ca-certificates git tini && \ curl -fsSL https://deb.nodesource.com/setup_22.x | bash - && \ apt-get install -y --no-install-recommends nodejs && \ rm -rf /var/lib/apt/lists/* -# Install hermes-agent (provides the `hermes` CLI) and pre-build its React -# dashboard so `hermes dashboard` has nothing to build at runtime. -# Deleting web/ afterwards makes hermes's internal _build_web_ui skip the -# rebuild step (it early-returns when package.json is absent), so container -# startup is fast and no runtime npm dependency is needed. +COPY --from=caddy-bin /usr/bin/caddy /usr/local/bin/caddy + RUN git clone --depth 1 --branch ${HERMES_REF} https://github.com/NousResearch/hermes-agent.git /opt/hermes-agent && \ cd /opt/hermes-agent && \ uv pip install --system --no-cache -e ".[all]" && \ @@ -43,34 +23,14 @@ RUN git clone --depth 1 --branch ${HERMES_REF} https://github.com/NousResearch/h npm run build && \ rm -rf /opt/hermes-agent/web /opt/hermes-agent/.git /root/.npm -# Why pre-build ui-tui (and why we don't delete it after): -# - The dashboard's embedded Chat tab spawns `node ui-tui/dist/entry.js` -# on every WebSocket connect to /api/pty. -# - hermes's _make_tui_argv runs `npm install` + `npm run build` via -# *synchronous* subprocess.run if dist/entry.js is missing or stale — -# that would block the dashboard's asyncio event loop for 30-60s on -# the first chat-open, freezing every other request. -# - Pre-building at image time costs ~200-300 MB of node_modules but -# makes first-chat-open instant and surfaces any build failure here -# instead of at user request time. -# - We keep ui-tui/ entirely (node_modules + dist + src) so hermes's -# freshness checks don't trigger a re-install at runtime. - -COPY requirements.txt /app/requirements.txt -RUN uv pip install --system --no-cache -r /app/requirements.txt - -RUN mkdir -p /data/.hermes +RUN mkdir -p /data/.hermes /app -COPY server.py /app/server.py -COPY templates/ /app/templates/ +COPY Caddyfile.tmpl /app/Caddyfile.tmpl COPY start.sh /app/start.sh RUN chmod +x /app/start.sh ENV HOME=/data ENV HERMES_HOME=/data/.hermes -# tini wraps start.sh so it runs as PID 1's child instead of as PID 1 itself. -# `-g` propagates signals to the whole process group so `docker stop` / -# Railway's SIGTERM cleanly terminates the entire tree, not just start.sh. ENTRYPOINT ["/usr/bin/tini", "-g", "--"] CMD ["/app/start.sh"] diff --git a/requirements.txt b/requirements.txt deleted file mode 100644 index e78a4e7..0000000 --- a/requirements.txt +++ /dev/null @@ -1,10 +0,0 @@ -starlette==1.0.0 -uvicorn==0.46.0 -jinja2==3.1.6 -python-multipart==0.0.27 -httpx==0.28.1 -# websockets is the upstream client our reverse proxy uses to connect to -# hermes's /api/pty, /api/ws, /api/events endpoints. httpx does not speak -# WebSockets, so we need a dedicated client library. -websockets==16.0 -pyyaml==6.0.2 diff --git a/server.py b/server.py deleted file mode 100644 index 3bbda7b..0000000 --- a/server.py +++ /dev/null @@ -1,1190 +0,0 @@ -""" -Hermes Agent — Railway admin server. - -Responsibilities: - - Admin UI / setup wizard at /setup (Starlette + Jinja, cookie-auth guarded) - - Management API at /setup/api/* (config, status, logs, gateway, pairing) - - Reverse proxy at / and /* → native Hermes dashboard (hermes_cli/web_server, on 127.0.0.1:9119) - - Managed subprocesses: `hermes gateway` (agent) and `hermes dashboard` (native UI) - - Cookie-based session auth at /login (HMAC-signed, 7-day expiry, httponly) - -Auth model: Basic Auth was dropped in favor of cookies because the Hermes React -SPA's plain fetch() calls do not reliably include basic-auth creds across browsers, -and basic-auth's per-directory protection space forced separate prompts for -/setup and /. Cookies auto-include on every same-origin request, so both the -setup UI and the proxied dashboard work with a single login. The cookie signing -secret is regenerated on every process start, so any ADMIN_PASSWORD change on -Railway (which triggers a redeploy) invalidates all existing sessions. - -First-visit behavior: if no provider+model config exists, GET / redirects to /setup. -Once configured, / proxies to the Hermes dashboard. A small "← Setup" widget is -injected into every proxied HTML response so users can always return to the wizard. -""" - -import asyncio -import json -import os -import re -import secrets -import signal -import time -from collections import deque -from contextlib import asynccontextmanager -from pathlib import Path - -import httpx -import websockets -import websockets.exceptions -from starlette.applications import Starlette -from starlette.requests import Request -from starlette.responses import ( - HTMLResponse, - JSONResponse, - RedirectResponse, - Response, -) -from starlette.routing import Route, WebSocketRoute -from starlette.templating import Jinja2Templates -from starlette.websockets import WebSocket, WebSocketDisconnect, WebSocketState - -ANSI_ESCAPE = re.compile(r"\x1b\[[0-9;]*m") -templates = Jinja2Templates(directory=str(Path(__file__).parent / "templates")) - -HERMES_HOME = os.environ.get("HERMES_HOME", str(Path.home() / ".hermes")) -ENV_FILE = Path(HERMES_HOME) / ".env" -PAIRING_DIR = Path(HERMES_HOME) / "pairing" -PAIRING_TTL = 3600 - -# Native Hermes dashboard — runs on loopback, fronted by our reverse proxy. -HERMES_DASHBOARD_HOST = "127.0.0.1" -HERMES_DASHBOARD_PORT = int(os.environ.get("HERMES_DASHBOARD_PORT", "9119")) -HERMES_DASHBOARD_URL = f"http://{HERMES_DASHBOARD_HOST}:{HERMES_DASHBOARD_PORT}" - -# Mirror dashboard-ref-only/auth_proxy.py: strip only `host` (httpx sets it) -# and `transfer-encoding` (httpx recomputes it from the body). Keep everything -# else — notably `authorization`, because the SPA uses Bearer tokens against -# hermes's own /api/env/reveal and OAuth endpoints, and keep `cookie` since -# some hermes endpoints read it. Aggressive stripping was masking requests in -# ways that produced spurious 401s. -HOP_BY_HOP = {"host", "transfer-encoding"} - -ADMIN_USERNAME = os.environ.get("ADMIN_USERNAME", "admin") -ADMIN_PASSWORD = os.environ.get("ADMIN_PASSWORD", "") -if not ADMIN_PASSWORD: - ADMIN_PASSWORD = secrets.token_urlsafe(16) - pw_file = Path(HERMES_HOME) / ".admin-password.txt" - pw_file.parent.mkdir(parents=True, exist_ok=True) - pw_file.write_text(ADMIN_PASSWORD) - try: - os.chmod(pw_file, 0o600) - except OSError: - pass - print( - f"[server] No ADMIN_PASSWORD set — generated one and wrote it to {pw_file} " - f"(mode 0600). Username: {ADMIN_USERNAME}. Read it once, then set ADMIN_PASSWORD " - "in Railway and redeploy.", - flush=True, - ) -else: - print(f"[server] Admin username: {ADMIN_USERNAME}", flush=True) - -# ── Env var registry ────────────────────────────────────────────────────────── -# (key, label, category, is_secret) -ENV_VARS = [ - ("LLM_MODEL", "Model", "model", False), - ("OPENROUTER_API_KEY", "OpenRouter", "provider", True), - ("DEEPSEEK_API_KEY", "DeepSeek", "provider", True), - ("DASHSCOPE_API_KEY", "DashScope", "provider", True), - ("GLM_API_KEY", "GLM / Z.AI", "provider", True), - ("KIMI_API_KEY", "Kimi", "provider", True), - ("MINIMAX_API_KEY", "MiniMax", "provider", True), - ("HF_TOKEN", "Hugging Face", "provider", True), - # Added in v2026.4.23 (hermes v0.11.0). All plain API-key auth — hermes - # auto-routes by env-var presence, no extra config needed on our side. - # OAuth-based providers (Gemini CLI, Qwen OAuth, Claude Code, Copilot) - # are reachable via the dashboard's Keys tab and not exposed here. - ("NVIDIA_API_KEY", "NVIDIA NIM", "provider", True), - ("ARCEE_API_KEY", "Arcee AI", "provider", True), - ("STEPFUN_API_KEY", "Step Plan", "provider", True), - ("AI_GATEWAY_API_KEY", "Vercel AI Gateway", "provider", True), - ("GEMINI_API_KEY", "Google AI Studio", "provider", True), - ("PARALLEL_API_KEY", "Parallel (search)", "tool", True), - ("FIRECRAWL_API_KEY", "Firecrawl (scrape)", "tool", True), - ("TAVILY_API_KEY", "Tavily (search)", "tool", True), - ("FAL_KEY", "FAL (image gen)", "tool", True), - ("BROWSERBASE_API_KEY", "Browserbase key", "tool", True), - ("BROWSERBASE_PROJECT_ID", "Browserbase project", "tool", False), - ("GITHUB_TOKEN", "GitHub token", "tool", True), - ("VOICE_TOOLS_OPENAI_KEY", "OpenAI (voice/TTS)", "tool", True), - ("HONCHO_API_KEY", "Honcho (memory)", "tool", True), - ("TELEGRAM_BOT_TOKEN", "Bot Token", "telegram", True), - ("TELEGRAM_ALLOWED_USERS", "Allowed User IDs", "telegram", False), - ("DISCORD_BOT_TOKEN", "Bot Token", "discord", True), - ("DISCORD_ALLOWED_USERS", "Allowed User IDs", "discord", False), - ("SLACK_BOT_TOKEN", "Bot Token (xoxb-...)", "slack", True), - ("SLACK_APP_TOKEN", "App Token (xapp-...)", "slack", True), - ("WHATSAPP_ENABLED", "Enable WhatsApp", "whatsapp", False), - ("EMAIL_ADDRESS", "Email Address", "email", False), - ("EMAIL_PASSWORD", "Email Password", "email", True), - ("EMAIL_IMAP_HOST", "IMAP Host", "email", False), - ("EMAIL_SMTP_HOST", "SMTP Host", "email", False), - ("MATTERMOST_URL", "Server URL", "mattermost",False), - ("MATTERMOST_TOKEN", "Bot Token", "mattermost",True), - ("MATRIX_HOMESERVER", "Homeserver URL", "matrix", False), - ("MATRIX_ACCESS_TOKEN", "Access Token", "matrix", True), - ("MATRIX_USER_ID", "User ID", "matrix", False), - ("GATEWAY_ALLOW_ALL_USERS", "Allow all users", "gateway", False), - ("ADMIN_USERNAME", "Admin username", "admin", False), - ("ADMIN_PASSWORD", "Admin password", "admin", True), -] - -SECRET_KEYS = {k for k, _, _, s in ENV_VARS if s} -PROVIDER_KEYS = [k for k, _, c, _ in ENV_VARS if c == "provider"] -CHANNEL_MAP = { - "Telegram": "TELEGRAM_BOT_TOKEN", - "Discord": "DISCORD_BOT_TOKEN", - "Slack": "SLACK_BOT_TOKEN", - "WhatsApp": "WHATSAPP_ENABLED", - "Email": "EMAIL_ADDRESS", - "Mattermost": "MATTERMOST_TOKEN", - "Matrix": "MATRIX_ACCESS_TOKEN", -} - - -# ── .env helpers ────────────────────────────────────────────────────────────── -def read_env(path: Path) -> dict[str, str]: - if not path.exists(): - return {} - out = {} - for line in path.read_text().splitlines(): - line = line.strip() - if not line or line.startswith("#") or "=" not in line: - continue - k, _, v = line.partition("=") - v = v.strip() - if len(v) >= 2 and v[0] == v[-1] and v[0] in ('"', "'"): - v = v[1:-1] - out[k.strip()] = v - return out - - -def write_config_yaml(data: dict[str, str]) -> None: - """Write a minimal config.yaml so hermes picks up the model and provider.""" - import yaml - config = { - "model": {"default": data.get("LLM_MODEL", ""), "provider": "auto"}, - "terminal": {"backend": "local", "timeout": 60, "cwd": "/tmp"}, - "agent": {"max_iterations": 50}, - "data_dir": HERMES_HOME, - } - config_path = Path(HERMES_HOME) / "config.yaml" - config_path.parent.mkdir(parents=True, exist_ok=True) - config_path.write_text(yaml.safe_dump(config, sort_keys=False, default_flow_style=False)) - - -def write_env(path: Path, data: dict[str, str]) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - cat_order = ["model", "provider", "tool", - "telegram", "discord", "slack", "whatsapp", - "email", "mattermost", "matrix", "gateway"] - cat_labels = { - "model": "Model", "provider": "Providers", "tool": "Tools", - "telegram": "Telegram", "discord": "Discord", "slack": "Slack", - "whatsapp": "WhatsApp", "email": "Email", - "mattermost": "Mattermost", "matrix": "Matrix", "gateway": "Gateway", - } - key_cat = {k: c for k, _, c, _ in ENV_VARS} - grouped: dict[str, list[str]] = {c: [] for c in cat_order} - grouped["other"] = [] - - for k, v in data.items(): - if not v: - continue - cat = key_cat.get(k, "other") - grouped.setdefault(cat, []).append(f"{k}={v}") - - lines: list[str] = [] - for cat in cat_order: - entries = sorted(grouped.get(cat, [])) - if entries: - lines.append(f"# {cat_labels.get(cat, cat)}") - lines.extend(entries) - lines.append("") - if grouped["other"]: - lines.append("# Other") - lines.extend(sorted(grouped["other"])) - lines.append("") - - path.write_text("\n".join(lines)) - try: os.chmod(path, 0o600) - except OSError: pass - - -def is_config_complete(data: dict[str, str] | None = None) -> bool: - """Single source of truth for 'ready to run the gateway'. - - Used by: GET / redirect, auto_start on boot, admin API status. - """ - if data is None: - data = read_env(ENV_FILE) - has_model = bool(data.get("LLM_MODEL")) - has_provider = any(data.get(k) for k in PROVIDER_KEYS) - return has_model and has_provider - - -def mask(data: dict[str, str]) -> dict[str, str]: - return { - k: (v[:8] + "***" if len(v) > 8 else "***") if k in SECRET_KEYS and v else v - for k, v in data.items() - } - - -def unmask(new: dict[str, str], existing: dict[str, str]) -> dict[str, str]: - return { - k: (existing.get(k, "") if k in SECRET_KEYS and v.endswith("***") else v) - for k, v in new.items() - } - - -# ── Auth (cookie-based) ─────────────────────────────────────────────────────── -# We use HMAC-signed cookies instead of HTTP Basic Auth because: -# 1. Basic auth's per-directory protection space means browsers cache creds -# for /setup/* separately from /*, forcing re-prompt on navigation. -# 2. Browser behavior for sending Basic auth on XHR/fetch is inconsistent; -# the Hermes React SPA's plain fetch() calls don't reliably include it, -# causing every proxied API call to 401. -# Cookies are auto-included on every same-origin request (navigation + XHR) -# so both the setup UI and the proxied Hermes dashboard work with one login. -# -# The SECRET is regenerated on every process start. That means any ADMIN_PASSWORD -# change via Railway → redeploy → all existing cookies invalidate → users re-login. -import hashlib as _hashlib -import hmac as _hmac -from urllib.parse import quote as _url_quote, urlparse as _urlparse - -COOKIE_NAME = "hermes_auth" -COOKIE_MAX_AGE = 7 * 86400 # 7 days -COOKIE_SECRET = secrets.token_bytes(32) - -# Public paths — no auth required. Everything else is behind the cookie gate. -PUBLIC_PATHS = {"/health", "/login", "/logout"} - - -def _make_auth_token() -> str: - """Build a cookie value: `.`.""" - expires = str(int(time.time()) + COOKIE_MAX_AGE) - sig = _hmac.new(COOKIE_SECRET, expires.encode(), _hashlib.sha256).hexdigest() - return f"{expires}.{sig}" - - -def _verify_auth_token(token: str) -> bool: - try: - expires_s, sig = token.rsplit(".", 1) - if int(expires_s) < time.time(): - return False - expected = _hmac.new(COOKIE_SECRET, expires_s.encode(), _hashlib.sha256).hexdigest() - return _hmac.compare_digest(sig, expected) - except Exception: - return False - - -def _is_authenticated(request: Request) -> bool: - return _verify_auth_token(request.cookies.get(COOKIE_NAME, "")) - - -# Per-IP failed-login tracker. Single-process app, so an in-memory deque is enough. -_LOGIN_FAILS: dict[str, deque] = {} -_LOGIN_LOCK_WINDOW_SHORT = (5, 60) # 5 fails / 60s -_LOGIN_LOCK_WINDOW_LONG = (10, 600) # 10 fails / 10min - - -def _client_ip(request: Request) -> str: - fwd = request.headers.get("x-forwarded-for", "") - if fwd: - return fwd.split(",")[0].strip() - return request.client.host if request.client else "?" - - -def _login_locked(ip: str) -> bool: - now = time.time() - fails = _LOGIN_FAILS.get(ip) - if not fails: - return False - while fails and now - fails[0] > _LOGIN_LOCK_WINDOW_LONG[1]: - fails.popleft() - short_n = sum(1 for t in fails if now - t <= _LOGIN_LOCK_WINDOW_SHORT[1]) - if short_n >= _LOGIN_LOCK_WINDOW_SHORT[0]: - return True - if len(fails) >= _LOGIN_LOCK_WINDOW_LONG[0]: - return True - return False - - -def _login_record_fail(ip: str) -> None: - _LOGIN_FAILS.setdefault(ip, deque(maxlen=20)).append(time.time()) - - -def _login_clear(ip: str) -> None: - _LOGIN_FAILS.pop(ip, None) - - -def _safe_return_to(value: str) -> str: - """Reject open-redirect attempts — only allow same-origin relative paths.""" - if not value or not value.startswith("/") or value.startswith("//"): - return "/" - # Strip any scheme/netloc that slipped through. - p = _urlparse(value) - if p.scheme or p.netloc: - return "/" - return value - - -def guard(request: Request) -> Response | None: - """Enforce auth on protected routes. - - - HTML navigation: 302 to /login?returnTo= - - API / XHR: 401 JSON (so the SPA's fetch() can surface it cleanly) - """ - if _is_authenticated(request): - return None - accept = request.headers.get("accept", "").lower() - wants_html = "text/html" in accept - if wants_html: - rt = request.url.path - if request.url.query: - rt = f"{rt}?{request.url.query}" - return RedirectResponse(f"/login?returnTo={_url_quote(rt)}", status_code=302) - return JSONResponse({"error": "Unauthorized"}, status_code=401) - - -LOGIN_PAGE_HTML = """ - - -Hermes Agent — Sign in - - - - - -
-
- -
Sign in to continue
-
- __ERROR__ -
- - - - - - -
-

Credentials are the ADMIN_USERNAME and ADMIN_PASSWORD
Railway service variables.

-
-""" - - -def _html_escape(s: str) -> str: - return (s.replace("&", "&").replace("<", "<").replace(">", ">") - .replace('"', """).replace("'", "'")) - - -async def page_login(request: Request) -> Response: - """GET /login — render the sign-in form.""" - # Already signed in? Bounce to returnTo (or /). - if _is_authenticated(request): - return RedirectResponse(_safe_return_to(request.query_params.get("returnTo", "/")), status_code=302) - rt = _safe_return_to(request.query_params.get("returnTo", "/")) - err = request.query_params.get("error", "") - if err == "ratelimit": - error_html = ('
Too many failed attempts. ' - 'Please wait a minute before trying again.
') - elif err: - error_html = '
Invalid username or password
' - else: - error_html = "" - html = (LOGIN_PAGE_HTML - .replace("__ERROR__", error_html) - .replace("__RETURN_TO__", _html_escape(rt))) - return HTMLResponse(html) - - -async def login_post(request: Request) -> Response: - """POST /login — validate creds and set the auth cookie.""" - form = await request.form() - username = str(form.get("username", "")) - password = str(form.get("password", "")) - return_to = _safe_return_to(str(form.get("returnTo", "/"))) - ip = _client_ip(request) - - if _login_locked(ip): - return RedirectResponse( - f"/login?returnTo={_url_quote(return_to)}&error=ratelimit", status_code=302 - ) - - valid_user = _hmac.compare_digest(username, ADMIN_USERNAME) - valid_pw = _hmac.compare_digest(password, ADMIN_PASSWORD) - if valid_user and valid_pw: - _login_clear(ip) - resp = RedirectResponse(return_to, status_code=302) - resp.set_cookie( - COOKIE_NAME, - _make_auth_token(), - max_age=COOKIE_MAX_AGE, - httponly=True, - samesite="lax", - path="/", - ) - return resp - _login_record_fail(ip) - return RedirectResponse(f"/login?returnTo={_url_quote(return_to)}&error=1", status_code=302) - - -async def logout(request: Request) -> Response: - """GET /logout — clear cookie and bounce to login.""" - resp = RedirectResponse("/login", status_code=302) - resp.delete_cookie(COOKIE_NAME, path="/") - return resp - - -# ── Gateway manager ─────────────────────────────────────────────────────────── -class Gateway: - def __init__(self): - self.proc: asyncio.subprocess.Process | None = None - self.state = "stopped" - self.logs: deque[str] = deque(maxlen=500) - self.started_at: float | None = None - self.restarts = 0 - - async def start(self): - if self.proc and self.proc.returncode is None: - return - self.state = "starting" - try: - # .env values take priority over Railway env vars. - # We build the env this way so hermes's own dotenv loading - # (which reads the same file) doesn't shadow our values. - env = {**os.environ, "HERMES_HOME": HERMES_HOME} - env.update(read_env(ENV_FILE)) - model = env.get("LLM_MODEL", "") - provider_key = next((env.get(k, "") for k in PROVIDER_KEYS if env.get(k)), "") - print(f"[gateway] model={model or '⚠ NOT SET'} | provider_key={'set' if provider_key else '⚠ NOT SET'}", flush=True) - # Write config.yaml so hermes picks up the model (env vars alone aren't always enough) - write_config_yaml(read_env(ENV_FILE)) - self.proc = await asyncio.create_subprocess_exec( - "hermes", "gateway", - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.STDOUT, - env=env, - ) - self.state = "running" - self.started_at = time.time() - asyncio.create_task(self._drain()) - except Exception as e: - self.state = "error" - self.logs.append(f"[error] Failed to start: {e}") - - async def stop(self): - if not self.proc or self.proc.returncode is not None: - self.state = "stopped" - return - self.state = "stopping" - self.proc.terminate() - try: - await asyncio.wait_for(self.proc.wait(), timeout=10) - except asyncio.TimeoutError: - self.proc.kill() - await self.proc.wait() - self.state = "stopped" - self.started_at = None - - async def restart(self): - await self.stop() - self.restarts += 1 - await self.start() - - async def _drain(self): - assert self.proc and self.proc.stdout - async for raw in self.proc.stdout: - line = ANSI_ESCAPE.sub("", raw.decode(errors="replace").rstrip()) - self.logs.append(line) - if self.state == "running": - self.state = "error" - self.logs.append(f"[error] Gateway exited (code {self.proc.returncode})") - - def status(self) -> dict: - uptime = int(time.time() - self.started_at) if self.started_at and self.state == "running" else None - return { - "state": self.state, - "pid": self.proc.pid if self.proc and self.proc.returncode is None else None, - "uptime": uptime, - "restarts": self.restarts, - } - - -gw = Gateway() -cfg_lock = asyncio.Lock() - - -# ── Hermes dashboard subprocess ─────────────────────────────────────────────── -class Dashboard: - """Manages the `hermes dashboard` subprocess (native Hermes web UI). - - Bound to loopback only — we expose it to the public internet through our - reverse proxy on $PORT, where edge basic auth guards every request. - The dashboard is independent of the gateway: it reads config files - directly and tolerates a stopped gateway. - - All subprocess output is streamed to our stdout (→ Railway logs) with a - `[dashboard]` prefix AND retained in a ring buffer for diagnostics. - Unexpected exits are explicitly logged with their return code. - """ - - def __init__(self): - self.proc: asyncio.subprocess.Process | None = None - self.logs: deque[str] = deque(maxlen=300) - self._drain_task: asyncio.Task | None = None - - async def start(self): - if self.proc and self.proc.returncode is None: - return - try: - self.proc = await asyncio.create_subprocess_exec( - "hermes", "dashboard", - "--host", HERMES_DASHBOARD_HOST, - "--port", str(HERMES_DASHBOARD_PORT), - "--no-open", - # --tui exposes /api/pty + /api/ws + /api/events so the - # dashboard's embedded Chat tab works end-to-end. Requires - # hermes >= v2026.4.23 — older releases exit immediately - # with "unrecognized arguments: --tui". The Dockerfile - # pre-builds ui-tui/dist/ so PTY spawn is instant. - "--tui", - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.STDOUT, - ) - print(f"[dashboard] spawned pid={self.proc.pid} → {HERMES_DASHBOARD_URL}", flush=True) - self._drain_task = asyncio.create_task(self._drain()) - except Exception as e: - print(f"[dashboard] FAILED to spawn: {e!r}", flush=True) - - async def _drain(self): - """Stream subprocess output to Railway logs (prefixed) and a ring buffer.""" - assert self.proc and self.proc.stdout - try: - async for raw in self.proc.stdout: - line = ANSI_ESCAPE.sub("", raw.decode(errors="replace").rstrip()) - self.logs.append(line) - print(f"[dashboard] {line}", flush=True) - except Exception as e: - print(f"[dashboard] drain error: {e!r}", flush=True) - finally: - rc = self.proc.returncode if self.proc else None - if rc is not None and rc != 0: - print(f"[dashboard] EXITED with code {rc} — reverse proxy will return 503 until restart", flush=True) - elif rc == 0: - print(f"[dashboard] exited cleanly (code 0)", flush=True) - - async def stop(self): - if not self.proc or self.proc.returncode is not None: - return - self.proc.terminate() - try: - await asyncio.wait_for(self.proc.wait(), timeout=5) - except asyncio.TimeoutError: - self.proc.kill() - await self.proc.wait() - - -dash = Dashboard() - -# Shared async HTTP client for the reverse proxy. Created lazily so we pick up -# the running event loop, torn down in lifespan. -_http_client: httpx.AsyncClient | None = None - - -def get_http_client() -> httpx.AsyncClient: - global _http_client - if _http_client is None: - _http_client = httpx.AsyncClient( - timeout=httpx.Timeout(30.0, connect=5.0), - follow_redirects=False, - ) - return _http_client - - -# ── Route handlers ──────────────────────────────────────────────────────────── -async def page_index(request: Request): - if err := guard(request): return err - return templates.TemplateResponse(request, "index.html") - - -async def route_health(request: Request): - return JSONResponse({"status": "ok", "gateway": gw.state}) - - -async def api_config_get(request: Request): - if err := guard(request): return err - async with cfg_lock: - data = read_env(ENV_FILE) - defs = [{"key": k, "label": l, "category": c, "secret": s} for k, l, c, s in ENV_VARS] - return JSONResponse({"vars": mask(data), "defs": defs}) - - -async def api_config_put(request: Request): - if err := guard(request): return err - try: - body = await request.json() - except Exception: - return JSONResponse({"error": "Invalid JSON"}, status_code=400) - try: - restart = body.pop("_restart", False) - new_vars = body.get("vars", {}) - async with cfg_lock: - existing = read_env(ENV_FILE) - merged = unmask(new_vars, existing) - for k, v in existing.items(): - if k not in merged: - merged[k] = v - write_env(ENV_FILE, merged) - write_config_yaml(merged) - if restart: - asyncio.create_task(gw.restart()) - return JSONResponse({"ok": True, "restarting": restart}) - except Exception as e: - return JSONResponse({"error": str(e)}, status_code=500) - - -async def api_status(request: Request): - if err := guard(request): return err - data = read_env(ENV_FILE) - providers = { - k.replace("_API_KEY","").replace("_TOKEN","").replace("HF_","HuggingFace ").replace("_"," ").title(): - {"configured": bool(data.get(k))} - for k in PROVIDER_KEYS - } - channels = { - name: {"configured": bool(v := data.get(key,"")) and v.lower() not in ("false","0","no")} - for name, key in CHANNEL_MAP.items() - } - return JSONResponse({"gateway": gw.status(), "providers": providers, "channels": channels}) - - -async def api_logs(request: Request): - if err := guard(request): return err - return JSONResponse({"lines": list(gw.logs)}) - - -async def api_gw_start(request: Request): - if err := guard(request): return err - asyncio.create_task(gw.start()) - return JSONResponse({"ok": True}) - - -async def api_gw_stop(request: Request): - if err := guard(request): return err - asyncio.create_task(gw.stop()) - return JSONResponse({"ok": True}) - - -async def api_gw_restart(request: Request): - if err := guard(request): return err - asyncio.create_task(gw.restart()) - return JSONResponse({"ok": True}) - - -async def api_config_reset(request: Request): - if err := guard(request): return err - asyncio.create_task(gw.stop()) - async with cfg_lock: - if ENV_FILE.exists(): - ENV_FILE.unlink() - write_config_yaml({}) - return JSONResponse({"ok": True}) - - -# ── Pairing ─────────────────────────────────────────────────────────────────── -def _pjson(path: Path) -> dict: - try: - return json.loads(path.read_text()) if path.exists() else {} - except Exception: - return {} - - -def _wjson(path: Path, data: dict): - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text(json.dumps(data, indent=2, ensure_ascii=False)) - try: os.chmod(path, 0o600) - except OSError: pass - - -def _platforms(suffix: str) -> list[str]: - if not PAIRING_DIR.exists(): return [] - return [f.stem.rsplit(f"-{suffix}", 1)[0] for f in PAIRING_DIR.glob(f"*-{suffix}.json")] - - -async def api_pairing_pending(request: Request): - if err := guard(request): return err - now = time.time() - out = [] - for p in _platforms("pending"): - for code, info in _pjson(PAIRING_DIR / f"{p}-pending.json").items(): - if now - info.get("created_at", now) <= PAIRING_TTL: - out.append({"platform": p, "code": code, - "user_id": info.get("user_id",""), "user_name": info.get("user_name",""), - "age_minutes": int((now - info.get("created_at", now)) / 60)}) - return JSONResponse({"pending": out}) - - -async def api_pairing_approve(request: Request): - if err := guard(request): return err - try: body = await request.json() - except Exception: return JSONResponse({"error": "Invalid JSON"}, status_code=400) - platform, code = body.get("platform",""), body.get("code","").upper().strip() - if not platform or not code: - return JSONResponse({"error": "platform and code required"}, status_code=400) - pending_path = PAIRING_DIR / f"{platform}-pending.json" - pending = _pjson(pending_path) - if code not in pending: - return JSONResponse({"error": "Code not found"}, status_code=404) - entry = pending.pop(code) - _wjson(pending_path, pending) - approved = _pjson(PAIRING_DIR / f"{platform}-approved.json") - approved[entry["user_id"]] = {"user_name": entry.get("user_name",""), "approved_at": time.time()} - _wjson(PAIRING_DIR / f"{platform}-approved.json", approved) - return JSONResponse({"ok": True}) - - -async def api_pairing_deny(request: Request): - if err := guard(request): return err - try: body = await request.json() - except Exception: return JSONResponse({"error": "Invalid JSON"}, status_code=400) - platform, code = body.get("platform",""), body.get("code","").upper().strip() - p = PAIRING_DIR / f"{platform}-pending.json" - pending = _pjson(p) - if code in pending: - del pending[code] - _wjson(p, pending) - return JSONResponse({"ok": True}) - - -async def api_pairing_approved(request: Request): - if err := guard(request): return err - out = [] - for p in _platforms("approved"): - for uid, info in _pjson(PAIRING_DIR / f"{p}-approved.json").items(): - out.append({"platform": p, "user_id": uid, - "user_name": info.get("user_name",""), "approved_at": info.get("approved_at",0)}) - return JSONResponse({"approved": out}) - - -async def api_pairing_revoke(request: Request): - if err := guard(request): return err - try: body = await request.json() - except Exception: return JSONResponse({"error": "Invalid JSON"}, status_code=400) - platform, uid = body.get("platform",""), body.get("user_id","") - if not platform or not uid: - return JSONResponse({"error": "platform and user_id required"}, status_code=400) - p = PAIRING_DIR / f"{platform}-approved.json" - approved = _pjson(p) - if uid in approved: - del approved[uid] - _wjson(p, approved) - return JSONResponse({"ok": True}) - - -# ── Reverse proxy → Hermes dashboard ────────────────────────────────────────── -_WIDGET_LINK_STYLE = ( - "background:rgba(20,24,31,0.92);backdrop-filter:blur(8px);" - "border:1px solid #252d3d;border-radius:6px;padding:6px 12px;" - "color:#c9d1d9;text-decoration:none;display:inline-flex;" - "align-items:center;gap:6px;" -) -BACK_TO_SETUP_WIDGET = ( - '
' - f'← Setup' - f'Sign out' - '
' -) - -DASHBOARD_UNAVAILABLE_HTML = """ -Dashboard starting… - -
-

⚠ Hermes dashboard unavailable

-

The native Hermes dashboard is not responding on port %d.
-It may still be starting up, or it may have crashed.

-

Try refreshing in a few seconds, or head back to setup.

-← Back to Setup -
- -""" % HERMES_DASHBOARD_PORT - - -async def _proxy_to_dashboard(request: Request) -> Response: - """Forward an authenticated request to the Hermes dashboard subprocess. - - Assumes edge auth (basic auth middleware) has already validated the caller. - HTTP-only: the native Hermes dashboard does not use WebSockets. - """ - client = get_http_client() - target = f"{HERMES_DASHBOARD_URL}{request.url.path}" - if request.url.query: - target = f"{target}?{request.url.query}" - - req_headers = { - k: v for k, v in request.headers.items() - if k.lower() not in HOP_BY_HOP - } - body = await request.body() - - try: - upstream = await client.request( - request.method, - target, - headers=req_headers, - content=body, - ) - except (httpx.ConnectError, httpx.ConnectTimeout): - return HTMLResponse(DASHBOARD_UNAVAILABLE_HTML, status_code=503) - except httpx.RequestError as e: - print(f"[proxy] upstream error for {request.method} {request.url.path}: {e}", flush=True) - return HTMLResponse(DASHBOARD_UNAVAILABLE_HTML, status_code=502) - - # Surface non-2xx responses from hermes into Railway logs so we can - # diagnose 401/500s without needing browser DevTools access. - if upstream.status_code >= 400: - body_snip = upstream.content[:200].decode("utf-8", errors="replace") - print( - f"[proxy] {request.method} {request.url.path} -> {upstream.status_code} " - f"body={body_snip!r}", - flush=True, - ) - - # Strip hop-by-hop and length/encoding headers — Starlette recomputes them. - resp_headers = { - k: v for k, v in upstream.headers.items() - if k.lower() not in HOP_BY_HOP - and k.lower() not in ("content-encoding", "content-length") - } - - content = upstream.content - content_type = upstream.headers.get("content-type", "").lower() - - # Inject the "← Setup" widget into HTML pages so users can always return. - if "text/html" in content_type and b"" in content: - try: - text = content.decode("utf-8", errors="replace") - text = text.replace("", BACK_TO_SETUP_WIDGET + "", 1) - content = text.encode("utf-8") - except Exception: - pass # on any error, fall back to raw upstream content - - return Response( - content=content, - status_code=upstream.status_code, - headers=resp_headers, - ) - - -async def route_root(request: Request) -> Response: - """GET /: first-visit smart redirect, otherwise proxy to the dashboard. - - - Unconfigured + bare GET `/` → bounce to `/setup` so new users land on - the wizard instead of a half-empty dashboard. - - Sidebar / in-app links pass `?force=1` to opt out of that redirect — - users who explicitly want the dashboard (e.g. to set providers via - the Keys tab) can still reach it without saving config first. - - Non-GET (SPA API calls, etc.) always proxy through. - """ - if err := guard(request): return err - if (request.method == "GET" - and request.query_params.get("force") != "1" - and not is_config_complete()): - return RedirectResponse("/setup", status_code=302) - return await _proxy_to_dashboard(request) - - -async def route_proxy(request: Request) -> Response: - """Catch-all: forward any unmatched path to the Hermes dashboard.""" - if err := guard(request): return err - return await _proxy_to_dashboard(request) - - -async def route_setup_404(request: Request) -> Response: - """Typos under /setup/* should 404 here — not fall through to the proxy.""" - if err := guard(request): return err - return Response("Not Found", status_code=404, media_type="text/plain") - - -# ── App lifecycle ───────────────────────────────────────────────────────────── -async def auto_start(): - if is_config_complete(): - asyncio.create_task(gw.start()) - else: - print("[server] Config incomplete — gateway not started. Configure provider + model in the admin UI.", flush=True) - - -@asynccontextmanager -async def lifespan(app): - # Dashboard runs always — it's the user-facing UI after setup is done, - # and it's independent of gateway state. - asyncio.create_task(dash.start()) - await auto_start() - try: - yield - finally: - await asyncio.gather( - gw.stop(), - dash.stop(), - return_exceptions=True, - ) - global _http_client - if _http_client is not None: - await _http_client.aclose() - _http_client = None - - -# ── WebSocket reverse proxy ────────────────────────────────────────────────── -# The hermes dashboard exposes 4 WebSocket endpoints when started with --tui. -# Three are opened by the browser SPA and need to flow through our reverse -# proxy; the fourth (/api/pub) is opened only by the PTY child against -# loopback and is intentionally NOT proxied — exposing it would let an -# authed user spam events into channels. -# -# /api/pty binary stream — embedded TUI keystrokes/output -# /api/ws JSON-RPC — gateway sidecar driving Chat metadata -# /api/events text frames — dashboard subscriber for /api/pub fan-out -# -# Auth model (matches the HTTP proxy): -# * Edge: our HMAC cookie via _is_authenticated. WebSocket inherits .cookies -# from starlette HTTPConnection so the same helper works unchanged. -# * Upstream: hermes's own ?token=<_SESSION_TOKEN> query param. The SPA -# fetches that token via /api/auth/session-token and includes it in the -# WS URL, so we just forward path + query verbatim. -PROXIED_WS_PATHS = ("/api/pty", "/api/ws", "/api/events") - - -async def _ws_pump_client_to_upstream( - client: WebSocket, - upstream: websockets.WebSocketClientProtocol, -) -> None: - """Forward client → upstream until the client side disconnects. - - Handles both binary (PTY bytes) and text (JSON-RPC) frames. - """ - try: - while True: - msg = await client.receive() - if msg.get("type") == "websocket.disconnect": - return - data = msg.get("bytes") - if data is not None: - await upstream.send(data) - continue - text = msg.get("text") - if text is not None: - await upstream.send(text) - except (WebSocketDisconnect, websockets.exceptions.ConnectionClosed): - return - except Exception as e: - print(f"[ws-proxy] client→upstream error on {client.url.path}: {e!r}", flush=True) - return - - -async def _ws_pump_upstream_to_client( - upstream: websockets.WebSocketClientProtocol, - client: WebSocket, -) -> None: - """Forward upstream → client until upstream closes.""" - try: - async for msg in upstream: - if isinstance(msg, bytes): - await client.send_bytes(msg) - else: - await client.send_text(msg) - except (websockets.exceptions.ConnectionClosed, WebSocketDisconnect): - return - except Exception as e: - print(f"[ws-proxy] upstream→client error on {client.url.path}: {e!r}", flush=True) - return - - -async def ws_proxy(websocket: WebSocket) -> None: - """Reverse-proxy a single WebSocket from browser → hermes dashboard. - - Order matters: connect upstream BEFORE accepting the client. If hermes - is wedged or rejects the upgrade, we close the client with a meaningful - code instead of accepting and then dropping silently. - - Connection lifecycle: - 1. Verify edge cookie auth → 4401 close on failure - 2. Open upstream WS with bounded open_timeout → 1011 on failure - 3. Accept client - 4. Spawn two pump tasks (bidirectional byte forwarding) - 5. When either direction ends (client navigates away, upstream PTY - exits, etc.), cancel the other task and close both sockets - """ - # 1. Edge auth. - if not _is_authenticated(websocket): - # Close before accept — browser sees the handshake fail (expected - # for unauthenticated calls). - await websocket.close(code=4401) - return - - # 2. Build upstream URL preserving the SPA's path + query (the query - # contains the hermes session token + channel id). - path = websocket.url.path - qs = websocket.url.query - upstream_url = f"ws://{HERMES_DASHBOARD_HOST}:{HERMES_DASHBOARD_PORT}{path}" - if qs: - upstream_url = f"{upstream_url}?{qs}" - - try: - upstream = await websockets.connect( - upstream_url, - open_timeout=5, - # Don't forward client cookies/headers — hermes WS auth is - # purely token-based via the URL, and forwarding random - # headers risks future upstream surprises. - ) - except (asyncio.TimeoutError, OSError, websockets.exceptions.WebSocketException) as e: - # Hermes dashboard down, restarting, or rejected the upgrade - # (e.g. bad/missing session token). - print(f"[ws-proxy] upstream connect failed for {path}: {e!r}", flush=True) - # 1011 = internal error; client SPA will surface a generic close. - await websocket.close(code=1011) - return - - # 3. Both sides ready — accept and start pumping. - await websocket.accept() - - pump_in = asyncio.create_task(_ws_pump_client_to_upstream(websocket, upstream)) - pump_out = asyncio.create_task(_ws_pump_upstream_to_client(upstream, websocket)) - - try: - # First side to finish wins; cancel the other. - done, pending = await asyncio.wait( - (pump_in, pump_out), - return_when=asyncio.FIRST_COMPLETED, - ) - for task in pending: - task.cancel() - try: - await task - except (asyncio.CancelledError, Exception): - pass - finally: - # websockets.connect() outside `async with` doesn't auto-close; - # do it explicitly. Same for the client side if still open. - try: - await upstream.close() - except Exception: - pass - if websocket.client_state == WebSocketState.CONNECTED: - try: - await websocket.close() - except Exception: - pass - - -ANY_METHOD = ["GET", "POST", "PUT", "DELETE", "PATCH", "HEAD", "OPTIONS"] - -routes = [ - # Public — no auth required. - Route("/health", route_health), - Route("/login", page_login, methods=["GET"]), - Route("/login", login_post, methods=["POST"]), - Route("/logout", logout), - - # Our setup wizard + management API, all under /setup/* (cookie-auth guarded). - Route("/setup", page_index), - Route("/setup/", page_index), - Route("/setup/api/config", api_config_get, methods=["GET"]), - Route("/setup/api/config", api_config_put, methods=["PUT"]), - Route("/setup/api/status", api_status), - Route("/setup/api/logs", api_logs), - Route("/setup/api/gateway/start", api_gw_start, methods=["POST"]), - Route("/setup/api/gateway/stop", api_gw_stop, methods=["POST"]), - Route("/setup/api/gateway/restart", api_gw_restart, methods=["POST"]), - Route("/setup/api/config/reset", api_config_reset, methods=["POST"]), - Route("/setup/api/pairing/pending", api_pairing_pending), - Route("/setup/api/pairing/approve", api_pairing_approve, methods=["POST"]), - Route("/setup/api/pairing/deny", api_pairing_deny, methods=["POST"]), - Route("/setup/api/pairing/approved", api_pairing_approved), - Route("/setup/api/pairing/revoke", api_pairing_revoke, methods=["POST"]), - - # /setup/* typos return a real 404 — not a silent proxy fallthrough. - Route("/setup/{path:path}", route_setup_404, methods=ANY_METHOD), - - # Reverse-proxy hermes's dashboard WebSockets (Chat tab + sidecar). - # WebSocketRoute is matched independently of HTTP routes, so order - # relative to the catch-all HTTP `Route("/{path:path}", ...)` below - # doesn't matter — but listing them as a group keeps the surface - # area auditable. Only paths in PROXIED_WS_PATHS are forwarded; - # /api/pub is intentionally omitted. - WebSocketRoute("/api/pty", ws_proxy), - WebSocketRoute("/api/ws", ws_proxy), - WebSocketRoute("/api/events", ws_proxy), - - # Root: redirect to /setup if unconfigured, otherwise proxy the dashboard. - Route("/", route_root, methods=ANY_METHOD), - - # Catch-all: everything else proxies to the Hermes dashboard subprocess. - Route("/{path:path}", route_proxy, methods=ANY_METHOD), -] - -# No middleware — auth is enforced per-handler via guard(). This keeps /health -# and /login truly unauthenticated without middleware gymnastics. -app = Starlette(routes=routes, lifespan=lifespan) - -if __name__ == "__main__": - import uvicorn - port = int(os.environ.get("PORT", "8080")) - loop = asyncio.new_event_loop() - asyncio.set_event_loop(loop) - config = uvicorn.Config(app, host="0.0.0.0", port=port, log_level="info", loop="asyncio") - server = uvicorn.Server(config) - - def _shutdown(): - loop.create_task(gw.stop()) - loop.create_task(dash.stop()) - server.should_exit = True - - for sig in (signal.SIGTERM, signal.SIGINT): - loop.add_signal_handler(sig, _shutdown) - - loop.run_until_complete(server.serve()) diff --git a/start.sh b/start.sh index 3edfbf7..6034def 100644 --- a/start.sh +++ b/start.sh @@ -1,10 +1,8 @@ #!/bin/bash -set -e +set -euo pipefail + +# Required: tini delivers SIGTERM here, we forward to the gateway via `exec`. -# Mirror dashboard-ref-only's startup: create every directory hermes expects -# and seed a default config.yaml if the volume is empty. Without these, -# `hermes dashboard` endpoints that hit logs/, sessions/, cron/, etc. can fail -# with opaque errors even though no auth is actually involved. mkdir -p /data/.hermes/cron /data/.hermes/sessions /data/.hermes/logs \ /data/.hermes/memories /data/.hermes/skills /data/.hermes/pairing \ /data/.hermes/hooks /data/.hermes/image_cache /data/.hermes/audio_cache \ @@ -13,16 +11,28 @@ mkdir -p /data/.hermes/cron /data/.hermes/sessions /data/.hermes/logs \ if [ ! -f /data/.hermes/config.yaml ] && [ -f /opt/hermes-agent/cli-config.yaml.example ]; then cp /opt/hermes-agent/cli-config.yaml.example /data/.hermes/config.yaml fi - [ ! -f /data/.hermes/.env ] && touch /data/.hermes/.env -# Clear any stale gateway PID file left over from the previous container. -# `hermes gateway` writes /data/.hermes/gateway.pid on start but does not -# remove it on SIGTERM. Since /data is a persistent volume, the file -# survives container restarts and causes every subsequent boot to exit with -# "ERROR gateway.run: PID file race lost to another gateway instance". -# No hermes process can be running at this point (we're pre-exec in a fresh -# container), so removing the file unconditionally is safe. -rm -f /data/.hermes/gateway.pid +# `hermes gateway run --replace` (added upstream) supersedes the manual +# stale-PID cleanup the old server.py needed. + +: "${ADMIN_USERNAME:?ADMIN_USERNAME must be set}" +: "${ADMIN_PASSWORD:?ADMIN_PASSWORD must be set}" +: "${PORT:=8080}" + +# bcrypt hash so plaintext never lands on disk +ADMIN_PASSWORD_HASH="$(caddy hash-password --plaintext "$ADMIN_PASSWORD")" +export ADMIN_USERNAME ADMIN_PASSWORD_HASH PORT + +# Caddy's own envsubst-equivalent is `{$VAR}` in Caddyfile syntax, so we +# can hand it the template directly. +cp /app/Caddyfile.tmpl /tmp/Caddyfile + +# Native dashboard on loopback — Caddy fronts it with basic auth at the edge. +hermes dashboard --host 127.0.0.1 --port 9119 --no-open --tui & + +caddy run --config /tmp/Caddyfile --adapter caddyfile & -exec python /app/server.py +# Foreground: tini → start.sh → exec → hermes gateway. SIGTERM reaches the +# gateway directly so its own shutdown handlers run. +exec hermes gateway run --replace diff --git a/templates/index.html b/templates/index.html deleted file mode 100644 index 96dc1a9..0000000 --- a/templates/index.html +++ /dev/null @@ -1,1469 +0,0 @@ - - - - - -Hermes Agent - - - - - - - - - - - - -
- - - - - -
- - -
-
-
Setup
-
Configure your Hermes Agent
-
-
- -
- New to Hermes? Head over to the How to Setup section for a step-by-step guide on getting started. -
- - - -
-

- The dropdown below has the most-used providers. Hermes adds new ones - often and this list won't always be perfectly in sync — but every - provider Hermes supports is configurable from the - Hermes Dashboard - → Keys tab. A typical flow: pick one here (e.g. - OpenRouter) to get the agent running, then open the dashboard to add - or switch to any other provider you need. -

-
- - -
- -
- - -
- -
- - -

OpenRouter format: provider/model-name

-
-
- - - -
- - -
- - ✈️ - Telegram -
-
-
-
-

Use * to allow all users

-
-
- - -
- - 🎮 - Discord -
-
-
-
-

Use * to allow all users

-
-
- - -
- - 💼 - Slack -
-
-
-
-
-
-
- - -
- - 📧 - Email -
-
-
-
-
-
-
-
-
- - -
- - 🔷 - Mattermost -
-
-
-
-
-
-
- - -
- - 🔢 - Matrix -
-
-
-
-
-
-
-
- - -
- - 💬 - WhatsApp - pairing via gateway logs on first run -
- -
- - -
- -
-

Allow all users

-

Skip per-platform allowlists — use with caution

-
-
- - - -
- -
- -
-
- -
- - -
-
-
- - -
-
-
Status
-
-
-
- -
-
-
Gateway State
-
-
-
-
Uptime
-
-
-
-
Pending Pairs
-
-
-
-
Model
-
-
-
- -
- - - -
- -
-
- -
- -
-
-
- -
- -
-
-
- -
-
- - -
-
-
Logs
-
-
-
- -
No logs yet.
-
-
- - -
-
-
Users
-
Pairing requests & approved
-
-
- - -
No pending requests. Unauthorized users who message your bot will appear here.
- - - -
No approved users yet.
-
- - - - - - - - - - - - -
PlatformUserApproved
-
- -
-
- - -
-
-
How to Setup
-
Get your Hermes Agent running in minutes
-
-
- - -
-
- 1 - Get a Free LLM API Key -
-
    -
  1. - Go to openrouter.ai and create a free account. -
  2. -
  3. - Once logged in, go to openrouter.ai/workspaces/default/keys and click Create Key. - Copy the generated API key — it starts with sk-or-... -
  4. -
  5. - OpenRouter offers many models, including free ones. To get started, you can pick a free model such as: - nvidia/llama-3.3-nemotron-super-49b-v1:free - Browse all available models at openrouter.ai/models — filter by "Free" to see no-cost options. -
  6. -
  7. - Head over to the Setup page (left sidebar), select OpenRouter as provider, paste your API key, and enter the model name. - - Want a different provider (Anthropic, DeepSeek, Groq, Cerebras, Mistral, etc.)? Open the - Hermes Dashboard from the sidebar — the - Keys tab lists every provider Hermes supports out of the box, plus - skills, toolsets, and analytics you can configure from there. - -
  8. -
-
- - -
-
- 2 - Connect a Messaging Platform -
-

- Hermes Agent works through messaging platforms. You need to connect at least one. The easiest to set up is Telegram. -

-
    -
  1. - Open Telegram and search for @BotFather. -
  2. -
  3. - Send /newbot and follow the prompts — give your bot a name and username. -
  4. -
  5. - BotFather will reply with a Bot Token — it looks like 1234567890:ABCdefGhIjKlMnOpQrStUvWxYz. - Copy this token. -
  6. -
  7. - Go to the Setup page, enable Telegram under Messaging Channels, and paste the Bot Token. -
  8. -
  9. - For Allowed User IDs, if you're just starting out, enter * to allow all users. - You can restrict access later by specifying individual Telegram user IDs. -
  10. -
  11. - Click Save & Start on the bottom right — your Hermes Agent is now live! -
  12. -
-
- - -
- Facing an issue or something not working as expected? Create a GitHub issue and we'll help you out. -
- -
-
- - -
-
-
Other Templates
-
Deploy more AI agents on Railway
-
-
-

- Explore other AI agent templates you can deploy on Railway with one click. -

- -
- -
-
OpenClaw
-
- Self-hosted OpenClaw agent — deploy and manage your own OpenClaw instance on Railway. -
- - Deploy on Railway - -
- -
-
Paperclip
-
- Paperclip AI Company — deploy your own Paperclip AI agent instance on Railway. -
- - Deploy on Railway - -
- -
-
-
- -
-
- - -
- - - - From 210fdc14866179df0365885f5e0ec0e0f856ef49 Mon Sep 17 00:00:00 2001 From: tomfuertes Date: Fri, 1 May 2026 13:32:10 -0500 Subject: [PATCH 2/2] fix: rewrite upstream Host header for hermes dashboard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The native dashboard's web_server enforces a Host check unless launched with --insecure (which we don't want — it disables every binding-related guard). Caddy's default reverse_proxy preserves the original public Host (e.g. your-app.up.railway.app), which the dashboard rejects with "Invalid Host header. Dashboard requests must use the hostname the server was bound to." `header_up Host {upstream_hostport}` rewrites Host to 127.0.0.1:9119 on the upstream hop so the dashboard sees what it expects. X-Forwarded-Host preserves the original for any downstream code that wants it (link generation, redirects). --- Caddyfile.tmpl | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/Caddyfile.tmpl b/Caddyfile.tmpl index 42d78ea..354fbdf 100644 --- a/Caddyfile.tmpl +++ b/Caddyfile.tmpl @@ -12,6 +12,9 @@ basic_auth { {$ADMIN_USERNAME} {$ADMIN_PASSWORD_HASH} } - reverse_proxy 127.0.0.1:9119 + reverse_proxy 127.0.0.1:9119 { + header_up Host {upstream_hostport} + header_up X-Forwarded-Host {host} + } } }