Files
Solve-Field-GUI/hermes_llm/hermes_llm_gui.py
T

1038 lines
42 KiB
Python

#!/usr/bin/env python3
"""
Remote control GUI for:
- llama.cpp server (LLM)
- Hermes backend server
The GUI runs on Windows and controls a remote Ubuntu host over SSH.
Dependency:
pip install asyncssh
"""
from __future__ import annotations
import asyncio
import json
import os
import posixpath
import queue
import shlex
import threading
import tkinter as tk
from pathlib import Path
from tkinter import messagebox, scrolledtext, ttk
try:
import asyncssh
HAS_ASYNCSSH = True
except ImportError:
HAS_ASYNCSSH = False
SCRIPT_DIR = Path(__file__).resolve().parent
SETTINGS_PATH = SCRIPT_DIR / "hermes_llm_gui_settings.json"
class Tooltip:
DELAY_MS = 500
def __init__(self, widget: tk.Widget, text: str):
self.widget = widget
self.text = text
self.timer_id = None
self.tip_window = None
widget.bind("<Enter>", self._schedule, add="+")
widget.bind("<Leave>", self._cancel, add="+")
widget.bind("<ButtonPress>", self._cancel, add="+")
def _schedule(self, _event=None):
self._cancel()
self.timer_id = self.widget.after(self.DELAY_MS, self._show)
def _cancel(self, _event=None):
if self.timer_id:
self.widget.after_cancel(self.timer_id)
self.timer_id = None
if self.tip_window:
self.tip_window.destroy()
self.tip_window = None
def _show(self):
if self.tip_window:
return
x = self.widget.winfo_rootx() + 18
y = self.widget.winfo_rooty() + self.widget.winfo_height() + 4
self.tip_window = win = tk.Toplevel(self.widget)
win.wm_overrideredirect(True)
win.wm_geometry(f"+{x}+{y}")
win.attributes("-topmost", True)
tk.Label(
win,
text=self.text,
justify=tk.LEFT,
background="#fff8dc",
foreground="#1a1a1a",
relief=tk.SOLID,
borderwidth=1,
wraplength=420,
padx=8,
pady=6,
font=("Segoe UI", 9),
).pack()
def tip(widget: tk.Widget, text: str) -> tk.Widget:
if text:
Tooltip(widget, text)
return widget
class AsyncSSH:
def __init__(self, host: str, user: str, password: str, port: int = 22):
self.host = host
self.user = user
self.password = password
self.port = port
self._conn: asyncssh.SSHClientConnection | None = None
async def connect(self):
self._conn = await asyncssh.connect(
self.host,
port=self.port,
username=self.user,
password=self.password,
known_hosts=None,
)
async def _ensure_conn(self):
if self._conn is None:
await self.connect()
async def exec(self, cmd: str) -> str:
await self._ensure_conn()
result = await self._conn.run(cmd, check=False)
return (result.stdout or "") + (result.stderr or "")
async def exec_stream(self, cmd: str, output_cb):
await self._ensure_conn()
async with self._conn.create_process(
cmd,
stderr=asyncssh.STDOUT,
encoding="utf-8",
errors="replace",
) as proc:
async for line in proc.stdout:
output_cb(line if line.endswith("\n") else line + "\n")
output_cb(f"\n[process exited with code {proc.returncode}]\n")
return proc.returncode
async def close(self):
if self._conn:
self._conn.close()
await self._conn.wait_closed()
self._conn = None
class HermesLLMGUI:
def __init__(self, root: tk.Tk):
self.root = root
self.root.title("Hermes + LLM Remote Control")
self.root.geometry("1180x900")
self.root.minsize(980, 760)
self._loop = asyncio.new_event_loop()
threading.Thread(target=self._loop.run_forever, daemon=True, name="hermes-llm-asyncio").start()
self._log_queue: queue.Queue[tuple[str, str]] = queue.Queue()
self._chat_history: list[tuple[str, str]] = []
self._init_vars()
self._load_settings()
self._build_ui()
self._poll_log()
if not HAS_ASYNCSSH:
self._log("[error] asyncssh is missing. Install it with: pip install asyncssh\n", "error")
def _init_vars(self):
self.v_host = tk.StringVar(value="10.42.44.25")
self.v_port = tk.IntVar(value=22)
self.v_user = tk.StringVar(value="eskimue")
self.v_pass = tk.StringVar(value="Preside")
self.v_status = tk.StringVar(value="not connected")
self.v_llm_bind_host = tk.StringVar(value="0.0.0.0")
self.v_llm_port = tk.IntVar(value=8012)
self.v_llm_alias = tk.StringVar(value="qwen-local")
self.v_llm_threads = tk.IntVar(value=4)
self.v_llm_ctx = tk.IntVar(value=2048)
self.v_llm_model = tk.StringVar(
value="/home/eskimue/local-ai/models/qwen2.5-0.5b-instruct/qwen2.5-0.5b-instruct-q4_k_m.gguf"
)
self.v_llm_workdir = tk.StringVar(value="/home/eskimue/local-ai/llama.cpp")
self.v_llm_log = tk.StringVar(value="/home/eskimue/local-ai/logs/llama-server-qwen05b.log")
self.v_llm_pid = tk.StringVar(value="/home/eskimue/local-ai/run/llama-server-qwen05b.pid")
self.v_llm_status = tk.StringVar(value="unknown")
self.v_test_prompt = tk.StringVar(value="Antworte nur mit OK")
self.v_chat_mode = tk.StringVar(value="llm")
self.v_hermes_host = tk.StringVar(value="127.0.0.1")
self.v_hermes_port = tk.IntVar(value=9119)
self.v_hermes_venv = tk.StringVar(value="/home/eskimue/local-ai/hermes-venv")
self.v_hermes_log = tk.StringVar(value="/home/eskimue/local-ai/logs/hermes-serve.log")
self.v_hermes_pid = tk.StringVar(value="/home/eskimue/local-ai/run/hermes-serve.pid")
self.v_hermes_extra = tk.StringVar(value="--skip-build")
self.v_hermes_status = tk.StringVar(value="unknown")
self.v_chat_system = tk.StringVar(
value="Du bist Hermes. Antworte hilfreich, klar und auf Deutsch."
)
self.v_chat_input = tk.StringVar()
def _build_ui(self):
top = ttk.Frame(self.root, padding=6)
top.pack(fill=tk.X)
self._build_connection_bar(top)
body = ttk.PanedWindow(self.root, orient=tk.VERTICAL)
body.pack(fill=tk.BOTH, expand=True, padx=6, pady=(0, 6))
upper = ttk.Frame(body, padding=2)
upper.columnconfigure(0, weight=1)
upper.columnconfigure(1, weight=1)
body.add(upper, weight=3)
self._build_llm_panel(upper)
self._build_hermes_panel(upper)
lower = ttk.Frame(body, padding=2)
lower.columnconfigure(0, weight=1)
lower.rowconfigure(0, weight=1)
body.add(lower, weight=2)
lower_tabs = ttk.Notebook(lower)
lower_tabs.grid(row=0, column=0, sticky=tk.NSEW)
actions_tab = ttk.Frame(lower_tabs, padding=4)
chat_tab = ttk.Frame(lower_tabs, padding=4)
lower_tabs.add(actions_tab, text="Actions")
lower_tabs.add(chat_tab, text="Hermes Chat")
self._build_actions(actions_tab)
self._build_chat_tab(chat_tab)
log_frame = ttk.LabelFrame(self.root, text="Output", padding=4)
log_frame.pack(fill=tk.BOTH, expand=True, padx=6, pady=(0, 6))
self.log = scrolledtext.ScrolledText(
log_frame,
font=("Consolas", 9),
bg="#1e1e1e",
fg="#d4d4d4",
insertbackground="white",
wrap=tk.WORD,
state=tk.DISABLED,
)
self.log.pack(fill=tk.BOTH, expand=True)
self.log.tag_config("error", foreground="#f48771")
self.log.tag_config("ok", foreground="#89d185")
self.log.tag_config("info", foreground="#9cdcfe")
def _build_connection_bar(self, parent):
frame = ttk.LabelFrame(parent, text="SSH connection", padding=6)
frame.pack(fill=tk.X)
fields = [
("Host", self.v_host, 18, "Host name or IP of the Linux server."),
("Port", self.v_port, 6, "SSH port, usually 22."),
("User", self.v_user, 14, "SSH user that starts and stops the services."),
("Password", self.v_pass, 16, "Password stays only in memory unless you save settings."),
]
for idx, (label, var, width, help_text) in enumerate(fields):
ttk.Label(frame, text=f"{label}:").grid(row=0, column=idx * 2, padx=(4, 2), pady=2, sticky=tk.W)
entry = ttk.Entry(frame, textvariable=var, width=width, show="*" if label == "Password" else "")
entry.grid(row=0, column=idx * 2 + 1, padx=(0, 8), pady=2)
tip(entry, help_text)
tip(
ttk.Button(frame, text="Connect & test", command=self._test_connection),
"Checks SSH access and whether the configured remote directories exist.",
).grid(row=0, column=8, padx=4)
tip(
ttk.Button(frame, text="Refresh status", command=self._refresh_all_status),
"Checks the current LLM and Hermes process status on the server.",
).grid(row=0, column=9, padx=4)
tip(
ttk.Button(frame, text="Save settings", command=self._save_settings),
"Stores the current GUI values in a local JSON file next to this script.",
).grid(row=0, column=10, padx=4)
ttk.Label(frame, textvariable=self.v_status, foreground="#555").grid(row=0, column=11, padx=6, sticky=tk.W)
def _build_llm_panel(self, parent):
frame = ttk.LabelFrame(parent, text="LLM server (llama.cpp)", padding=8)
frame.grid(row=0, column=0, sticky=tk.NSEW, padx=(0, 6))
frame.columnconfigure(1, weight=1)
rows = [
("Bind host", self.v_llm_bind_host, "The host address llama-server listens on. Use 0.0.0.0 for external access."),
("Port", self.v_llm_port, "HTTP port for the OpenAI-compatible API."),
("Alias", self.v_llm_alias, "Public model alias used by the API, for example qwen-local."),
("Threads", self.v_llm_threads, "CPU threads used by llama-server."),
("Context", self.v_llm_ctx, "Context size passed to llama-server."),
("Model path", self.v_llm_model, "Absolute path to the remote GGUF file."),
("Work dir", self.v_llm_workdir, "Directory containing the llama.cpp build."),
("Log path", self.v_llm_log, "Remote file where stdout and stderr are written."),
("PID path", self.v_llm_pid, "Remote file which stores the llama-server process ID."),
]
for row, (label, var, help_text) in enumerate(rows):
ttk.Label(frame, text=f"{label}:").grid(row=row, column=0, sticky=tk.W, padx=4, pady=3)
ttk.Entry(frame, textvariable=var).grid(row=row, column=1, sticky=tk.EW, padx=4, pady=3)
tip(ttk.Label(frame, text="?"), help_text).grid(row=row, column=2, sticky=tk.W)
ttk.Label(frame, text="Status:").grid(row=len(rows), column=0, sticky=tk.W, padx=4, pady=(8, 3))
self.llm_status_label = ttk.Label(frame, textvariable=self.v_llm_status, foreground="#555")
self.llm_status_label.grid(row=len(rows), column=1, sticky=tk.W, padx=4, pady=(8, 3))
buttons = ttk.Frame(frame)
buttons.grid(row=len(rows) + 1, column=0, columnspan=3, sticky=tk.W, padx=4, pady=(8, 0))
tip(ttk.Button(buttons, text="Start LLM", command=self._start_llm), "Starts the remote llama-server process.").pack(
side=tk.LEFT, padx=(0, 6)
)
tip(ttk.Button(buttons, text="Stop LLM", command=self._stop_llm), "Stops the remote llama-server process.").pack(
side=tk.LEFT, padx=6
)
tip(
ttk.Button(buttons, text="LLM status", command=self._refresh_llm_status),
"Checks the current process state and whether the HTTP endpoint is healthy.",
).pack(side=tk.LEFT, padx=6)
tip(
ttk.Button(buttons, text="Tail LLM log", command=self._tail_llm_log),
"Shows the last lines from the remote llama-server log.",
).pack(side=tk.LEFT, padx=6)
def _build_hermes_panel(self, parent):
frame = ttk.LabelFrame(parent, text="Hermes backend", padding=8)
frame.grid(row=0, column=1, sticky=tk.NSEW)
frame.columnconfigure(1, weight=1)
rows = [
("Bind host", self.v_hermes_host, "Hermes serve host. Keep 127.0.0.1 unless you really need remote access."),
("Port", self.v_hermes_port, "HTTP/WebSocket port for hermes serve."),
("Venv path", self.v_hermes_venv, "Python virtual environment where Hermes Agent is installed."),
("Log path", self.v_hermes_log, "Remote file where Hermes serve writes its output."),
("PID path", self.v_hermes_pid, "Remote file which stores the Hermes process ID."),
("Extra args", self.v_hermes_extra, "Additional command line flags appended to 'hermes serve'."),
]
for row, (label, var, help_text) in enumerate(rows):
ttk.Label(frame, text=f"{label}:").grid(row=row, column=0, sticky=tk.W, padx=4, pady=3)
ttk.Entry(frame, textvariable=var).grid(row=row, column=1, sticky=tk.EW, padx=4, pady=3)
tip(ttk.Label(frame, text="?"), help_text).grid(row=row, column=2, sticky=tk.W)
ttk.Label(frame, text="Default mode:").grid(row=len(rows), column=0, sticky=tk.NW, padx=4, pady=(8, 3))
ttk.Label(
frame,
text=(
"This GUI starts 'hermes serve' as a background backend.\n"
"That is the most stable long-running Hermes process for this server."
),
justify=tk.LEFT,
foreground="#555",
).grid(row=len(rows), column=1, sticky=tk.W, padx=4, pady=(8, 3))
ttk.Label(frame, text="Status:").grid(row=len(rows) + 1, column=0, sticky=tk.W, padx=4, pady=(8, 3))
self.hermes_status_label = ttk.Label(frame, textvariable=self.v_hermes_status, foreground="#555")
self.hermes_status_label.grid(row=len(rows) + 1, column=1, sticky=tk.W, padx=4, pady=(8, 3))
buttons = ttk.Frame(frame)
buttons.grid(row=len(rows) + 2, column=0, columnspan=3, sticky=tk.W, padx=4, pady=(8, 0))
tip(ttk.Button(buttons, text="Start Hermes", command=self._start_hermes), "Starts 'hermes serve' in the background.").pack(
side=tk.LEFT, padx=(0, 6)
)
tip(ttk.Button(buttons, text="Stop Hermes", command=self._stop_hermes), "Stops the Hermes backend process.").pack(
side=tk.LEFT, padx=6
)
tip(
ttk.Button(buttons, text="Hermes status", command=self._refresh_hermes_status),
"Checks whether the Hermes backend process is currently running.",
).pack(side=tk.LEFT, padx=6)
tip(
ttk.Button(buttons, text="Tail Hermes log", command=self._tail_hermes_log),
"Shows the last lines from the remote Hermes log.",
).pack(side=tk.LEFT, padx=6)
def _build_actions(self, parent):
frame = ttk.LabelFrame(parent, text="Combined actions", padding=8)
frame.pack(fill=tk.BOTH, expand=True)
row = ttk.Frame(frame)
row.pack(fill=tk.X)
tip(
ttk.Button(row, text="Start both", command=self._start_both),
"Starts the llama.cpp server first and then the Hermes backend.",
).pack(side=tk.LEFT, padx=(0, 6))
tip(
ttk.Button(row, text="Stop both", command=self._stop_both),
"Stops Hermes first and then the llama.cpp server.",
).pack(side=tk.LEFT, padx=6)
tip(
ttk.Button(row, text="Health check", command=self._health_check),
"Runs a small remote probe against the configured LLM and Hermes ports.",
).pack(side=tk.LEFT, padx=6)
tip(
ttk.Button(row, text="Test chat", command=self._test_chat),
"Sends a valid POST request to /v1/chat/completions with the configured model alias.",
).pack(side=tk.LEFT, padx=6)
tip(ttk.Button(row, text="Clear output", command=self._clear_log), "Clears the output panel.").pack(
side=tk.RIGHT, padx=(6, 0)
)
prompt_row = ttk.Frame(frame)
prompt_row.pack(fill=tk.X, pady=(8, 0))
ttk.Label(prompt_row, text="Test prompt:").pack(side=tk.LEFT)
ttk.Entry(prompt_row, textvariable=self.v_test_prompt).pack(side=tk.LEFT, fill=tk.X, expand=True, padx=(6, 0))
ttk.Label(
frame,
text=(
"Note: Hermes itself currently has stricter model requirements than this small local model can satisfy.\n"
"So the Hermes backend can be started and stopped here, but not every Hermes workflow will be useful on this machine."
),
justify=tk.LEFT,
foreground="#555",
).pack(anchor=tk.W, pady=(8, 0))
def _build_chat_tab(self, parent):
parent.columnconfigure(0, weight=1)
parent.rowconfigure(2, weight=1)
config = ttk.LabelFrame(parent, text="Chat setup", padding=8)
config.grid(row=0, column=0, sticky=tk.EW, pady=(0, 6))
config.columnconfigure(1, weight=1)
ttk.Label(config, text="System prompt:").grid(row=0, column=0, sticky=tk.W, padx=4, pady=3)
ttk.Entry(config, textvariable=self.v_chat_system).grid(row=0, column=1, sticky=tk.EW, padx=4, pady=3)
ttk.Label(config, text="Mode:").grid(row=1, column=0, sticky=tk.W, padx=4, pady=3)
mode_row = ttk.Frame(config)
mode_row.grid(row=1, column=1, sticky=tk.W, padx=4, pady=3)
ttk.Radiobutton(mode_row, text="Direct LLM", variable=self.v_chat_mode, value="llm").pack(side=tk.LEFT)
ttk.Radiobutton(mode_row, text="Hermes", variable=self.v_chat_mode, value="hermes").pack(side=tk.LEFT, padx=(12, 0))
ttk.Label(
config,
text="Direct LLM is the recommended mode on this server. Hermes mode stays available for experiments.",
foreground="#555",
).grid(row=2, column=0, columnspan=2, sticky=tk.W, padx=4, pady=(2, 0))
buttons = ttk.Frame(parent)
buttons.grid(row=1, column=0, sticky=tk.W, pady=(0, 6))
tip(
ttk.Button(buttons, text="Send to Hermes", command=self._send_chat_message),
"Sends the current chat input using the selected mode.",
).pack(side=tk.LEFT, padx=(0, 6))
tip(
ttk.Button(buttons, text="Clear chat", command=self._clear_chat_history),
"Clears only the chat tab history in this GUI.",
).pack(side=tk.LEFT, padx=6)
history_frame = ttk.LabelFrame(parent, text="Conversation", padding=4)
history_frame.grid(row=2, column=0, sticky=tk.NSEW, pady=(0, 6))
history_frame.columnconfigure(0, weight=1)
history_frame.rowconfigure(0, weight=1)
self.chat_text = scrolledtext.ScrolledText(
history_frame,
wrap=tk.WORD,
font=("Segoe UI", 10),
state=tk.DISABLED,
height=12,
)
self.chat_text.grid(row=0, column=0, sticky=tk.NSEW)
input_frame = ttk.Frame(parent)
input_frame.grid(row=3, column=0, sticky=tk.EW)
input_frame.columnconfigure(0, weight=1)
entry = ttk.Entry(input_frame, textvariable=self.v_chat_input)
entry.grid(row=0, column=0, sticky=tk.EW, padx=(0, 6))
entry.bind("<Return>", lambda _event: self._send_chat_message())
tip(
ttk.Button(input_frame, text="Send", command=self._send_chat_message),
"Submits the current message to Hermes.",
).grid(row=0, column=1)
def _submit(self, coro):
return asyncio.run_coroutine_threadsafe(coro, self._loop)
def _make_ssh(self) -> AsyncSSH:
return AsyncSSH(
self.v_host.get().strip(),
self.v_user.get().strip(),
self.v_pass.get(),
int(self.v_port.get()),
)
def _load_settings(self):
if not SETTINGS_PATH.is_file():
return
try:
data = json.loads(SETTINGS_PATH.read_text(encoding="utf-8"))
except Exception:
return
for name, value in data.items():
var = getattr(self, name, None)
if var is not None and hasattr(var, "set"):
var.set(value)
def _save_settings(self):
data = {}
for name, value in self.__dict__.items():
if name.startswith("v_") and hasattr(value, "get"):
data[name] = value.get()
try:
SETTINGS_PATH.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
self._log(f"[ok] settings saved to {SETTINGS_PATH}\n", "ok")
except Exception as exc:
self._log(f"[error] saving settings failed: {exc}\n", "error")
def _test_connection(self):
if not HAS_ASYNCSSH:
messagebox.showerror("Missing dependency", "asyncssh is not installed.\nRun: pip install asyncssh")
return
self.v_status.set("connecting...")
self._log("[info] testing SSH connection\n", "info")
self._submit(self._async_test_connection())
async def _async_test_connection(self):
ssh = self._make_ssh()
try:
await ssh.connect()
cmd = (
"bash -lc "
+ shlex.quote(
"hostname && echo --- && "
f"test -d {shlex.quote(self.v_llm_workdir.get().strip())} && echo llama-dir-ok || echo llama-dir-missing; "
f"test -d {shlex.quote(self.v_hermes_venv.get().strip())} && echo hermes-venv-ok || echo hermes-venv-missing"
)
)
await ssh.exec_stream(cmd, self._log)
self.root.after(0, lambda: self.v_status.set("connected"))
except Exception as exc:
self._log(f"[error] {exc}\n", "error")
self.root.after(0, lambda: self.v_status.set("connection failed"))
finally:
await ssh.close()
def _llm_start_script(self) -> str:
workdir = self.v_llm_workdir.get().strip()
model = self.v_llm_model.get().strip()
alias = self.v_llm_alias.get().strip()
bind_host = self.v_llm_bind_host.get().strip()
port = int(self.v_llm_port.get())
threads = int(self.v_llm_threads.get())
ctx = int(self.v_llm_ctx.get())
log_path = self.v_llm_log.get().strip()
pid_path = self.v_llm_pid.get().strip()
return f"""
set -e
mkdir -p {shlex.quote(os.path.dirname(log_path))} {shlex.quote(os.path.dirname(pid_path))}
pkill -f "llama-server.*{model}" || true
cd {shlex.quote(workdir)}
nohup ./build/bin/llama-server \\
-m {shlex.quote(model)} \\
-a {shlex.quote(alias)} \\
-ngl 0 -t {threads} -c {ctx} --host {shlex.quote(bind_host)} --port {port} \\
> {shlex.quote(log_path)} 2>&1 &
echo $! > {shlex.quote(pid_path)}
sleep 3
cat {shlex.quote(pid_path)}
"""
def _llm_stop_script(self) -> str:
pid_path = self.v_llm_pid.get().strip()
model = self.v_llm_model.get().strip()
return f"""
set -e
if [ -f {shlex.quote(pid_path)} ]; then
pid=$(cat {shlex.quote(pid_path)})
kill "$pid" 2>/dev/null || true
rm -f {shlex.quote(pid_path)}
fi
pkill -f "llama-server.*{model}" || true
"""
def _llm_status_script(self) -> str:
pid_path = self.v_llm_pid.get().strip()
port = int(self.v_llm_port.get())
return f"""
if [ -f {shlex.quote(pid_path)} ]; then
pid=$(cat {shlex.quote(pid_path)})
if ps -p "$pid" > /dev/null 2>&1; then
echo "running: pid=$pid"
ps -p "$pid" -o pid=,etime=,cmd=
else
echo "stale pid file: $pid"
fi
else
echo "not running"
fi
printf '%s\\n' '--- health ---'
curl -s http://127.0.0.1:{port}/health || true
"""
def _hermes_start_script(self) -> str:
venv = self.v_hermes_venv.get().strip()
bind_host = self.v_hermes_host.get().strip()
port = int(self.v_hermes_port.get())
log_path = self.v_hermes_log.get().strip()
pid_path = self.v_hermes_pid.get().strip()
extra = self.v_hermes_extra.get().strip()
extra_suffix = f" {extra}" if extra else ""
return f"""
set -e
mkdir -p {shlex.quote(os.path.dirname(log_path))} {shlex.quote(os.path.dirname(pid_path))}
if [ -f {shlex.quote(pid_path)} ]; then
oldpid=$(cat {shlex.quote(pid_path)})
kill "$oldpid" 2>/dev/null || true
rm -f {shlex.quote(pid_path)}
fi
. {shlex.quote(posixpath.join(venv, "bin", "activate"))}
nohup hermes serve --host {shlex.quote(bind_host)} --port {port}{extra_suffix} > {shlex.quote(log_path)} 2>&1 &
echo $! > {shlex.quote(pid_path)}
sleep 3
cat {shlex.quote(pid_path)}
"""
def _hermes_stop_script(self) -> str:
pid_path = self.v_hermes_pid.get().strip()
port = int(self.v_hermes_port.get())
return f"""
set -e
if [ -f {shlex.quote(pid_path)} ]; then
pid=$(cat {shlex.quote(pid_path)})
kill "$pid" 2>/dev/null || true
rm -f {shlex.quote(pid_path)}
fi
pkill -f "hermes serve --host .* --port {port}" || true
"""
def _hermes_status_script(self) -> str:
pid_path = self.v_hermes_pid.get().strip()
port = int(self.v_hermes_port.get())
return f"""
if [ -f {shlex.quote(pid_path)} ]; then
pid=$(cat {shlex.quote(pid_path)})
if ps -p "$pid" > /dev/null 2>&1; then
echo "running: pid=$pid"
ps -p "$pid" -o pid=,etime=,cmd=
else
echo "stale pid file: $pid"
fi
else
echo "not running"
fi
printf '%s\\n' '--- port probe ---'
curl -sI http://127.0.0.1:{port} | head -1 || true
"""
def _start_llm(self):
self._log("[info] starting remote llama.cpp server\n", "info")
self._submit(self._async_run_script(self._llm_start_script(), "LLM started", self._refresh_llm_status))
def _stop_llm(self):
self._log("[info] stopping remote llama.cpp server\n", "info")
self._submit(self._async_run_script(self._llm_stop_script(), "LLM stopped", self._refresh_llm_status))
def _start_hermes(self):
self._log("[info] starting remote Hermes backend\n", "info")
self._submit(self._async_run_script(self._hermes_start_script(), "Hermes started", self._refresh_hermes_status))
def _stop_hermes(self):
self._log("[info] stopping remote Hermes backend\n", "info")
self._submit(self._async_run_script(self._hermes_stop_script(), "Hermes stopped", self._refresh_hermes_status))
def _start_both(self):
self._log("[info] starting LLM and Hermes\n", "info")
self._submit(self._async_start_both())
def _stop_both(self):
self._log("[info] stopping Hermes and LLM\n", "info")
self._submit(self._async_stop_both())
async def _async_start_both(self):
self._log("[info] starting LLM first\n", "info")
await self._async_run_script(self._llm_start_script(), "LLM started")
self._log("[info] starting Hermes next\n", "info")
await self._async_run_script(self._hermes_start_script(), "Hermes started")
self.root.after(0, self._refresh_all_status)
async def _async_stop_both(self):
await self._async_run_script(self._hermes_stop_script(), "Hermes stopped")
await self._async_run_script(self._llm_stop_script(), "LLM stopped")
self.root.after(0, self._refresh_all_status)
def _refresh_llm_status(self):
self._submit(self._async_refresh_service_status("llm"))
def _refresh_hermes_status(self):
self._submit(self._async_refresh_service_status("hermes"))
def _refresh_all_status(self):
self._refresh_llm_status()
self._refresh_hermes_status()
async def _async_refresh_service_status(self, service: str):
ssh = self._make_ssh()
try:
await ssh.connect()
if service == "llm":
output = await ssh.exec(f"bash -lc {shlex.quote(self._llm_status_script())}")
self._log("[info] LLM status\n" + output + "\n", "info")
first_line = output.strip().splitlines()[0] if output.strip() else "unknown"
self.root.after(0, lambda: self._set_llm_status(first_line))
else:
output = await ssh.exec(f"bash -lc {shlex.quote(self._hermes_status_script())}")
self._log("[info] Hermes status\n" + output + "\n", "info")
first_line = output.strip().splitlines()[0] if output.strip() else "unknown"
self.root.after(0, lambda: self._set_hermes_status(first_line))
except Exception as exc:
self._log(f"[error] status check failed: {exc}\n", "error")
finally:
await ssh.close()
def _tail_llm_log(self):
self._submit(self._async_tail_log(self.v_llm_log.get().strip(), "LLM"))
def _tail_hermes_log(self):
self._submit(self._async_tail_log(self.v_hermes_log.get().strip(), "Hermes"))
async def _async_tail_log(self, log_path: str, label: str):
ssh = self._make_ssh()
try:
await ssh.connect()
cmd = f"bash -lc {shlex.quote(f'tail -n 80 {shlex.quote(log_path)}')}"
output = await ssh.exec(cmd)
self._log(f"[info] tail {label} log\n{output}\n", "info")
except Exception as exc:
self._log(f"[error] tail log failed: {exc}\n", "error")
finally:
await ssh.close()
def _health_check(self):
self._submit(self._async_health_check())
def _test_chat(self):
self._log("[info] running test chat request\n", "info")
self._submit(self._async_test_chat())
def _send_chat_message(self):
text = self.v_chat_input.get().strip()
if not text:
return
self.v_chat_input.set("")
self._append_chat_message("You", text)
self._chat_history.append(("user", text))
if self.v_chat_mode.get() == "hermes":
self._submit(self._async_send_chat_message())
else:
self._submit(self._async_send_direct_llm_chat_message())
def _clear_chat_history(self):
self._chat_history.clear()
self.chat_text.config(state=tk.NORMAL)
self.chat_text.delete("1.0", tk.END)
self.chat_text.config(state=tk.DISABLED)
def _append_chat_message(self, speaker: str, text: str):
self.chat_text.config(state=tk.NORMAL)
self.chat_text.insert(tk.END, f"{speaker}:\n{text.strip()}\n\n")
self.chat_text.see(tk.END)
self.chat_text.config(state=tk.DISABLED)
def _set_llm_status(self, text: str):
self.v_llm_status.set(text)
color = "#555"
lower = text.lower()
if "running" in lower:
color = "#188038"
elif "not running" in lower or "stale" in lower:
color = "#c5221f"
self.llm_status_label.configure(foreground=color)
def _set_hermes_status(self, text: str):
self.v_hermes_status.set(text)
color = "#555"
lower = text.lower()
if "running" in lower:
color = "#188038"
elif "not running" in lower or "stale" in lower:
color = "#c5221f"
self.hermes_status_label.configure(foreground=color)
def _render_hermes_prompt(self) -> str:
lines = [self.v_chat_system.get().strip(), "", "Conversation so far:"]
for role, content in self._chat_history[-12:]:
label = "User" if role == "user" else "Hermes"
lines.append(f"{label}: {content}")
lines.append("")
lines.append("Answer now as Hermes in German.")
return "\n".join(lines)
async def _ensure_llm_ready(self, ssh: AsyncSSH) -> bool:
llm_port = int(self.v_llm_port.get())
llm_host = self.v_host.get().strip()
health_cmd = "bash -lc " + shlex.quote(f"curl -s http://{llm_host}:{llm_port}/health")
health_output = await ssh.exec(health_cmd)
if '"status":"ok"' in health_output.replace(" ", ""):
return True
self._log(
f"[info] LLM not reachable on {llm_host}:{llm_port}, trying automatic restart\n",
"info",
)
start_output = await ssh.exec(f"bash -lc {shlex.quote(self._llm_start_script())}")
if start_output.strip():
self._log(start_output if start_output.endswith("\n") else start_output + "\n", "info")
await asyncio.sleep(12)
health_output = await ssh.exec(health_cmd)
if '"status":"ok"' in health_output.replace(" ", ""):
self._log(f"[ok] LLM restart succeeded on {llm_host}:{llm_port}\n", "ok")
return True
self._log(
f"[error] LLM is still not reachable on {llm_host}:{llm_port} after restart\n",
"error",
)
if health_output.strip():
self._log(health_output if health_output.endswith("\n") else health_output + "\n", "error")
return False
async def _async_health_check(self):
ssh = self._make_ssh()
try:
await ssh.connect()
llm_port = int(self.v_llm_port.get())
hermes_port = int(self.v_hermes_port.get())
llm_host = self.v_host.get().strip()
hermes_host = self.v_hermes_host.get().strip()
cmd = (
"bash -lc "
+ shlex.quote(
f"printf '%s\\n' '--- LLM ---'; "
f"curl -s http://{llm_host}:{llm_port}/health || true; "
f"printf '\\n%s\\n' '--- Hermes ---'; "
f"curl -sI http://{hermes_host}:{hermes_port} | head -1 || true"
)
)
output = await ssh.exec(cmd)
self._log(f"[info] health check\n{output}\n", "info")
except Exception as exc:
self._log(f"[error] health check failed: {exc}\n", "error")
finally:
await ssh.close()
async def _async_test_chat(self):
ssh = self._make_ssh()
try:
await ssh.connect()
llm_port = int(self.v_llm_port.get())
llm_host = self.v_host.get().strip()
alias = self.v_llm_alias.get().strip()
prompt = self.v_test_prompt.get().strip() or "Antworte nur mit OK"
if not await self._ensure_llm_ready(ssh):
self._log(
f"[error] test chat aborted: the LLM server is not reachable on {llm_host}:{llm_port}\n",
"error",
)
return
payload = json.dumps(
{
"model": alias,
"messages": [{"role": "user", "content": prompt}],
"max_tokens": 32,
"temperature": 0,
"stream": False,
},
ensure_ascii=False,
)
python_script = (
"import json, sys, urllib.request\n"
f"payload = {payload!r}\n"
"req = urllib.request.Request("
f"'http://{llm_host}:{llm_port}/v1/chat/completions', "
"data=payload.encode('utf-8'), "
"headers={'Content-Type': 'application/json'}, "
"method='POST')\n"
"with urllib.request.urlopen(req, timeout=180) as resp:\n"
" print(resp.read().decode('utf-8'))\n"
)
cmd = "bash -lc " + shlex.quote(f"python3 - <<'PY'\n{python_script}PY")
await ssh._ensure_conn()
result = await ssh._conn.run(cmd, check=False)
output = (result.stdout or "") + (result.stderr or "")
if result.exit_status == 0:
self._log(f"[ok] test chat response\n{output}\n", "ok")
else:
self._log("[error] test chat request failed\n", "error")
if output.strip():
self._log(output if output.endswith("\n") else output + "\n", "error")
except Exception as exc:
self._log(f"[error] test chat failed: {exc}\n", "error")
finally:
await ssh.close()
async def _async_send_chat_message(self):
ssh = self._make_ssh()
try:
await ssh.connect()
if not await self._ensure_llm_ready(ssh):
self.root.after(
0,
lambda: self._append_chat_message(
"Hermes Error",
f"Der LLM-Server auf {self.v_host.get().strip()}:{int(self.v_llm_port.get())} ist nicht erreichbar.",
),
)
return
venv = self.v_hermes_venv.get().strip()
prompt = self._render_hermes_prompt()
cmd = (
"bash -lc "
+ shlex.quote(
f". {shlex.quote(posixpath.join(venv, 'bin', 'activate'))} && "
f"hermes chat -q {shlex.quote(prompt)} -Q --cli"
)
)
await ssh._ensure_conn()
result = await ssh._conn.run(cmd, check=False)
output = ((result.stdout or "") + (result.stderr or "")).strip()
if result.exit_status == 0 and output:
self._chat_history.append(("assistant", output))
self.root.after(0, lambda: self._append_chat_message("Hermes", output))
else:
error_text = output or f"Hermes exited with code {result.exit_status}"
self.root.after(0, lambda: self._append_chat_message("Hermes Error", error_text))
self._log(f"[error] Hermes chat failed\n{error_text}\n", "error")
except Exception as exc:
error_text = str(exc)
self.root.after(0, lambda: self._append_chat_message("Hermes Error", error_text))
self._log(f"[error] Hermes chat failed: {exc}\n", "error")
finally:
await ssh.close()
async def _async_send_direct_llm_chat_message(self):
ssh = self._make_ssh()
try:
await ssh.connect()
if not await self._ensure_llm_ready(ssh):
self.root.after(
0,
lambda: self._append_chat_message(
"LLM Error",
f"Der LLM-Server auf {self.v_host.get().strip()}:{int(self.v_llm_port.get())} ist nicht erreichbar.",
),
)
return
llm_port = int(self.v_llm_port.get())
llm_host = self.v_host.get().strip()
alias = self.v_llm_alias.get().strip()
messages = []
system = self.v_chat_system.get().strip()
if system:
messages.append({"role": "system", "content": system})
for role, content in self._chat_history[-12:]:
messages.append({"role": role, "content": content})
payload = json.dumps(
{
"model": alias,
"messages": messages,
"max_tokens": 192,
"temperature": 0.4,
"stream": False,
},
ensure_ascii=False,
)
python_script = (
"import json, urllib.request\n"
f"payload = {payload!r}\n"
"req = urllib.request.Request("
f"'http://{llm_host}:{llm_port}/v1/chat/completions', "
"data=payload.encode('utf-8'), "
"headers={'Content-Type': 'application/json'}, "
"method='POST')\n"
"with urllib.request.urlopen(req, timeout=180) as resp:\n"
" data = json.loads(resp.read().decode('utf-8'))\n"
" print(data['choices'][0]['message']['content'])\n"
)
cmd = "bash -lc " + shlex.quote(f"python3 - <<'PY'\n{python_script}PY")
await ssh._ensure_conn()
result = await ssh._conn.run(cmd, check=False)
output = ((result.stdout or "") + (result.stderr or "")).strip()
if result.exit_status == 0 and output:
self._chat_history.append(("assistant", output))
self.root.after(0, lambda: self._append_chat_message("LLM", output))
else:
error_text = output or f"LLM exited with code {result.exit_status}"
self.root.after(0, lambda: self._append_chat_message("LLM Error", error_text))
self._log(f"[error] LLM chat failed\n{error_text}\n", "error")
except Exception as exc:
error_text = str(exc)
self.root.after(0, lambda: self._append_chat_message("LLM Error", error_text))
self._log(f"[error] LLM chat failed: {exc}\n", "error")
finally:
await ssh.close()
async def _async_run_script(self, script: str, ok_message: str = "", callback=None):
ssh = self._make_ssh()
try:
await ssh.connect()
result = await ssh.exec(f"bash -lc {shlex.quote(script)}")
if result.strip():
self._log(result if result.endswith("\n") else result + "\n", "info")
if ok_message:
self._log(f"[ok] {ok_message}\n", "ok")
except Exception as exc:
self._log(f"[error] {exc}\n", "error")
finally:
await ssh.close()
if callback:
self.root.after(0, callback)
def _log(self, text: str, tag: str = ""):
self._log_queue.put((text, tag))
def _poll_log(self):
try:
while True:
text, tag = self._log_queue.get_nowait()
self.log.config(state=tk.NORMAL)
self.log.insert(tk.END, text, tag or "")
self.log.see(tk.END)
self.log.config(state=tk.DISABLED)
except queue.Empty:
pass
self.root.after(100, self._poll_log)
def _clear_log(self):
self.log.config(state=tk.NORMAL)
self.log.delete("1.0", tk.END)
self.log.config(state=tk.DISABLED)
def on_close(self):
self._loop.call_soon_threadsafe(self._loop.stop)
self.root.destroy()
def main():
root = tk.Tk()
style = ttk.Style(root)
for preferred in ("clam", "alt", "default"):
if preferred in style.theme_names():
style.theme_use(preferred)
break
app = HermesLLMGUI(root)
root.protocol("WM_DELETE_WINDOW", app.on_close)
root.mainloop()
if __name__ == "__main__":
main()