#!/usr/bin/env python3 """ Remote control GUI for: - llama.cpp server (LLM) - Hermes backend server The GUI runs on Windows and controls a remote Ubuntu host over SSH. Dependency: pip install asyncssh """ from __future__ import annotations import asyncio import json import os import posixpath import queue import shlex import threading import tkinter as tk from pathlib import Path from tkinter import messagebox, scrolledtext, ttk try: import asyncssh HAS_ASYNCSSH = True except ImportError: HAS_ASYNCSSH = False SCRIPT_DIR = Path(__file__).resolve().parent SETTINGS_PATH = SCRIPT_DIR / "hermes_llm_gui_settings.json" class Tooltip: DELAY_MS = 500 def __init__(self, widget: tk.Widget, text: str): self.widget = widget self.text = text self.timer_id = None self.tip_window = None widget.bind("", self._schedule, add="+") widget.bind("", self._cancel, add="+") widget.bind("", self._cancel, add="+") def _schedule(self, _event=None): self._cancel() self.timer_id = self.widget.after(self.DELAY_MS, self._show) def _cancel(self, _event=None): if self.timer_id: self.widget.after_cancel(self.timer_id) self.timer_id = None if self.tip_window: self.tip_window.destroy() self.tip_window = None def _show(self): if self.tip_window: return x = self.widget.winfo_rootx() + 18 y = self.widget.winfo_rooty() + self.widget.winfo_height() + 4 self.tip_window = win = tk.Toplevel(self.widget) win.wm_overrideredirect(True) win.wm_geometry(f"+{x}+{y}") win.attributes("-topmost", True) tk.Label( win, text=self.text, justify=tk.LEFT, background="#fff8dc", foreground="#1a1a1a", relief=tk.SOLID, borderwidth=1, wraplength=420, padx=8, pady=6, font=("Segoe UI", 9), ).pack() def tip(widget: tk.Widget, text: str) -> tk.Widget: if text: Tooltip(widget, text) return widget class AsyncSSH: def __init__(self, host: str, user: str, password: str, port: int = 22): self.host = host self.user = user self.password = password self.port = port self._conn: asyncssh.SSHClientConnection | None = None async def connect(self): self._conn = await asyncssh.connect( self.host, port=self.port, username=self.user, password=self.password, known_hosts=None, ) async def _ensure_conn(self): if self._conn is None: await self.connect() async def exec(self, cmd: str) -> str: await self._ensure_conn() result = await self._conn.run(cmd, check=False) return (result.stdout or "") + (result.stderr or "") async def exec_stream(self, cmd: str, output_cb): await self._ensure_conn() async with self._conn.create_process( cmd, stderr=asyncssh.STDOUT, encoding="utf-8", errors="replace", ) as proc: async for line in proc.stdout: output_cb(line if line.endswith("\n") else line + "\n") output_cb(f"\n[process exited with code {proc.returncode}]\n") return proc.returncode async def close(self): if self._conn: self._conn.close() await self._conn.wait_closed() self._conn = None class HermesLLMGUI: def __init__(self, root: tk.Tk): self.root = root self.root.title("Hermes + LLM Remote Control") self.root.geometry("1180x900") self.root.minsize(980, 760) self._loop = asyncio.new_event_loop() threading.Thread(target=self._loop.run_forever, daemon=True, name="hermes-llm-asyncio").start() self._log_queue: queue.Queue[tuple[str, str]] = queue.Queue() self._chat_history: list[tuple[str, str]] = [] self._init_vars() self._load_settings() self._build_ui() self._poll_log() if not HAS_ASYNCSSH: self._log("[error] asyncssh is missing. Install it with: pip install asyncssh\n", "error") def _init_vars(self): self.v_host = tk.StringVar(value="10.42.44.25") self.v_port = tk.IntVar(value=22) self.v_user = tk.StringVar(value="eskimue") self.v_pass = tk.StringVar(value="Preside") self.v_status = tk.StringVar(value="not connected") self.v_llm_bind_host = tk.StringVar(value="0.0.0.0") self.v_llm_port = tk.IntVar(value=8012) self.v_llm_alias = tk.StringVar(value="qwen-local") self.v_llm_threads = tk.IntVar(value=4) self.v_llm_ctx = tk.IntVar(value=2048) self.v_llm_model = tk.StringVar( value="/home/eskimue/local-ai/models/qwen2.5-0.5b-instruct/qwen2.5-0.5b-instruct-q4_k_m.gguf" ) self.v_llm_workdir = tk.StringVar(value="/home/eskimue/local-ai/llama.cpp") self.v_llm_log = tk.StringVar(value="/home/eskimue/local-ai/logs/llama-server-qwen05b.log") self.v_llm_pid = tk.StringVar(value="/home/eskimue/local-ai/run/llama-server-qwen05b.pid") self.v_llm_status = tk.StringVar(value="unknown") self.v_test_prompt = tk.StringVar(value="Antworte nur mit OK") self.v_chat_mode = tk.StringVar(value="llm") self.v_hermes_host = tk.StringVar(value="127.0.0.1") self.v_hermes_port = tk.IntVar(value=9119) self.v_hermes_venv = tk.StringVar(value="/home/eskimue/local-ai/hermes-venv") self.v_hermes_log = tk.StringVar(value="/home/eskimue/local-ai/logs/hermes-serve.log") self.v_hermes_pid = tk.StringVar(value="/home/eskimue/local-ai/run/hermes-serve.pid") self.v_hermes_extra = tk.StringVar(value="--skip-build") self.v_hermes_status = tk.StringVar(value="unknown") self.v_chat_system = tk.StringVar( value="Du bist Hermes. Antworte hilfreich, klar und auf Deutsch." ) self.v_chat_input = tk.StringVar() def _build_ui(self): top = ttk.Frame(self.root, padding=6) top.pack(fill=tk.X) self._build_connection_bar(top) body = ttk.PanedWindow(self.root, orient=tk.VERTICAL) body.pack(fill=tk.BOTH, expand=True, padx=6, pady=(0, 6)) top_area = ttk.PanedWindow(body, orient=tk.VERTICAL) body.add(top_area, weight=5) upper = ttk.Frame(top_area, padding=2) upper.columnconfigure(0, weight=1) upper.columnconfigure(1, weight=1) top_area.add(upper, weight=4) self._build_llm_panel(upper) self._build_hermes_panel(upper) lower = ttk.Frame(top_area, padding=2) lower.columnconfigure(0, weight=1) lower.rowconfigure(0, weight=1) top_area.add(lower, weight=3) lower_tabs = ttk.Notebook(lower) lower_tabs.grid(row=0, column=0, sticky=tk.NSEW) actions_tab = ttk.Frame(lower_tabs, padding=4) chat_tab = ttk.Frame(lower_tabs, padding=4) lower_tabs.add(actions_tab, text="Actions") lower_tabs.add(chat_tab, text="Chat") self._build_actions(actions_tab) self._build_chat_tab(chat_tab) log_frame = ttk.LabelFrame(body, text="Output", padding=4) body.add(log_frame, weight=4) log_frame.columnconfigure(0, weight=1) log_frame.rowconfigure(0, weight=1) self.log = scrolledtext.ScrolledText( log_frame, font=("Consolas", 10), bg="#1e1e1e", fg="#d4d4d4", insertbackground="white", wrap=tk.WORD, state=tk.DISABLED, height=18, ) self.log.grid(row=0, column=0, sticky=tk.NSEW) self.log.tag_config("error", foreground="#f48771") self.log.tag_config("ok", foreground="#89d185") self.log.tag_config("info", foreground="#9cdcfe") self.root.after(100, lambda: self._set_initial_pane_sizes(body, top_area)) def _build_connection_bar(self, parent): frame = ttk.LabelFrame(parent, text="SSH connection", padding=6) frame.pack(fill=tk.X) fields = [ ("Host", self.v_host, 18, "Host name or IP of the Linux server."), ("Port", self.v_port, 6, "SSH port, usually 22."), ("User", self.v_user, 14, "SSH user that starts and stops the services."), ("Password", self.v_pass, 16, "Password stays only in memory unless you save settings."), ] for idx, (label, var, width, help_text) in enumerate(fields): ttk.Label(frame, text=f"{label}:").grid(row=0, column=idx * 2, padx=(4, 2), pady=2, sticky=tk.W) entry = ttk.Entry(frame, textvariable=var, width=width, show="*" if label == "Password" else "") entry.grid(row=0, column=idx * 2 + 1, padx=(0, 8), pady=2) tip(entry, help_text) tip( ttk.Button(frame, text="Connect & test", command=self._test_connection), "Checks SSH access and whether the configured remote directories exist.", ).grid(row=0, column=8, padx=4) tip( ttk.Button(frame, text="Refresh status", command=self._refresh_all_status), "Checks the current LLM and Hermes process status on the server.", ).grid(row=0, column=9, padx=4) tip( ttk.Button(frame, text="Save settings", command=self._save_settings), "Stores the current GUI values in a local JSON file next to this script.", ).grid(row=0, column=10, padx=4) ttk.Label(frame, textvariable=self.v_status, foreground="#555").grid(row=0, column=11, padx=6, sticky=tk.W) def _build_llm_panel(self, parent): frame = ttk.LabelFrame(parent, text="LLM server (llama.cpp)", padding=8) frame.grid(row=0, column=0, sticky=tk.NSEW, padx=(0, 6)) frame.columnconfigure(1, weight=1) rows = [ ("Bind host", self.v_llm_bind_host, "The host address llama-server listens on. Use 0.0.0.0 for external access."), ("Port", self.v_llm_port, "HTTP port for the OpenAI-compatible API."), ("Alias", self.v_llm_alias, "Public model alias used by the API, for example qwen-local."), ("Threads", self.v_llm_threads, "CPU threads used by llama-server."), ("Context", self.v_llm_ctx, "Context size passed to llama-server."), ("Model path", self.v_llm_model, "Absolute path to the remote GGUF file."), ("Work dir", self.v_llm_workdir, "Directory containing the llama.cpp build."), ("Log path", self.v_llm_log, "Remote file where stdout and stderr are written."), ("PID path", self.v_llm_pid, "Remote file which stores the llama-server process ID."), ] for row, (label, var, help_text) in enumerate(rows): ttk.Label(frame, text=f"{label}:").grid(row=row, column=0, sticky=tk.W, padx=4, pady=3) ttk.Entry(frame, textvariable=var).grid(row=row, column=1, sticky=tk.EW, padx=4, pady=3) tip(ttk.Label(frame, text="?"), help_text).grid(row=row, column=2, sticky=tk.W) ttk.Label(frame, text="Status:").grid(row=len(rows), column=0, sticky=tk.W, padx=4, pady=(8, 3)) self.llm_status_label = ttk.Label(frame, textvariable=self.v_llm_status, foreground="#555") self.llm_status_label.grid(row=len(rows), column=1, sticky=tk.W, padx=4, pady=(8, 3)) self.llm_status_label.bind("", lambda _event: self._copy_status("LLM", self.v_llm_status.get())) buttons = ttk.Frame(frame) buttons.grid(row=len(rows) + 1, column=0, columnspan=3, sticky=tk.W, padx=4, pady=(8, 0)) self.start_llm_button = tip( ttk.Button(buttons, text="Start LLM", command=self._start_llm), "Starts the remote llama-server process.", ) self.start_llm_button.pack(side=tk.LEFT, padx=(0, 6)) self.stop_llm_button = tip( ttk.Button(buttons, text="Stop LLM", command=self._stop_llm), "Stops the remote llama-server process.", ) self.stop_llm_button.pack(side=tk.LEFT, padx=6) tip( ttk.Button(buttons, text="LLM status", command=self._refresh_llm_status), "Checks the current process state and whether the HTTP endpoint is healthy.", ).pack(side=tk.LEFT, padx=6) tip( ttk.Button(buttons, text="Tail LLM log", command=self._tail_llm_log), "Shows the last lines from the remote llama-server log.", ).pack(side=tk.LEFT, padx=6) def _build_hermes_panel(self, parent): frame = ttk.LabelFrame(parent, text="Hermes backend", padding=8) frame.grid(row=0, column=1, sticky=tk.NSEW) frame.columnconfigure(1, weight=1) rows = [ ("Bind host", self.v_hermes_host, "Hermes serve host. Keep 127.0.0.1 unless you really need remote access."), ("Port", self.v_hermes_port, "HTTP/WebSocket port for hermes serve."), ("Venv path", self.v_hermes_venv, "Python virtual environment where Hermes Agent is installed."), ("Log path", self.v_hermes_log, "Remote file where Hermes serve writes its output."), ("PID path", self.v_hermes_pid, "Remote file which stores the Hermes process ID."), ("Extra args", self.v_hermes_extra, "Additional command line flags appended to 'hermes serve'."), ] for row, (label, var, help_text) in enumerate(rows): ttk.Label(frame, text=f"{label}:").grid(row=row, column=0, sticky=tk.W, padx=4, pady=3) ttk.Entry(frame, textvariable=var).grid(row=row, column=1, sticky=tk.EW, padx=4, pady=3) tip(ttk.Label(frame, text="?"), help_text).grid(row=row, column=2, sticky=tk.W) ttk.Label(frame, text="Default mode:").grid(row=len(rows), column=0, sticky=tk.NW, padx=4, pady=(8, 3)) ttk.Label( frame, text=( "This GUI starts 'hermes serve' as a background backend.\n" "That is the most stable long-running Hermes process for this server." ), justify=tk.LEFT, foreground="#555", ).grid(row=len(rows), column=1, sticky=tk.W, padx=4, pady=(8, 3)) ttk.Label(frame, text="Status:").grid(row=len(rows) + 1, column=0, sticky=tk.W, padx=4, pady=(8, 3)) self.hermes_status_label = ttk.Label(frame, textvariable=self.v_hermes_status, foreground="#555") self.hermes_status_label.grid(row=len(rows) + 1, column=1, sticky=tk.W, padx=4, pady=(8, 3)) self.hermes_status_label.bind( "", lambda _event: self._copy_status("Hermes", self.v_hermes_status.get()), ) buttons = ttk.Frame(frame) buttons.grid(row=len(rows) + 2, column=0, columnspan=3, sticky=tk.W, padx=4, pady=(8, 0)) tip(ttk.Button(buttons, text="Start Hermes", command=self._start_hermes), "Starts 'hermes serve' in the background.").pack( side=tk.LEFT, padx=(0, 6) ) tip(ttk.Button(buttons, text="Stop Hermes", command=self._stop_hermes), "Stops the Hermes backend process.").pack( side=tk.LEFT, padx=6 ) tip( ttk.Button(buttons, text="Hermes status", command=self._refresh_hermes_status), "Checks whether the Hermes backend process is currently running.", ).pack(side=tk.LEFT, padx=6) tip( ttk.Button(buttons, text="Tail Hermes log", command=self._tail_hermes_log), "Shows the last lines from the remote Hermes log.", ).pack(side=tk.LEFT, padx=6) def _build_actions(self, parent): frame = ttk.LabelFrame(parent, text="Combined actions", padding=8) frame.pack(fill=tk.BOTH, expand=True) row = ttk.Frame(frame) row.pack(fill=tk.X) tip( ttk.Button(row, text="Start both", command=self._start_both), "Starts the llama.cpp server first and then the Hermes backend.", ).pack(side=tk.LEFT, padx=(0, 6)) tip( ttk.Button(row, text="Stop both", command=self._stop_both), "Stops Hermes first and then the llama.cpp server.", ).pack(side=tk.LEFT, padx=6) tip( ttk.Button(row, text="Health check", command=self._health_check), "Runs a small remote probe against the configured LLM and Hermes ports.", ).pack(side=tk.LEFT, padx=6) tip( ttk.Button(row, text="Test chat", command=self._test_chat), "Sends a valid POST request to /v1/chat/completions with the configured model alias.", ).pack(side=tk.LEFT, padx=6) tip(ttk.Button(row, text="Clear output", command=self._clear_log), "Clears the output panel.").pack( side=tk.RIGHT, padx=(6, 0) ) prompt_row = ttk.Frame(frame) prompt_row.pack(fill=tk.X, pady=(8, 0)) ttk.Label(prompt_row, text="Test prompt:").pack(side=tk.LEFT) ttk.Entry(prompt_row, textvariable=self.v_test_prompt).pack(side=tk.LEFT, fill=tk.X, expand=True, padx=(6, 0)) ttk.Label( frame, text=( "Note: Hermes itself currently has stricter model requirements than this small local model can satisfy.\n" "So the Hermes backend can be started and stopped here, but not every Hermes workflow will be useful on this machine." ), justify=tk.LEFT, foreground="#555", ).pack(anchor=tk.W, pady=(8, 0)) def _build_chat_tab(self, parent): parent.columnconfigure(0, weight=1) parent.rowconfigure(2, weight=1) config = ttk.LabelFrame(parent, text="Chat setup", padding=8) config.grid(row=0, column=0, sticky=tk.EW, pady=(0, 6)) config.columnconfigure(1, weight=1) ttk.Label(config, text="System prompt:").grid(row=0, column=0, sticky=tk.W, padx=4, pady=3) ttk.Entry(config, textvariable=self.v_chat_system).grid(row=0, column=1, sticky=tk.EW, padx=4, pady=3) ttk.Label(config, text="Mode:").grid(row=1, column=0, sticky=tk.W, padx=4, pady=3) mode_row = ttk.Frame(config) mode_row.grid(row=1, column=1, sticky=tk.W, padx=4, pady=3) ttk.Radiobutton(mode_row, text="Direct LLM", variable=self.v_chat_mode, value="llm").pack(side=tk.LEFT) ttk.Radiobutton(mode_row, text="Hermes", variable=self.v_chat_mode, value="hermes").pack(side=tk.LEFT, padx=(12, 0)) ttk.Label( config, text="Direct LLM is the recommended mode on this server. Hermes mode stays available for experiments.", foreground="#555", ).grid(row=2, column=0, columnspan=2, sticky=tk.W, padx=4, pady=(2, 0)) buttons = ttk.Frame(parent) buttons.grid(row=1, column=0, sticky=tk.W, pady=(0, 6)) tip( ttk.Button(buttons, text="Send", command=self._send_chat_message), "Sends the current chat input using the selected mode.", ).pack(side=tk.LEFT, padx=(0, 6)) tip( ttk.Button(buttons, text="New chat", command=self._clear_chat_history), "Starts a new local chat session in this GUI.", ).pack(side=tk.LEFT, padx=6) ttk.Label( buttons, text="Enter = senden, Shift+Enter = Zeilenumbruch", foreground="#555", ).pack(side=tk.LEFT, padx=(12, 0)) history_frame = ttk.LabelFrame(parent, text="Conversation", padding=4) history_frame.grid(row=2, column=0, sticky=tk.NSEW, pady=(0, 6)) history_frame.columnconfigure(0, weight=1) history_frame.rowconfigure(0, weight=1) self.chat_text = scrolledtext.ScrolledText( history_frame, wrap=tk.WORD, font=("Segoe UI", 10), state=tk.DISABLED, height=12, ) self.chat_text.grid(row=0, column=0, sticky=tk.NSEW) self.chat_text.tag_config("chat_user", foreground="#0b57d0", spacing1=8, spacing3=10) self.chat_text.tag_config("chat_assistant", foreground="#188038", spacing1=8, spacing3=10) self.chat_text.tag_config("chat_error", foreground="#c5221f", spacing1=8, spacing3=10) self.chat_text.tag_config("chat_meta", foreground="#666666", spacing1=4, spacing3=8) input_frame = ttk.Frame(parent) input_frame.grid(row=3, column=0, sticky=tk.EW) input_frame.columnconfigure(0, weight=1) self.chat_input = tk.Text(input_frame, height=4, wrap=tk.WORD, font=("Segoe UI", 10)) self.chat_input.grid(row=0, column=0, sticky=tk.EW, padx=(0, 6)) self.chat_input.bind("", self._on_chat_return) self.chat_input.bind("", self._on_chat_shift_return) self.chat_send_button = tip( ttk.Button(input_frame, text="Send", command=self._send_chat_message), "Submits the current message using the selected chat mode.", ) self.chat_send_button.grid(row=0, column=1, sticky=tk.N) self._append_chat_meta("Neuer Chat gestartet.") def _submit(self, coro): return asyncio.run_coroutine_threadsafe(coro, self._loop) def _set_initial_pane_sizes(self, outer_pane: ttk.PanedWindow, top_pane: ttk.PanedWindow): try: total = max(self.root.winfo_height(), 760) outer_pane.sashpos(0, int(total * 0.62)) top_pane.sashpos(0, int(total * 0.36)) except Exception: pass def _make_ssh(self) -> AsyncSSH: return AsyncSSH( self.v_host.get().strip(), self.v_user.get().strip(), self.v_pass.get(), int(self.v_port.get()), ) def _load_settings(self): if not SETTINGS_PATH.is_file(): return try: data = json.loads(SETTINGS_PATH.read_text(encoding="utf-8")) except Exception: return for name, value in data.items(): var = getattr(self, name, None) if var is not None and hasattr(var, "set"): var.set(value) def _save_settings(self): data = {} for name, value in self.__dict__.items(): if name.startswith("v_") and hasattr(value, "get"): data[name] = value.get() try: SETTINGS_PATH.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8") self._log(f"[ok] settings saved to {SETTINGS_PATH}\n", "ok") except Exception as exc: self._log(f"[error] saving settings failed: {exc}\n", "error") def _test_connection(self): if not HAS_ASYNCSSH: messagebox.showerror("Missing dependency", "asyncssh is not installed.\nRun: pip install asyncssh") return self.v_status.set("connecting...") self._log("[info] testing SSH connection\n", "info") self._submit(self._async_test_connection()) async def _async_test_connection(self): ssh = self._make_ssh() try: await ssh.connect() cmd = ( "bash -lc " + shlex.quote( "hostname && echo --- && " f"test -d {shlex.quote(self.v_llm_workdir.get().strip())} && echo llama-dir-ok || echo llama-dir-missing; " f"test -d {shlex.quote(self.v_hermes_venv.get().strip())} && echo hermes-venv-ok || echo hermes-venv-missing" ) ) await ssh.exec_stream(cmd, self._log) self.root.after(0, lambda: self.v_status.set("connected")) except Exception as exc: self._log(f"[error] {exc}\n", "error") self.root.after(0, lambda: self.v_status.set("connection failed")) finally: await ssh.close() def _llm_start_script(self) -> str: workdir = self.v_llm_workdir.get().strip() model = self.v_llm_model.get().strip() alias = self.v_llm_alias.get().strip() bind_host = self.v_llm_bind_host.get().strip() port = int(self.v_llm_port.get()) threads = int(self.v_llm_threads.get()) ctx = int(self.v_llm_ctx.get()) log_path = self.v_llm_log.get().strip() pid_path = self.v_llm_pid.get().strip() return f""" set -e mkdir -p {shlex.quote(os.path.dirname(log_path))} {shlex.quote(os.path.dirname(pid_path))} if [ -f {shlex.quote(pid_path)} ]; then oldpid=$(cat {shlex.quote(pid_path)}) if ps -p "$oldpid" > /dev/null 2>&1; then kill "$oldpid" 2>/dev/null || true sleep 1 fi rm -f {shlex.quote(pid_path)} fi pkill -x llama-server 2>/dev/null || true cd {shlex.quote(workdir)} setsid nohup ./build/bin/llama-server \\ -m {shlex.quote(model)} \\ -a {shlex.quote(alias)} \\ -ngl 0 -t {threads} -c {ctx} --host {shlex.quote(bind_host)} --port {port} \\ < /dev/null >> {shlex.quote(log_path)} 2>&1 & pid=$! echo "$pid" > {shlex.quote(pid_path)} sleep 3 cat {shlex.quote(pid_path)} """ def _llm_stop_script(self) -> str: pid_path = self.v_llm_pid.get().strip() return f""" set -e if [ -f {shlex.quote(pid_path)} ]; then pid=$(cat {shlex.quote(pid_path)}) kill "$pid" 2>/dev/null || true rm -f {shlex.quote(pid_path)} fi pkill -x llama-server 2>/dev/null || true """ def _llm_status_script(self) -> str: pid_path = self.v_llm_pid.get().strip() port = int(self.v_llm_port.get()) host = self.v_host.get().strip() return f""" if [ -f {shlex.quote(pid_path)} ]; then pid=$(cat {shlex.quote(pid_path)}) if ps -p "$pid" > /dev/null 2>&1; then echo "running: pid=$pid" ps -p "$pid" -o pid=,etime=,cmd= else rm -f {shlex.quote(pid_path)} echo "not running (removed stale pid: $pid)" fi else echo "not running" fi printf '%s\\n' '--- health ---' curl -s http://{host}:{port}/health || true """ def _hermes_start_script(self) -> str: venv = self.v_hermes_venv.get().strip() bind_host = self.v_hermes_host.get().strip() port = int(self.v_hermes_port.get()) log_path = self.v_hermes_log.get().strip() pid_path = self.v_hermes_pid.get().strip() extra = self.v_hermes_extra.get().strip() extra_suffix = f" {extra}" if extra else "" return f""" set -e mkdir -p {shlex.quote(os.path.dirname(log_path))} {shlex.quote(os.path.dirname(pid_path))} if [ -f {shlex.quote(pid_path)} ]; then oldpid=$(cat {shlex.quote(pid_path)}) kill "$oldpid" 2>/dev/null || true rm -f {shlex.quote(pid_path)} fi . {shlex.quote(posixpath.join(venv, "bin", "activate"))} setsid bash -lc {shlex.quote( f". {shlex.quote(posixpath.join(venv, 'bin', 'activate'))} && " + f"hermes serve --host {shlex.quote(bind_host)} --port {port}{extra_suffix} " + f"< /dev/null >> {shlex.quote(log_path)} 2>&1" )} & pid=$! echo "$pid" > {shlex.quote(pid_path)} sleep 3 cat {shlex.quote(pid_path)} """ def _hermes_stop_script(self) -> str: pid_path = self.v_hermes_pid.get().strip() port = int(self.v_hermes_port.get()) return f""" set -e if [ -f {shlex.quote(pid_path)} ]; then pid=$(cat {shlex.quote(pid_path)}) kill "$pid" 2>/dev/null || true rm -f {shlex.quote(pid_path)} fi pkill -f "hermes serve --host .* --port {port}" || true """ def _hermes_status_script(self) -> str: pid_path = self.v_hermes_pid.get().strip() port = int(self.v_hermes_port.get()) return f""" if [ -f {shlex.quote(pid_path)} ]; then pid=$(cat {shlex.quote(pid_path)}) if ps -p "$pid" > /dev/null 2>&1; then echo "running: pid=$pid" ps -p "$pid" -o pid=,etime=,cmd= else echo "stale pid file: $pid" fi else echo "not running" fi printf '%s\\n' '--- port probe ---' curl -sI http://127.0.0.1:{port} | head -1 || true """ def _start_llm(self): self._log("[info] starting remote llama.cpp server\n", "info") self._set_llm_starting(True) self._submit(self._async_start_llm()) def _stop_llm(self): self._log("[info] stopping remote llama.cpp server\n", "info") self._submit(self._async_stop_llm()) def _start_hermes(self): self._log("[info] starting remote Hermes backend\n", "info") self._submit(self._async_run_script(self._hermes_start_script(), "Hermes started", self._refresh_hermes_status)) def _stop_hermes(self): self._log("[info] stopping remote Hermes backend\n", "info") self._submit(self._async_run_script(self._hermes_stop_script(), "Hermes stopped", self._refresh_hermes_status)) def _start_both(self): self._log("[info] starting LLM and Hermes\n", "info") self._submit(self._async_start_both()) def _stop_both(self): self._log("[info] stopping Hermes and LLM\n", "info") self._submit(self._async_stop_both()) async def _async_start_both(self): self._log("[info] starting LLM first\n", "info") await self._async_start_llm() self._log("[info] starting Hermes next\n", "info") await self._async_run_script(self._hermes_start_script(), "Hermes started") self.root.after(0, self._refresh_all_status) async def _async_stop_both(self): await self._async_run_script(self._hermes_stop_script(), "Hermes stopped") await self._async_stop_llm() self.root.after(0, self._refresh_all_status) def _refresh_llm_status(self): self._submit(self._async_refresh_service_status("llm")) def _refresh_hermes_status(self): self._submit(self._async_refresh_service_status("hermes")) def _refresh_all_status(self): self._refresh_llm_status() self._refresh_hermes_status() async def _async_refresh_service_status(self, service: str): ssh = self._make_ssh() try: await ssh.connect() if service == "llm": output = await ssh.exec(f"bash -lc {shlex.quote(self._llm_status_script())}") self._log("[info] LLM status\n" + output + "\n", "info") first_line = output.strip().splitlines()[0] if output.strip() else "unknown" self.root.after(0, lambda: self._set_llm_status(first_line)) else: output = await ssh.exec(f"bash -lc {shlex.quote(self._hermes_status_script())}") self._log("[info] Hermes status\n" + output + "\n", "info") first_line = output.strip().splitlines()[0] if output.strip() else "unknown" self.root.after(0, lambda: self._set_hermes_status(first_line)) except Exception as exc: self._log(f"[error] status check failed: {exc}\n", "error") finally: await ssh.close() async def _async_start_llm(self): ssh = self._make_ssh() try: await ssh.connect() result = await ssh.exec(f"bash -lc {shlex.quote(self._llm_start_script())}") if result.strip(): self._log(result if result.endswith("\n") else result + "\n", "info") llm_port = int(self.v_llm_port.get()) llm_host = self.v_host.get().strip() health_cmd = "bash -lc " + shlex.quote(f"curl -s http://{llm_host}:{llm_port}/health") for _ in range(20): health_output = await ssh.exec(health_cmd) if '"status":"ok"' in health_output.replace(" ", ""): self._log(f"[ok] LLM started and healthy on {llm_host}:{llm_port}\n", "ok") self.root.after(0, lambda: self._set_llm_status("running")) return await asyncio.sleep(1) self._log( f"[error] LLM start command ran, but no healthy endpoint appeared on {llm_host}:{llm_port}\n", "error", ) log_output = await ssh.exec( f"bash -lc {shlex.quote(f'tail -n 60 {shlex.quote(self.v_llm_log.get().strip())}')}" ) if log_output.strip(): self._log(log_output if log_output.endswith("\n") else log_output + "\n", "error") self.root.after(0, lambda: self._set_llm_status("not running")) except Exception as exc: self._log(f"[error] LLM start failed: {exc}\n", "error") self.root.after(0, lambda: self._set_llm_status("not running")) finally: await ssh.close() self.root.after(0, lambda: self._set_llm_starting(False)) self.root.after(0, self._refresh_llm_status) async def _async_stop_llm(self): await self._async_run_script(self._llm_stop_script(), "LLM stopped", self._refresh_llm_status) def _tail_llm_log(self): self._submit(self._async_tail_log(self.v_llm_log.get().strip(), "LLM")) def _tail_hermes_log(self): self._submit(self._async_tail_log(self.v_hermes_log.get().strip(), "Hermes")) async def _async_tail_log(self, log_path: str, label: str): ssh = self._make_ssh() try: await ssh.connect() cmd = f"bash -lc {shlex.quote(f'tail -n 80 {shlex.quote(log_path)}')}" output = await ssh.exec(cmd) self._log(f"[info] tail {label} log\n{output}\n", "info") except Exception as exc: self._log(f"[error] tail log failed: {exc}\n", "error") finally: await ssh.close() def _health_check(self): self._submit(self._async_health_check()) def _test_chat(self): self._log("[info] running test chat request\n", "info") self._submit(self._async_test_chat()) def _send_chat_message(self): if hasattr(self, "chat_send_button") and str(self.chat_send_button["state"]) == "disabled": return text = self.chat_input.get("1.0", tk.END).strip() if not text: return self.chat_input.delete("1.0", tk.END) self._set_chat_busy(True) self._append_chat_message("You", text) self._chat_history.append(("user", text)) if self.v_chat_mode.get() == "hermes": self._submit(self._async_send_chat_message()) else: self._submit(self._async_send_direct_llm_chat_message()) def _clear_chat_history(self): self._chat_history.clear() self.chat_text.config(state=tk.NORMAL) self.chat_text.delete("1.0", tk.END) self.chat_text.config(state=tk.DISABLED) if hasattr(self, "chat_input"): self.chat_input.delete("1.0", tk.END) self._append_chat_meta("Neuer Chat gestartet.") def _append_chat_message(self, speaker: str, text: str): tag = "chat_assistant" if speaker == "You": tag = "chat_user" elif "Error" in speaker: tag = "chat_error" self.chat_text.config(state=tk.NORMAL) self.chat_text.insert(tk.END, f"{speaker}:\n", tag) self.chat_text.insert(tk.END, f"{text.strip()}\n\n") self.chat_text.see(tk.END) self.chat_text.config(state=tk.DISABLED) def _append_chat_meta(self, text: str): self.chat_text.config(state=tk.NORMAL) self.chat_text.insert(tk.END, f"{text.strip()}\n\n", "chat_meta") self.chat_text.see(tk.END) self.chat_text.config(state=tk.DISABLED) def _copy_status(self, label: str, value: str): text = f"{label} status: {value or 'unknown'}" self.root.clipboard_clear() self.root.clipboard_append(text) self._log(f"[info] copied to clipboard: {text}\n", "info") def _set_chat_busy(self, busy: bool): if hasattr(self, "chat_send_button"): self.chat_send_button.configure(state=tk.DISABLED if busy else tk.NORMAL) if hasattr(self, "chat_input"): self.chat_input.configure(state=tk.DISABLED if busy else tk.NORMAL) if not busy: self.chat_input.focus_set() def _on_chat_return(self, _event): self._send_chat_message() return "break" def _on_chat_shift_return(self, _event): self.chat_input.insert(tk.INSERT, "\n") return "break" def _set_llm_status(self, text: str): self.v_llm_status.set(text) color = "#555" lower = text.lower() if "running" in lower: color = "#188038" elif "not running" in lower or "stale" in lower: color = "#c5221f" self.llm_status_label.configure(foreground=color) def _set_llm_starting(self, starting: bool): if hasattr(self, "start_llm_button"): self.start_llm_button.configure( text="Starting..." if starting else "Start LLM", state=tk.DISABLED if starting else tk.NORMAL, ) if hasattr(self, "stop_llm_button"): self.stop_llm_button.configure(state=tk.DISABLED if starting else tk.NORMAL) def _set_hermes_status(self, text: str): self.v_hermes_status.set(text) color = "#555" lower = text.lower() if "running" in lower: color = "#188038" elif "not running" in lower or "stale" in lower: color = "#c5221f" self.hermes_status_label.configure(foreground=color) def _render_hermes_prompt(self) -> str: lines = [self.v_chat_system.get().strip(), "", "Conversation so far:"] for role, content in self._chat_history[-12:]: label = "User" if role == "user" else "Hermes" lines.append(f"{label}: {content}") lines.append("") lines.append("Answer now as Hermes in German.") return "\n".join(lines) async def _ensure_llm_ready(self, ssh: AsyncSSH) -> bool: llm_port = int(self.v_llm_port.get()) llm_host = self.v_host.get().strip() health_cmd = "bash -lc " + shlex.quote(f"curl -s http://{llm_host}:{llm_port}/health") health_output = await ssh.exec(health_cmd) if '"status":"ok"' in health_output.replace(" ", ""): return True self._log( f"[info] LLM not reachable on {llm_host}:{llm_port}, trying automatic restart\n", "info", ) start_output = await ssh.exec(f"bash -lc {shlex.quote(self._llm_start_script())}") if start_output.strip(): self._log(start_output if start_output.endswith("\n") else start_output + "\n", "info") await asyncio.sleep(12) health_output = await ssh.exec(health_cmd) if '"status":"ok"' in health_output.replace(" ", ""): self._log(f"[ok] LLM restart succeeded on {llm_host}:{llm_port}\n", "ok") return True self._log( f"[error] LLM is still not reachable on {llm_host}:{llm_port} after restart\n", "error", ) if health_output.strip(): self._log(health_output if health_output.endswith("\n") else health_output + "\n", "error") return False async def _async_health_check(self): ssh = self._make_ssh() try: await ssh.connect() llm_port = int(self.v_llm_port.get()) hermes_port = int(self.v_hermes_port.get()) llm_host = self.v_host.get().strip() hermes_host = self.v_hermes_host.get().strip() cmd = ( "bash -lc " + shlex.quote( f"printf '%s\\n' '--- LLM ---'; " f"curl -s http://{llm_host}:{llm_port}/health || true; " f"printf '\\n%s\\n' '--- Hermes ---'; " f"curl -sI http://{hermes_host}:{hermes_port} | head -1 || true" ) ) output = await ssh.exec(cmd) self._log(f"[info] health check\n{output}\n", "info") except Exception as exc: self._log(f"[error] health check failed: {exc}\n", "error") finally: await ssh.close() async def _async_test_chat(self): ssh = self._make_ssh() try: await ssh.connect() llm_port = int(self.v_llm_port.get()) llm_host = self.v_host.get().strip() alias = self.v_llm_alias.get().strip() prompt = self.v_test_prompt.get().strip() or "Antworte nur mit OK" if not await self._ensure_llm_ready(ssh): self._log( f"[error] test chat aborted: the LLM server is not reachable on {llm_host}:{llm_port}\n", "error", ) return payload = json.dumps( { "model": alias, "messages": [{"role": "user", "content": prompt}], "max_tokens": 32, "temperature": 0, "stream": False, }, ensure_ascii=False, ) python_script = ( "import json, sys, urllib.request\n" f"payload = {payload!r}\n" "req = urllib.request.Request(" f"'http://{llm_host}:{llm_port}/v1/chat/completions', " "data=payload.encode('utf-8'), " "headers={'Content-Type': 'application/json'}, " "method='POST')\n" "with urllib.request.urlopen(req, timeout=180) as resp:\n" " print(resp.read().decode('utf-8'))\n" ) cmd = "bash -lc " + shlex.quote(f"python3 - <<'PY'\n{python_script}PY") await ssh._ensure_conn() result = await ssh._conn.run(cmd, check=False) output = (result.stdout or "") + (result.stderr or "") if result.exit_status == 0: self._log(f"[ok] test chat response\n{output}\n", "ok") else: self._log("[error] test chat request failed\n", "error") if output.strip(): self._log(output if output.endswith("\n") else output + "\n", "error") except Exception as exc: self._log(f"[error] test chat failed: {exc}\n", "error") finally: await ssh.close() async def _async_send_chat_message(self): ssh = self._make_ssh() try: await ssh.connect() if not await self._ensure_llm_ready(ssh): self.root.after( 0, lambda: self._append_chat_message( "Hermes Error", f"Der LLM-Server auf {self.v_host.get().strip()}:{int(self.v_llm_port.get())} ist nicht erreichbar.", ), ) return venv = self.v_hermes_venv.get().strip() prompt = self._render_hermes_prompt() cmd = ( "bash -lc " + shlex.quote( f". {shlex.quote(posixpath.join(venv, 'bin', 'activate'))} && " f"hermes chat -q {shlex.quote(prompt)} -Q --cli" ) ) await ssh._ensure_conn() result = await ssh._conn.run(cmd, check=False) output = ((result.stdout or "") + (result.stderr or "")).strip() if result.exit_status == 0 and output: self._chat_history.append(("assistant", output)) self.root.after(0, lambda: self._append_chat_message("Hermes", output)) else: error_text = output or f"Hermes exited with code {result.exit_status}" self.root.after(0, lambda: self._append_chat_message("Hermes Error", error_text)) self._log(f"[error] Hermes chat failed\n{error_text}\n", "error") except Exception as exc: error_text = str(exc) self.root.after(0, lambda: self._append_chat_message("Hermes Error", error_text)) self._log(f"[error] Hermes chat failed: {exc}\n", "error") finally: await ssh.close() self.root.after(0, lambda: self._set_chat_busy(False)) async def _async_send_direct_llm_chat_message(self): ssh = self._make_ssh() try: await ssh.connect() if not await self._ensure_llm_ready(ssh): self.root.after( 0, lambda: self._append_chat_message( "LLM Error", f"Der LLM-Server auf {self.v_host.get().strip()}:{int(self.v_llm_port.get())} ist nicht erreichbar.", ), ) return llm_port = int(self.v_llm_port.get()) llm_host = self.v_host.get().strip() alias = self.v_llm_alias.get().strip() messages = [] system = self.v_chat_system.get().strip() if system: messages.append({"role": "system", "content": system}) for role, content in self._chat_history[-12:]: messages.append({"role": role, "content": content}) payload = json.dumps( { "model": alias, "messages": messages, "max_tokens": 192, "temperature": 0.4, "stream": False, }, ensure_ascii=False, ) python_script = ( "import json, urllib.request\n" f"payload = {payload!r}\n" "req = urllib.request.Request(" f"'http://{llm_host}:{llm_port}/v1/chat/completions', " "data=payload.encode('utf-8'), " "headers={'Content-Type': 'application/json'}, " "method='POST')\n" "with urllib.request.urlopen(req, timeout=180) as resp:\n" " data = json.loads(resp.read().decode('utf-8'))\n" " print(data['choices'][0]['message']['content'])\n" ) cmd = "bash -lc " + shlex.quote(f"python3 - <<'PY'\n{python_script}PY") await ssh._ensure_conn() result = await ssh._conn.run(cmd, check=False) output = ((result.stdout or "") + (result.stderr or "")).strip() if result.exit_status == 0 and output: self._chat_history.append(("assistant", output)) self.root.after(0, lambda: self._append_chat_message("LLM", output)) else: error_text = output or f"LLM exited with code {result.exit_status}" self.root.after(0, lambda: self._append_chat_message("LLM Error", error_text)) self._log(f"[error] LLM chat failed\n{error_text}\n", "error") except Exception as exc: error_text = str(exc) self.root.after(0, lambda: self._append_chat_message("LLM Error", error_text)) self._log(f"[error] LLM chat failed: {exc}\n", "error") finally: await ssh.close() self.root.after(0, lambda: self._set_chat_busy(False)) async def _async_run_script(self, script: str, ok_message: str = "", callback=None): ssh = self._make_ssh() try: await ssh.connect() result = await ssh.exec(f"bash -lc {shlex.quote(script)}") if result.strip(): self._log(result if result.endswith("\n") else result + "\n", "info") if ok_message: self._log(f"[ok] {ok_message}\n", "ok") except Exception as exc: self._log(f"[error] {exc}\n", "error") finally: await ssh.close() if callback: self.root.after(0, callback) def _log(self, text: str, tag: str = ""): self._log_queue.put((text, tag)) def _poll_log(self): try: while True: text, tag = self._log_queue.get_nowait() self.log.config(state=tk.NORMAL) self.log.insert(tk.END, text, tag or "") self.log.see(tk.END) self.log.config(state=tk.DISABLED) except queue.Empty: pass self.root.after(100, self._poll_log) def _clear_log(self): self.log.config(state=tk.NORMAL) self.log.delete("1.0", tk.END) self.log.config(state=tk.DISABLED) def on_close(self): self._loop.call_soon_threadsafe(self._loop.stop) self.root.destroy() def main(): root = tk.Tk() style = ttk.Style(root) for preferred in ("clam", "alt", "default"): if preferred in style.theme_names(): style.theme_use(preferred) break app = HermesLLMGUI(root) root.protocol("WM_DELETE_WINDOW", app.on_close) root.mainloop() if __name__ == "__main__": main()