Dies und das

This commit is contained in:
Eskimue
2026-07-14 12:22:30 +02:00
parent e94488def7
commit 2bd5b92802
8 changed files with 2619 additions and 0 deletions
+698
View File
@@ -0,0 +1,698 @@
#!/usr/bin/env python3
"""
Speech AI GUI with two tabs:
- Audio -> Text via faster-whisper on the remote server
- Text -> Audio via edge-tts on the remote server
Dependency:
pip install asyncssh
"""
from __future__ import annotations
import asyncio
import json
import os
import posixpath
import queue
import shlex
import threading
import tkinter as tk
from tkinter import filedialog, messagebox, scrolledtext, ttk
try:
import asyncssh
HAS_ASYNCSSH = True
except ImportError:
HAS_ASYNCSSH = False
TRANSCRIBE_SCRIPT = """#!/usr/bin/env python3
import json
import sys
from faster_whisper import WhisperModel
input_path, output_txt, output_json, model_name, language, task, device, compute_type = sys.argv[1:9]
language = None if language == "-" else language
model = WhisperModel(model_name, device=device, compute_type=compute_type)
segments, info = model.transcribe(input_path, language=language, task=task, vad_filter=True)
segments_list = []
text_parts = []
for segment in segments:
text_parts.append(segment.text.strip())
segments_list.append({
"start": segment.start,
"end": segment.end,
"text": segment.text.strip(),
})
text = "\\n".join(part for part in text_parts if part)
with open(output_txt, "w", encoding="utf-8") as handle:
handle.write(text + ("\\n" if text else ""))
with open(output_json, "w", encoding="utf-8") as handle:
json.dump({
"language": getattr(info, "language", language),
"duration": getattr(info, "duration", None),
"text": text,
"segments": segments_list,
}, handle, ensure_ascii=False, indent=2)
print(text)
"""
TTS_SCRIPT = """#!/usr/bin/env python3
import asyncio
import sys
import edge_tts
text_file, output_audio, voice, rate, volume = sys.argv[1:6]
with open(text_file, "r", encoding="utf-8") as handle:
text = handle.read().strip()
if not text:
raise SystemExit("No text provided.")
async def main():
communicate = edge_tts.Communicate(text, voice=voice, rate=rate, volume=volume)
await communicate.save(output_audio)
asyncio.run(main())
print(output_audio)
"""
class Tooltip:
DELAY_MS = 500
def __init__(self, widget: tk.Widget, text: str):
self.widget = widget
self.text = text
self.timer_id = None
self.tip_window = None
widget.bind("<Enter>", self._schedule, add="+")
widget.bind("<Leave>", self._cancel, add="+")
widget.bind("<ButtonPress>", self._cancel, add="+")
def _schedule(self, _event=None):
self._cancel()
self.timer_id = self.widget.after(self.DELAY_MS, self._show)
def _cancel(self, _event=None):
if self.timer_id:
self.widget.after_cancel(self.timer_id)
self.timer_id = None
if self.tip_window:
self.tip_window.destroy()
self.tip_window = None
def _show(self):
if self.tip_window:
return
x = self.widget.winfo_rootx() + 18
y = self.widget.winfo_rooty() + self.widget.winfo_height() + 4
self.tip_window = win = tk.Toplevel(self.widget)
win.wm_overrideredirect(True)
win.wm_geometry(f"+{x}+{y}")
win.attributes("-topmost", True)
tk.Label(
win,
text=self.text,
justify=tk.LEFT,
background="#fff8dc",
foreground="#1a1a1a",
relief=tk.SOLID,
borderwidth=1,
wraplength=420,
padx=8,
pady=6,
font=("Segoe UI", 9),
).pack()
def tip(widget: tk.Widget, text: str) -> tk.Widget:
if text:
Tooltip(widget, text)
return widget
class AsyncSSH:
def __init__(self, host: str, user: str, password: str, port: int = 22):
self.host = host
self.user = user
self.password = password
self.port = port
self._conn: asyncssh.SSHClientConnection | None = None
async def connect(self):
self._conn = await asyncssh.connect(
self.host,
port=self.port,
username=self.user,
password=self.password,
known_hosts=None,
)
async def _ensure_conn(self):
if self._conn is None:
await self.connect()
async def exec(self, cmd: str) -> str:
await self._ensure_conn()
result = await self._conn.run(cmd, check=False)
return (result.stdout or "") + (result.stderr or "")
async def exec_stream(self, cmd: str, output_cb):
await self._ensure_conn()
async with self._conn.create_process(
cmd,
stderr=asyncssh.STDOUT,
encoding="utf-8",
errors="replace",
) as proc:
async for line in proc.stdout:
output_cb(line if line.endswith("\n") else line + "\n")
output_cb(f"\n[process exited with code {proc.returncode}]\n")
return proc.returncode
async def mkdir_p(self, remote_dir: str):
await self._ensure_conn()
if remote_dir in ("", "."):
return
async with self._conn.start_sftp_client() as sftp:
parts = [part for part in remote_dir.strip("/").split("/") if part]
path = ""
for part in parts:
path = f"/{path}/{part}" if path else f"/{part}"
try:
await sftp.stat(path)
except asyncssh.SFTPError:
await sftp.mkdir(path)
async def write_text(self, remote_path: str, text: str):
await self._ensure_conn()
await self.mkdir_p(posixpath.dirname(remote_path))
async with self._conn.start_sftp_client() as sftp:
async with sftp.open(remote_path, "w", encoding="utf-8") as handle:
await handle.write(text)
async def read_text(self, remote_path: str) -> str:
await self._ensure_conn()
async with self._conn.start_sftp_client() as sftp:
async with sftp.open(remote_path, "r", encoding="utf-8", errors="replace") as handle:
return await handle.read()
async def upload(self, local_path: str, remote_path: str):
await self._ensure_conn()
await self.mkdir_p(posixpath.dirname(remote_path))
async with self._conn.start_sftp_client() as sftp:
await sftp.put(local_path, remote_path)
async def download(self, remote_path: str, local_path: str):
await self._ensure_conn()
async with self._conn.start_sftp_client() as sftp:
await sftp.get(remote_path, local_path)
async def close(self):
if self._conn:
self._conn.close()
await self._conn.wait_closed()
self._conn = None
class SpeechAIGUI:
def __init__(self, root: tk.Tk):
self.root = root
self.root.title("Speech AI GUI")
self.root.geometry("1180x900")
self.root.minsize(980, 720)
self._loop = asyncio.new_event_loop()
threading.Thread(target=self._loop.run_forever, daemon=True, name="speech-ai-asyncio").start()
self._log_queue: queue.Queue[tuple[str, str]] = queue.Queue()
self._current_task = None
self.latest_transcript_remote = ""
self.latest_tts_remote = ""
self._init_vars()
self._build_ui()
self._poll_log()
if not HAS_ASYNCSSH:
self._log("[error] asyncssh is missing. Install it with: pip install asyncssh\n", "error")
def _init_vars(self):
self.v_host = tk.StringVar(value="10.42.44.25")
self.v_port = tk.IntVar(value=22)
self.v_user = tk.StringVar(value="eskimue")
self.v_pass = tk.StringVar(value="Preside")
self.v_status = tk.StringVar(value="not connected")
self.v_remote_dir = tk.StringVar(value="/home/eskimue/speech-ai")
self.v_python = tk.StringVar(value="/home/eskimue/whisper-env/bin/python")
self.v_local_audio = tk.StringVar()
self.v_remote_audio = tk.StringVar(value="/home/eskimue/speech-ai/input.wav")
self.v_stt_model = tk.StringVar(value="base")
self.v_stt_language = tk.StringVar(value="de")
self.v_stt_task = tk.StringVar(value="transcribe")
self.v_stt_device = tk.StringVar(value="cpu")
self.v_stt_compute = tk.StringVar(value="int8")
self.v_tts_voice = tk.StringVar(value="de-DE-KatjaNeural")
self.v_tts_rate = tk.StringVar(value="+0%")
self.v_tts_volume = tk.StringVar(value="+0%")
self.v_tts_output = tk.StringVar(value="speech_output.mp3")
def _build_ui(self):
top = ttk.Frame(self.root, padding=6)
top.pack(fill=tk.X)
self._build_connection_bar(top)
main = ttk.Frame(self.root, padding=6)
main.pack(fill=tk.BOTH, expand=True)
main.columnconfigure(0, weight=3)
main.columnconfigure(1, weight=2)
main.rowconfigure(1, weight=1)
self._build_server_panel(main)
self._build_tabs(main)
self._build_side_panel(main)
log_frame = ttk.LabelFrame(self.root, text="Output", padding=4)
log_frame.pack(fill=tk.BOTH, expand=True, padx=6, pady=(0, 6))
self.log = scrolledtext.ScrolledText(
log_frame,
font=("Consolas", 9),
bg="#1e1e1e",
fg="#d4d4d4",
insertbackground="white",
wrap=tk.WORD,
state=tk.DISABLED,
)
self.log.pack(fill=tk.BOTH, expand=True)
self.log.tag_config("error", foreground="#f48771")
self.log.tag_config("ok", foreground="#89d185")
self.log.tag_config("info", foreground="#9cdcfe")
def _build_connection_bar(self, parent):
frame = ttk.LabelFrame(parent, text="SSH connection", padding=6)
frame.pack(fill=tk.X)
fields = [
("Host", self.v_host, 18, "Host name or IP of the Linux server."),
("Port", self.v_port, 6, "SSH port, usually 22."),
("User", self.v_user, 14, "SSH user used for Whisper and TTS."),
("Password", self.v_pass, 16, "Password stays only in memory while this GUI runs."),
]
for idx, (label, var, width, help_text) in enumerate(fields):
ttk.Label(frame, text=f"{label}:").grid(row=0, column=idx * 2, padx=(4, 2), pady=2, sticky=tk.W)
entry = ttk.Entry(frame, textvariable=var, width=width, show="*" if label == "Password" else "")
entry.grid(row=0, column=idx * 2 + 1, padx=(0, 8), pady=2)
tip(entry, help_text)
tip(ttk.Button(frame, text="Connect & test", command=self._test_connection), "Check the server, Python environment and the speech work directory.").grid(row=0, column=8, padx=4)
ttk.Label(frame, textvariable=self.v_status, foreground="#555").grid(row=0, column=9, padx=6, sticky=tk.W)
def _build_server_panel(self, parent):
frame = ttk.LabelFrame(parent, text="Remote setup", padding=6)
frame.grid(row=0, column=0, sticky=tk.EW, pady=(0, 6))
frame.columnconfigure(1, weight=1)
rows = [
("Remote work dir", self.v_remote_dir, "Directory on the server where helper scripts, uploads and generated files are stored."),
("Speech Python", self.v_python, "Python executable inside the server-side speech environment."),
]
for row, (label, var, help_text) in enumerate(rows):
ttk.Label(frame, text=f"{label}:").grid(row=row, column=0, padx=4, pady=3, sticky=tk.W)
entry = ttk.Entry(frame, textvariable=var)
entry.grid(row=row, column=1, sticky=tk.EW, padx=4, pady=3)
tip(entry, help_text)
def _build_tabs(self, parent):
notebook = ttk.Notebook(parent)
notebook.grid(row=1, column=0, sticky=tk.NSEW, padx=(0, 6))
stt_tab = ttk.Frame(notebook, padding=8)
tts_tab = ttk.Frame(notebook, padding=8)
notebook.add(stt_tab, text="Audio -> Text")
notebook.add(tts_tab, text="Text -> Audio")
self._build_stt_tab(stt_tab)
self._build_tts_tab(tts_tab)
def _build_stt_tab(self, parent):
parent.columnconfigure(1, weight=1)
parent.rowconfigure(6, weight=1)
ttk.Label(parent, text="Local audio file:").grid(row=0, column=0, sticky=tk.W, padx=4, pady=3)
ttk.Entry(parent, textvariable=self.v_local_audio).grid(row=0, column=1, sticky=tk.EW, padx=4, pady=3)
tip(ttk.Button(parent, text="Browse...", command=self._browse_audio), "Select a local audio file such as WAV, MP3 or M4A.").grid(row=0, column=2, padx=4, pady=3)
ttk.Label(parent, text="Remote audio file:").grid(row=1, column=0, sticky=tk.W, padx=4, pady=3)
ttk.Entry(parent, textvariable=self.v_remote_audio).grid(row=1, column=1, sticky=tk.EW, padx=4, pady=3)
ttk.Label(parent, text="The upload target on the server.").grid(row=1, column=2, sticky=tk.W, padx=4, pady=3)
ttk.Label(parent, text="Model:").grid(row=2, column=0, sticky=tk.W, padx=4, pady=3)
model_box = ttk.Combobox(parent, textvariable=self.v_stt_model, state="readonly", values=["tiny", "base", "small", "medium"])
model_box.grid(row=2, column=1, sticky=tk.W, padx=4, pady=3)
tip(model_box, "Whisper model size. On this server, tiny or base are the safest choices.")
ttk.Label(parent, text="Language:").grid(row=3, column=0, sticky=tk.W, padx=4, pady=3)
ttk.Entry(parent, textvariable=self.v_stt_language, width=10).grid(row=3, column=1, sticky=tk.W, padx=4, pady=3)
ttk.Label(parent, text="Use de, en, fr or leave empty for auto-detect.").grid(row=3, column=2, sticky=tk.W, padx=4, pady=3)
options = ttk.Frame(parent)
options.grid(row=4, column=0, columnspan=3, sticky=tk.W, padx=4, pady=3)
ttk.Label(options, text="Task:").pack(side=tk.LEFT)
task_box = ttk.Combobox(options, textvariable=self.v_stt_task, state="readonly", width=12, values=["transcribe", "translate"])
task_box.pack(side=tk.LEFT, padx=(4, 12))
tip(task_box, "Transcribe keeps the source language. Translate asks Whisper to return English text.")
ttk.Label(options, text="Device:").pack(side=tk.LEFT)
device_box = ttk.Combobox(options, textvariable=self.v_stt_device, state="readonly", width=10, values=["cpu", "auto"])
device_box.pack(side=tk.LEFT, padx=(4, 12))
tip(device_box, "CPU is the reliable default on this server. Auto may try GPU.")
ttk.Label(options, text="Compute:").pack(side=tk.LEFT)
compute_box = ttk.Combobox(options, textvariable=self.v_stt_compute, state="readonly", width=10, values=["int8", "float32"])
compute_box.pack(side=tk.LEFT, padx=(4, 0))
tip(compute_box, "int8 is the lightest option and works well on small machines.")
button_row = ttk.Frame(parent)
button_row.grid(row=5, column=0, columnspan=3, sticky=tk.W, padx=4, pady=(6, 6))
tip(ttk.Button(button_row, text="Upload audio", command=self._upload_audio_only), "Upload the selected local audio file to the server.").pack(side=tk.LEFT, padx=(0, 6))
tip(ttk.Button(button_row, text="Upload + transcribe", command=self._transcribe_audio), "Upload the file and run faster-whisper on the server.").pack(side=tk.LEFT, padx=6)
tip(ttk.Button(button_row, text="Download transcript...", command=self._download_transcript), "Save the latest transcript text from the server to a local file.").pack(side=tk.LEFT, padx=6)
self.transcript_text = scrolledtext.ScrolledText(parent, wrap=tk.WORD, font=("Consolas", 10))
self.transcript_text.grid(row=6, column=0, columnspan=3, sticky=tk.NSEW, padx=4, pady=4)
def _build_tts_tab(self, parent):
parent.columnconfigure(0, weight=1)
parent.rowconfigure(3, weight=1)
top = ttk.Frame(parent)
top.grid(row=0, column=0, sticky=tk.EW, pady=(0, 6))
ttk.Label(top, text="Voice:").pack(side=tk.LEFT)
voice_box = ttk.Combobox(
top,
textvariable=self.v_tts_voice,
width=28,
values=[
"de-DE-KatjaNeural",
"de-DE-ConradNeural",
"en-US-AriaNeural",
"en-US-GuyNeural",
],
)
voice_box.pack(side=tk.LEFT, padx=(4, 12))
tip(voice_box, "Edge TTS voice name. The German voices are a good starting point.")
ttk.Label(top, text="Rate:").pack(side=tk.LEFT)
rate_box = ttk.Combobox(top, textvariable=self.v_tts_rate, width=8, values=["-20%", "-10%", "+0%", "+10%", "+20%"])
rate_box.pack(side=tk.LEFT, padx=(4, 12))
tip(rate_box, "Speech speed relative to the default voice speed.")
ttk.Label(top, text="Volume:").pack(side=tk.LEFT)
volume_box = ttk.Combobox(top, textvariable=self.v_tts_volume, width=8, values=["-20%", "-10%", "+0%", "+10%", "+20%"])
volume_box.pack(side=tk.LEFT, padx=(4, 12))
tip(volume_box, "Speech volume relative to the default voice volume.")
ttk.Label(top, text="Output file:").pack(side=tk.LEFT)
ttk.Entry(top, textvariable=self.v_tts_output, width=20).pack(side=tk.LEFT, padx=(4, 0))
button_row = ttk.Frame(parent)
button_row.grid(row=1, column=0, sticky=tk.W, pady=(0, 6))
tip(ttk.Button(button_row, text="Generate audio", command=self._generate_tts), "Send the text to the server and synthesize an audio file.").pack(side=tk.LEFT, padx=(0, 6))
tip(ttk.Button(button_row, text="Download audio...", command=self._download_tts_audio), "Save the generated audio file locally after synthesis.").pack(side=tk.LEFT, padx=6)
tip(ttk.Button(button_row, text="Download + open", command=self._download_and_open_tts_audio), "Download the generated audio file and open it locally.").pack(side=tk.LEFT, padx=6)
ttk.Label(parent, text="Text for speech synthesis:").grid(row=2, column=0, sticky=tk.W)
self.tts_text = scrolledtext.ScrolledText(parent, wrap=tk.WORD, font=("Segoe UI", 11))
self.tts_text.grid(row=3, column=0, sticky=tk.NSEW)
def _build_side_panel(self, parent):
side = ttk.LabelFrame(parent, text="Notes", padding=6)
side.grid(row=0, column=1, rowspan=2, sticky=tk.NSEW)
text = (
"Tab 1: Choose an audio file, then run 'Upload + transcribe'.\n\n"
"Tab 2: Enter text, generate speech on the server, then download the MP3.\n\n"
"The server stores temporary files in the speech work directory."
)
ttk.Label(side, text=text, justify=tk.LEFT, wraplength=300).pack(fill=tk.BOTH, expand=True)
def _submit(self, coro):
return asyncio.run_coroutine_threadsafe(coro, self._loop)
def _make_ssh(self) -> AsyncSSH:
return AsyncSSH(
self.v_host.get().strip(),
self.v_user.get().strip(),
self.v_pass.get(),
int(self.v_port.get()),
)
def _browse_audio(self):
path = filedialog.askopenfilename(
title="Select audio file",
filetypes=[
("Audio files", "*.wav *.mp3 *.m4a *.ogg *.flac *.aac"),
("All files", "*.*"),
],
)
if not path:
return
self.v_local_audio.set(path)
remote_path = posixpath.join(self.v_remote_dir.get().strip(), os.path.basename(path))
self.v_remote_audio.set(remote_path)
def _test_connection(self):
if not HAS_ASYNCSSH:
messagebox.showerror("Missing dependency", "asyncssh is not installed.\nRun: pip install asyncssh")
return
self.v_status.set("connecting...")
self._log("[info] testing SSH connection\n", "info")
self._submit(self._async_test_connection())
async def _async_test_connection(self):
ssh = self._make_ssh()
try:
await ssh.connect()
await self._ensure_remote_ready(ssh)
work_dir = self.v_remote_dir.get().strip()
py = self.v_python.get().strip()
cmd = (
f"bash -lc {shlex.quote(f'cd {shlex.quote(work_dir)} && {shlex.quote(py)} --version && ls -1')}"
)
await ssh.exec_stream(cmd, self._log)
self.root.after(0, lambda: self.v_status.set("connected"))
except Exception as exc:
self._log(f"[error] {exc}\n", "error")
self.root.after(0, lambda: self.v_status.set("connection failed"))
finally:
await ssh.close()
async def _ensure_remote_ready(self, ssh: AsyncSSH):
work_dir = self.v_remote_dir.get().strip()
await ssh.mkdir_p(work_dir)
await ssh.write_text(posixpath.join(work_dir, "transcribe_remote.py"), TRANSCRIBE_SCRIPT)
await ssh.write_text(posixpath.join(work_dir, "tts_remote.py"), TTS_SCRIPT)
def _upload_audio_only(self):
local_path = self.v_local_audio.get().strip()
if not local_path or not os.path.isfile(local_path):
messagebox.showwarning("Missing file", "Please select a local audio file first.")
return
self._submit(self._async_upload_audio(local_path))
async def _async_upload_audio(self, local_path: str):
ssh = self._make_ssh()
try:
await ssh.connect()
await self._ensure_remote_ready(ssh)
remote_path = self.v_remote_audio.get().strip()
await ssh.upload(local_path, remote_path)
self._log(f"[ok] uploaded audio to {remote_path}\n", "ok")
except Exception as exc:
self._log(f"[error] upload failed: {exc}\n", "error")
finally:
await ssh.close()
def _transcribe_audio(self):
local_path = self.v_local_audio.get().strip()
if not local_path or not os.path.isfile(local_path):
messagebox.showwarning("Missing file", "Please select a local audio file first.")
return
self._current_task = self._submit(self._async_transcribe_audio(local_path))
async def _async_transcribe_audio(self, local_path: str):
ssh = self._make_ssh()
try:
await ssh.connect()
await self._ensure_remote_ready(ssh)
remote_audio = self.v_remote_audio.get().strip()
await ssh.upload(local_path, remote_audio)
self._log(f"[info] uploaded audio to {remote_audio}\n", "info")
work_dir = self.v_remote_dir.get().strip()
py = self.v_python.get().strip()
output_txt = posixpath.join(work_dir, "transcript.txt")
output_json = posixpath.join(work_dir, "transcript.json")
language = self.v_stt_language.get().strip() or "-"
cmd_parts = [
shlex.quote(py),
shlex.quote(posixpath.join(work_dir, "transcribe_remote.py")),
shlex.quote(remote_audio),
shlex.quote(output_txt),
shlex.quote(output_json),
shlex.quote(self.v_stt_model.get().strip()),
shlex.quote(language),
shlex.quote(self.v_stt_task.get().strip()),
shlex.quote(self.v_stt_device.get().strip()),
shlex.quote(self.v_stt_compute.get().strip()),
]
shell_cmd = f"bash -lc {shlex.quote(f'cd {shlex.quote(work_dir)} && ' + ' '.join(cmd_parts))}"
self._log(f"[info] transcribing with model {self.v_stt_model.get().strip()}\n", "info")
exit_code = await ssh.exec_stream(shell_cmd, self._log)
if exit_code != 0:
self._log("[error] transcription failed\n", "error")
return
transcript = await ssh.read_text(output_txt)
self.latest_transcript_remote = output_txt
self.root.after(0, lambda: self._set_transcript_text(transcript))
self._log(f"[ok] transcript saved at {output_txt}\n", "ok")
except Exception as exc:
self._log(f"[error] transcription failed: {exc}\n", "error")
finally:
await ssh.close()
def _set_transcript_text(self, text: str):
self.transcript_text.delete("1.0", tk.END)
self.transcript_text.insert("1.0", text)
def _download_transcript(self):
if not self.latest_transcript_remote:
messagebox.showinfo("No transcript", "No transcript is available yet. Run a transcription first.")
return
target = filedialog.asksaveasfilename(
title="Save transcript",
defaultextension=".txt",
filetypes=[("Text files", "*.txt"), ("All files", "*.*")],
)
if not target:
return
self._submit(self._async_download_file(self.latest_transcript_remote, target, "transcript"))
def _generate_tts(self):
text = self.tts_text.get("1.0", tk.END).strip()
if not text:
messagebox.showwarning("Missing text", "Please enter text for speech synthesis first.")
return
self._current_task = self._submit(self._async_generate_tts(text))
async def _async_generate_tts(self, text: str):
ssh = self._make_ssh()
try:
await ssh.connect()
await self._ensure_remote_ready(ssh)
work_dir = self.v_remote_dir.get().strip()
py = self.v_python.get().strip()
remote_text = posixpath.join(work_dir, "tts_input.txt")
remote_audio = posixpath.join(work_dir, self.v_tts_output.get().strip())
await ssh.write_text(remote_text, text)
cmd_parts = [
shlex.quote(py),
shlex.quote(posixpath.join(work_dir, "tts_remote.py")),
shlex.quote(remote_text),
shlex.quote(remote_audio),
shlex.quote(self.v_tts_voice.get().strip()),
shlex.quote(self.v_tts_rate.get().strip()),
shlex.quote(self.v_tts_volume.get().strip()),
]
shell_cmd = f"bash -lc {shlex.quote(f'cd {shlex.quote(work_dir)} && ' + ' '.join(cmd_parts))}"
self._log(f"[info] generating audio with voice {self.v_tts_voice.get().strip()}\n", "info")
exit_code = await ssh.exec_stream(shell_cmd, self._log)
if exit_code != 0:
self._log("[error] audio generation failed\n", "error")
return
self.latest_tts_remote = remote_audio
self._log(f"[ok] audio saved at {remote_audio}\n", "ok")
except Exception as exc:
self._log(f"[error] audio generation failed: {exc}\n", "error")
finally:
await ssh.close()
def _download_tts_audio(self):
if not self.latest_tts_remote:
messagebox.showinfo("No audio", "No generated audio is available yet. Run synthesis first.")
return
target = filedialog.asksaveasfilename(
title="Save audio",
defaultextension=".mp3",
filetypes=[("MP3 audio", "*.mp3"), ("All files", "*.*")],
)
if not target:
return
self._submit(self._async_download_file(self.latest_tts_remote, target, "audio"))
def _download_and_open_tts_audio(self):
if not self.latest_tts_remote:
messagebox.showinfo("No audio", "No generated audio is available yet. Run synthesis first.")
return
target = filedialog.asksaveasfilename(
title="Save and open audio",
defaultextension=".mp3",
filetypes=[("MP3 audio", "*.mp3"), ("All files", "*.*")],
)
if not target:
return
self._submit(self._async_download_file(self.latest_tts_remote, target, "audio", open_after=True))
async def _async_download_file(self, remote_path: str, local_path: str, label: str, open_after: bool = False):
ssh = self._make_ssh()
try:
await ssh.connect()
await ssh.download(remote_path, local_path)
self._log(f"[ok] downloaded {label} to {local_path}\n", "ok")
if open_after:
self.root.after(0, lambda: os.startfile(local_path))
except Exception as exc:
self._log(f"[error] download failed: {exc}\n", "error")
finally:
await ssh.close()
def _log(self, text: str, tag: str = ""):
self._log_queue.put((text, tag))
def _poll_log(self):
try:
while True:
text, tag = self._log_queue.get_nowait()
self.log.config(state=tk.NORMAL)
self.log.insert(tk.END, text, tag or "")
self.log.see(tk.END)
self.log.config(state=tk.DISABLED)
except queue.Empty:
pass
self.root.after(100, self._poll_log)
def on_close(self):
self._loop.call_soon_threadsafe(self._loop.stop)
self.root.destroy()
def main():
root = tk.Tk()
style = ttk.Style(root)
for preferred in ("clam", "alt", "default"):
if preferred in style.theme_names():
style.theme_use(preferred)
break
app = SpeechAIGUI(root)
root.protocol("WM_DELETE_WINDOW", app.on_close)
root.mainloop()
if __name__ == "__main__":
main()