diff --git a/__pycache__/speech_ai_gui.cpython-312.pyc b/__pycache__/speech_ai_gui.cpython-312.pyc new file mode 100644 index 0000000..21fb96d Binary files /dev/null and b/__pycache__/speech_ai_gui.cpython-312.pyc differ diff --git a/speech_ai_gui.py b/speech_ai_gui.py index b19a84d..36ec2cb 100644 --- a/speech_ai_gui.py +++ b/speech_ai_gui.py @@ -16,8 +16,10 @@ import os import posixpath import queue import shlex +import stat import threading import tkinter as tk +import tkinter.font as tkfont from tkinter import filedialog, messagebox, scrolledtext, ttk try: @@ -200,17 +202,35 @@ class AsyncSSH: async with sftp.open(remote_path, "r", encoding="utf-8", errors="replace") as handle: return await handle.read() - async def upload(self, local_path: str, remote_path: str): + async def upload(self, local_path: str, remote_path: str, progress_cb=None): await self._ensure_conn() await self.mkdir_p(posixpath.dirname(remote_path)) + total_bytes = os.path.getsize(local_path) + sent_bytes = 0 async with self._conn.start_sftp_client() as sftp: - await sftp.put(local_path, remote_path) + async with sftp.open(remote_path, "wb") as remote_handle: + with open(local_path, "rb") as local_handle: + while True: + chunk = local_handle.read(1024 * 1024) + if not chunk: + break + await remote_handle.write(chunk) + sent_bytes += len(chunk) + if progress_cb: + progress_cb(sent_bytes, total_bytes) + if progress_cb: + progress_cb(total_bytes, total_bytes) async def download(self, remote_path: str, local_path: str): await self._ensure_conn() async with self._conn.start_sftp_client() as sftp: await sftp.get(remote_path, local_path) + async def stat(self, remote_path: str): + await self._ensure_conn() + async with self._conn.start_sftp_client() as sftp: + return await sftp.stat(remote_path) + async def close(self): if self._conn: self._conn.close() @@ -232,6 +252,7 @@ class SpeechAIGUI: self.latest_transcript_remote = "" self.latest_tts_remote = "" + self.results_files: dict[str, dict[str, str]] = {} self._init_vars() self._build_ui() @@ -257,6 +278,8 @@ class SpeechAIGUI: self.v_stt_task = tk.StringVar(value="transcribe") self.v_stt_device = tk.StringVar(value="cpu") self.v_stt_compute = tk.StringVar(value="int8") + self.v_upload_status = tk.StringVar(value="Kein Upload aktiv") + self.v_upload_percent = tk.DoubleVar(value=0.0) self.v_tts_voice = tk.StringVar(value="de-DE-KatjaNeural") self.v_tts_rate = tk.StringVar(value="+0%") @@ -330,16 +353,24 @@ class SpeechAIGUI: tip(entry, help_text) def _build_tabs(self, parent): - notebook = ttk.Notebook(parent) - notebook.grid(row=1, column=0, sticky=tk.NSEW, padx=(0, 6)) + self.nb = ttk.Notebook(parent) + self.nb.grid(row=1, column=0, sticky=tk.NSEW, padx=(0, 6)) + self.nb.bind("<>", self._on_tab_changed) - stt_tab = ttk.Frame(notebook, padding=8) - tts_tab = ttk.Frame(notebook, padding=8) - notebook.add(stt_tab, text="Audio -> Text") - notebook.add(tts_tab, text="Text -> Audio") + stt_tab = ttk.Frame(self.nb, padding=8) + tts_tab = ttk.Frame(self.nb, padding=8) + self.results_tab = ttk.Frame(self.nb, padding=8) + self.nb.add(stt_tab, text="Audio -> Text") + self.nb.add(tts_tab, text="Text -> Audio") + self.nb.add(self.results_tab, text="Remote-Dateien") self._build_stt_tab(stt_tab) self._build_tts_tab(tts_tab) + self._build_results_tab(self.results_tab) + + def _on_tab_changed(self, _event=None): + if self.nb.select() == str(self.results_tab): + self._refresh_results_tab() def _build_stt_tab(self, parent): parent.columnconfigure(1, weight=1) @@ -381,10 +412,18 @@ class SpeechAIGUI: button_row.grid(row=5, column=0, columnspan=3, sticky=tk.W, padx=4, pady=(6, 6)) tip(ttk.Button(button_row, text="Upload audio", command=self._upload_audio_only), "Upload the selected local audio file to the server.").pack(side=tk.LEFT, padx=(0, 6)) tip(ttk.Button(button_row, text="Upload + transcribe", command=self._transcribe_audio), "Upload the file and run faster-whisper on the server.").pack(side=tk.LEFT, padx=6) + tip(ttk.Button(button_row, text="Nur transkribieren", command=self._transcribe_remote_audio), "Run faster-whisper for the audio file that is already stored on the server.").pack(side=tk.LEFT, padx=6) tip(ttk.Button(button_row, text="Download transcript...", command=self._download_transcript), "Save the latest transcript text from the server to a local file.").pack(side=tk.LEFT, padx=6) + progress_row = ttk.Frame(parent) + progress_row.grid(row=6, column=0, columnspan=3, sticky=tk.EW, padx=4, pady=(0, 4)) + progress_row.columnconfigure(0, weight=1) + self.upload_progress = ttk.Progressbar(progress_row, maximum=100, variable=self.v_upload_percent) + self.upload_progress.grid(row=0, column=0, sticky=tk.EW, padx=(0, 8)) + ttk.Label(progress_row, textvariable=self.v_upload_status, width=28).grid(row=0, column=1, sticky=tk.W) + self.transcript_text = scrolledtext.ScrolledText(parent, wrap=tk.WORD, font=("Consolas", 10)) - self.transcript_text.grid(row=6, column=0, columnspan=3, sticky=tk.NSEW, padx=4, pady=4) + self.transcript_text.grid(row=7, column=0, columnspan=3, sticky=tk.NSEW, padx=4, pady=4) def _build_tts_tab(self, parent): parent.columnconfigure(0, weight=1) @@ -430,6 +469,31 @@ class SpeechAIGUI: self.tts_text = scrolledtext.ScrolledText(parent, wrap=tk.WORD, font=("Segoe UI", 11)) self.tts_text.grid(row=3, column=0, sticky=tk.NSEW) + def _build_results_tab(self, parent): + head = ttk.Frame(parent) + head.pack(fill=tk.X, pady=(0, 4)) + + tip(ttk.Label(head, text="Dateien im Speech-Arbeitsverzeichnis:"), "Zeigt den aktuellen Inhalt des Remote-Verzeichnisses an.").pack(side=tk.LEFT) + tip(ttk.Button(head, text="Aktualisieren", command=self._refresh_results_tab), "Liest den Inhalt des Remote-Verzeichnisses erneut vom Server ein.").pack(side=tk.RIGHT, padx=4) + tip(ttk.Button(head, text="Remote-Inhalt löschen", command=self._delete_remote_dir), "Löscht alle Dateien und Unterordner im Remote-Verzeichnis, aber nicht das Verzeichnis selbst.").pack(side=tk.RIGHT, padx=4) + tip(ttk.Button(head, text="Herunterladen", command=self._download_selected_result), "Lädt die ausgewählte Remote-Datei per SFTP herunter.").pack(side=tk.RIGHT, padx=4) + + self.results_info = ttk.Label(parent, text="Noch keine Remote-Dateien geladen.", foreground="gray") + self.results_info.pack(anchor=tk.W, pady=(0, 4)) + + cols = ("name", "size", "modified", "remote") + self.results_tree = ttk.Treeview(parent, columns=cols, show="headings", height=14, selectmode="extended") + self.results_tree.heading("name", text="Datei") + self.results_tree.heading("size", text="Größe") + self.results_tree.heading("modified", text="Geändert") + self.results_tree.heading("remote", text="Remote-Pfad") + self.results_tree.column("name", width=220, anchor=tk.W) + self.results_tree.column("size", width=110, anchor=tk.E) + self.results_tree.column("modified", width=150, anchor=tk.W) + self.results_tree.column("remote", width=520, anchor=tk.W) + self.results_tree.pack(fill=tk.BOTH, expand=True) + self.results_tree.bind("", lambda _e: self._download_selected_result()) + def _build_side_panel(self, parent): side = ttk.LabelFrame(parent, text="Notes", padding=6) side.grid(row=0, column=1, rowspan=2, sticky=tk.NSEW) @@ -497,6 +561,23 @@ class SpeechAIGUI: await ssh.write_text(posixpath.join(work_dir, "transcribe_remote.py"), TRANSCRIBE_SCRIPT) await ssh.write_text(posixpath.join(work_dir, "tts_remote.py"), TTS_SCRIPT) + def _set_upload_progress(self, sent_bytes: int, total_bytes: int): + total = max(total_bytes, 1) + percent = (sent_bytes / total) * 100 + sent_mb = sent_bytes / (1024 * 1024) + total_mb = total_bytes / (1024 * 1024) + self.root.after(0, lambda: self.v_upload_percent.set(percent)) + self.root.after(0, lambda: self.v_upload_status.set(f"Upload {percent:.0f}% ({sent_mb:.1f}/{total_mb:.1f} MB)")) + + def _reset_upload_progress(self, status: str = "Kein Upload aktiv"): + self.root.after(0, lambda: self.v_upload_percent.set(0.0)) + self.root.after(0, lambda: self.v_upload_status.set(status)) + + async def _upload_audio(self, ssh: AsyncSSH, local_path: str, remote_path: str): + self._reset_upload_progress("Upload startet...") + await ssh.upload(local_path, remote_path, self._set_upload_progress) + self.root.after(0, lambda: self.v_upload_status.set("Upload abgeschlossen")) + def _upload_audio_only(self): local_path = self.v_local_audio.get().strip() if not local_path or not os.path.isfile(local_path): @@ -510,9 +591,10 @@ class SpeechAIGUI: await ssh.connect() await self._ensure_remote_ready(ssh) remote_path = self.v_remote_audio.get().strip() - await ssh.upload(local_path, remote_path) + await self._upload_audio(ssh, local_path, remote_path) self._log(f"[ok] uploaded audio to {remote_path}\n", "ok") except Exception as exc: + self._reset_upload_progress("Upload fehlgeschlagen") self._log(f"[error] upload failed: {exc}\n", "error") finally: await ssh.close() @@ -530,42 +612,74 @@ class SpeechAIGUI: await ssh.connect() await self._ensure_remote_ready(ssh) remote_audio = self.v_remote_audio.get().strip() - await ssh.upload(local_path, remote_audio) + await self._upload_audio(ssh, local_path, remote_audio) self._log(f"[info] uploaded audio to {remote_audio}\n", "info") + await self._run_remote_transcription(ssh, remote_audio) + except Exception as exc: + self._reset_upload_progress("Upload fehlgeschlagen") + self._log(f"[error] transcription failed: {exc}\n", "error") + finally: + await ssh.close() - work_dir = self.v_remote_dir.get().strip() - py = self.v_python.get().strip() - output_txt = posixpath.join(work_dir, "transcript.txt") - output_json = posixpath.join(work_dir, "transcript.json") - language = self.v_stt_language.get().strip() or "-" - cmd_parts = [ - shlex.quote(py), - shlex.quote(posixpath.join(work_dir, "transcribe_remote.py")), - shlex.quote(remote_audio), - shlex.quote(output_txt), - shlex.quote(output_json), - shlex.quote(self.v_stt_model.get().strip()), - shlex.quote(language), - shlex.quote(self.v_stt_task.get().strip()), - shlex.quote(self.v_stt_device.get().strip()), - shlex.quote(self.v_stt_compute.get().strip()), - ] - shell_cmd = f"bash -lc {shlex.quote(f'cd {shlex.quote(work_dir)} && ' + ' '.join(cmd_parts))}" - self._log(f"[info] transcribing with model {self.v_stt_model.get().strip()}\n", "info") - exit_code = await ssh.exec_stream(shell_cmd, self._log) - if exit_code != 0: - self._log("[error] transcription failed\n", "error") + def _transcribe_remote_audio(self): + remote_audio = self.v_remote_audio.get().strip() + if not remote_audio: + messagebox.showwarning("Missing file", "Please enter the remote audio file path first.") + return + self._current_task = self._submit(self._async_transcribe_remote_audio()) + + async def _async_transcribe_remote_audio(self): + ssh = self._make_ssh() + try: + await ssh.connect() + await self._ensure_remote_ready(ssh) + remote_audio = self.v_remote_audio.get().strip() + try: + remote_stat = await ssh.stat(remote_audio) + except asyncssh.SFTPError: + self._log(f"[error] remote audio file not found: {remote_audio}\n", "error") return - - transcript = await ssh.read_text(output_txt) - self.latest_transcript_remote = output_txt - self.root.after(0, lambda: self._set_transcript_text(transcript)) - self._log(f"[ok] transcript saved at {output_txt}\n", "ok") + if not stat.S_ISREG(remote_stat.permissions): + self._log(f"[error] remote path is not a file: {remote_audio}\n", "error") + return + self._reset_upload_progress("Kein Upload aktiv") + await self._run_remote_transcription(ssh, remote_audio) except Exception as exc: self._log(f"[error] transcription failed: {exc}\n", "error") finally: await ssh.close() + async def _run_remote_transcription(self, ssh: AsyncSSH, remote_audio: str): + work_dir = self.v_remote_dir.get().strip() + py = self.v_python.get().strip() + output_txt = posixpath.join(work_dir, "transcript.txt") + output_json = posixpath.join(work_dir, "transcript.json") + language = self.v_stt_language.get().strip() or "-" + cmd_parts = [ + shlex.quote(py), + shlex.quote(posixpath.join(work_dir, "transcribe_remote.py")), + shlex.quote(remote_audio), + shlex.quote(output_txt), + shlex.quote(output_json), + shlex.quote(self.v_stt_model.get().strip()), + shlex.quote(language), + shlex.quote(self.v_stt_task.get().strip()), + shlex.quote(self.v_stt_device.get().strip()), + shlex.quote(self.v_stt_compute.get().strip()), + ] + shell_cmd = f"bash -lc {shlex.quote(f'cd {shlex.quote(work_dir)} && ' + ' '.join(cmd_parts))}" + self._log(f"[info] transcribing with model {self.v_stt_model.get().strip()}\n", "info") + exit_code = await ssh.exec_stream(shell_cmd, self._log) + if exit_code != 0: + self._log("[error] transcription failed\n", "error") + return + + transcript = await ssh.read_text(output_txt) + self.latest_transcript_remote = output_txt + self.root.after(0, lambda: self._set_transcript_text(transcript)) + self._log(f"[ok] transcript saved at {output_txt}\n", "ok") + await self._async_refresh_results_tab(work_dir) + def _set_transcript_text(self, text: str): self.transcript_text.delete("1.0", tk.END) self.transcript_text.insert("1.0", text) @@ -618,6 +732,7 @@ class SpeechAIGUI: return self.latest_tts_remote = remote_audio self._log(f"[ok] audio saved at {remote_audio}\n", "ok") + await self._async_refresh_results_tab(work_dir) except Exception as exc: self._log(f"[error] audio generation failed: {exc}\n", "error") finally: @@ -677,6 +792,194 @@ class SpeechAIGUI: pass self.root.after(100, self._poll_log) + def _refresh_results_tab(self): + remote_dir = self.v_remote_dir.get().strip() + self.root.config(cursor="watch") + self.nb.config(cursor="watch") + self.results_tree.config(cursor="watch") + self._submit(self._async_refresh_results_tab(remote_dir)) + + async def _async_refresh_results_tab(self, remote_dir: str): + if not remote_dir: + self.root.after(0, lambda: self._set_results_info("Kein Remote-Verzeichnis gesetzt.")) + self.root.after(0, lambda: self._set_results_items([])) + self.root.after(0, self._clear_wait_cursor) + return + + quoted_dir = shlex.quote(remote_dir) + cmd = ( + f"dir={quoted_dir}; " + "if [ ! -e \"$dir\" ]; then " + "echo 'Remote-Verzeichnis existiert nicht.'; exit 0; " + "fi; " + "find \"$dir\" -maxdepth 2 -type f -printf '%s\t%TY-%Tm-%Td %TH:%TM:%TS\t%P\t%p\\n' | sort" + ) + self.root.after(0, lambda: self._set_results_info(f"Remote-Verzeichnis: {remote_dir}")) + self.root.after(0, lambda: self._set_results_items([])) + + entries: list[dict[str, str]] = [] + + def _capture(line: str): + stripped = line.strip() + if not stripped or stripped.startswith("[process exited"): + return + if "\t" not in stripped: + return + size, modified, rel_path, full_path = stripped.split("\t", 3) + entries.append({ + "name": rel_path or posixpath.basename(full_path), + "size": size, + "modified": modified.split(".")[0], + "remote": full_path, + }) + + try: + ssh = self._make_ssh() + await ssh.connect() + await ssh.exec_stream(cmd, _capture) + await ssh.close() + self.root.after(0, lambda: self._set_results_items(entries)) + except Exception as exc: + self.root.after(0, lambda: self._set_results_info(f"[error] {exc}")) + self.root.after(0, lambda: self._set_results_items([])) + finally: + self.root.after(0, self._clear_wait_cursor) + + def _set_results_info(self, text: str): + self.results_info.config(text=text, foreground="gray") + + def _set_results_items(self, entries: list[dict[str, str]]): + self.results_files = {} + for item in self.results_tree.get_children(): + self.results_tree.delete(item) + for i, entry in enumerate(entries): + item_id = f"result_{i}" + self.results_files[item_id] = entry + self.results_tree.insert( + "", + tk.END, + iid=item_id, + values=( + entry["name"], + self._format_size(entry["size"]), + entry["modified"], + entry["remote"], + ), + ) + self._autosize_results_columns() + + def _autosize_results_columns(self): + body_font = tkfont.nametofont("TkDefaultFont") + head_font = tkfont.nametofont("TkHeadingFont") + padding = 24 + + for col in ("name", "size", "modified", "remote"): + header = self.results_tree.heading(col, "text") or col + width = head_font.measure(header) + padding + for item_id in self.results_tree.get_children(): + value = self.results_tree.set(item_id, col) + width = max(width, body_font.measure(str(value)) + padding) + self.results_tree.column(col, width=width) + + def _clear_wait_cursor(self): + self.root.config(cursor="") + self.nb.config(cursor="") + self.results_tree.config(cursor="") + + def _format_size(self, size_text: str) -> str: + try: + size = int(size_text) + except Exception: + return size_text + units = ["B", "KB", "MB", "GB", "TB"] + value = float(size) + for unit in units: + if value < 1024 or unit == units[-1]: + if unit == "B": + return f"{int(value)} {unit}" + return f"{value:.1f} {unit}" + value /= 1024 + return size_text + + def _download_selected_result(self): + selection = self.results_tree.selection() + if not selection: + messagebox.showwarning("Keine Datei ausgewählt", "Bitte zuerst eine oder mehrere Dateien im Tab 'Remote-Dateien' auswählen.") + return + entries = [] + for item_id in selection: + entry = self.results_files.get(item_id) + if entry: + entries.append(entry) + if not entries: + messagebox.showwarning("Datei fehlt", "Die ausgewählten Dateien konnten nicht gefunden werden.") + return + + source_dir = os.path.dirname(self.v_local_audio.get().strip()) if self.v_local_audio.get().strip() else "" + initial_dir = source_dir if source_dir and os.path.isdir(source_dir) else os.getcwd() + local_dir = filedialog.askdirectory(title="Zielordner für Download auswählen", initialdir=initial_dir) + if not local_dir: + return + + downloads = [] + for entry in entries: + local_path = os.path.normpath(os.path.join(local_dir, os.path.basename(entry["remote"]))) + downloads.append((entry["remote"], local_path)) + self._submit(self._async_download_results(downloads)) + + async def _async_download_results(self, downloads: list[tuple[str, str]]): + try: + ssh = self._make_ssh() + await ssh.connect() + for remote_path, local_path in downloads: + self._log(f"[info] downloading file: {remote_path} -> {local_path}\n", "info") + os.makedirs(os.path.dirname(local_path), exist_ok=True) + await ssh.download(remote_path, local_path) + if not os.path.isfile(local_path): + raise FileNotFoundError(f"Datei nach Download nicht gefunden: {local_path}") + self._log(f"[ok] download completed: {local_path}\n", "ok") + await ssh.close() + except Exception as exc: + self._log(f"[error] download failed: {exc}\n", "error") + + def _delete_remote_dir(self): + remote_dir = self.v_remote_dir.get().strip() + if not remote_dir: + messagebox.showwarning("Remote-Verzeichnis fehlt", "Bitte zuerst ein Remote-Verzeichnis angeben.") + return + if remote_dir == "/": + messagebox.showerror("Ungültiges Verzeichnis", "Das Root-Verzeichnis '/' darf nicht geleert werden.") + return + if not messagebox.askyesno( + "Remote-Inhalt löschen", + f"Soll der gesamte Inhalt dieses Verzeichnisses auf dem Server gelöscht werden?\n\n{remote_dir}\n\nDas Verzeichnis selbst bleibt erhalten." + ): + return + self._submit(self._async_delete_remote_dir(remote_dir)) + + async def _async_delete_remote_dir(self, remote_dir: str): + quoted_dir = shlex.quote(remote_dir) + cmd = ( + f"dir={quoted_dir}; " + "if [ -z \"$dir\" ] || [ \"$dir\" = \"/\" ]; then " + "echo '[error] unsafe remote directory'; exit 1; " + "fi; " + "if [ ! -e \"$dir\" ]; then " + "echo '[info] remote directory does not exist'; exit 0; " + "fi; " + "find \"$dir\" -mindepth 1 -maxdepth 1 -exec rm -rf -- {} +; " + "echo \"[ok] remote directory content deleted: $dir\"" + ) + self._log(f"[info] deleting remote directory content: {remote_dir}\n", "info") + try: + ssh = self._make_ssh() + await ssh.connect() + await ssh.exec_stream(cmd, self._log) + await ssh.close() + await self._async_refresh_results_tab(remote_dir) + except Exception as exc: + self._log(f"[error] delete failed: {exc}\n", "error") + def on_close(self): self._loop.call_soon_threadsafe(self._loop.stop) self.root.destroy()