From 4248d337af042517a6f9910d618c535bd4726b0c Mon Sep 17 00:00:00 2001 From: Dita Aji Pratama Date: Tue, 6 Oct 2026 16:29:31 +0700 Subject: [PATCH] WiP ASR --- config.default.yaml | 14 ++ config.py | 25 ++- interfaces/tui/actions.py | 25 +++ interfaces/tui/app.py | 45 +++++- interfaces/tui/keycodes.py | 24 +++ interfaces/tui/moopui.py | 1 + interfaces/tui/render.py | 24 ++- interfaces/tui/talk.py | 232 ++++++++++++++++++++++++++++ lib/moop.py | 9 +- plan/asr-push-to-talk.md | 50 ++++++ tools/asr.py | 302 +++++++++++++++++++++++++++++++++++++ 11 files changed, 742 insertions(+), 9 deletions(-) create mode 100644 interfaces/tui/talk.py create mode 100644 plan/asr-push-to-talk.md create mode 100644 tools/asr.py diff --git a/config.default.yaml b/config.default.yaml index 824b4b7..b70c25a 100644 --- a/config.default.yaml +++ b/config.default.yaml @@ -15,6 +15,20 @@ llm: moop: # Model Option db_path: "~/.config/hendrik/moop.sqlite3" +asr: # Speech-to-Text (push-to-talk TUI). Model dikelola via moop tipe 'asr'. + record_dir : "~/.config/hendrik/recordings" # lokasi file hasil rekaman + record_key : f9 # tombol push-to-talk (f1..f12 / ctrl-) + record_backend : pulse # ffmpeg input: pulse | alsa | oss | windows + device : default # nama device input + windows_mic : "" # backend=windows: nama device dshow, + # contoh "Microphone Array (Realtek(R) Audio)" + format : wav # wav | ogg (opus) + sample_rate : 16000 # Hz + channels : 1 + max_seconds : 60 # auto-stop pengaman + transcribe_timeout: 60 # detik timeout request ASR + keep_recordings : true # false = hapus file setelah ditranskrip + session: # Chat session db_path: "~/.config/hendrik/sessions.json" diff --git a/config.py b/config.py index c873de3..5e0845f 100644 --- a/config.py +++ b/config.py @@ -39,6 +39,29 @@ moop_db_path = os.path.expanduser( llm_timeout = int(_yaml_get("llm", "timeout", default=600)) +# ─── ASR / Push-to-Talk (TUI) ───────────────────────────────────────────────── +# Model ASR dikelola via moop (tipe 'asr'); bagian ini mengatur rekam mic. + +def _as_bool(value, default=False): + return default if value is None else str(value).strip().lower() in ("true", "1", "yes") + +ASR_RECORD_DIR = os.path.expanduser( + _yaml_get("asr", "record_dir", default="~/.config/hendrik/recordings") +) +ASR_RECORD_BACKEND = str(_yaml_get("asr", "record_backend", default="pulse")).strip().lower() # pulse | alsa | oss | windows +ASR_DEVICE = str(_yaml_get("asr", "device", default="default")).strip() or "default" +ASR_WINDOWS_MIC = str(_yaml_get("asr", "windows_mic", default="")).strip() # nama device dshow (backend: windows) +ASR_FORMAT = str(_yaml_get("asr", "format", default="wav")).strip().lower() # wav | ogg +ASR_SAMPLE_RATE = int(_yaml_get("asr", "sample_rate", default=16000)) +ASR_CHANNELS = int(_yaml_get("asr", "channels", default=1)) +ASR_MAX_SECONDS = int(_yaml_get("asr", "max_seconds", default=60)) +ASR_TIMEOUT = int(_yaml_get("asr", "transcribe_timeout", default=60)) +ASR_KEEP_RECORDINGS = _as_bool(_yaml_get("asr", "keep_recordings", default=True), default=True) + +# Tombol push-to-talk: "f1".."f12" atau "ctrl-". Default F9. +ASR_RECORD_KEY = str(_yaml_get("asr", "record_key", default="f9")).strip().lower() + + XMPP_USERNAME = _yaml_get("xmpp", "username", default="") XMPP_PASSWORD = _yaml_get("xmpp", "password", default="") @@ -262,7 +285,7 @@ def load_model_config(char_override=None): """Terapkan model.yaml milik karakter aktif ke default moop (sqlite). Jika file agent/characters//model.yaml ada, tiap tipe yang terisi - (llm/embedding/imagegen/imagevision) akan dijadikan default model set + (asr/embedding/imagegen/imagevision/llm) akan dijadikan default model set untuk tipe tersebut. Tipe kosong/missing tidak diubah; karakter tanpa model.yaml akan memakai default yang sudah ada. diff --git a/interfaces/tui/actions.py b/interfaces/tui/actions.py index 775286c..ff4c73e 100644 --- a/interfaces/tui/actions.py +++ b/interfaces/tui/actions.py @@ -29,6 +29,10 @@ from .moopui import ( providers_manage_popup, ) from .agent import submit, log +from .keycodes import resolve_key +from . import talk as talk_mod + +import config as _config # ---------------------------------------------------------------- handlers --- @@ -46,6 +50,16 @@ def _send_prompt(app, stdscr): submit(app, stdscr) +def _record_talk(app, stdscr): + talk_mod.talk_toggle(app, stdscr) + + +def _talk_configured(app): + # Tombol rekam valid (di-resolve dari config). Ketersediaan chain 'asr' + # divalidasi di handler supaya bisa memberi pesan panduan, bukan diam. + return getattr(app, "talk_key", None) is not None + + def _ready(app): return not app.processing @@ -104,6 +118,15 @@ ACTIONS = { "label": "Change Workspace", "mnemonic": "C", "shortcut": "F6", "handler": workspace_popup, "enabled": _ready, }, + # Push-to-Talk (ASR): tombol berasal dari config.asr.record_key (default F9). + # Handler talk_toggle memvalidasi chain moop 'asr' + ketersediaan mic. + "push_to_talk": { + "key": resolve_key(getattr(_config, "ASR_RECORD_KEY", "f9")), + "label": "Push-to-Talk", "mnemonic": "T", + "shortcut": (getattr(_config, "ASR_RECORD_KEY", "f9") or "").upper(), + "handler": _record_talk, + "enabled": lambda app: _ready(app) and _talk_configured(app), + }, } @@ -144,6 +167,8 @@ MENUS = [ "sep", _item("manage_sets"), _item("manage_providers"), + "sep", + _item("push_to_talk"), ], }, { diff --git a/interfaces/tui/app.py b/interfaces/tui/app.py index 509f244..44adf8c 100644 --- a/interfaces/tui/app.py +++ b/interfaces/tui/app.py @@ -7,6 +7,7 @@ from .theme import init_colors from .render import draw from .input import handle_key from . import keycodes, menubar +from . import talk as talk_mod from .agent import log, WELCOME_ART from services.session_manager_neo import NeoSessionManager, NeoSession from lib import ragroleplay, moop @@ -49,6 +50,26 @@ class HendrikTUI: self.agent_thread: threading.Thread | None = None self.agent_done = threading.Event() + # State push-to-talk (ASR) — lihat interfaces/tui/talk.py + self.talk_recorder = None # Recorder aktif saat merekam + self.talk_transcribing = False # request ASR berjalan + self.talk_cancelled = False # user batal saat transcribing + self.talk_done = threading.Event() + self._talk_thread: threading.Thread | None = None + self._talk_result: dict | None = None + # Kode tombol rekam dari config (None = tombol tidak valid/fitur mati) + from .actions import ACTIONS as _ACTIONS + self.talk_key = _ACTIONS["push_to_talk"]["key"] + self.talk_label = _ACTIONS["push_to_talk"]["shortcut"] + # Cache ketersediaan chain moop 'asr' (diisi saat start/run & switch set) + self.asr_ready = self._check_asr_ready() + + def _check_asr_ready(self) -> bool: + try: + return bool(moop.default_chain(None, "asr")) + except Exception: + return False + self.session_mgr = NeoSessionManager() self.current_session: NeoSession | None = None @@ -95,6 +116,9 @@ class HendrikTUI: self.current_session.doc_id, self._model_info() ) + if type_name == "asr": + self.asr_ready = bool(chain) + if type_name == "llm": self.model_set_name = target.get("name") or self._lookup_model_set_name() @@ -235,6 +259,7 @@ class HendrikTUI: try: self._run_loop(stdscr) finally: + talk_mod.shutdown(self) keycodes.disable_extended_keys() def _run_loop(self, stdscr): @@ -250,7 +275,8 @@ class HendrikTUI: draw(self, stdscr) curses.curs_set(2) - timeout_ms = 100 if self.processing else -1 + timeout_ms = 100 if (self.processing or getattr(self, "talk_recorder", None) + or getattr(self, "talk_transcribing", False)) else -1 stdscr.timeout(timeout_ms) try: @@ -264,12 +290,18 @@ class HendrikTUI: key = -1 # Routing key: + # 0. mode talk aktif → key dikhususkan utk push-to-talk # 1. menu bar fokus → navigasi menu bar - # 2. shortcut global → aksi fitur (Ctrl+N, F2, F4, F6, ...) + # 2. shortcut global → aksi fitur (Ctrl+N, F2, F4, F6, F9, ...) # 3. mnemonic Alt+ → fokus menu # 4. sisanya → editing input if self.menu_active: menubar.handle_menubar_key(self, stdscr, key) + elif talk_mod.talking(self) and key != self.talk_key \ + and talk_mod.talk_key(self, stdscr, key): + # Mode talk: Enter=stop+transkrip, Backspace/Esc=batal, key lain + # diabaikan. talk_key mengembalikan False → jatuh ke routing normal. + pass else: if not menubar.dispatch_global_key(self, stdscr, key): if not menubar.handle_alt_menu_key(self, stdscr, key): @@ -280,3 +312,12 @@ class HendrikTUI: self.agent_done.clear() self.processing = False self.agent_thread = None + + if self.talk_done.is_set(): + if self._talk_thread: + self._talk_thread.join(timeout=1) + self._talk_thread = None + self.talk_done.clear() + talk_mod.finish_talk(self, stdscr) + + talk_mod.poll_talk(self, stdscr) diff --git a/interfaces/tui/keycodes.py b/interfaces/tui/keycodes.py index 61ec6e3..de4c4c3 100644 --- a/interfaces/tui/keycodes.py +++ b/interfaces/tui/keycodes.py @@ -263,6 +263,30 @@ def read_key(stdscr, timeout_ms: int = -1) -> int: return KEY_ESC +def resolve_key(name) -> int | None: + """Petakan nama tombol dari config menjadi kode kunci. + + Format: "f1".."f12", "ctrl-", atau "esc"/"enter". + Mengembalikan None jika tidak dikenal.""" + n = str(name or "").strip().lower().replace("+", "-").replace(" ", "") + if not n: + return None + _F_KEYS = {f"f{i}": k for i, k in enumerate( + [KEY_F1, KEY_F2, KEY_F3, KEY_F4, KEY_F5, KEY_F6, + KEY_F7, KEY_F8, KEY_F9, KEY_F10, KEY_F11, KEY_F12], start=1)} + if n in _F_KEYS: + return _F_KEYS[n] + if n == "esc": + return KEY_ESC + if n == "enter": + return 13 + if n.startswith("ctrl-") and len(n) == 6: + ch = n[-1] + if "a" <= ch <= "z": + return ord(ch) & 0x1F + return None + + def is_alt_key(key: int) -> bool: return _KEY_ALT_BASE <= key <= _KEY_ALT_BASE + 126 diff --git a/interfaces/tui/moopui.py b/interfaces/tui/moopui.py index 48df60a..226f7ee 100644 --- a/interfaces/tui/moopui.py +++ b/interfaces/tui/moopui.py @@ -244,6 +244,7 @@ def _first_model_desc(set_id): # ---------------------------------------------------------- select model --- _TYPE_LABELS = { + "asr": "Speech Transcription", "llm": "LLM", "embedding": "Embedding", "imagegen": "Image Generation", diff --git a/interfaces/tui/render.py b/interfaces/tui/render.py index 64083a5..aff6627 100644 --- a/interfaces/tui/render.py +++ b/interfaces/tui/render.py @@ -358,13 +358,23 @@ def draw_status(app, stdscr): h, w = app.h, app.w y = h - 9 - mode = " PROCESSING " if app.processing else " READY " + rec = getattr(app, "talk_recorder", None) + transcribing = getattr(app, "talk_transcribing", False) + if rec is not None: + mode = " \u25cf REC " + elif transcribing: + mode = " TRANSCRIBE " + else: + mode = " PROCESSING " if app.processing else " READY " set_name = (getattr(app, "model_set_name", "") or "-").strip() or "-" model = (app.llm.model or "-").strip() or "-" ws = workspace.get_current_workspace() or "-" send_key = ACTIONS["send_prompt"].get("shortcut") or "Ctrl+Return" - right = f" {app.character_name} {send_key}:Send " + hints = f"{send_key}:Send" + if getattr(app, "asr_ready", False) and rec is None and not transcribing and not app.processing: + hints += f" {getattr(app, 'talk_label', 'F9')}:Talk" + right = f" {app.character_name} {hints} " # Sisi kiri hanya boleh memakai ruang selebar w - len(right) - 1 (pemisah). avail = max(0, w - len(right) - 1) @@ -388,8 +398,14 @@ def draw_status(app, stdscr): except curses.error: pass - # Highlight mode dengan warna berbeda (Hijau/Kuning) - mode_attr = curses.color_pair(C_STATUS_READY) if not app.processing else curses.color_pair(C_STATUS_PROC) + # Highlight mode dengan warna berbeda (Hijau/Kuning; merah saat REC) + if rec is not None: + mode_attr = curses.color_pair(C_ERROR) | curses.A_BOLD | curses.A_BLINK + elif transcribing: + mode_attr = curses.color_pair(C_STATUS_PROC) | curses.A_BOLD + else: + mode_attr = (curses.color_pair(C_STATUS_READY) if not app.processing + else curses.color_pair(C_STATUS_PROC)) | curses.A_BOLD highlight_attr = curses.color_pair(C_STATUS_INFO) | curses.A_BOLD try: diff --git a/interfaces/tui/talk.py b/interfaces/tui/talk.py new file mode 100644 index 0000000..8594708 --- /dev/null +++ b/interfaces/tui/talk.py @@ -0,0 +1,232 @@ +# talk.py — Push-to-Talk (ASR/STT) untuk TUI. +# +# Alur: +# * F9 (configurable) → mulai rekam (ffmpeg subprocess, non-blocking). +# * F9 lagi / Enter → stop rekam → transkrip (thread) → hasil DISISIPKAN +# ke form input pada posisi kursor, TIDAK submit. +# * Backspace / Esc → batal: rekaman dibuang / hasil transkrip diabaikan. +# +# State disimpan di app: +# app.talk_recorder — instance Recorder saat merekam, else None +# app.talk_transcribing — True saat request ASR berjalan (thread) +# app.talk_cancelled — True jika user membatalkan saat transcribing +# app.talk_done/_thread/_talk_result — pola selesai-thread (mirip agent_done) + +import os +import threading + +import config +from tools import asr as asr_mod + +from .agent import log + + +def _shortcut_label(): + """Label tombol rekam untuk pesan UI (dari config.asr.record_key).""" + n = (config.ASR_RECORD_KEY or "f9").strip().lower() + if n.startswith("ctrl-") and len(n) == 6: + return f"Ctrl+{n[5].upper()}" + return n.upper() + + +# ------------------------------------------------------------------ helpers --- + +def talking(app): + """True jika sedang merekam atau sedang mentranskrip.""" + return getattr(app, "talk_recorder", None) is not None or \ + getattr(app, "talk_transcribing", False) + + +def insert_text(app, text) -> bool: + """Sisipkan text ke input buffer pada posisi kursor (multi-line aman). + Tidak mengubah isi lain dan tidak submit.""" + if not text: + return False + buf = app.input_buffer + li, col = app.input_line, app.input_col + if "\n" not in text: + cur = buf[li] + buf[li] = cur[:col] + text + cur[col:] + app.input_col = col + len(text) + return True + lines = text.split("\n") + cur = buf[li] + head, tail = cur[:col], cur[col:] + block = [head + lines[0]] + lines[1:-1] + [lines[-1] + tail] + buf[li:li + 1] = block + app.input_line = li + len(lines) - 1 + app.input_col = len(lines[-1]) + return True + + +# ------------------------------------------------------------------- actions --- + +def start_talk(app, stdscr): + # Validasi awal: model set 'asr' harus sudah ada sebelum mikrofon dibuka. + if not getattr(app, "asr_ready", False): + import lib.moop as _moop + if _moop.default_chain(None, "asr"): + app.asr_ready = True + if not getattr(app, "asr_ready", False): + log(app, "error", + "Push-to-Talk: model set 'asr' belum diset. " + "Model > Select Model → Speech Transcription.") + return False + try: + rec = asr_mod.start_recording( + record_dir=config.ASR_RECORD_DIR, + backend=config.ASR_RECORD_BACKEND, + device=config.ASR_DEVICE, + fmt=config.ASR_FORMAT, + sample_rate=config.ASR_SAMPLE_RATE, + channels=config.ASR_CHANNELS, + max_seconds=config.ASR_MAX_SECONDS, + windows_mic=getattr(config, "ASR_WINDOWS_MIC", None), + ) + except asr_mod.RecorderError as e: + log(app, "error", f"Push-to-Talk: {e}") + return False + app.talk_recorder = rec + app.talk_cancelled = False + log(app, "system", + f" \u25cf REC — bicara sekarang. {_shortcut_label()}: stop+transkrip, " + "Backspace: batal") + return True + + +def _transcribe_worker(app, path): + res = asr_mod.transcribe_file(path, timeout=config.ASR_TIMEOUT) + if not config.ASR_KEEP_RECORDINGS: + try: + if os.path.isfile(path): + os.remove(path) + except OSError: + pass + app._talk_result = res + app.talk_transcribing = False + app.talk_done.set() + + +def stop_and_transcribe(app, stdscr): + """Stop rekaman lalu transkrip di thread. Hasil masuk form (tanpa submit).""" + rec = getattr(app, "talk_recorder", None) + if rec is None: + if getattr(app, "talk_transcribing", False): + log(app, "system", " Transkripsi masih berjalan, tunggu sebentar...") + return + path = rec.path + try: + path, duration = rec.stop() + except asr_mod.RecorderError as e: + app.talk_recorder = None + log(app, "error", f"Push-to-Talk: {e}") + return + app.talk_recorder = None + app.talk_transcribing = True + app.talk_cancelled = False + log(app, "system", f" Transcribing... ({duration:.1f}s)") + + app.talk_done.clear() + app._talk_thread = threading.Thread( + target=_transcribe_worker, args=(app, path), daemon=True + ) + app._talk_thread.start() + + +def cancel_talk(app): + """Backspace/Esc: batalkan rekaman (buang file) atau transkripsi (abaikan hasil). + + Kondisi 'pending' sengaja lebar: selama transcribing ATAU hasil sudah + menunggu untuk disisipkan (_talk_result / talk_done), pembatalan tetap + berlaku — menutup race saat ASR selesai sesaat sebelum Backspace.""" + rec = getattr(app, "talk_recorder", None) + if rec is not None: + rec.cancel() + app.talk_recorder = None + log(app, "system", " Rekaman dibatalkan.") + return + pending = (getattr(app, "talk_transcribing", False) + or getattr(app, "_talk_result", None) is not None + or app.talk_done.is_set()) + if pending: + app.talk_cancelled = True + log(app, "system", " Hasil transkripsi akan diabaikan.") + + +def talk_toggle(app, stdscr): + """Handler aksi F9: mulai rekam / stop+transkrip (toggle).""" + if getattr(app, "talk_transcribing", False): + return + if getattr(app, "talk_recorder", None) is not None: + stop_and_transcribe(app, stdscr) + else: + start_talk(app, stdscr) + + +def talk_key(app, stdscr, key) -> bool: + """Konsumsi satu key saat mode talk. Returns True bila key dikonsumsi.""" + from . import keycodes + if key in (keycodes.KEY_ENTER, 10, 13): + # Enter: stop + transkrip, hasil HANYA mengisi form (tidak submit). + stop_and_transcribe(app, stdscr) + return True + if key in (keycodes.KEY_BACKSPACE, 127, keycodes.KEY_ESC): + cancel_talk(app) + return True + if key == keycodes.KEY_CTRL_C: + # Ctrl+C tetap jadi exit aplikasi (bukan konsumsi mode talk). + return False + # Semua key lain diabaikan selama mode talk. + return True + + +def poll_talk(app, stdscr): + """Dipanggil tiap iterasi loop: tangani auto-stop saat ffmpeg mencapai + max_seconds (prosesnya exit sendiri).""" + rec = getattr(app, "talk_recorder", None) + if rec is not None and not rec.alive(): + log(app, "system", " Batas waktu rekaman tercapai.") + stop_and_transcribe(app, stdscr) + + +def shutdown(app): + """Dipanggil saat TUI keluar: pastikan proses ffmpeg rekaman mati & file dibuang.""" + rec = getattr(app, "talk_recorder", None) + if rec is not None: + try: + rec.cancel() + except Exception: + pass + app.talk_recorder = None + th = getattr(app, "_talk_thread", None) + if th is not None: + app.talk_cancelled = True + try: + th.join(timeout=1) + except Exception: + pass + app._talk_thread = None + + +def finish_talk(app, stdscr): + """Dipanggil main loop saat app.talk_done ter-set: sisipkan hasil.""" + res = getattr(app, "_talk_result", None) + app._talk_result = None + if getattr(app, "talk_cancelled", False): + app.talk_cancelled = False + return + if not res: + return + if res.get("ok"): + text = res.get("text", "") + if not text: + log(app, "system", " Tidak ada ucapan terdeteksi.") + return + if insert_text(app, text): + log(app, "system", + f" \u266a [{res.get('provider', '?')} / {res.get('model', '?')}] " + "teks masuk ke form (Ctrl+Enter untuk kirim)") + if app.scroll_follow: + app.scroll = 999999 + else: + log(app, "error", f"ASR: {res.get('error', 'unknown error')}") diff --git a/lib/moop.py b/lib/moop.py index 9ce656b..4e30824 100644 --- a/lib/moop.py +++ b/lib/moop.py @@ -4,7 +4,7 @@ # sqlite. Modul ini bertanggung jawab atas: # * inisialisasi schema sql # * CRUD model set, provider (moop_api), key, dan model -# * default model set per tipe (llm/embedding/imagegen/imagevision) +# * default model set per tipe (llm/embedding/imagegen/imagevision/asr) # * resolve "chain" kandidat (base_url, model, api_key) untuk auto-switch # # Prioritas auto-switch di dalam satu model set: @@ -19,7 +19,7 @@ import uuid import config -MODEL_TYPES = ("llm", "embedding", "imagegen", "imagevision") +MODEL_TYPES = ("asr", "embedding", "imagegen", "imagevision", "llm") _SCHEMA = """ PRAGMA foreign_keys = ON; @@ -675,3 +675,8 @@ def embedding_endpoint(path=None): def imagegen_endpoint(path=None): """Endpoint image generation (url, model, api_key) dari default chain 'imagegen'.""" return first_endpoint(path, "imagegen") + + +def asr_endpoint(path=None): + """Endpoint speech-to-text (url, model, api_key) dari default chain 'asr'.""" + return first_endpoint(path, "asr") diff --git a/plan/asr-push-to-talk.md b/plan/asr-push-to-talk.md new file mode 100644 index 0000000..47116fe --- /dev/null +++ b/plan/asr-push-to-talk.md @@ -0,0 +1,50 @@ +# Fitur ASR / Push-to-Talk (TUI) + +Menambahkan jenis model `asr` ke moop dan fitur push-to-talk di TUI: +tekan tombol → bicara → stop → teks hasil transkripsi masuk ke form input +(tidak auto-submit). + +## moop +- `MODEL_TYPES` += `asr` (self-healing: `INSERT OR IGNORE` saat init, DB lama aman). +- `asr_endpoint()` helper ditambahkan. +- TUI: tipe 'asr' muncul di Model > Select Model (label "Speech Transcription") + dan di Manage Sets (add type / set default). + +## Konfigurasi (config.yaml, blok `asr:`) + asr: + record_dir : "~/.config/hendrik/recordings" # lokasi file rekaman + record_key : f9 # f1..f12 / ctrl- + record_backend : pulse # pulse | alsa | oss | windows (dshow) + device : default + windows_mic : "" # backend=windows: nama device dshow + format : wav # wav | ogg(libopus) + sample_rate : 16000 + channels : 1 + max_seconds : 60 # auto-stop pengaman + transcribe_timeout: 60 + keep_recordings : true + +## Alur push-to-talk (interfaces/tui/talk.py) +- `F9` mulai rekam (ffmpeg subprocess, non-blocking). Status bar: `● REC` (merah blink). +- `F9` lagi ATAU `Enter` → stop → transkrip (thread) → hasil disisipkan ke form + di posisi kursor. Status bar: `TRANSCRIBE`. `Enter` TIDAK submit; submit `Ctrl+Enter`. +- `Backspace`/`Esc` → batal (rekaman dibuang / transkrip diabaikan, incl. race saat + hasil sudah menunggu). +- `Ctrl+C` tetap exit kapan pun; saat exit, rekaman aktif otomatis dibersihkan. +- ffmpeg mencapai `max_seconds` → auto stop+transkrip. +- Validasi: `asr` model set belum diset → pesan panduan; device bisu (peak RMS + dari wav di bawah ambang) → "Rekaman kosong", transkrip dilewati. + +## Transkripsi (tools/asr.py) +- `POST {base_url}/audio/transcriptions` multipart (file + model), gaya OpenAI. +- Memakai seluruh chain moop `asr` (rotasi model/provider ikut priority), pola + sama dengan tools/imagegen.py. HTTP 200 + text kosong = valid ("no speech"). +- Provider/Model dicontohkan: OpenRouter `qwen/qwen3-asr-1.7b` + (modality `audio->transcription`; endpoint `/audio/transcriptions` dikonfirmasi + hidup; teruji HTTP 200 dengan usage.seconds dari tone 2 detik). + +## Catatan WSL2 +- Output audio WSLg (sink) bekerja; input mic (RDPSource) TIDAK tersambung pada + WSL 3.0.1/WSLg 1.0.79 (handshake RDP hanya accept `rdpsnd`, tanpa kanal audio-in; + source RDPSource SUSPENDED, rekam hang). Jalur cadangan yang disiapkan: + `record_backend: windows` (ffmpeg.exe + dshow via interop, simpan ke /mnt/c). diff --git a/tools/asr.py b/tools/asr.py new file mode 100644 index 0000000..296f930 --- /dev/null +++ b/tools/asr.py @@ -0,0 +1,302 @@ +# asr.py — Automatic Speech Recognition (STT). +# +# Dua tanggung jawab: +# * start_recording() — spawn `ffmpeg` merekam mic (push-to-talk). +# Backend input: pulse | alsa | oss | windows (ffmpeg.exe + dshow). +# Stop graceful kirim 'q' ke stdin ffmpeg supaya header file tertutup +# rapi (jangan pakai kill/terminate bila masih bisa). +# * transcribe_file() — POST {base_url}/audio/transcriptions (multipart, +# gaya OpenAI Whisper) ke semua kandidat chain moop tipe 'asr', +# rotasi model/provider mengikuti priority (pola tools/imagegen.py). +# +# Parameter rekam (dir, format, device, dst) sengaja di-pass eksplisit dari +# pemanggil (TUI membaca config.asr_*), agar modul ini tidak mengikat config. + +import os +import shutil +import subprocess +import time +from datetime import datetime + +import requests + +from lib import moop + +MIME_BY_EXT = {".wav": "audio/wav", ".ogg": "audio/ogg", ".mp3": "audio/mpeg", + ".m4a": "audio/mp4", ".flac": "audio/flac", ".opus": "audio/ogg"} + + +class RecorderError(Exception): + """Gagal menyiapkan/menghentikan proses rekam.""" + + +# ---------------------------------------------------------------- recorder --- + +def _resolve_ffmpeg(backend, ffmpeg_bin): + """ffmpeg binary untuk backend tertentu + cek ketersediaan.""" + if backend == "windows": + exe = (ffmpeg_bin or "ffmpeg.exe").strip() + found = shutil.which(exe) + if not found: + # Coba lokasi lazim bila PATH sesi belum di-refresh. + for guess in ("ffmpeg.exe", "/mnt/c/ffmpeg/bin/ffmpeg.exe"): + found = shutil.which(guess) + if found: + break + if not found: + import glob + hits = glob.glob("/mnt/c/Program Files*/ffmpeg*/bin/ffmpeg.exe") + found = hits[0] if hits else None + if not found: + raise RecorderError( + "ffmpeg.exe (Windows) tidak ditemukan. Install: winget install Gyan.FFmpeg" + ) + return found + found = shutil.which(ffmpeg_bin or "ffmpeg") + if not found: + raise RecorderError("ffmpeg tidak ditemukan di PATH. Install ffmpeg untuk rekam mic.") + return found + + +def _input_args(backend, device, windows_mic): + """Argument ffmpeg untuk sumber audio per backend.""" + if backend == "windows": + if not windows_mic: + raise RecorderError( + "asr.windows_mic belum diisi (nama device dshow, contoh: " + "\"Microphone Array (Realtek(R) Audio)\")." + ) + return ["-f", "dshow", "-i", f"audio={windows_mic}"] + if backend not in ("pulse", "alsa", "oss"): + raise RecorderError(f"Backend rekam tidak dikenal: '{backend}' (pulse|alsa|oss|windows)") + dev = device or "default" + if backend == "oss" and dev == "default": + dev = "/dev/dsp" + return ["-f", backend, "-i", dev] + + +def _encode_args(fmt, sample_rate, channels, opus_bitrate): + """Argument pemrosesan/encoding sesuai format output.""" + base = ["-ac", str(channels), "-ar", str(sample_rate)] + if fmt == "ogg": + return base + ["-c:a", "libopus", "-b:a", f"{opus_bitrate}k"] + return base + ["-c:a", "pcm_s16le"] + + +def build_record_command(ffmpeg_bin, backend, device, windows_mic, fmt, + sample_rate, channels, max_seconds, out_path, opus_bitrate=24): + cmd = [ffmpeg_bin, "-hide_banner", "-loglevel", "error"] + cmd += _input_args(backend, device, windows_mic) + cmd += _encode_args(fmt, sample_rate, channels, opus_bitrate) + cmd += ["-t", str(max_seconds), "-y", out_path] + return cmd + + +class Recorder: + """Handle satu proses rekaman ffmpeg.""" + + def __init__(self, proc, path, started, min_bytes=1024): + self.proc = proc + self.path = path + self.started = started + self.min_bytes = min_bytes # ambang "rekaman kosong" (header saja) + self._closed = False + + def elapsed(self): + return time.time() - self.started + + def alive(self): + return self.proc.poll() is None + + def _graceful_stop(self, timeout=3.0): + """Kirim 'q' ke stdin (ffmpeg menutup file dengan benar).""" + if self._closed: + return + self._closed = True + if self.proc.poll() is not None: + return + try: + if self.proc.stdin: + self.proc.stdin.write(b"q\n") + self.proc.stdin.flush() + except Exception: + pass + try: + self.proc.wait(timeout=timeout) + except subprocess.TimeoutExpired: + self.proc.terminate() + try: + self.proc.wait(timeout=2) + except subprocess.TimeoutExpired: + self.proc.kill() + + @staticmethod + def _peak_amplitude(path, max_bytes=512 * 1024): + """Peak |sample| dari wav PCM16 (stdlib wave). None bila tidak bisa + dibaca/bukan wav (mis. ogg) → pemanggil menganggapnya 'tidak diketahui'.""" + try: + import wave + with wave.open(path, "rb") as w: + if w.getsampwidth() != 2: + return None + raw = w.readframes(max_bytes // 2) + if not raw: + return None + import array + a = array.array("h") + a.frombytes(raw[: len(raw) & ~1]) + return max((abs(s) for s in a), default=0) + except Exception: + return None + + def stop(self, min_seconds=0.3, silence_peak=128): + """Hentikan rekaman. Returns (path, duration). RecorderError bila file + hilang, terlalu pendek, atau kosong (device hidup tapi tanpa sinyal).""" + duration = max(self.elapsed(), 0.0) + self._graceful_stop() + if not os.path.isfile(self.path): + raise RecorderError("File rekaman tidak terbentuk (mic/perangkat tidak tersedia?).") + size = os.path.getsize(self.path) + peak = self._peak_amplitude(self.path) if self.path.endswith(".wav") else None + too_short = duration < min_seconds or size <= self.min_bytes + silent = peak is not None and peak < silence_peak + if too_short or silent: + try: + os.remove(self.path) + except OSError: + pass + if silent: + raise RecorderError( + "Rekaman kosong (tidak ada sinyal mic). " + "Cek device input / izin mikrofon." + ) + raise RecorderError("Rekaman terlalu pendek, dibatalkan.") + return self.path, duration + + def cancel(self): + """Batalkan: bunuh proses, buang file.""" + if self.alive(): + try: + self.proc.terminate() + self.proc.wait(timeout=2) + except Exception: + try: + self.proc.kill() + except Exception: + pass + self._closed = True + try: + if os.path.isfile(self.path): + os.remove(self.path) + except OSError: + pass + + +def start_recording(record_dir, backend="pulse", device="default", fmt="wav", + sample_rate=16000, channels=1, max_seconds=60, + ffmpeg_bin=None, windows_mic=None, opus_bitrate=24): + """Mulai rekam mic (non-blocking). Returns Recorder.""" + fmt = (fmt or "wav").strip().lower() + if fmt not in ("wav", "ogg"): + raise RecorderError(f"Format rekam tidak dikenal: '{fmt}' (wav|ogg)") + backend = (backend or "pulse").strip().lower() + + exe = _resolve_ffmpeg(backend, ffmpeg_bin) + + record_dir = os.path.expanduser(record_dir) + try: + os.makedirs(record_dir, exist_ok=True) + except OSError as e: + raise RecorderError(f"Gagal membuat folder rekam '{record_dir}': {e}") + + out_path = os.path.join( + record_dir, f"talk_{datetime.now().strftime('%Y%m%d_%H%M%S')}.{fmt}" + ) + cmd = build_record_command( + exe, backend, device, windows_mic, fmt, + sample_rate, channels, max_seconds, out_path, opus_bitrate, + ) + try: + proc = subprocess.Popen( + cmd, + stdin=subprocess.PIPE, # untuk 'q' saat stop + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, # hindari pipe penuh memblok ffmpeg + ) + except Exception as e: + raise RecorderError(f"Gagal menjalankan ffmpeg: {e}") + + # Beri kesempatan gagal instan (device tidak ada/izin ditolak) supaya + # bisa langsung dilaporkan, bukan setelah user menekan stop. + time.sleep(0.35) + if proc.poll() is not None: + raise RecorderError( + f"ffmpeg langsung exit (kode {proc.returncode}) — device input " + f"'{backend}:{device or 'default'}' tidak tersedia/tertunda." + ) + return Recorder(proc, out_path, time.time()) + + +# -------------------------------------------------------------- transcribe --- + +def transcribe_url(base_url): + base = (base_url or "").rstrip("/") + if base.endswith("/audio/transcriptions"): + return base + return base + "/audio/transcriptions" + + +def transcribe_file(audio_path, timeout=60): + """Transkrip satu file audio memakai default chain moop 'asr'. + + Returns dict: {"ok": True, "text": ..., "provider":..., "model":...} + atau {"ok": False, "error": "..."}.""" + if not audio_path or not os.path.isfile(audio_path): + return {"ok": False, "error": f"File audio tidak ditemukan: {audio_path}"} + + chain = moop.default_chain(None, "asr") + if not chain: + return {"ok": False, "error": + "Belum ada model set untuk tipe 'asr'. " + "Atur via Model > Select Model → Speech Transcription (Model Option)."} + + try: + with open(audio_path, "rb") as f: + data = f.read() + except OSError as e: + return {"ok": False, "error": f"Gagal membaca file: {e}"} + + ext = os.path.splitext(audio_path)[1].lower() + mime = MIME_BY_EXT.get(ext, "application/octet-stream") + + last_err = "" + for cand in chain: + url = transcribe_url(cand["base_url"]) + headers = {} + if cand.get("api_key"): + headers["Authorization"] = f"Bearer {cand['api_key']}" + try: + resp = requests.post( + url, + headers=headers, + files={"file": (os.path.basename(audio_path), data, mime)}, + data={"model": cand["model"]}, + timeout=timeout, + ) + except Exception as e: + last_err = f"{cand['provider']}: {e}" + continue + + if resp.status_code == 200: + try: + text = (resp.json() or {}).get("text", "") + except ValueError: + text = (resp.text or "").strip() + # HTTP 200 + text kosong = valid (mis. audio tanpa ucapan) → jangan + # dianggap kegagalan; pemanggil yang memutuskan tampil 'no speech'. + return {"ok": True, "text": (text or "").strip(), + "provider": cand["provider"], "model": cand["model"]} + body = (resp.text or "")[:300].replace("\n", " ") + last_err = f"{cand['provider']} [{cand['model']}]: HTTP {resp.status_code} {body}" + + return {"ok": False, "error": f"Semua kandidat model ASR gagal. {last_err}"}