Merge branch 'master' into product/public
This commit is contained in:
commit
2c87061564
@ -15,6 +15,20 @@ llm:
|
|||||||
moop: # Model Option
|
moop: # Model Option
|
||||||
db_path: "~/.config/hendrik/moop.sqlite3"
|
db_path: "~/.config/hendrik/moop.sqlite3"
|
||||||
|
|
||||||
|
asr: # Speech-to-Text (push-to-talk TUI). Model dikelola via moop tipe 'asr'.
|
||||||
|
record_dir : "~/.config/hendrik/recordings" # lokasi file hasil rekaman
|
||||||
|
record_key : f9 # tombol push-to-talk (f1..f12 / ctrl-<huruf>)
|
||||||
|
record_backend : pulse # ffmpeg input: pulse | alsa | oss | windows
|
||||||
|
device : default # nama device input
|
||||||
|
windows_mic : "" # backend=windows: nama device dshow,
|
||||||
|
# contoh "Microphone Array (Realtek(R) Audio)"
|
||||||
|
format : wav # wav | ogg (opus)
|
||||||
|
sample_rate : 16000 # Hz
|
||||||
|
channels : 1
|
||||||
|
max_seconds : 60 # auto-stop pengaman
|
||||||
|
transcribe_timeout: 60 # detik timeout request ASR
|
||||||
|
keep_recordings : true # false = hapus file setelah ditranskrip
|
||||||
|
|
||||||
session: # Chat session
|
session: # Chat session
|
||||||
db_path: "~/.config/hendrik/sessions.json"
|
db_path: "~/.config/hendrik/sessions.json"
|
||||||
|
|
||||||
|
|||||||
25
config.py
25
config.py
@ -39,6 +39,29 @@ moop_db_path = os.path.expanduser(
|
|||||||
llm_timeout = int(_yaml_get("llm", "timeout", default=600))
|
llm_timeout = int(_yaml_get("llm", "timeout", default=600))
|
||||||
|
|
||||||
|
|
||||||
|
# ─── ASR / Push-to-Talk (TUI) ─────────────────────────────────────────────────
|
||||||
|
# Model ASR dikelola via moop (tipe 'asr'); bagian ini mengatur rekam mic.
|
||||||
|
|
||||||
|
def _as_bool(value, default=False):
|
||||||
|
return default if value is None else str(value).strip().lower() in ("true", "1", "yes")
|
||||||
|
|
||||||
|
ASR_RECORD_DIR = os.path.expanduser(
|
||||||
|
_yaml_get("asr", "record_dir", default="~/.config/hendrik/recordings")
|
||||||
|
)
|
||||||
|
ASR_RECORD_BACKEND = str(_yaml_get("asr", "record_backend", default="pulse")).strip().lower() # pulse | alsa | oss | windows
|
||||||
|
ASR_DEVICE = str(_yaml_get("asr", "device", default="default")).strip() or "default"
|
||||||
|
ASR_WINDOWS_MIC = str(_yaml_get("asr", "windows_mic", default="")).strip() # nama device dshow (backend: windows)
|
||||||
|
ASR_FORMAT = str(_yaml_get("asr", "format", default="wav")).strip().lower() # wav | ogg
|
||||||
|
ASR_SAMPLE_RATE = int(_yaml_get("asr", "sample_rate", default=16000))
|
||||||
|
ASR_CHANNELS = int(_yaml_get("asr", "channels", default=1))
|
||||||
|
ASR_MAX_SECONDS = int(_yaml_get("asr", "max_seconds", default=60))
|
||||||
|
ASR_TIMEOUT = int(_yaml_get("asr", "transcribe_timeout", default=60))
|
||||||
|
ASR_KEEP_RECORDINGS = _as_bool(_yaml_get("asr", "keep_recordings", default=True), default=True)
|
||||||
|
|
||||||
|
# Tombol push-to-talk: "f1".."f12" atau "ctrl-<huruf>". Default F9.
|
||||||
|
ASR_RECORD_KEY = str(_yaml_get("asr", "record_key", default="f9")).strip().lower()
|
||||||
|
|
||||||
|
|
||||||
XMPP_USERNAME = _yaml_get("xmpp", "username", default="")
|
XMPP_USERNAME = _yaml_get("xmpp", "username", default="")
|
||||||
XMPP_PASSWORD = _yaml_get("xmpp", "password", default="")
|
XMPP_PASSWORD = _yaml_get("xmpp", "password", default="")
|
||||||
|
|
||||||
@ -262,7 +285,7 @@ def load_model_config(char_override=None):
|
|||||||
"""Terapkan model.yaml milik karakter aktif ke default moop (sqlite).
|
"""Terapkan model.yaml milik karakter aktif ke default moop (sqlite).
|
||||||
|
|
||||||
Jika file agent/characters/<char>/model.yaml ada, tiap tipe yang terisi
|
Jika file agent/characters/<char>/model.yaml ada, tiap tipe yang terisi
|
||||||
(llm/embedding/imagegen/imagevision) akan dijadikan default model set
|
(asr/embedding/imagegen/imagevision/llm) akan dijadikan default model set
|
||||||
untuk tipe tersebut. Tipe kosong/missing tidak diubah; karakter tanpa
|
untuk tipe tersebut. Tipe kosong/missing tidak diubah; karakter tanpa
|
||||||
model.yaml akan memakai default yang sudah ada.
|
model.yaml akan memakai default yang sudah ada.
|
||||||
|
|
||||||
|
|||||||
@ -29,6 +29,10 @@ from .moopui import (
|
|||||||
providers_manage_popup,
|
providers_manage_popup,
|
||||||
)
|
)
|
||||||
from .agent import submit, log
|
from .agent import submit, log
|
||||||
|
from .keycodes import resolve_key
|
||||||
|
from . import talk as talk_mod
|
||||||
|
|
||||||
|
import config as _config
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------- handlers ---
|
# ---------------------------------------------------------------- handlers ---
|
||||||
@ -46,6 +50,16 @@ def _send_prompt(app, stdscr):
|
|||||||
submit(app, stdscr)
|
submit(app, stdscr)
|
||||||
|
|
||||||
|
|
||||||
|
def _record_talk(app, stdscr):
|
||||||
|
talk_mod.talk_toggle(app, stdscr)
|
||||||
|
|
||||||
|
|
||||||
|
def _talk_configured(app):
|
||||||
|
# Tombol rekam valid (di-resolve dari config). Ketersediaan chain 'asr'
|
||||||
|
# divalidasi di handler supaya bisa memberi pesan panduan, bukan diam.
|
||||||
|
return getattr(app, "talk_key", None) is not None
|
||||||
|
|
||||||
|
|
||||||
def _ready(app):
|
def _ready(app):
|
||||||
return not app.processing
|
return not app.processing
|
||||||
|
|
||||||
@ -104,6 +118,15 @@ ACTIONS = {
|
|||||||
"label": "Change Workspace", "mnemonic": "C", "shortcut": "F6",
|
"label": "Change Workspace", "mnemonic": "C", "shortcut": "F6",
|
||||||
"handler": workspace_popup, "enabled": _ready,
|
"handler": workspace_popup, "enabled": _ready,
|
||||||
},
|
},
|
||||||
|
# Push-to-Talk (ASR): tombol berasal dari config.asr.record_key (default F9).
|
||||||
|
# Handler talk_toggle memvalidasi chain moop 'asr' + ketersediaan mic.
|
||||||
|
"push_to_talk": {
|
||||||
|
"key": resolve_key(getattr(_config, "ASR_RECORD_KEY", "f9")),
|
||||||
|
"label": "Push-to-Talk", "mnemonic": "T",
|
||||||
|
"shortcut": (getattr(_config, "ASR_RECORD_KEY", "f9") or "").upper(),
|
||||||
|
"handler": _record_talk,
|
||||||
|
"enabled": lambda app: _ready(app) and _talk_configured(app),
|
||||||
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@ -144,6 +167,8 @@ MENUS = [
|
|||||||
"sep",
|
"sep",
|
||||||
_item("manage_sets"),
|
_item("manage_sets"),
|
||||||
_item("manage_providers"),
|
_item("manage_providers"),
|
||||||
|
"sep",
|
||||||
|
_item("push_to_talk"),
|
||||||
],
|
],
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
|
|||||||
@ -7,6 +7,7 @@ from .theme import init_colors
|
|||||||
from .render import draw
|
from .render import draw
|
||||||
from .input import handle_key
|
from .input import handle_key
|
||||||
from . import keycodes, menubar
|
from . import keycodes, menubar
|
||||||
|
from . import talk as talk_mod
|
||||||
from .agent import log, WELCOME_ART
|
from .agent import log, WELCOME_ART
|
||||||
from services.session_manager_neo import NeoSessionManager, NeoSession
|
from services.session_manager_neo import NeoSessionManager, NeoSession
|
||||||
from lib import ragroleplay, moop
|
from lib import ragroleplay, moop
|
||||||
@ -52,6 +53,26 @@ class HendrikTUI:
|
|||||||
self.session_mgr = NeoSessionManager()
|
self.session_mgr = NeoSessionManager()
|
||||||
self.current_session: NeoSession | None = None
|
self.current_session: NeoSession | None = None
|
||||||
|
|
||||||
|
# State push-to-talk (ASR) — lihat interfaces/tui/talk.py
|
||||||
|
self.talk_recorder = None # Recorder aktif saat merekam
|
||||||
|
self.talk_transcribing = False # request ASR berjalan
|
||||||
|
self.talk_cancelled = False # user batal saat transcribing
|
||||||
|
self.talk_done = threading.Event()
|
||||||
|
self._talk_thread: threading.Thread | None = None
|
||||||
|
self._talk_result: dict | None = None
|
||||||
|
# Kode tombol rekam dari config (None = tombol tidak valid/fitur mati)
|
||||||
|
from .actions import ACTIONS as _ACTIONS
|
||||||
|
self.talk_key = _ACTIONS["push_to_talk"]["key"]
|
||||||
|
self.talk_label = _ACTIONS["push_to_talk"]["shortcut"]
|
||||||
|
# Cache ketersediaan chain moop 'asr' (diisi saat start/run & switch set)
|
||||||
|
self.asr_ready = self._check_asr_ready()
|
||||||
|
|
||||||
|
def _check_asr_ready(self) -> bool:
|
||||||
|
try:
|
||||||
|
return bool(moop.default_chain(None, "asr"))
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
def _lookup_model_set_name(self) -> str:
|
def _lookup_model_set_name(self) -> str:
|
||||||
"""Nama model set yang sedang aktif (untuk status bar).
|
"""Nama model set yang sedang aktif (untuk status bar).
|
||||||
|
|
||||||
@ -95,6 +116,9 @@ class HendrikTUI:
|
|||||||
self.current_session.doc_id, self._model_info()
|
self.current_session.doc_id, self._model_info()
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if type_name == "asr":
|
||||||
|
self.asr_ready = bool(chain)
|
||||||
|
|
||||||
if type_name == "llm":
|
if type_name == "llm":
|
||||||
self.model_set_name = target.get("name") or self._lookup_model_set_name()
|
self.model_set_name = target.get("name") or self._lookup_model_set_name()
|
||||||
|
|
||||||
@ -235,6 +259,7 @@ class HendrikTUI:
|
|||||||
try:
|
try:
|
||||||
self._run_loop(stdscr)
|
self._run_loop(stdscr)
|
||||||
finally:
|
finally:
|
||||||
|
talk_mod.shutdown(self)
|
||||||
keycodes.disable_extended_keys()
|
keycodes.disable_extended_keys()
|
||||||
|
|
||||||
def _run_loop(self, stdscr):
|
def _run_loop(self, stdscr):
|
||||||
@ -250,7 +275,8 @@ class HendrikTUI:
|
|||||||
draw(self, stdscr)
|
draw(self, stdscr)
|
||||||
curses.curs_set(2)
|
curses.curs_set(2)
|
||||||
|
|
||||||
timeout_ms = 100 if self.processing else -1
|
timeout_ms = 100 if (self.processing or getattr(self, "talk_recorder", None)
|
||||||
|
or getattr(self, "talk_transcribing", False)) else -1
|
||||||
stdscr.timeout(timeout_ms)
|
stdscr.timeout(timeout_ms)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@ -264,12 +290,18 @@ class HendrikTUI:
|
|||||||
key = -1
|
key = -1
|
||||||
|
|
||||||
# Routing key:
|
# Routing key:
|
||||||
|
# 0. mode talk aktif → key dikhususkan utk push-to-talk
|
||||||
# 1. menu bar fokus → navigasi menu bar
|
# 1. menu bar fokus → navigasi menu bar
|
||||||
# 2. shortcut global → aksi fitur (Ctrl+N, F2, F4, F6, ...)
|
# 2. shortcut global → aksi fitur (Ctrl+N, F2, F4, F6, F9, ...)
|
||||||
# 3. mnemonic Alt+ → fokus menu
|
# 3. mnemonic Alt+ → fokus menu
|
||||||
# 4. sisanya → editing input
|
# 4. sisanya → editing input
|
||||||
if self.menu_active:
|
if self.menu_active:
|
||||||
menubar.handle_menubar_key(self, stdscr, key)
|
menubar.handle_menubar_key(self, stdscr, key)
|
||||||
|
elif talk_mod.talking(self) and key != self.talk_key \
|
||||||
|
and talk_mod.talk_key(self, stdscr, key):
|
||||||
|
# Mode talk: Enter=stop+transkrip, Backspace/Esc=batal, key lain
|
||||||
|
# diabaikan. talk_key mengembalikan False → jatuh ke routing normal.
|
||||||
|
pass
|
||||||
else:
|
else:
|
||||||
if not menubar.dispatch_global_key(self, stdscr, key):
|
if not menubar.dispatch_global_key(self, stdscr, key):
|
||||||
if not menubar.handle_alt_menu_key(self, stdscr, key):
|
if not menubar.handle_alt_menu_key(self, stdscr, key):
|
||||||
@ -280,3 +312,12 @@ class HendrikTUI:
|
|||||||
self.agent_done.clear()
|
self.agent_done.clear()
|
||||||
self.processing = False
|
self.processing = False
|
||||||
self.agent_thread = None
|
self.agent_thread = None
|
||||||
|
|
||||||
|
if self.talk_done.is_set():
|
||||||
|
if self._talk_thread:
|
||||||
|
self._talk_thread.join(timeout=1)
|
||||||
|
self._talk_thread = None
|
||||||
|
self.talk_done.clear()
|
||||||
|
talk_mod.finish_talk(self, stdscr)
|
||||||
|
|
||||||
|
talk_mod.poll_talk(self, stdscr)
|
||||||
|
|||||||
@ -263,6 +263,30 @@ def read_key(stdscr, timeout_ms: int = -1) -> int:
|
|||||||
return KEY_ESC
|
return KEY_ESC
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_key(name) -> int | None:
|
||||||
|
"""Petakan nama tombol dari config menjadi kode kunci.
|
||||||
|
|
||||||
|
Format: "f1".."f12", "ctrl-<huruf>", atau "esc"/"enter".
|
||||||
|
Mengembalikan None jika tidak dikenal."""
|
||||||
|
n = str(name or "").strip().lower().replace("+", "-").replace(" ", "")
|
||||||
|
if not n:
|
||||||
|
return None
|
||||||
|
_F_KEYS = {f"f{i}": k for i, k in enumerate(
|
||||||
|
[KEY_F1, KEY_F2, KEY_F3, KEY_F4, KEY_F5, KEY_F6,
|
||||||
|
KEY_F7, KEY_F8, KEY_F9, KEY_F10, KEY_F11, KEY_F12], start=1)}
|
||||||
|
if n in _F_KEYS:
|
||||||
|
return _F_KEYS[n]
|
||||||
|
if n == "esc":
|
||||||
|
return KEY_ESC
|
||||||
|
if n == "enter":
|
||||||
|
return 13
|
||||||
|
if n.startswith("ctrl-") and len(n) == 6:
|
||||||
|
ch = n[-1]
|
||||||
|
if "a" <= ch <= "z":
|
||||||
|
return ord(ch) & 0x1F
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def is_alt_key(key: int) -> bool:
|
def is_alt_key(key: int) -> bool:
|
||||||
return _KEY_ALT_BASE <= key <= _KEY_ALT_BASE + 126
|
return _KEY_ALT_BASE <= key <= _KEY_ALT_BASE + 126
|
||||||
|
|
||||||
|
|||||||
@ -244,6 +244,7 @@ def _first_model_desc(set_id):
|
|||||||
# ---------------------------------------------------------- select model ---
|
# ---------------------------------------------------------- select model ---
|
||||||
|
|
||||||
_TYPE_LABELS = {
|
_TYPE_LABELS = {
|
||||||
|
"asr": "Speech Transcription",
|
||||||
"llm": "LLM",
|
"llm": "LLM",
|
||||||
"embedding": "Embedding",
|
"embedding": "Embedding",
|
||||||
"imagegen": "Image Generation",
|
"imagegen": "Image Generation",
|
||||||
|
|||||||
@ -358,13 +358,23 @@ def draw_status(app, stdscr):
|
|||||||
h, w = app.h, app.w
|
h, w = app.h, app.w
|
||||||
y = h - 9
|
y = h - 9
|
||||||
|
|
||||||
mode = " PROCESSING " if app.processing else " READY "
|
rec = getattr(app, "talk_recorder", None)
|
||||||
|
transcribing = getattr(app, "talk_transcribing", False)
|
||||||
|
if rec is not None:
|
||||||
|
mode = " \u25cf REC "
|
||||||
|
elif transcribing:
|
||||||
|
mode = " TRANSCRIBE "
|
||||||
|
else:
|
||||||
|
mode = " PROCESSING " if app.processing else " READY "
|
||||||
set_name = (getattr(app, "model_set_name", "") or "-").strip() or "-"
|
set_name = (getattr(app, "model_set_name", "") or "-").strip() or "-"
|
||||||
model = (app.llm.model or "-").strip() or "-"
|
model = (app.llm.model or "-").strip() or "-"
|
||||||
ws = workspace.get_current_workspace() or "-"
|
ws = workspace.get_current_workspace() or "-"
|
||||||
|
|
||||||
send_key = ACTIONS["send_prompt"].get("shortcut") or "Ctrl+Return"
|
send_key = ACTIONS["send_prompt"].get("shortcut") or "Ctrl+Return"
|
||||||
right = f" {app.character_name} {send_key}:Send "
|
hints = f"{send_key}:Send"
|
||||||
|
if getattr(app, "asr_ready", False) and rec is None and not transcribing and not app.processing:
|
||||||
|
hints += f" {getattr(app, 'talk_label', 'F9')}:Talk"
|
||||||
|
right = f" {app.character_name} {hints} "
|
||||||
|
|
||||||
# Sisi kiri hanya boleh memakai ruang selebar w - len(right) - 1 (pemisah).
|
# Sisi kiri hanya boleh memakai ruang selebar w - len(right) - 1 (pemisah).
|
||||||
avail = max(0, w - len(right) - 1)
|
avail = max(0, w - len(right) - 1)
|
||||||
@ -388,8 +398,14 @@ def draw_status(app, stdscr):
|
|||||||
except curses.error:
|
except curses.error:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
# Highlight mode dengan warna berbeda (Hijau/Kuning)
|
# Highlight mode dengan warna berbeda (Hijau/Kuning; merah saat REC)
|
||||||
mode_attr = curses.color_pair(C_STATUS_READY) if not app.processing else curses.color_pair(C_STATUS_PROC)
|
if rec is not None:
|
||||||
|
mode_attr = curses.color_pair(C_ERROR) | curses.A_BOLD | curses.A_BLINK
|
||||||
|
elif transcribing:
|
||||||
|
mode_attr = curses.color_pair(C_STATUS_PROC) | curses.A_BOLD
|
||||||
|
else:
|
||||||
|
mode_attr = (curses.color_pair(C_STATUS_READY) if not app.processing
|
||||||
|
else curses.color_pair(C_STATUS_PROC)) | curses.A_BOLD
|
||||||
highlight_attr = curses.color_pair(C_STATUS_INFO) | curses.A_BOLD
|
highlight_attr = curses.color_pair(C_STATUS_INFO) | curses.A_BOLD
|
||||||
|
|
||||||
try:
|
try:
|
||||||
|
|||||||
232
interfaces/tui/talk.py
Normal file
232
interfaces/tui/talk.py
Normal file
@ -0,0 +1,232 @@
|
|||||||
|
# talk.py — Push-to-Talk (ASR/STT) untuk TUI.
|
||||||
|
#
|
||||||
|
# Alur:
|
||||||
|
# * F9 (configurable) → mulai rekam (ffmpeg subprocess, non-blocking).
|
||||||
|
# * F9 lagi / Enter → stop rekam → transkrip (thread) → hasil DISISIPKAN
|
||||||
|
# ke form input pada posisi kursor, TIDAK submit.
|
||||||
|
# * Backspace / Esc → batal: rekaman dibuang / hasil transkrip diabaikan.
|
||||||
|
#
|
||||||
|
# State disimpan di app:
|
||||||
|
# app.talk_recorder — instance Recorder saat merekam, else None
|
||||||
|
# app.talk_transcribing — True saat request ASR berjalan (thread)
|
||||||
|
# app.talk_cancelled — True jika user membatalkan saat transcribing
|
||||||
|
# app.talk_done/_thread/_talk_result — pola selesai-thread (mirip agent_done)
|
||||||
|
|
||||||
|
import os
|
||||||
|
import threading
|
||||||
|
|
||||||
|
import config
|
||||||
|
from tools import asr as asr_mod
|
||||||
|
|
||||||
|
from .agent import log
|
||||||
|
|
||||||
|
|
||||||
|
def _shortcut_label():
|
||||||
|
"""Label tombol rekam untuk pesan UI (dari config.asr.record_key)."""
|
||||||
|
n = (config.ASR_RECORD_KEY or "f9").strip().lower()
|
||||||
|
if n.startswith("ctrl-") and len(n) == 6:
|
||||||
|
return f"Ctrl+{n[5].upper()}"
|
||||||
|
return n.upper()
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------ helpers ---
|
||||||
|
|
||||||
|
def talking(app):
|
||||||
|
"""True jika sedang merekam atau sedang mentranskrip."""
|
||||||
|
return getattr(app, "talk_recorder", None) is not None or \
|
||||||
|
getattr(app, "talk_transcribing", False)
|
||||||
|
|
||||||
|
|
||||||
|
def insert_text(app, text) -> bool:
|
||||||
|
"""Sisipkan text ke input buffer pada posisi kursor (multi-line aman).
|
||||||
|
Tidak mengubah isi lain dan tidak submit."""
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
buf = app.input_buffer
|
||||||
|
li, col = app.input_line, app.input_col
|
||||||
|
if "\n" not in text:
|
||||||
|
cur = buf[li]
|
||||||
|
buf[li] = cur[:col] + text + cur[col:]
|
||||||
|
app.input_col = col + len(text)
|
||||||
|
return True
|
||||||
|
lines = text.split("\n")
|
||||||
|
cur = buf[li]
|
||||||
|
head, tail = cur[:col], cur[col:]
|
||||||
|
block = [head + lines[0]] + lines[1:-1] + [lines[-1] + tail]
|
||||||
|
buf[li:li + 1] = block
|
||||||
|
app.input_line = li + len(lines) - 1
|
||||||
|
app.input_col = len(lines[-1])
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------- actions ---
|
||||||
|
|
||||||
|
def start_talk(app, stdscr):
|
||||||
|
# Validasi awal: model set 'asr' harus sudah ada sebelum mikrofon dibuka.
|
||||||
|
if not getattr(app, "asr_ready", False):
|
||||||
|
import lib.moop as _moop
|
||||||
|
if _moop.default_chain(None, "asr"):
|
||||||
|
app.asr_ready = True
|
||||||
|
if not getattr(app, "asr_ready", False):
|
||||||
|
log(app, "error",
|
||||||
|
"Push-to-Talk: model set 'asr' belum diset. "
|
||||||
|
"Model > Select Model → Speech Transcription.")
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
rec = asr_mod.start_recording(
|
||||||
|
record_dir=config.ASR_RECORD_DIR,
|
||||||
|
backend=config.ASR_RECORD_BACKEND,
|
||||||
|
device=config.ASR_DEVICE,
|
||||||
|
fmt=config.ASR_FORMAT,
|
||||||
|
sample_rate=config.ASR_SAMPLE_RATE,
|
||||||
|
channels=config.ASR_CHANNELS,
|
||||||
|
max_seconds=config.ASR_MAX_SECONDS,
|
||||||
|
windows_mic=getattr(config, "ASR_WINDOWS_MIC", None),
|
||||||
|
)
|
||||||
|
except asr_mod.RecorderError as e:
|
||||||
|
log(app, "error", f"Push-to-Talk: {e}")
|
||||||
|
return False
|
||||||
|
app.talk_recorder = rec
|
||||||
|
app.talk_cancelled = False
|
||||||
|
log(app, "system",
|
||||||
|
f" \u25cf REC — bicara sekarang. {_shortcut_label()}: stop+transkrip, "
|
||||||
|
"Backspace: batal")
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def _transcribe_worker(app, path):
|
||||||
|
res = asr_mod.transcribe_file(path, timeout=config.ASR_TIMEOUT)
|
||||||
|
if not config.ASR_KEEP_RECORDINGS:
|
||||||
|
try:
|
||||||
|
if os.path.isfile(path):
|
||||||
|
os.remove(path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
app._talk_result = res
|
||||||
|
app.talk_transcribing = False
|
||||||
|
app.talk_done.set()
|
||||||
|
|
||||||
|
|
||||||
|
def stop_and_transcribe(app, stdscr):
|
||||||
|
"""Stop rekaman lalu transkrip di thread. Hasil masuk form (tanpa submit)."""
|
||||||
|
rec = getattr(app, "talk_recorder", None)
|
||||||
|
if rec is None:
|
||||||
|
if getattr(app, "talk_transcribing", False):
|
||||||
|
log(app, "system", " Transkripsi masih berjalan, tunggu sebentar...")
|
||||||
|
return
|
||||||
|
path = rec.path
|
||||||
|
try:
|
||||||
|
path, duration = rec.stop()
|
||||||
|
except asr_mod.RecorderError as e:
|
||||||
|
app.talk_recorder = None
|
||||||
|
log(app, "error", f"Push-to-Talk: {e}")
|
||||||
|
return
|
||||||
|
app.talk_recorder = None
|
||||||
|
app.talk_transcribing = True
|
||||||
|
app.talk_cancelled = False
|
||||||
|
log(app, "system", f" Transcribing... ({duration:.1f}s)")
|
||||||
|
|
||||||
|
app.talk_done.clear()
|
||||||
|
app._talk_thread = threading.Thread(
|
||||||
|
target=_transcribe_worker, args=(app, path), daemon=True
|
||||||
|
)
|
||||||
|
app._talk_thread.start()
|
||||||
|
|
||||||
|
|
||||||
|
def cancel_talk(app):
|
||||||
|
"""Backspace/Esc: batalkan rekaman (buang file) atau transkripsi (abaikan hasil).
|
||||||
|
|
||||||
|
Kondisi 'pending' sengaja lebar: selama transcribing ATAU hasil sudah
|
||||||
|
menunggu untuk disisipkan (_talk_result / talk_done), pembatalan tetap
|
||||||
|
berlaku — menutup race saat ASR selesai sesaat sebelum Backspace."""
|
||||||
|
rec = getattr(app, "talk_recorder", None)
|
||||||
|
if rec is not None:
|
||||||
|
rec.cancel()
|
||||||
|
app.talk_recorder = None
|
||||||
|
log(app, "system", " Rekaman dibatalkan.")
|
||||||
|
return
|
||||||
|
pending = (getattr(app, "talk_transcribing", False)
|
||||||
|
or getattr(app, "_talk_result", None) is not None
|
||||||
|
or app.talk_done.is_set())
|
||||||
|
if pending:
|
||||||
|
app.talk_cancelled = True
|
||||||
|
log(app, "system", " Hasil transkripsi akan diabaikan.")
|
||||||
|
|
||||||
|
|
||||||
|
def talk_toggle(app, stdscr):
|
||||||
|
"""Handler aksi F9: mulai rekam / stop+transkrip (toggle)."""
|
||||||
|
if getattr(app, "talk_transcribing", False):
|
||||||
|
return
|
||||||
|
if getattr(app, "talk_recorder", None) is not None:
|
||||||
|
stop_and_transcribe(app, stdscr)
|
||||||
|
else:
|
||||||
|
start_talk(app, stdscr)
|
||||||
|
|
||||||
|
|
||||||
|
def talk_key(app, stdscr, key) -> bool:
|
||||||
|
"""Konsumsi satu key saat mode talk. Returns True bila key dikonsumsi."""
|
||||||
|
from . import keycodes
|
||||||
|
if key in (keycodes.KEY_ENTER, 10, 13):
|
||||||
|
# Enter: stop + transkrip, hasil HANYA mengisi form (tidak submit).
|
||||||
|
stop_and_transcribe(app, stdscr)
|
||||||
|
return True
|
||||||
|
if key in (keycodes.KEY_BACKSPACE, 127, keycodes.KEY_ESC):
|
||||||
|
cancel_talk(app)
|
||||||
|
return True
|
||||||
|
if key == keycodes.KEY_CTRL_C:
|
||||||
|
# Ctrl+C tetap jadi exit aplikasi (bukan konsumsi mode talk).
|
||||||
|
return False
|
||||||
|
# Semua key lain diabaikan selama mode talk.
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def poll_talk(app, stdscr):
|
||||||
|
"""Dipanggil tiap iterasi loop: tangani auto-stop saat ffmpeg mencapai
|
||||||
|
max_seconds (prosesnya exit sendiri)."""
|
||||||
|
rec = getattr(app, "talk_recorder", None)
|
||||||
|
if rec is not None and not rec.alive():
|
||||||
|
log(app, "system", " Batas waktu rekaman tercapai.")
|
||||||
|
stop_and_transcribe(app, stdscr)
|
||||||
|
|
||||||
|
|
||||||
|
def shutdown(app):
|
||||||
|
"""Dipanggil saat TUI keluar: pastikan proses ffmpeg rekaman mati & file dibuang."""
|
||||||
|
rec = getattr(app, "talk_recorder", None)
|
||||||
|
if rec is not None:
|
||||||
|
try:
|
||||||
|
rec.cancel()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
app.talk_recorder = None
|
||||||
|
th = getattr(app, "_talk_thread", None)
|
||||||
|
if th is not None:
|
||||||
|
app.talk_cancelled = True
|
||||||
|
try:
|
||||||
|
th.join(timeout=1)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
app._talk_thread = None
|
||||||
|
|
||||||
|
|
||||||
|
def finish_talk(app, stdscr):
|
||||||
|
"""Dipanggil main loop saat app.talk_done ter-set: sisipkan hasil."""
|
||||||
|
res = getattr(app, "_talk_result", None)
|
||||||
|
app._talk_result = None
|
||||||
|
if getattr(app, "talk_cancelled", False):
|
||||||
|
app.talk_cancelled = False
|
||||||
|
return
|
||||||
|
if not res:
|
||||||
|
return
|
||||||
|
if res.get("ok"):
|
||||||
|
text = res.get("text", "")
|
||||||
|
if not text:
|
||||||
|
log(app, "system", " Tidak ada ucapan terdeteksi.")
|
||||||
|
return
|
||||||
|
if insert_text(app, text):
|
||||||
|
log(app, "system",
|
||||||
|
f" \u266a [{res.get('provider', '?')} / {res.get('model', '?')}] "
|
||||||
|
"teks masuk ke form (Ctrl+Enter untuk kirim)")
|
||||||
|
if app.scroll_follow:
|
||||||
|
app.scroll = 999999
|
||||||
|
else:
|
||||||
|
log(app, "error", f"ASR: {res.get('error', 'unknown error')}")
|
||||||
@ -4,7 +4,7 @@
|
|||||||
# sqlite. Modul ini bertanggung jawab atas:
|
# sqlite. Modul ini bertanggung jawab atas:
|
||||||
# * inisialisasi schema sql
|
# * inisialisasi schema sql
|
||||||
# * CRUD model set, provider (moop_api), key, dan model
|
# * CRUD model set, provider (moop_api), key, dan model
|
||||||
# * default model set per tipe (llm/embedding/imagegen/imagevision)
|
# * default model set per tipe (llm/embedding/imagegen/imagevision/asr)
|
||||||
# * resolve "chain" kandidat (base_url, model, api_key) untuk auto-switch
|
# * resolve "chain" kandidat (base_url, model, api_key) untuk auto-switch
|
||||||
#
|
#
|
||||||
# Prioritas auto-switch di dalam satu model set:
|
# Prioritas auto-switch di dalam satu model set:
|
||||||
@ -19,7 +19,7 @@ import uuid
|
|||||||
|
|
||||||
import config
|
import config
|
||||||
|
|
||||||
MODEL_TYPES = ("llm", "embedding", "imagegen", "imagevision")
|
MODEL_TYPES = ("asr", "embedding", "imagegen", "imagevision", "llm")
|
||||||
|
|
||||||
_SCHEMA = """
|
_SCHEMA = """
|
||||||
PRAGMA foreign_keys = ON;
|
PRAGMA foreign_keys = ON;
|
||||||
@ -675,3 +675,8 @@ def embedding_endpoint(path=None):
|
|||||||
def imagegen_endpoint(path=None):
|
def imagegen_endpoint(path=None):
|
||||||
"""Endpoint image generation (url, model, api_key) dari default chain 'imagegen'."""
|
"""Endpoint image generation (url, model, api_key) dari default chain 'imagegen'."""
|
||||||
return first_endpoint(path, "imagegen")
|
return first_endpoint(path, "imagegen")
|
||||||
|
|
||||||
|
|
||||||
|
def asr_endpoint(path=None):
|
||||||
|
"""Endpoint speech-to-text (url, model, api_key) dari default chain 'asr'."""
|
||||||
|
return first_endpoint(path, "asr")
|
||||||
|
|||||||
50
plan/asr-push-to-talk.md
Normal file
50
plan/asr-push-to-talk.md
Normal file
@ -0,0 +1,50 @@
|
|||||||
|
# Fitur ASR / Push-to-Talk (TUI)
|
||||||
|
|
||||||
|
Menambahkan jenis model `asr` ke moop dan fitur push-to-talk di TUI:
|
||||||
|
tekan tombol → bicara → stop → teks hasil transkripsi masuk ke form input
|
||||||
|
(tidak auto-submit).
|
||||||
|
|
||||||
|
## moop
|
||||||
|
- `MODEL_TYPES` += `asr` (self-healing: `INSERT OR IGNORE` saat init, DB lama aman).
|
||||||
|
- `asr_endpoint()` helper ditambahkan.
|
||||||
|
- TUI: tipe 'asr' muncul di Model > Select Model (label "Speech Transcription")
|
||||||
|
dan di Manage Sets (add type / set default).
|
||||||
|
|
||||||
|
## Konfigurasi (config.yaml, blok `asr:`)
|
||||||
|
asr:
|
||||||
|
record_dir : "~/.config/hendrik/recordings" # lokasi file rekaman
|
||||||
|
record_key : f9 # f1..f12 / ctrl-<huruf>
|
||||||
|
record_backend : pulse # pulse | alsa | oss | windows (dshow)
|
||||||
|
device : default
|
||||||
|
windows_mic : "" # backend=windows: nama device dshow
|
||||||
|
format : wav # wav | ogg(libopus)
|
||||||
|
sample_rate : 16000
|
||||||
|
channels : 1
|
||||||
|
max_seconds : 60 # auto-stop pengaman
|
||||||
|
transcribe_timeout: 60
|
||||||
|
keep_recordings : true
|
||||||
|
|
||||||
|
## Alur push-to-talk (interfaces/tui/talk.py)
|
||||||
|
- `F9` mulai rekam (ffmpeg subprocess, non-blocking). Status bar: `● REC` (merah blink).
|
||||||
|
- `F9` lagi ATAU `Enter` → stop → transkrip (thread) → hasil disisipkan ke form
|
||||||
|
di posisi kursor. Status bar: `TRANSCRIBE`. `Enter` TIDAK submit; submit `Ctrl+Enter`.
|
||||||
|
- `Backspace`/`Esc` → batal (rekaman dibuang / transkrip diabaikan, incl. race saat
|
||||||
|
hasil sudah menunggu).
|
||||||
|
- `Ctrl+C` tetap exit kapan pun; saat exit, rekaman aktif otomatis dibersihkan.
|
||||||
|
- ffmpeg mencapai `max_seconds` → auto stop+transkrip.
|
||||||
|
- Validasi: `asr` model set belum diset → pesan panduan; device bisu (peak RMS
|
||||||
|
dari wav di bawah ambang) → "Rekaman kosong", transkrip dilewati.
|
||||||
|
|
||||||
|
## Transkripsi (tools/asr.py)
|
||||||
|
- `POST {base_url}/audio/transcriptions` multipart (file + model), gaya OpenAI.
|
||||||
|
- Memakai seluruh chain moop `asr` (rotasi model/provider ikut priority), pola
|
||||||
|
sama dengan tools/imagegen.py. HTTP 200 + text kosong = valid ("no speech").
|
||||||
|
- Provider/Model dicontohkan: OpenRouter `qwen/qwen3-asr-1.7b`
|
||||||
|
(modality `audio->transcription`; endpoint `/audio/transcriptions` dikonfirmasi
|
||||||
|
hidup; teruji HTTP 200 dengan usage.seconds dari tone 2 detik).
|
||||||
|
|
||||||
|
## Catatan WSL2
|
||||||
|
- Output audio WSLg (sink) bekerja; input mic (RDPSource) TIDAK tersambung pada
|
||||||
|
WSL 3.0.1/WSLg 1.0.79 (handshake RDP hanya accept `rdpsnd`, tanpa kanal audio-in;
|
||||||
|
source RDPSource SUSPENDED, rekam hang). Jalur cadangan yang disiapkan:
|
||||||
|
`record_backend: windows` (ffmpeg.exe + dshow via interop, simpan ke /mnt/c).
|
||||||
302
tools/asr.py
Normal file
302
tools/asr.py
Normal file
@ -0,0 +1,302 @@
|
|||||||
|
# asr.py — Automatic Speech Recognition (STT).
|
||||||
|
#
|
||||||
|
# Dua tanggung jawab:
|
||||||
|
# * start_recording() — spawn `ffmpeg` merekam mic (push-to-talk).
|
||||||
|
# Backend input: pulse | alsa | oss | windows (ffmpeg.exe + dshow).
|
||||||
|
# Stop graceful kirim 'q' ke stdin ffmpeg supaya header file tertutup
|
||||||
|
# rapi (jangan pakai kill/terminate bila masih bisa).
|
||||||
|
# * transcribe_file() — POST {base_url}/audio/transcriptions (multipart,
|
||||||
|
# gaya OpenAI Whisper) ke semua kandidat chain moop tipe 'asr',
|
||||||
|
# rotasi model/provider mengikuti priority (pola tools/imagegen.py).
|
||||||
|
#
|
||||||
|
# Parameter rekam (dir, format, device, dst) sengaja di-pass eksplisit dari
|
||||||
|
# pemanggil (TUI membaca config.asr_*), agar modul ini tidak mengikat config.
|
||||||
|
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import time
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
import requests
|
||||||
|
|
||||||
|
from lib import moop
|
||||||
|
|
||||||
|
MIME_BY_EXT = {".wav": "audio/wav", ".ogg": "audio/ogg", ".mp3": "audio/mpeg",
|
||||||
|
".m4a": "audio/mp4", ".flac": "audio/flac", ".opus": "audio/ogg"}
|
||||||
|
|
||||||
|
|
||||||
|
class RecorderError(Exception):
|
||||||
|
"""Gagal menyiapkan/menghentikan proses rekam."""
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------- recorder ---
|
||||||
|
|
||||||
|
def _resolve_ffmpeg(backend, ffmpeg_bin):
|
||||||
|
"""ffmpeg binary untuk backend tertentu + cek ketersediaan."""
|
||||||
|
if backend == "windows":
|
||||||
|
exe = (ffmpeg_bin or "ffmpeg.exe").strip()
|
||||||
|
found = shutil.which(exe)
|
||||||
|
if not found:
|
||||||
|
# Coba lokasi lazim bila PATH sesi belum di-refresh.
|
||||||
|
for guess in ("ffmpeg.exe", "/mnt/c/ffmpeg/bin/ffmpeg.exe"):
|
||||||
|
found = shutil.which(guess)
|
||||||
|
if found:
|
||||||
|
break
|
||||||
|
if not found:
|
||||||
|
import glob
|
||||||
|
hits = glob.glob("/mnt/c/Program Files*/ffmpeg*/bin/ffmpeg.exe")
|
||||||
|
found = hits[0] if hits else None
|
||||||
|
if not found:
|
||||||
|
raise RecorderError(
|
||||||
|
"ffmpeg.exe (Windows) tidak ditemukan. Install: winget install Gyan.FFmpeg"
|
||||||
|
)
|
||||||
|
return found
|
||||||
|
found = shutil.which(ffmpeg_bin or "ffmpeg")
|
||||||
|
if not found:
|
||||||
|
raise RecorderError("ffmpeg tidak ditemukan di PATH. Install ffmpeg untuk rekam mic.")
|
||||||
|
return found
|
||||||
|
|
||||||
|
|
||||||
|
def _input_args(backend, device, windows_mic):
|
||||||
|
"""Argument ffmpeg untuk sumber audio per backend."""
|
||||||
|
if backend == "windows":
|
||||||
|
if not windows_mic:
|
||||||
|
raise RecorderError(
|
||||||
|
"asr.windows_mic belum diisi (nama device dshow, contoh: "
|
||||||
|
"\"Microphone Array (Realtek(R) Audio)\")."
|
||||||
|
)
|
||||||
|
return ["-f", "dshow", "-i", f"audio={windows_mic}"]
|
||||||
|
if backend not in ("pulse", "alsa", "oss"):
|
||||||
|
raise RecorderError(f"Backend rekam tidak dikenal: '{backend}' (pulse|alsa|oss|windows)")
|
||||||
|
dev = device or "default"
|
||||||
|
if backend == "oss" and dev == "default":
|
||||||
|
dev = "/dev/dsp"
|
||||||
|
return ["-f", backend, "-i", dev]
|
||||||
|
|
||||||
|
|
||||||
|
def _encode_args(fmt, sample_rate, channels, opus_bitrate):
|
||||||
|
"""Argument pemrosesan/encoding sesuai format output."""
|
||||||
|
base = ["-ac", str(channels), "-ar", str(sample_rate)]
|
||||||
|
if fmt == "ogg":
|
||||||
|
return base + ["-c:a", "libopus", "-b:a", f"{opus_bitrate}k"]
|
||||||
|
return base + ["-c:a", "pcm_s16le"]
|
||||||
|
|
||||||
|
|
||||||
|
def build_record_command(ffmpeg_bin, backend, device, windows_mic, fmt,
|
||||||
|
sample_rate, channels, max_seconds, out_path, opus_bitrate=24):
|
||||||
|
cmd = [ffmpeg_bin, "-hide_banner", "-loglevel", "error"]
|
||||||
|
cmd += _input_args(backend, device, windows_mic)
|
||||||
|
cmd += _encode_args(fmt, sample_rate, channels, opus_bitrate)
|
||||||
|
cmd += ["-t", str(max_seconds), "-y", out_path]
|
||||||
|
return cmd
|
||||||
|
|
||||||
|
|
||||||
|
class Recorder:
|
||||||
|
"""Handle satu proses rekaman ffmpeg."""
|
||||||
|
|
||||||
|
def __init__(self, proc, path, started, min_bytes=1024):
|
||||||
|
self.proc = proc
|
||||||
|
self.path = path
|
||||||
|
self.started = started
|
||||||
|
self.min_bytes = min_bytes # ambang "rekaman kosong" (header saja)
|
||||||
|
self._closed = False
|
||||||
|
|
||||||
|
def elapsed(self):
|
||||||
|
return time.time() - self.started
|
||||||
|
|
||||||
|
def alive(self):
|
||||||
|
return self.proc.poll() is None
|
||||||
|
|
||||||
|
def _graceful_stop(self, timeout=3.0):
|
||||||
|
"""Kirim 'q' ke stdin (ffmpeg menutup file dengan benar)."""
|
||||||
|
if self._closed:
|
||||||
|
return
|
||||||
|
self._closed = True
|
||||||
|
if self.proc.poll() is not None:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
if self.proc.stdin:
|
||||||
|
self.proc.stdin.write(b"q\n")
|
||||||
|
self.proc.stdin.flush()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
self.proc.wait(timeout=timeout)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
self.proc.terminate()
|
||||||
|
try:
|
||||||
|
self.proc.wait(timeout=2)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
self.proc.kill()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _peak_amplitude(path, max_bytes=512 * 1024):
|
||||||
|
"""Peak |sample| dari wav PCM16 (stdlib wave). None bila tidak bisa
|
||||||
|
dibaca/bukan wav (mis. ogg) → pemanggil menganggapnya 'tidak diketahui'."""
|
||||||
|
try:
|
||||||
|
import wave
|
||||||
|
with wave.open(path, "rb") as w:
|
||||||
|
if w.getsampwidth() != 2:
|
||||||
|
return None
|
||||||
|
raw = w.readframes(max_bytes // 2)
|
||||||
|
if not raw:
|
||||||
|
return None
|
||||||
|
import array
|
||||||
|
a = array.array("h")
|
||||||
|
a.frombytes(raw[: len(raw) & ~1])
|
||||||
|
return max((abs(s) for s in a), default=0)
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
def stop(self, min_seconds=0.3, silence_peak=128):
|
||||||
|
"""Hentikan rekaman. Returns (path, duration). RecorderError bila file
|
||||||
|
hilang, terlalu pendek, atau kosong (device hidup tapi tanpa sinyal)."""
|
||||||
|
duration = max(self.elapsed(), 0.0)
|
||||||
|
self._graceful_stop()
|
||||||
|
if not os.path.isfile(self.path):
|
||||||
|
raise RecorderError("File rekaman tidak terbentuk (mic/perangkat tidak tersedia?).")
|
||||||
|
size = os.path.getsize(self.path)
|
||||||
|
peak = self._peak_amplitude(self.path) if self.path.endswith(".wav") else None
|
||||||
|
too_short = duration < min_seconds or size <= self.min_bytes
|
||||||
|
silent = peak is not None and peak < silence_peak
|
||||||
|
if too_short or silent:
|
||||||
|
try:
|
||||||
|
os.remove(self.path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
if silent:
|
||||||
|
raise RecorderError(
|
||||||
|
"Rekaman kosong (tidak ada sinyal mic). "
|
||||||
|
"Cek device input / izin mikrofon."
|
||||||
|
)
|
||||||
|
raise RecorderError("Rekaman terlalu pendek, dibatalkan.")
|
||||||
|
return self.path, duration
|
||||||
|
|
||||||
|
def cancel(self):
|
||||||
|
"""Batalkan: bunuh proses, buang file."""
|
||||||
|
if self.alive():
|
||||||
|
try:
|
||||||
|
self.proc.terminate()
|
||||||
|
self.proc.wait(timeout=2)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
self.proc.kill()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
self._closed = True
|
||||||
|
try:
|
||||||
|
if os.path.isfile(self.path):
|
||||||
|
os.remove(self.path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def start_recording(record_dir, backend="pulse", device="default", fmt="wav",
|
||||||
|
sample_rate=16000, channels=1, max_seconds=60,
|
||||||
|
ffmpeg_bin=None, windows_mic=None, opus_bitrate=24):
|
||||||
|
"""Mulai rekam mic (non-blocking). Returns Recorder."""
|
||||||
|
fmt = (fmt or "wav").strip().lower()
|
||||||
|
if fmt not in ("wav", "ogg"):
|
||||||
|
raise RecorderError(f"Format rekam tidak dikenal: '{fmt}' (wav|ogg)")
|
||||||
|
backend = (backend or "pulse").strip().lower()
|
||||||
|
|
||||||
|
exe = _resolve_ffmpeg(backend, ffmpeg_bin)
|
||||||
|
|
||||||
|
record_dir = os.path.expanduser(record_dir)
|
||||||
|
try:
|
||||||
|
os.makedirs(record_dir, exist_ok=True)
|
||||||
|
except OSError as e:
|
||||||
|
raise RecorderError(f"Gagal membuat folder rekam '{record_dir}': {e}")
|
||||||
|
|
||||||
|
out_path = os.path.join(
|
||||||
|
record_dir, f"talk_{datetime.now().strftime('%Y%m%d_%H%M%S')}.{fmt}"
|
||||||
|
)
|
||||||
|
cmd = build_record_command(
|
||||||
|
exe, backend, device, windows_mic, fmt,
|
||||||
|
sample_rate, channels, max_seconds, out_path, opus_bitrate,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
proc = subprocess.Popen(
|
||||||
|
cmd,
|
||||||
|
stdin=subprocess.PIPE, # untuk 'q' saat stop
|
||||||
|
stdout=subprocess.DEVNULL,
|
||||||
|
stderr=subprocess.DEVNULL, # hindari pipe penuh memblok ffmpeg
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
raise RecorderError(f"Gagal menjalankan ffmpeg: {e}")
|
||||||
|
|
||||||
|
# Beri kesempatan gagal instan (device tidak ada/izin ditolak) supaya
|
||||||
|
# bisa langsung dilaporkan, bukan setelah user menekan stop.
|
||||||
|
time.sleep(0.35)
|
||||||
|
if proc.poll() is not None:
|
||||||
|
raise RecorderError(
|
||||||
|
f"ffmpeg langsung exit (kode {proc.returncode}) — device input "
|
||||||
|
f"'{backend}:{device or 'default'}' tidak tersedia/tertunda."
|
||||||
|
)
|
||||||
|
return Recorder(proc, out_path, time.time())
|
||||||
|
|
||||||
|
|
||||||
|
# -------------------------------------------------------------- transcribe ---
|
||||||
|
|
||||||
|
def transcribe_url(base_url):
|
||||||
|
base = (base_url or "").rstrip("/")
|
||||||
|
if base.endswith("/audio/transcriptions"):
|
||||||
|
return base
|
||||||
|
return base + "/audio/transcriptions"
|
||||||
|
|
||||||
|
|
||||||
|
def transcribe_file(audio_path, timeout=60):
|
||||||
|
"""Transkrip satu file audio memakai default chain moop 'asr'.
|
||||||
|
|
||||||
|
Returns dict: {"ok": True, "text": ..., "provider":..., "model":...}
|
||||||
|
atau {"ok": False, "error": "..."}."""
|
||||||
|
if not audio_path or not os.path.isfile(audio_path):
|
||||||
|
return {"ok": False, "error": f"File audio tidak ditemukan: {audio_path}"}
|
||||||
|
|
||||||
|
chain = moop.default_chain(None, "asr")
|
||||||
|
if not chain:
|
||||||
|
return {"ok": False, "error":
|
||||||
|
"Belum ada model set untuk tipe 'asr'. "
|
||||||
|
"Atur via Model > Select Model → Speech Transcription (Model Option)."}
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(audio_path, "rb") as f:
|
||||||
|
data = f.read()
|
||||||
|
except OSError as e:
|
||||||
|
return {"ok": False, "error": f"Gagal membaca file: {e}"}
|
||||||
|
|
||||||
|
ext = os.path.splitext(audio_path)[1].lower()
|
||||||
|
mime = MIME_BY_EXT.get(ext, "application/octet-stream")
|
||||||
|
|
||||||
|
last_err = ""
|
||||||
|
for cand in chain:
|
||||||
|
url = transcribe_url(cand["base_url"])
|
||||||
|
headers = {}
|
||||||
|
if cand.get("api_key"):
|
||||||
|
headers["Authorization"] = f"Bearer {cand['api_key']}"
|
||||||
|
try:
|
||||||
|
resp = requests.post(
|
||||||
|
url,
|
||||||
|
headers=headers,
|
||||||
|
files={"file": (os.path.basename(audio_path), data, mime)},
|
||||||
|
data={"model": cand["model"]},
|
||||||
|
timeout=timeout,
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
last_err = f"{cand['provider']}: {e}"
|
||||||
|
continue
|
||||||
|
|
||||||
|
if resp.status_code == 200:
|
||||||
|
try:
|
||||||
|
text = (resp.json() or {}).get("text", "")
|
||||||
|
except ValueError:
|
||||||
|
text = (resp.text or "").strip()
|
||||||
|
# HTTP 200 + text kosong = valid (mis. audio tanpa ucapan) → jangan
|
||||||
|
# dianggap kegagalan; pemanggil yang memutuskan tampil 'no speech'.
|
||||||
|
return {"ok": True, "text": (text or "").strip(),
|
||||||
|
"provider": cand["provider"], "model": cand["model"]}
|
||||||
|
body = (resp.text or "")[:300].replace("\n", " ")
|
||||||
|
last_err = f"{cand['provider']} [{cand['model']}]: HTTP {resp.status_code} {body}"
|
||||||
|
|
||||||
|
return {"ok": False, "error": f"Semua kandidat model ASR gagal. {last_err}"}
|
||||||
Loading…
Reference in New Issue
Block a user