# asr.py — Automatic Speech Recognition (STT). # # Dua tanggung jawab: # * start_recording() — spawn `ffmpeg` merekam mic (push-to-talk). # Backend input: pulse | alsa | oss | windows (ffmpeg.exe + dshow). # Stop graceful kirim 'q' ke stdin ffmpeg supaya header file tertutup # rapi (jangan pakai kill/terminate bila masih bisa). # * transcribe_file() — POST {base_url}/audio/transcriptions (multipart, # gaya OpenAI Whisper) ke semua kandidat chain moop tipe 'asr', # rotasi model/provider mengikuti priority (pola tools/imagegen.py). # # Parameter rekam (dir, format, device, dst) sengaja di-pass eksplisit dari # pemanggil (TUI membaca config.asr_*), agar modul ini tidak mengikat config. import os import shutil import subprocess import time from datetime import datetime import requests from lib import moop MIME_BY_EXT = {".wav": "audio/wav", ".ogg": "audio/ogg", ".mp3": "audio/mpeg", ".m4a": "audio/mp4", ".flac": "audio/flac", ".opus": "audio/ogg"} class RecorderError(Exception): """Gagal menyiapkan/menghentikan proses rekam.""" # ---------------------------------------------------------------- recorder --- def _resolve_ffmpeg(backend, ffmpeg_bin): """ffmpeg binary untuk backend tertentu + cek ketersediaan.""" if backend == "windows": exe = (ffmpeg_bin or "ffmpeg.exe").strip() found = shutil.which(exe) if not found: # Coba lokasi lazim bila PATH sesi belum di-refresh. for guess in ("ffmpeg.exe", "/mnt/c/ffmpeg/bin/ffmpeg.exe"): found = shutil.which(guess) if found: break if not found: import glob hits = glob.glob("/mnt/c/Program Files*/ffmpeg*/bin/ffmpeg.exe") found = hits[0] if hits else None if not found: raise RecorderError( "ffmpeg.exe (Windows) tidak ditemukan. Install: winget install Gyan.FFmpeg" ) return found found = shutil.which(ffmpeg_bin or "ffmpeg") if not found: raise RecorderError("ffmpeg tidak ditemukan di PATH. Install ffmpeg untuk rekam mic.") return found def _input_args(backend, device, windows_mic): """Argument ffmpeg untuk sumber audio per backend.""" if backend == "windows": if not windows_mic: raise RecorderError( "asr.windows_mic belum diisi (nama device dshow, contoh: " "\"Microphone Array (Realtek(R) Audio)\")." ) return ["-f", "dshow", "-i", f"audio={windows_mic}"] if backend not in ("pulse", "alsa", "oss"): raise RecorderError(f"Backend rekam tidak dikenal: '{backend}' (pulse|alsa|oss|windows)") dev = device or "default" if backend == "oss" and dev == "default": dev = "/dev/dsp" return ["-f", backend, "-i", dev] def _encode_args(fmt, sample_rate, channels, opus_bitrate): """Argument pemrosesan/encoding sesuai format output.""" base = ["-ac", str(channels), "-ar", str(sample_rate)] if fmt == "ogg": return base + ["-c:a", "libopus", "-b:a", f"{opus_bitrate}k"] return base + ["-c:a", "pcm_s16le"] def build_record_command(ffmpeg_bin, backend, device, windows_mic, fmt, sample_rate, channels, max_seconds, out_path, opus_bitrate=24): cmd = [ffmpeg_bin, "-hide_banner", "-loglevel", "error"] cmd += _input_args(backend, device, windows_mic) cmd += _encode_args(fmt, sample_rate, channels, opus_bitrate) cmd += ["-t", str(max_seconds), "-y", out_path] return cmd class Recorder: """Handle satu proses rekaman ffmpeg.""" def __init__(self, proc, path, started, min_bytes=1024): self.proc = proc self.path = path self.started = started self.min_bytes = min_bytes # ambang "rekaman kosong" (header saja) self._closed = False def elapsed(self): return time.time() - self.started def alive(self): return self.proc.poll() is None def _graceful_stop(self, timeout=3.0): """Kirim 'q' ke stdin (ffmpeg menutup file dengan benar).""" if self._closed: return self._closed = True if self.proc.poll() is not None: return try: if self.proc.stdin: self.proc.stdin.write(b"q\n") self.proc.stdin.flush() except Exception: pass try: self.proc.wait(timeout=timeout) except subprocess.TimeoutExpired: self.proc.terminate() try: self.proc.wait(timeout=2) except subprocess.TimeoutExpired: self.proc.kill() @staticmethod def _peak_amplitude(path, max_bytes=512 * 1024): """Peak |sample| dari wav PCM16 (stdlib wave). None bila tidak bisa dibaca/bukan wav (mis. ogg) → pemanggil menganggapnya 'tidak diketahui'.""" try: import wave with wave.open(path, "rb") as w: if w.getsampwidth() != 2: return None raw = w.readframes(max_bytes // 2) if not raw: return None import array a = array.array("h") a.frombytes(raw[: len(raw) & ~1]) return max((abs(s) for s in a), default=0) except Exception: return None def stop(self, min_seconds=0.3, silence_peak=128): """Hentikan rekaman. Returns (path, duration). RecorderError bila file hilang, terlalu pendek, atau kosong (device hidup tapi tanpa sinyal).""" duration = max(self.elapsed(), 0.0) self._graceful_stop() if not os.path.isfile(self.path): raise RecorderError("File rekaman tidak terbentuk (mic/perangkat tidak tersedia?).") size = os.path.getsize(self.path) peak = self._peak_amplitude(self.path) if self.path.endswith(".wav") else None too_short = duration < min_seconds or size <= self.min_bytes silent = peak is not None and peak < silence_peak if too_short or silent: try: os.remove(self.path) except OSError: pass if silent: raise RecorderError( "Rekaman kosong (tidak ada sinyal mic). " "Cek device input / izin mikrofon." ) raise RecorderError("Rekaman terlalu pendek, dibatalkan.") return self.path, duration def cancel(self): """Batalkan: bunuh proses, buang file.""" if self.alive(): try: self.proc.terminate() self.proc.wait(timeout=2) except Exception: try: self.proc.kill() except Exception: pass self._closed = True try: if os.path.isfile(self.path): os.remove(self.path) except OSError: pass def start_recording(record_dir, backend="pulse", device="default", fmt="wav", sample_rate=16000, channels=1, max_seconds=60, ffmpeg_bin=None, windows_mic=None, opus_bitrate=24): """Mulai rekam mic (non-blocking). Returns Recorder.""" fmt = (fmt or "wav").strip().lower() if fmt not in ("wav", "ogg"): raise RecorderError(f"Format rekam tidak dikenal: '{fmt}' (wav|ogg)") backend = (backend or "pulse").strip().lower() exe = _resolve_ffmpeg(backend, ffmpeg_bin) record_dir = os.path.expanduser(record_dir) try: os.makedirs(record_dir, exist_ok=True) except OSError as e: raise RecorderError(f"Gagal membuat folder rekam '{record_dir}': {e}") out_path = os.path.join( record_dir, f"talk_{datetime.now().strftime('%Y%m%d_%H%M%S')}.{fmt}" ) cmd = build_record_command( exe, backend, device, windows_mic, fmt, sample_rate, channels, max_seconds, out_path, opus_bitrate, ) try: proc = subprocess.Popen( cmd, stdin=subprocess.PIPE, # untuk 'q' saat stop stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, # hindari pipe penuh memblok ffmpeg ) except Exception as e: raise RecorderError(f"Gagal menjalankan ffmpeg: {e}") # Beri kesempatan gagal instan (device tidak ada/izin ditolak) supaya # bisa langsung dilaporkan, bukan setelah user menekan stop. time.sleep(0.35) if proc.poll() is not None: raise RecorderError( f"ffmpeg langsung exit (kode {proc.returncode}) — device input " f"'{backend}:{device or 'default'}' tidak tersedia/tertunda." ) return Recorder(proc, out_path, time.time()) # -------------------------------------------------------------- transcribe --- def transcribe_url(base_url): base = (base_url or "").rstrip("/") if base.endswith("/audio/transcriptions"): return base return base + "/audio/transcriptions" def transcribe_file(audio_path, timeout=60): """Transkrip satu file audio memakai default chain moop 'asr'. Returns dict: {"ok": True, "text": ..., "provider":..., "model":...} atau {"ok": False, "error": "..."}.""" if not audio_path or not os.path.isfile(audio_path): return {"ok": False, "error": f"File audio tidak ditemukan: {audio_path}"} chain = moop.default_chain(None, "asr") if not chain: return {"ok": False, "error": "Belum ada model set untuk tipe 'asr'. " "Atur via Model > Select Model → Speech Transcription (Model Option)."} try: with open(audio_path, "rb") as f: data = f.read() except OSError as e: return {"ok": False, "error": f"Gagal membaca file: {e}"} ext = os.path.splitext(audio_path)[1].lower() mime = MIME_BY_EXT.get(ext, "application/octet-stream") last_err = "" for cand in chain: url = transcribe_url(cand["base_url"]) headers = {} if cand.get("api_key"): headers["Authorization"] = f"Bearer {cand['api_key']}" try: resp = requests.post( url, headers=headers, files={"file": (os.path.basename(audio_path), data, mime)}, data={"model": cand["model"]}, timeout=timeout, ) except Exception as e: last_err = f"{cand['provider']}: {e}" continue if resp.status_code == 200: try: text = (resp.json() or {}).get("text", "") except ValueError: text = (resp.text or "").strip() # HTTP 200 + text kosong = valid (mis. audio tanpa ucapan) → jangan # dianggap kegagalan; pemanggil yang memutuskan tampil 'no speech'. return {"ok": True, "text": (text or "").strip(), "provider": cand["provider"], "model": cand["model"]} body = (resp.text or "")[:300].replace("\n", " ") last_err = f"{cand['provider']} [{cand['model']}]: HTTP {resp.status_code} {body}" return {"ok": False, "error": f"Semua kandidat model ASR gagal. {last_err}"}