hendrik/tools/asr.py
2026-10-06 16:29:31 +07:00

303 lines
11 KiB
Python

# asr.py — Automatic Speech Recognition (STT).
#
# Dua tanggung jawab:
# * start_recording() — spawn `ffmpeg` merekam mic (push-to-talk).
# Backend input: pulse | alsa | oss | windows (ffmpeg.exe + dshow).
# Stop graceful kirim 'q' ke stdin ffmpeg supaya header file tertutup
# rapi (jangan pakai kill/terminate bila masih bisa).
# * transcribe_file() — POST {base_url}/audio/transcriptions (multipart,
# gaya OpenAI Whisper) ke semua kandidat chain moop tipe 'asr',
# rotasi model/provider mengikuti priority (pola tools/imagegen.py).
#
# Parameter rekam (dir, format, device, dst) sengaja di-pass eksplisit dari
# pemanggil (TUI membaca config.asr_*), agar modul ini tidak mengikat config.
import os
import shutil
import subprocess
import time
from datetime import datetime
import requests
from lib import moop
MIME_BY_EXT = {".wav": "audio/wav", ".ogg": "audio/ogg", ".mp3": "audio/mpeg",
".m4a": "audio/mp4", ".flac": "audio/flac", ".opus": "audio/ogg"}
class RecorderError(Exception):
"""Gagal menyiapkan/menghentikan proses rekam."""
# ---------------------------------------------------------------- recorder ---
def _resolve_ffmpeg(backend, ffmpeg_bin):
"""ffmpeg binary untuk backend tertentu + cek ketersediaan."""
if backend == "windows":
exe = (ffmpeg_bin or "ffmpeg.exe").strip()
found = shutil.which(exe)
if not found:
# Coba lokasi lazim bila PATH sesi belum di-refresh.
for guess in ("ffmpeg.exe", "/mnt/c/ffmpeg/bin/ffmpeg.exe"):
found = shutil.which(guess)
if found:
break
if not found:
import glob
hits = glob.glob("/mnt/c/Program Files*/ffmpeg*/bin/ffmpeg.exe")
found = hits[0] if hits else None
if not found:
raise RecorderError(
"ffmpeg.exe (Windows) tidak ditemukan. Install: winget install Gyan.FFmpeg"
)
return found
found = shutil.which(ffmpeg_bin or "ffmpeg")
if not found:
raise RecorderError("ffmpeg tidak ditemukan di PATH. Install ffmpeg untuk rekam mic.")
return found
def _input_args(backend, device, windows_mic):
"""Argument ffmpeg untuk sumber audio per backend."""
if backend == "windows":
if not windows_mic:
raise RecorderError(
"asr.windows_mic belum diisi (nama device dshow, contoh: "
"\"Microphone Array (Realtek(R) Audio)\")."
)
return ["-f", "dshow", "-i", f"audio={windows_mic}"]
if backend not in ("pulse", "alsa", "oss"):
raise RecorderError(f"Backend rekam tidak dikenal: '{backend}' (pulse|alsa|oss|windows)")
dev = device or "default"
if backend == "oss" and dev == "default":
dev = "/dev/dsp"
return ["-f", backend, "-i", dev]
def _encode_args(fmt, sample_rate, channels, opus_bitrate):
"""Argument pemrosesan/encoding sesuai format output."""
base = ["-ac", str(channels), "-ar", str(sample_rate)]
if fmt == "ogg":
return base + ["-c:a", "libopus", "-b:a", f"{opus_bitrate}k"]
return base + ["-c:a", "pcm_s16le"]
def build_record_command(ffmpeg_bin, backend, device, windows_mic, fmt,
sample_rate, channels, max_seconds, out_path, opus_bitrate=24):
cmd = [ffmpeg_bin, "-hide_banner", "-loglevel", "error"]
cmd += _input_args(backend, device, windows_mic)
cmd += _encode_args(fmt, sample_rate, channels, opus_bitrate)
cmd += ["-t", str(max_seconds), "-y", out_path]
return cmd
class Recorder:
"""Handle satu proses rekaman ffmpeg."""
def __init__(self, proc, path, started, min_bytes=1024):
self.proc = proc
self.path = path
self.started = started
self.min_bytes = min_bytes # ambang "rekaman kosong" (header saja)
self._closed = False
def elapsed(self):
return time.time() - self.started
def alive(self):
return self.proc.poll() is None
def _graceful_stop(self, timeout=3.0):
"""Kirim 'q' ke stdin (ffmpeg menutup file dengan benar)."""
if self._closed:
return
self._closed = True
if self.proc.poll() is not None:
return
try:
if self.proc.stdin:
self.proc.stdin.write(b"q\n")
self.proc.stdin.flush()
except Exception:
pass
try:
self.proc.wait(timeout=timeout)
except subprocess.TimeoutExpired:
self.proc.terminate()
try:
self.proc.wait(timeout=2)
except subprocess.TimeoutExpired:
self.proc.kill()
@staticmethod
def _peak_amplitude(path, max_bytes=512 * 1024):
"""Peak |sample| dari wav PCM16 (stdlib wave). None bila tidak bisa
dibaca/bukan wav (mis. ogg) → pemanggil menganggapnya 'tidak diketahui'."""
try:
import wave
with wave.open(path, "rb") as w:
if w.getsampwidth() != 2:
return None
raw = w.readframes(max_bytes // 2)
if not raw:
return None
import array
a = array.array("h")
a.frombytes(raw[: len(raw) & ~1])
return max((abs(s) for s in a), default=0)
except Exception:
return None
def stop(self, min_seconds=0.3, silence_peak=128):
"""Hentikan rekaman. Returns (path, duration). RecorderError bila file
hilang, terlalu pendek, atau kosong (device hidup tapi tanpa sinyal)."""
duration = max(self.elapsed(), 0.0)
self._graceful_stop()
if not os.path.isfile(self.path):
raise RecorderError("File rekaman tidak terbentuk (mic/perangkat tidak tersedia?).")
size = os.path.getsize(self.path)
peak = self._peak_amplitude(self.path) if self.path.endswith(".wav") else None
too_short = duration < min_seconds or size <= self.min_bytes
silent = peak is not None and peak < silence_peak
if too_short or silent:
try:
os.remove(self.path)
except OSError:
pass
if silent:
raise RecorderError(
"Rekaman kosong (tidak ada sinyal mic). "
"Cek device input / izin mikrofon."
)
raise RecorderError("Rekaman terlalu pendek, dibatalkan.")
return self.path, duration
def cancel(self):
"""Batalkan: bunuh proses, buang file."""
if self.alive():
try:
self.proc.terminate()
self.proc.wait(timeout=2)
except Exception:
try:
self.proc.kill()
except Exception:
pass
self._closed = True
try:
if os.path.isfile(self.path):
os.remove(self.path)
except OSError:
pass
def start_recording(record_dir, backend="pulse", device="default", fmt="wav",
sample_rate=16000, channels=1, max_seconds=60,
ffmpeg_bin=None, windows_mic=None, opus_bitrate=24):
"""Mulai rekam mic (non-blocking). Returns Recorder."""
fmt = (fmt or "wav").strip().lower()
if fmt not in ("wav", "ogg"):
raise RecorderError(f"Format rekam tidak dikenal: '{fmt}' (wav|ogg)")
backend = (backend or "pulse").strip().lower()
exe = _resolve_ffmpeg(backend, ffmpeg_bin)
record_dir = os.path.expanduser(record_dir)
try:
os.makedirs(record_dir, exist_ok=True)
except OSError as e:
raise RecorderError(f"Gagal membuat folder rekam '{record_dir}': {e}")
out_path = os.path.join(
record_dir, f"talk_{datetime.now().strftime('%Y%m%d_%H%M%S')}.{fmt}"
)
cmd = build_record_command(
exe, backend, device, windows_mic, fmt,
sample_rate, channels, max_seconds, out_path, opus_bitrate,
)
try:
proc = subprocess.Popen(
cmd,
stdin=subprocess.PIPE, # untuk 'q' saat stop
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL, # hindari pipe penuh memblok ffmpeg
)
except Exception as e:
raise RecorderError(f"Gagal menjalankan ffmpeg: {e}")
# Beri kesempatan gagal instan (device tidak ada/izin ditolak) supaya
# bisa langsung dilaporkan, bukan setelah user menekan stop.
time.sleep(0.35)
if proc.poll() is not None:
raise RecorderError(
f"ffmpeg langsung exit (kode {proc.returncode}) — device input "
f"'{backend}:{device or 'default'}' tidak tersedia/tertunda."
)
return Recorder(proc, out_path, time.time())
# -------------------------------------------------------------- transcribe ---
def transcribe_url(base_url):
base = (base_url or "").rstrip("/")
if base.endswith("/audio/transcriptions"):
return base
return base + "/audio/transcriptions"
def transcribe_file(audio_path, timeout=60):
"""Transkrip satu file audio memakai default chain moop 'asr'.
Returns dict: {"ok": True, "text": ..., "provider":..., "model":...}
atau {"ok": False, "error": "..."}."""
if not audio_path or not os.path.isfile(audio_path):
return {"ok": False, "error": f"File audio tidak ditemukan: {audio_path}"}
chain = moop.default_chain(None, "asr")
if not chain:
return {"ok": False, "error":
"Belum ada model set untuk tipe 'asr'. "
"Atur via Model > Select Model → Speech Transcription (Model Option)."}
try:
with open(audio_path, "rb") as f:
data = f.read()
except OSError as e:
return {"ok": False, "error": f"Gagal membaca file: {e}"}
ext = os.path.splitext(audio_path)[1].lower()
mime = MIME_BY_EXT.get(ext, "application/octet-stream")
last_err = ""
for cand in chain:
url = transcribe_url(cand["base_url"])
headers = {}
if cand.get("api_key"):
headers["Authorization"] = f"Bearer {cand['api_key']}"
try:
resp = requests.post(
url,
headers=headers,
files={"file": (os.path.basename(audio_path), data, mime)},
data={"model": cand["model"]},
timeout=timeout,
)
except Exception as e:
last_err = f"{cand['provider']}: {e}"
continue
if resp.status_code == 200:
try:
text = (resp.json() or {}).get("text", "")
except ValueError:
text = (resp.text or "").strip()
# HTTP 200 + text kosong = valid (mis. audio tanpa ucapan) → jangan
# dianggap kegagalan; pemanggil yang memutuskan tampil 'no speech'.
return {"ok": True, "text": (text or "").strip(),
"provider": cand["provider"], "model": cand["model"]}
body = (resp.text or "")[:300].replace("\n", " ")
last_err = f"{cand['provider']} [{cand['model']}]: HTTP {resp.status_code} {body}"
return {"ok": False, "error": f"Semua kandidat model ASR gagal. {last_err}"}