From 896a3d4c17073e7eeca4cbc17cd4d12e92358ff1 Mon Sep 17 00:00:00 2001 From: Dita Aji Pratama Date: Mon, 21 Sep 2026 10:52:53 +0700 Subject: [PATCH] Add LLM and Embedding runner too --- README.md | 34 ++++++++++++++++++- config/embed.py | 12 +++++++ config/llm.py | 19 +++++++++++ core/embed_room.py | 5 +++ core/llm_room.py | 6 ++++ embed_runner.py | 37 ++++++++++++++++++++ llm_runner.py | 84 ++++++++++++++++++++++++++++++++++++++++++++++ requirements.txt | 3 +- usage-embed.sh | 19 +++++++++++ usage-llm.sh | 12 +++++++ 10 files changed, 229 insertions(+), 2 deletions(-) create mode 100644 config/embed.py create mode 100644 config/llm.py create mode 100644 core/embed_room.py create mode 100644 core/llm_room.py create mode 100644 embed_runner.py create mode 100644 llm_runner.py create mode 100755 usage-embed.sh create mode 100755 usage-llm.sh diff --git a/README.md b/README.md index 5063b56..46bbbdb 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # STT Runner -Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, plus Text-to-Speech with ZipVoice (zero-shot voice cloning). +Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, Text-to-Speech with ZipVoice (zero-shot voice cloning), and LLM/embedding inference with llama.cpp (Granite-4.2 + nomic-embed-text-v1.5). ## Installation @@ -25,6 +25,20 @@ python tts_runner.py [--output=out.wav] [--ref-audio=ref.wav] [--ref-text="..."] Output defaults to `output.wav`. The reference audio/text (voice to clone) is set in `config/tts.py` and can be overridden per-run with `--ref-audio` / `--ref-text` (the text must match the audio exactly). +### LLM Chat (llama.cpp) + +```bash +python llm_runner.py "What is the capital of France?" # single prompt +python llm_runner.py # interactive chat +``` + +### Embedding (llama.cpp) + +```bash +python embed_runner.py "text to embed" "another text" +python embed_runner.py "What is TSNE?" --prefix "search_query: " +``` + ## Configuration Model paths and inference parameters are hardcoded in `config/`: @@ -32,6 +46,8 @@ Model paths and inference parameters are hardcoded in `config/`: - `config/model.py` — model paths (conv_frontend, encoder, decoder, tokenizer under `models/`) - `config/asr.py` — inference params: `LANGUAGE`, `HOTWORDS`, `NUM_THREADS`, `PROVIDER`, `SAMPLE_RATE`, `FEATURE_DIM`, `MAX_TOTAL_LEN`, `MAX_NEW_TOKENS` - `config/tts.py` — TTS model paths, `REFERENCE_AUDIO`, `REFERENCE_TEXT`, `OUTPUT_FILE`, `NUM_THREADS`, `PROVIDER`, `NUM_STEPS` +- `config/llm.py` — Granite-4.2 model path, `N_CTX`, `N_THREADS`, `N_GPU_LAYERS`, `MAX_TOKENS`, `TEMPERATURE`, `TOP_P`, `TOP_K`, `SYSTEM_PROMPT`, `CHAT_TEMPLATE` +- `config/embed.py` — nomic-embed-text-v1.5 model path, `N_CTX`, `N_THREADS`, `PREFIX_QUERY`, `PREFIX_DOCUMENT`, `DEFAULT_PREFIX` `LANGUAGE` defaults to `""` (all languages / auto-detect). Passing `--language` on the CLI overrides it. @@ -56,4 +72,20 @@ wget -qO- https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models/she | tar xjf - -C models/zipvoice --strip-components=1 wget -O models/zipvoice/vocos_24khz.onnx \ https://github.com/k2-fsa/sherpa-onnx/releases/download/vocoder-models/vocos_24khz.onnx +``` + +## Download Model (Granite-4.2 8B Q4_K_M) + +```bash +mkdir -p models/granite-4.2 +wget -O models/granite-4.2/granite-4.2-8b-Q4_K_M.gguf \ + "https://huggingface.co/ibm-granite/granite-4.2-8b-GGUF/resolve/main/granite-4.2-8b-Q4_K_M.gguf" +``` + +## Download Model (nomic-embed-text-v1.5) + +```bash +mkdir -p models/nomic-embed-text-v1.5 +wget -O models/nomic-embed-text-v1.5/nomic-embed-text-v1.5.Q4_K_M.gguf \ + "https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf" ``` \ No newline at end of file diff --git a/config/embed.py b/config/embed.py new file mode 100644 index 0000000..9d5c5a5 --- /dev/null +++ b/config/embed.py @@ -0,0 +1,12 @@ +from config.model import MODEL_DIR + +MODEL = MODEL_DIR / "nomic-embed-text-v1.5" / "nomic-embed-text-v1.5.Q4_K_M.gguf" + +N_THREADS = 10 +N_CTX = 2048 # nomic GGUF defaults to 2048; 8192 needs YaRN scaling + +# nomic-embed-text needs a task instruction prefix per text. +# Use a query prefix for search/user input, a document prefix for stored text. +PREFIX_QUERY = "search_query: " +PREFIX_DOCUMENT = "search_document: " +DEFAULT_PREFIX = PREFIX_QUERY \ No newline at end of file diff --git a/config/llm.py b/config/llm.py new file mode 100644 index 0000000..cb47c55 --- /dev/null +++ b/config/llm.py @@ -0,0 +1,19 @@ +from config.model import MODEL_DIR + +MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf" + +N_THREADS = 10 +N_CTX = 8192 # Context window (tokens) +N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU +MAX_TOKENS = 1024 # Max tokens per reply +STREAM = True # Stream tokens as they are generated + +TEMPERATURE = 0.7 +TOP_P = 0.9 +TOP_K = 40 + +# System prompt for the chat session. +SYSTEM_PROMPT = "You are a helpful, concise assistant." + +# Default chat template. Leave empty to use the template embedded in the GGUF. +CHAT_TEMPLATE = "" \ No newline at end of file diff --git a/core/embed_room.py b/core/embed_room.py new file mode 100644 index 0000000..3e72d22 --- /dev/null +++ b/core/embed_room.py @@ -0,0 +1,5 @@ +import argparse + +parser = argparse.ArgumentParser() +parser.add_argument("texts", nargs="+", help="Text(s) to embed") +parser.add_argument("--prefix", type=str, default=None, help="Task prefix, overrides config DEFAULT_PREFIX") \ No newline at end of file diff --git a/core/llm_room.py b/core/llm_room.py new file mode 100644 index 0000000..2f1fc78 --- /dev/null +++ b/core/llm_room.py @@ -0,0 +1,6 @@ +import argparse + +parser = argparse.ArgumentParser() +parser.add_argument("prompt", nargs="?", default=None, help="Single prompt to answer, then exit. Omit for interactive chat") +parser.add_argument("--system", type=str, default=None, help="Override system prompt from config") +parser.add_argument("--max-tokens", type=int, default=None, help="Override config MAX_TOKENS") \ No newline at end of file diff --git a/embed_runner.py b/embed_runner.py new file mode 100644 index 0000000..96d424b --- /dev/null +++ b/embed_runner.py @@ -0,0 +1,37 @@ +import sys +import json +from core import embed_room as ap +from config.embed import MODEL, N_THREADS, N_CTX, DEFAULT_PREFIX +from llama_cpp import Llama + +def embed_run(args): + + if not MODEL.is_file(): + print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr) + return + + prefix = args.prefix if args.prefix is not None else DEFAULT_PREFIX + + print("Loading model...") + llm = Llama( + model_path = str(MODEL), + n_ctx = N_CTX, + n_threads = N_THREADS, + embedding = True, + verbose = False, + ) + print("Model ready!") + + inputs = [prefix + t for t in args.texts] + + resp = llm.create_embedding(inputs) + vectors = [d["embedding"] for d in resp["data"]] + + out = [ + {"text": text, "dim": len(vec), "embedding": vec} + for text, vec in zip(args.texts, vectors) + ] + print(json.dumps(out, ensure_ascii=False, indent=2)) + +if __name__ == "__main__": + embed_run( ap.parser.parse_args() ) \ No newline at end of file diff --git a/llm_runner.py b/llm_runner.py new file mode 100644 index 0000000..017cd3c --- /dev/null +++ b/llm_runner.py @@ -0,0 +1,84 @@ +import sys +from core import llm_room as ap +from config.llm import ( + MODEL, N_THREADS, N_CTX, N_GPU_LAYERS, MAX_TOKENS, STREAM, + TEMPERATURE, TOP_P, TOP_K, SYSTEM_PROMPT, CHAT_TEMPLATE, +) +from llama_cpp import Llama + +def load_llm(): + print("Loading model...") + llm = Llama( + model_path = str(MODEL), + n_ctx = N_CTX, + n_threads = N_THREADS, + n_gpu_layers = N_GPU_LAYERS, + chat_format = CHAT_TEMPLATE or None, + verbose = False, + ) + print("Model ready!") + return llm + +def chat_once(llm, messages, max_tokens): + if STREAM: + text = "" + for chunk in llm.create_chat_completion( + messages, + stream = True, + temperature = TEMPERATURE, + top_p = TOP_P, + top_k = TOP_K, + max_tokens = max_tokens, + ): + delta = chunk["choices"][0]["delta"].get("content") + if delta: + print(delta, end="", flush=True) + text += delta + print() + return text + + resp = llm.create_chat_completion( + messages, + stream = False, + temperature = TEMPERATURE, + top_p = TOP_P, + top_k = TOP_K, + max_tokens = max_tokens, + ) + return resp["choices"][0]["message"]["content"] + +def llm_run(args): + + if not MODEL.is_file(): + print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr) + return + + llm = load_llm() + system = args.system if args.system is not None else SYSTEM_PROMPT + max_tokens = args.max_tokens if args.max_tokens is not None else MAX_TOKENS + + if args.prompt is not None: + messages = [{"role": "system", "content": system}] if system else [] + messages.append({"role": "user", "content": args.prompt}) + chat_once(llm, messages, max_tokens) + return + + messages = [{"role": "system", "content": system}] if system else [] + print("Interactive chat. Type 'exit' or 'quit' to leave.") + try: + while True: + user = input("\nYou : ").strip() + if user.lower() in ("exit", "quit", "q"): + print("Bye.") + break + if not user: + continue + messages.append({"role": "user", "content": user}) + print("Assistant: ", end="", flush=True) + reply = chat_once(llm, messages, max_tokens) + messages.append({"role": "assistant", "content": reply}) + except (KeyboardInterrupt, EOFError): + print("\nBye.") + +if __name__ == "__main__": + llm_run( ap.parser.parse_args() ) \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index 016ccbc..e37cdc4 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,3 @@ sherpa-onnx -soundfile \ No newline at end of file +soundfile +llama-cpp-python \ No newline at end of file diff --git a/usage-embed.sh b/usage-embed.sh new file mode 100755 index 0000000..2aa6aa0 --- /dev/null +++ b/usage-embed.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +set -euo pipefail + +BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}" + +if [ ! -x "$PYTHON" ]; then + echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2 + exit 1 +fi + +if [ $# -lt 1 ]; then + echo "Usage: $0 [--prefix ]" >&2 + echo " Embed text(s) with nomic-embed-text-v1.5 (llama.cpp)." >&2 + echo " Default prefix and model path are set in config/embed.py." >&2 + exit 1 +fi + +"$PYTHON" "$BASE_DIR/embed_runner.py" "$@" \ No newline at end of file diff --git a/usage-llm.sh b/usage-llm.sh new file mode 100755 index 0000000..1a1f6e6 --- /dev/null +++ b/usage-llm.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +set -euo pipefail + +BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}" + +if [ ! -x "$PYTHON" ]; then + echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2 + exit 1 +fi + +"$PYTHON" "$BASE_DIR/llm_runner.py" "$@" \ No newline at end of file