Add LLM and Embedding runner too

This commit is contained in:
Dita Aji Pratama 2026-09-21 10:52:53 +07:00
parent 7d67b4af6c
commit 896a3d4c17
10 changed files with 229 additions and 2 deletions

View File

@ -1,6 +1,6 @@
# STT Runner
Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, plus Text-to-Speech with ZipVoice (zero-shot voice cloning).
Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, Text-to-Speech with ZipVoice (zero-shot voice cloning), and LLM/embedding inference with llama.cpp (Granite-4.2 + nomic-embed-text-v1.5).
## Installation
@ -25,6 +25,20 @@ python tts_runner.py [--output=out.wav] [--ref-audio=ref.wav] [--ref-text="..."]
Output defaults to `output.wav`. The reference audio/text (voice to clone) is set in `config/tts.py` and can be overridden per-run with `--ref-audio` / `--ref-text` (the text must match the audio exactly).
### LLM Chat (llama.cpp)
```bash
python llm_runner.py "What is the capital of France?" # single prompt
python llm_runner.py # interactive chat
```
### Embedding (llama.cpp)
```bash
python embed_runner.py "text to embed" "another text"
python embed_runner.py "What is TSNE?" --prefix "search_query: "
```
## Configuration
Model paths and inference parameters are hardcoded in `config/`:
@ -32,6 +46,8 @@ Model paths and inference parameters are hardcoded in `config/`:
- `config/model.py` — model paths (conv_frontend, encoder, decoder, tokenizer under `models/`)
- `config/asr.py` — inference params: `LANGUAGE`, `HOTWORDS`, `NUM_THREADS`, `PROVIDER`, `SAMPLE_RATE`, `FEATURE_DIM`, `MAX_TOTAL_LEN`, `MAX_NEW_TOKENS`
- `config/tts.py` — TTS model paths, `REFERENCE_AUDIO`, `REFERENCE_TEXT`, `OUTPUT_FILE`, `NUM_THREADS`, `PROVIDER`, `NUM_STEPS`
- `config/llm.py` — Granite-4.2 model path, `N_CTX`, `N_THREADS`, `N_GPU_LAYERS`, `MAX_TOKENS`, `TEMPERATURE`, `TOP_P`, `TOP_K`, `SYSTEM_PROMPT`, `CHAT_TEMPLATE`
- `config/embed.py` — nomic-embed-text-v1.5 model path, `N_CTX`, `N_THREADS`, `PREFIX_QUERY`, `PREFIX_DOCUMENT`, `DEFAULT_PREFIX`
`LANGUAGE` defaults to `""` (all languages / auto-detect). Passing `--language` on the CLI overrides it.
@ -56,4 +72,20 @@ wget -qO- https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models/she
| tar xjf - -C models/zipvoice --strip-components=1
wget -O models/zipvoice/vocos_24khz.onnx \
https://github.com/k2-fsa/sherpa-onnx/releases/download/vocoder-models/vocos_24khz.onnx
```
## Download Model (Granite-4.2 8B Q4_K_M)
```bash
mkdir -p models/granite-4.2
wget -O models/granite-4.2/granite-4.2-8b-Q4_K_M.gguf \
"https://huggingface.co/ibm-granite/granite-4.2-8b-GGUF/resolve/main/granite-4.2-8b-Q4_K_M.gguf"
```
## Download Model (nomic-embed-text-v1.5)
```bash
mkdir -p models/nomic-embed-text-v1.5
wget -O models/nomic-embed-text-v1.5/nomic-embed-text-v1.5.Q4_K_M.gguf \
"https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
```

12
config/embed.py Normal file
View File

@ -0,0 +1,12 @@
from config.model import MODEL_DIR
MODEL = MODEL_DIR / "nomic-embed-text-v1.5" / "nomic-embed-text-v1.5.Q4_K_M.gguf"
N_THREADS = 10
N_CTX = 2048 # nomic GGUF defaults to 2048; 8192 needs YaRN scaling
# nomic-embed-text needs a task instruction prefix per text.
# Use a query prefix for search/user input, a document prefix for stored text.
PREFIX_QUERY = "search_query: "
PREFIX_DOCUMENT = "search_document: "
DEFAULT_PREFIX = PREFIX_QUERY

19
config/llm.py Normal file
View File

@ -0,0 +1,19 @@
from config.model import MODEL_DIR
MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf"
N_THREADS = 10
N_CTX = 8192 # Context window (tokens)
N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU
MAX_TOKENS = 1024 # Max tokens per reply
STREAM = True # Stream tokens as they are generated
TEMPERATURE = 0.7
TOP_P = 0.9
TOP_K = 40
# System prompt for the chat session.
SYSTEM_PROMPT = "You are a helpful, concise assistant."
# Default chat template. Leave empty to use the template embedded in the GGUF.
CHAT_TEMPLATE = ""

5
core/embed_room.py Normal file
View File

@ -0,0 +1,5 @@
import argparse
parser = argparse.ArgumentParser()
parser.add_argument("texts", nargs="+", help="Text(s) to embed")
parser.add_argument("--prefix", type=str, default=None, help="Task prefix, overrides config DEFAULT_PREFIX")

6
core/llm_room.py Normal file
View File

@ -0,0 +1,6 @@
import argparse
parser = argparse.ArgumentParser()
parser.add_argument("prompt", nargs="?", default=None, help="Single prompt to answer, then exit. Omit for interactive chat")
parser.add_argument("--system", type=str, default=None, help="Override system prompt from config")
parser.add_argument("--max-tokens", type=int, default=None, help="Override config MAX_TOKENS")

37
embed_runner.py Normal file
View File

@ -0,0 +1,37 @@
import sys
import json
from core import embed_room as ap
from config.embed import MODEL, N_THREADS, N_CTX, DEFAULT_PREFIX
from llama_cpp import Llama
def embed_run(args):
if not MODEL.is_file():
print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr)
return
prefix = args.prefix if args.prefix is not None else DEFAULT_PREFIX
print("Loading model...")
llm = Llama(
model_path = str(MODEL),
n_ctx = N_CTX,
n_threads = N_THREADS,
embedding = True,
verbose = False,
)
print("Model ready!")
inputs = [prefix + t for t in args.texts]
resp = llm.create_embedding(inputs)
vectors = [d["embedding"] for d in resp["data"]]
out = [
{"text": text, "dim": len(vec), "embedding": vec}
for text, vec in zip(args.texts, vectors)
]
print(json.dumps(out, ensure_ascii=False, indent=2))
if __name__ == "__main__":
embed_run( ap.parser.parse_args() )

84
llm_runner.py Normal file
View File

@ -0,0 +1,84 @@
import sys
from core import llm_room as ap
from config.llm import (
MODEL, N_THREADS, N_CTX, N_GPU_LAYERS, MAX_TOKENS, STREAM,
TEMPERATURE, TOP_P, TOP_K, SYSTEM_PROMPT, CHAT_TEMPLATE,
)
from llama_cpp import Llama
def load_llm():
print("Loading model...")
llm = Llama(
model_path = str(MODEL),
n_ctx = N_CTX,
n_threads = N_THREADS,
n_gpu_layers = N_GPU_LAYERS,
chat_format = CHAT_TEMPLATE or None,
verbose = False,
)
print("Model ready!")
return llm
def chat_once(llm, messages, max_tokens):
if STREAM:
text = ""
for chunk in llm.create_chat_completion(
messages,
stream = True,
temperature = TEMPERATURE,
top_p = TOP_P,
top_k = TOP_K,
max_tokens = max_tokens,
):
delta = chunk["choices"][0]["delta"].get("content")
if delta:
print(delta, end="", flush=True)
text += delta
print()
return text
resp = llm.create_chat_completion(
messages,
stream = False,
temperature = TEMPERATURE,
top_p = TOP_P,
top_k = TOP_K,
max_tokens = max_tokens,
)
return resp["choices"][0]["message"]["content"]
def llm_run(args):
if not MODEL.is_file():
print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr)
return
llm = load_llm()
system = args.system if args.system is not None else SYSTEM_PROMPT
max_tokens = args.max_tokens if args.max_tokens is not None else MAX_TOKENS
if args.prompt is not None:
messages = [{"role": "system", "content": system}] if system else []
messages.append({"role": "user", "content": args.prompt})
chat_once(llm, messages, max_tokens)
return
messages = [{"role": "system", "content": system}] if system else []
print("Interactive chat. Type 'exit' or 'quit' to leave.")
try:
while True:
user = input("\nYou : ").strip()
if user.lower() in ("exit", "quit", "q"):
print("Bye.")
break
if not user:
continue
messages.append({"role": "user", "content": user})
print("Assistant: ", end="", flush=True)
reply = chat_once(llm, messages, max_tokens)
messages.append({"role": "assistant", "content": reply})
except (KeyboardInterrupt, EOFError):
print("\nBye.")
if __name__ == "__main__":
llm_run( ap.parser.parse_args() )

View File

@ -1,2 +1,3 @@
sherpa-onnx
soundfile
soundfile
llama-cpp-python

19
usage-embed.sh Executable file
View File

@ -0,0 +1,19 @@
#!/usr/bin/env bash
set -euo pipefail
BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}"
if [ ! -x "$PYTHON" ]; then
echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2
exit 1
fi
if [ $# -lt 1 ]; then
echo "Usage: $0 <text...> [--prefix <task prefix>]" >&2
echo " Embed text(s) with nomic-embed-text-v1.5 (llama.cpp)." >&2
echo " Default prefix and model path are set in config/embed.py." >&2
exit 1
fi
"$PYTHON" "$BASE_DIR/embed_runner.py" "$@"

12
usage-llm.sh Executable file
View File

@ -0,0 +1,12 @@
#!/usr/bin/env bash
set -euo pipefail
BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}"
if [ ! -x "$PYTHON" ]; then
echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2
exit 1
fi
"$PYTHON" "$BASE_DIR/llm_runner.py" "$@"