Add LLM and Embedding runner too
This commit is contained in:
parent
7d67b4af6c
commit
896a3d4c17
34
README.md
34
README.md
@ -1,6 +1,6 @@
|
|||||||
# STT Runner
|
# STT Runner
|
||||||
|
|
||||||
Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, plus Text-to-Speech with ZipVoice (zero-shot voice cloning).
|
Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, Text-to-Speech with ZipVoice (zero-shot voice cloning), and LLM/embedding inference with llama.cpp (Granite-4.2 + nomic-embed-text-v1.5).
|
||||||
|
|
||||||
## Installation
|
## Installation
|
||||||
|
|
||||||
@ -25,6 +25,20 @@ python tts_runner.py [--output=out.wav] [--ref-audio=ref.wav] [--ref-text="..."]
|
|||||||
|
|
||||||
Output defaults to `output.wav`. The reference audio/text (voice to clone) is set in `config/tts.py` and can be overridden per-run with `--ref-audio` / `--ref-text` (the text must match the audio exactly).
|
Output defaults to `output.wav`. The reference audio/text (voice to clone) is set in `config/tts.py` and can be overridden per-run with `--ref-audio` / `--ref-text` (the text must match the audio exactly).
|
||||||
|
|
||||||
|
### LLM Chat (llama.cpp)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python llm_runner.py "What is the capital of France?" # single prompt
|
||||||
|
python llm_runner.py # interactive chat
|
||||||
|
```
|
||||||
|
|
||||||
|
### Embedding (llama.cpp)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python embed_runner.py "text to embed" "another text"
|
||||||
|
python embed_runner.py "What is TSNE?" --prefix "search_query: "
|
||||||
|
```
|
||||||
|
|
||||||
## Configuration
|
## Configuration
|
||||||
|
|
||||||
Model paths and inference parameters are hardcoded in `config/`:
|
Model paths and inference parameters are hardcoded in `config/`:
|
||||||
@ -32,6 +46,8 @@ Model paths and inference parameters are hardcoded in `config/`:
|
|||||||
- `config/model.py` — model paths (conv_frontend, encoder, decoder, tokenizer under `models/`)
|
- `config/model.py` — model paths (conv_frontend, encoder, decoder, tokenizer under `models/`)
|
||||||
- `config/asr.py` — inference params: `LANGUAGE`, `HOTWORDS`, `NUM_THREADS`, `PROVIDER`, `SAMPLE_RATE`, `FEATURE_DIM`, `MAX_TOTAL_LEN`, `MAX_NEW_TOKENS`
|
- `config/asr.py` — inference params: `LANGUAGE`, `HOTWORDS`, `NUM_THREADS`, `PROVIDER`, `SAMPLE_RATE`, `FEATURE_DIM`, `MAX_TOTAL_LEN`, `MAX_NEW_TOKENS`
|
||||||
- `config/tts.py` — TTS model paths, `REFERENCE_AUDIO`, `REFERENCE_TEXT`, `OUTPUT_FILE`, `NUM_THREADS`, `PROVIDER`, `NUM_STEPS`
|
- `config/tts.py` — TTS model paths, `REFERENCE_AUDIO`, `REFERENCE_TEXT`, `OUTPUT_FILE`, `NUM_THREADS`, `PROVIDER`, `NUM_STEPS`
|
||||||
|
- `config/llm.py` — Granite-4.2 model path, `N_CTX`, `N_THREADS`, `N_GPU_LAYERS`, `MAX_TOKENS`, `TEMPERATURE`, `TOP_P`, `TOP_K`, `SYSTEM_PROMPT`, `CHAT_TEMPLATE`
|
||||||
|
- `config/embed.py` — nomic-embed-text-v1.5 model path, `N_CTX`, `N_THREADS`, `PREFIX_QUERY`, `PREFIX_DOCUMENT`, `DEFAULT_PREFIX`
|
||||||
|
|
||||||
`LANGUAGE` defaults to `""` (all languages / auto-detect). Passing `--language` on the CLI overrides it.
|
`LANGUAGE` defaults to `""` (all languages / auto-detect). Passing `--language` on the CLI overrides it.
|
||||||
|
|
||||||
@ -57,3 +73,19 @@ wget -qO- https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models/she
|
|||||||
wget -O models/zipvoice/vocos_24khz.onnx \
|
wget -O models/zipvoice/vocos_24khz.onnx \
|
||||||
https://github.com/k2-fsa/sherpa-onnx/releases/download/vocoder-models/vocos_24khz.onnx
|
https://github.com/k2-fsa/sherpa-onnx/releases/download/vocoder-models/vocos_24khz.onnx
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Download Model (Granite-4.2 8B Q4_K_M)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
mkdir -p models/granite-4.2
|
||||||
|
wget -O models/granite-4.2/granite-4.2-8b-Q4_K_M.gguf \
|
||||||
|
"https://huggingface.co/ibm-granite/granite-4.2-8b-GGUF/resolve/main/granite-4.2-8b-Q4_K_M.gguf"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Download Model (nomic-embed-text-v1.5)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
mkdir -p models/nomic-embed-text-v1.5
|
||||||
|
wget -O models/nomic-embed-text-v1.5/nomic-embed-text-v1.5.Q4_K_M.gguf \
|
||||||
|
"https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
|
||||||
|
```
|
||||||
12
config/embed.py
Normal file
12
config/embed.py
Normal file
@ -0,0 +1,12 @@
|
|||||||
|
from config.model import MODEL_DIR
|
||||||
|
|
||||||
|
MODEL = MODEL_DIR / "nomic-embed-text-v1.5" / "nomic-embed-text-v1.5.Q4_K_M.gguf"
|
||||||
|
|
||||||
|
N_THREADS = 10
|
||||||
|
N_CTX = 2048 # nomic GGUF defaults to 2048; 8192 needs YaRN scaling
|
||||||
|
|
||||||
|
# nomic-embed-text needs a task instruction prefix per text.
|
||||||
|
# Use a query prefix for search/user input, a document prefix for stored text.
|
||||||
|
PREFIX_QUERY = "search_query: "
|
||||||
|
PREFIX_DOCUMENT = "search_document: "
|
||||||
|
DEFAULT_PREFIX = PREFIX_QUERY
|
||||||
19
config/llm.py
Normal file
19
config/llm.py
Normal file
@ -0,0 +1,19 @@
|
|||||||
|
from config.model import MODEL_DIR
|
||||||
|
|
||||||
|
MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf"
|
||||||
|
|
||||||
|
N_THREADS = 10
|
||||||
|
N_CTX = 8192 # Context window (tokens)
|
||||||
|
N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU
|
||||||
|
MAX_TOKENS = 1024 # Max tokens per reply
|
||||||
|
STREAM = True # Stream tokens as they are generated
|
||||||
|
|
||||||
|
TEMPERATURE = 0.7
|
||||||
|
TOP_P = 0.9
|
||||||
|
TOP_K = 40
|
||||||
|
|
||||||
|
# System prompt for the chat session.
|
||||||
|
SYSTEM_PROMPT = "You are a helpful, concise assistant."
|
||||||
|
|
||||||
|
# Default chat template. Leave empty to use the template embedded in the GGUF.
|
||||||
|
CHAT_TEMPLATE = ""
|
||||||
5
core/embed_room.py
Normal file
5
core/embed_room.py
Normal file
@ -0,0 +1,5 @@
|
|||||||
|
import argparse
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("texts", nargs="+", help="Text(s) to embed")
|
||||||
|
parser.add_argument("--prefix", type=str, default=None, help="Task prefix, overrides config DEFAULT_PREFIX")
|
||||||
6
core/llm_room.py
Normal file
6
core/llm_room.py
Normal file
@ -0,0 +1,6 @@
|
|||||||
|
import argparse
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("prompt", nargs="?", default=None, help="Single prompt to answer, then exit. Omit for interactive chat")
|
||||||
|
parser.add_argument("--system", type=str, default=None, help="Override system prompt from config")
|
||||||
|
parser.add_argument("--max-tokens", type=int, default=None, help="Override config MAX_TOKENS")
|
||||||
37
embed_runner.py
Normal file
37
embed_runner.py
Normal file
@ -0,0 +1,37 @@
|
|||||||
|
import sys
|
||||||
|
import json
|
||||||
|
from core import embed_room as ap
|
||||||
|
from config.embed import MODEL, N_THREADS, N_CTX, DEFAULT_PREFIX
|
||||||
|
from llama_cpp import Llama
|
||||||
|
|
||||||
|
def embed_run(args):
|
||||||
|
|
||||||
|
if not MODEL.is_file():
|
||||||
|
print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr)
|
||||||
|
return
|
||||||
|
|
||||||
|
prefix = args.prefix if args.prefix is not None else DEFAULT_PREFIX
|
||||||
|
|
||||||
|
print("Loading model...")
|
||||||
|
llm = Llama(
|
||||||
|
model_path = str(MODEL),
|
||||||
|
n_ctx = N_CTX,
|
||||||
|
n_threads = N_THREADS,
|
||||||
|
embedding = True,
|
||||||
|
verbose = False,
|
||||||
|
)
|
||||||
|
print("Model ready!")
|
||||||
|
|
||||||
|
inputs = [prefix + t for t in args.texts]
|
||||||
|
|
||||||
|
resp = llm.create_embedding(inputs)
|
||||||
|
vectors = [d["embedding"] for d in resp["data"]]
|
||||||
|
|
||||||
|
out = [
|
||||||
|
{"text": text, "dim": len(vec), "embedding": vec}
|
||||||
|
for text, vec in zip(args.texts, vectors)
|
||||||
|
]
|
||||||
|
print(json.dumps(out, ensure_ascii=False, indent=2))
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
embed_run( ap.parser.parse_args() )
|
||||||
84
llm_runner.py
Normal file
84
llm_runner.py
Normal file
@ -0,0 +1,84 @@
|
|||||||
|
import sys
|
||||||
|
from core import llm_room as ap
|
||||||
|
from config.llm import (
|
||||||
|
MODEL, N_THREADS, N_CTX, N_GPU_LAYERS, MAX_TOKENS, STREAM,
|
||||||
|
TEMPERATURE, TOP_P, TOP_K, SYSTEM_PROMPT, CHAT_TEMPLATE,
|
||||||
|
)
|
||||||
|
from llama_cpp import Llama
|
||||||
|
|
||||||
|
def load_llm():
|
||||||
|
print("Loading model...")
|
||||||
|
llm = Llama(
|
||||||
|
model_path = str(MODEL),
|
||||||
|
n_ctx = N_CTX,
|
||||||
|
n_threads = N_THREADS,
|
||||||
|
n_gpu_layers = N_GPU_LAYERS,
|
||||||
|
chat_format = CHAT_TEMPLATE or None,
|
||||||
|
verbose = False,
|
||||||
|
)
|
||||||
|
print("Model ready!")
|
||||||
|
return llm
|
||||||
|
|
||||||
|
def chat_once(llm, messages, max_tokens):
|
||||||
|
if STREAM:
|
||||||
|
text = ""
|
||||||
|
for chunk in llm.create_chat_completion(
|
||||||
|
messages,
|
||||||
|
stream = True,
|
||||||
|
temperature = TEMPERATURE,
|
||||||
|
top_p = TOP_P,
|
||||||
|
top_k = TOP_K,
|
||||||
|
max_tokens = max_tokens,
|
||||||
|
):
|
||||||
|
delta = chunk["choices"][0]["delta"].get("content")
|
||||||
|
if delta:
|
||||||
|
print(delta, end="", flush=True)
|
||||||
|
text += delta
|
||||||
|
print()
|
||||||
|
return text
|
||||||
|
|
||||||
|
resp = llm.create_chat_completion(
|
||||||
|
messages,
|
||||||
|
stream = False,
|
||||||
|
temperature = TEMPERATURE,
|
||||||
|
top_p = TOP_P,
|
||||||
|
top_k = TOP_K,
|
||||||
|
max_tokens = max_tokens,
|
||||||
|
)
|
||||||
|
return resp["choices"][0]["message"]["content"]
|
||||||
|
|
||||||
|
def llm_run(args):
|
||||||
|
|
||||||
|
if not MODEL.is_file():
|
||||||
|
print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr)
|
||||||
|
return
|
||||||
|
|
||||||
|
llm = load_llm()
|
||||||
|
system = args.system if args.system is not None else SYSTEM_PROMPT
|
||||||
|
max_tokens = args.max_tokens if args.max_tokens is not None else MAX_TOKENS
|
||||||
|
|
||||||
|
if args.prompt is not None:
|
||||||
|
messages = [{"role": "system", "content": system}] if system else []
|
||||||
|
messages.append({"role": "user", "content": args.prompt})
|
||||||
|
chat_once(llm, messages, max_tokens)
|
||||||
|
return
|
||||||
|
|
||||||
|
messages = [{"role": "system", "content": system}] if system else []
|
||||||
|
print("Interactive chat. Type 'exit' or 'quit' to leave.")
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
user = input("\nYou : ").strip()
|
||||||
|
if user.lower() in ("exit", "quit", "q"):
|
||||||
|
print("Bye.")
|
||||||
|
break
|
||||||
|
if not user:
|
||||||
|
continue
|
||||||
|
messages.append({"role": "user", "content": user})
|
||||||
|
print("Assistant: ", end="", flush=True)
|
||||||
|
reply = chat_once(llm, messages, max_tokens)
|
||||||
|
messages.append({"role": "assistant", "content": reply})
|
||||||
|
except (KeyboardInterrupt, EOFError):
|
||||||
|
print("\nBye.")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
llm_run( ap.parser.parse_args() )
|
||||||
@ -1,2 +1,3 @@
|
|||||||
sherpa-onnx
|
sherpa-onnx
|
||||||
soundfile
|
soundfile
|
||||||
|
llama-cpp-python
|
||||||
19
usage-embed.sh
Executable file
19
usage-embed.sh
Executable file
@ -0,0 +1,19 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}"
|
||||||
|
|
||||||
|
if [ ! -x "$PYTHON" ]; then
|
||||||
|
echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ $# -lt 1 ]; then
|
||||||
|
echo "Usage: $0 <text...> [--prefix <task prefix>]" >&2
|
||||||
|
echo " Embed text(s) with nomic-embed-text-v1.5 (llama.cpp)." >&2
|
||||||
|
echo " Default prefix and model path are set in config/embed.py." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
"$PYTHON" "$BASE_DIR/embed_runner.py" "$@"
|
||||||
12
usage-llm.sh
Executable file
12
usage-llm.sh
Executable file
@ -0,0 +1,12 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}"
|
||||||
|
|
||||||
|
if [ ! -x "$PYTHON" ]; then
|
||||||
|
echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
"$PYTHON" "$BASE_DIR/llm_runner.py" "$@"
|
||||||
Loading…
Reference in New Issue
Block a user