Add LLM and Embedding runner too
This commit is contained in:
parent
7d67b4af6c
commit
896a3d4c17
34
README.md
34
README.md
@ -1,6 +1,6 @@
|
||||
# STT Runner
|
||||
|
||||
Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, plus Text-to-Speech with ZipVoice (zero-shot voice cloning).
|
||||
Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, Text-to-Speech with ZipVoice (zero-shot voice cloning), and LLM/embedding inference with llama.cpp (Granite-4.2 + nomic-embed-text-v1.5).
|
||||
|
||||
## Installation
|
||||
|
||||
@ -25,6 +25,20 @@ python tts_runner.py [--output=out.wav] [--ref-audio=ref.wav] [--ref-text="..."]
|
||||
|
||||
Output defaults to `output.wav`. The reference audio/text (voice to clone) is set in `config/tts.py` and can be overridden per-run with `--ref-audio` / `--ref-text` (the text must match the audio exactly).
|
||||
|
||||
### LLM Chat (llama.cpp)
|
||||
|
||||
```bash
|
||||
python llm_runner.py "What is the capital of France?" # single prompt
|
||||
python llm_runner.py # interactive chat
|
||||
```
|
||||
|
||||
### Embedding (llama.cpp)
|
||||
|
||||
```bash
|
||||
python embed_runner.py "text to embed" "another text"
|
||||
python embed_runner.py "What is TSNE?" --prefix "search_query: "
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
Model paths and inference parameters are hardcoded in `config/`:
|
||||
@ -32,6 +46,8 @@ Model paths and inference parameters are hardcoded in `config/`:
|
||||
- `config/model.py` — model paths (conv_frontend, encoder, decoder, tokenizer under `models/`)
|
||||
- `config/asr.py` — inference params: `LANGUAGE`, `HOTWORDS`, `NUM_THREADS`, `PROVIDER`, `SAMPLE_RATE`, `FEATURE_DIM`, `MAX_TOTAL_LEN`, `MAX_NEW_TOKENS`
|
||||
- `config/tts.py` — TTS model paths, `REFERENCE_AUDIO`, `REFERENCE_TEXT`, `OUTPUT_FILE`, `NUM_THREADS`, `PROVIDER`, `NUM_STEPS`
|
||||
- `config/llm.py` — Granite-4.2 model path, `N_CTX`, `N_THREADS`, `N_GPU_LAYERS`, `MAX_TOKENS`, `TEMPERATURE`, `TOP_P`, `TOP_K`, `SYSTEM_PROMPT`, `CHAT_TEMPLATE`
|
||||
- `config/embed.py` — nomic-embed-text-v1.5 model path, `N_CTX`, `N_THREADS`, `PREFIX_QUERY`, `PREFIX_DOCUMENT`, `DEFAULT_PREFIX`
|
||||
|
||||
`LANGUAGE` defaults to `""` (all languages / auto-detect). Passing `--language` on the CLI overrides it.
|
||||
|
||||
@ -56,4 +72,20 @@ wget -qO- https://github.com/k2-fsa/sherpa-onnx/releases/download/tts-models/she
|
||||
| tar xjf - -C models/zipvoice --strip-components=1
|
||||
wget -O models/zipvoice/vocos_24khz.onnx \
|
||||
https://github.com/k2-fsa/sherpa-onnx/releases/download/vocoder-models/vocos_24khz.onnx
|
||||
```
|
||||
|
||||
## Download Model (Granite-4.2 8B Q4_K_M)
|
||||
|
||||
```bash
|
||||
mkdir -p models/granite-4.2
|
||||
wget -O models/granite-4.2/granite-4.2-8b-Q4_K_M.gguf \
|
||||
"https://huggingface.co/ibm-granite/granite-4.2-8b-GGUF/resolve/main/granite-4.2-8b-Q4_K_M.gguf"
|
||||
```
|
||||
|
||||
## Download Model (nomic-embed-text-v1.5)
|
||||
|
||||
```bash
|
||||
mkdir -p models/nomic-embed-text-v1.5
|
||||
wget -O models/nomic-embed-text-v1.5/nomic-embed-text-v1.5.Q4_K_M.gguf \
|
||||
"https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
|
||||
```
|
||||
12
config/embed.py
Normal file
12
config/embed.py
Normal file
@ -0,0 +1,12 @@
|
||||
from config.model import MODEL_DIR
|
||||
|
||||
MODEL = MODEL_DIR / "nomic-embed-text-v1.5" / "nomic-embed-text-v1.5.Q4_K_M.gguf"
|
||||
|
||||
N_THREADS = 10
|
||||
N_CTX = 2048 # nomic GGUF defaults to 2048; 8192 needs YaRN scaling
|
||||
|
||||
# nomic-embed-text needs a task instruction prefix per text.
|
||||
# Use a query prefix for search/user input, a document prefix for stored text.
|
||||
PREFIX_QUERY = "search_query: "
|
||||
PREFIX_DOCUMENT = "search_document: "
|
||||
DEFAULT_PREFIX = PREFIX_QUERY
|
||||
19
config/llm.py
Normal file
19
config/llm.py
Normal file
@ -0,0 +1,19 @@
|
||||
from config.model import MODEL_DIR
|
||||
|
||||
MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf"
|
||||
|
||||
N_THREADS = 10
|
||||
N_CTX = 8192 # Context window (tokens)
|
||||
N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU
|
||||
MAX_TOKENS = 1024 # Max tokens per reply
|
||||
STREAM = True # Stream tokens as they are generated
|
||||
|
||||
TEMPERATURE = 0.7
|
||||
TOP_P = 0.9
|
||||
TOP_K = 40
|
||||
|
||||
# System prompt for the chat session.
|
||||
SYSTEM_PROMPT = "You are a helpful, concise assistant."
|
||||
|
||||
# Default chat template. Leave empty to use the template embedded in the GGUF.
|
||||
CHAT_TEMPLATE = ""
|
||||
5
core/embed_room.py
Normal file
5
core/embed_room.py
Normal file
@ -0,0 +1,5 @@
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("texts", nargs="+", help="Text(s) to embed")
|
||||
parser.add_argument("--prefix", type=str, default=None, help="Task prefix, overrides config DEFAULT_PREFIX")
|
||||
6
core/llm_room.py
Normal file
6
core/llm_room.py
Normal file
@ -0,0 +1,6 @@
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("prompt", nargs="?", default=None, help="Single prompt to answer, then exit. Omit for interactive chat")
|
||||
parser.add_argument("--system", type=str, default=None, help="Override system prompt from config")
|
||||
parser.add_argument("--max-tokens", type=int, default=None, help="Override config MAX_TOKENS")
|
||||
37
embed_runner.py
Normal file
37
embed_runner.py
Normal file
@ -0,0 +1,37 @@
|
||||
import sys
|
||||
import json
|
||||
from core import embed_room as ap
|
||||
from config.embed import MODEL, N_THREADS, N_CTX, DEFAULT_PREFIX
|
||||
from llama_cpp import Llama
|
||||
|
||||
def embed_run(args):
|
||||
|
||||
if not MODEL.is_file():
|
||||
print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr)
|
||||
return
|
||||
|
||||
prefix = args.prefix if args.prefix is not None else DEFAULT_PREFIX
|
||||
|
||||
print("Loading model...")
|
||||
llm = Llama(
|
||||
model_path = str(MODEL),
|
||||
n_ctx = N_CTX,
|
||||
n_threads = N_THREADS,
|
||||
embedding = True,
|
||||
verbose = False,
|
||||
)
|
||||
print("Model ready!")
|
||||
|
||||
inputs = [prefix + t for t in args.texts]
|
||||
|
||||
resp = llm.create_embedding(inputs)
|
||||
vectors = [d["embedding"] for d in resp["data"]]
|
||||
|
||||
out = [
|
||||
{"text": text, "dim": len(vec), "embedding": vec}
|
||||
for text, vec in zip(args.texts, vectors)
|
||||
]
|
||||
print(json.dumps(out, ensure_ascii=False, indent=2))
|
||||
|
||||
if __name__ == "__main__":
|
||||
embed_run( ap.parser.parse_args() )
|
||||
84
llm_runner.py
Normal file
84
llm_runner.py
Normal file
@ -0,0 +1,84 @@
|
||||
import sys
|
||||
from core import llm_room as ap
|
||||
from config.llm import (
|
||||
MODEL, N_THREADS, N_CTX, N_GPU_LAYERS, MAX_TOKENS, STREAM,
|
||||
TEMPERATURE, TOP_P, TOP_K, SYSTEM_PROMPT, CHAT_TEMPLATE,
|
||||
)
|
||||
from llama_cpp import Llama
|
||||
|
||||
def load_llm():
|
||||
print("Loading model...")
|
||||
llm = Llama(
|
||||
model_path = str(MODEL),
|
||||
n_ctx = N_CTX,
|
||||
n_threads = N_THREADS,
|
||||
n_gpu_layers = N_GPU_LAYERS,
|
||||
chat_format = CHAT_TEMPLATE or None,
|
||||
verbose = False,
|
||||
)
|
||||
print("Model ready!")
|
||||
return llm
|
||||
|
||||
def chat_once(llm, messages, max_tokens):
|
||||
if STREAM:
|
||||
text = ""
|
||||
for chunk in llm.create_chat_completion(
|
||||
messages,
|
||||
stream = True,
|
||||
temperature = TEMPERATURE,
|
||||
top_p = TOP_P,
|
||||
top_k = TOP_K,
|
||||
max_tokens = max_tokens,
|
||||
):
|
||||
delta = chunk["choices"][0]["delta"].get("content")
|
||||
if delta:
|
||||
print(delta, end="", flush=True)
|
||||
text += delta
|
||||
print()
|
||||
return text
|
||||
|
||||
resp = llm.create_chat_completion(
|
||||
messages,
|
||||
stream = False,
|
||||
temperature = TEMPERATURE,
|
||||
top_p = TOP_P,
|
||||
top_k = TOP_K,
|
||||
max_tokens = max_tokens,
|
||||
)
|
||||
return resp["choices"][0]["message"]["content"]
|
||||
|
||||
def llm_run(args):
|
||||
|
||||
if not MODEL.is_file():
|
||||
print(f"Model not found: {MODEL}. Download it first (see README).", file=sys.stderr)
|
||||
return
|
||||
|
||||
llm = load_llm()
|
||||
system = args.system if args.system is not None else SYSTEM_PROMPT
|
||||
max_tokens = args.max_tokens if args.max_tokens is not None else MAX_TOKENS
|
||||
|
||||
if args.prompt is not None:
|
||||
messages = [{"role": "system", "content": system}] if system else []
|
||||
messages.append({"role": "user", "content": args.prompt})
|
||||
chat_once(llm, messages, max_tokens)
|
||||
return
|
||||
|
||||
messages = [{"role": "system", "content": system}] if system else []
|
||||
print("Interactive chat. Type 'exit' or 'quit' to leave.")
|
||||
try:
|
||||
while True:
|
||||
user = input("\nYou : ").strip()
|
||||
if user.lower() in ("exit", "quit", "q"):
|
||||
print("Bye.")
|
||||
break
|
||||
if not user:
|
||||
continue
|
||||
messages.append({"role": "user", "content": user})
|
||||
print("Assistant: ", end="", flush=True)
|
||||
reply = chat_once(llm, messages, max_tokens)
|
||||
messages.append({"role": "assistant", "content": reply})
|
||||
except (KeyboardInterrupt, EOFError):
|
||||
print("\nBye.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
llm_run( ap.parser.parse_args() )
|
||||
@ -1,2 +1,3 @@
|
||||
sherpa-onnx
|
||||
soundfile
|
||||
soundfile
|
||||
llama-cpp-python
|
||||
19
usage-embed.sh
Executable file
19
usage-embed.sh
Executable file
@ -0,0 +1,19 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}"
|
||||
|
||||
if [ ! -x "$PYTHON" ]; then
|
||||
echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ $# -lt 1 ]; then
|
||||
echo "Usage: $0 <text...> [--prefix <task prefix>]" >&2
|
||||
echo " Embed text(s) with nomic-embed-text-v1.5 (llama.cpp)." >&2
|
||||
echo " Default prefix and model path are set in config/embed.py." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
"$PYTHON" "$BASE_DIR/embed_runner.py" "$@"
|
||||
12
usage-llm.sh
Executable file
12
usage-llm.sh
Executable file
@ -0,0 +1,12 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}"
|
||||
|
||||
if [ ! -x "$PYTHON" ]; then
|
||||
echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
"$PYTHON" "$BASE_DIR/llm_runner.py" "$@"
|
||||
Loading…
Reference in New Issue
Block a user