diff --git a/.gitignore b/.gitignore index e7ef4ca..54546a3 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,11 @@ .env .venv **/__pycache__ -*.pyc \ No newline at end of file +*.pyc + +# Model weights and the laya.cpp source/build tree (see README). +models/ +third_party/ + +# Runner output. +output.wav diff --git a/CLASSIFY.md b/CLASSIFY.md new file mode 100644 index 0000000..2e9c4e6 --- /dev/null +++ b/CLASSIFY.md @@ -0,0 +1,549 @@ +# Panduan Classification Runner (Laya / laya.cpp) + +Dokumen ini menjelaskan cara pakai `classify_runner.py` — runner klasifikasi teks pakai +**Laya**, model keputusan (*decision model*) multilingual 322M parameter, yang dijalankan lewat +runtime native **laya.cpp** di CPU. + +Untuk ringkasan proyek dan setup runner lain, lihat [README.md](README.md). + +--- + +## Daftar isi + +1. [Cara kerja singkat](#1-cara-kerja-singkat) +2. [Menjalankan](#2-menjalankan) +3. [Cara baca output](#3-cara-baca-output) +4. [Tiga tipe pertanyaan](#4-tiga-tipe-pertanyaan) +5. [Pertanyaan sendiri](#5-pertanyaan-sendiri) +6. [Mode server (lebih cepat)](#6-mode-server-lebih-cepat) +7. [Output JSON](#7-output-json) +8. [Semua opsi CLI](#8-semua-opsi-cli) +9. [Konfigurasi](#9-konfigurasi) +10. [Setup dari nol](#10-setup-dari-nol) +11. [Catatan penting](#11-catatan-penting) + +--- + +## 1. Cara kerja singkat + +Laya **bukan** classifier satu-label biasa. Kamu memberi dia sebuah *state* (teks, email, ticket, +atau JSON) plus beberapa **pertanyaan bertipe**. Tiap pertanyaan mendeklarasikan ruang jawabannya +sendiri, jadi **tidak perlu training/finetune** untuk memakainya di kasus baru. + +Semua pertanyaan dijawab dalam **satu forward pass**, dengan probabilitas — jadi tidak ada teks +yang perlu diparse dan tidak ada risiko hallucination seperti pada LLM generatif. + +``` +state + pertanyaan → satu forward pass → jawaban bertipe + probabilitas +``` + +Runner mendukung tiga mode pemanggilan: + +| Mode | Kapan dipakai | Kondisi | Kecepatan | +|---|---|---|---| +| CLI | satu kali jalan | tanpa setup | ~5–7 detik | +| Server | dipakai berulang | harus nyalakan server | ~0,6 detik | + +--- + +## 2. Menjalankan + +### Cara paling dasar + +```bash +./usage-classify.sh "Saya dikenakan biaya dua kali untuk invoice 4411, tolong kembalikan uang saya." +``` + +### Kalau teksnya panjang, pakai file + +```bash +./usage-classify.sh --state-file ticket.txt +``` + +### Kalau state-nya punya beberapa kolom + +Kalau teks diawali `{` atau `[`, runner otomatis mengirimnya sebagai **struktur JSON**, bukan teks +biasa. Jadi email dengan kolom subject/body bisa langsung: + +```bash +./usage-classify.sh '{"subject":"Refund not received","body":"I cancelled two weeks ago and still have no refund."}' +``` + +Atau dari file JSON: + +```bash +./usage-classify.sh --state-file ticket.json +``` + +--- + +## 3. Cara baca output + +Contoh output nyata: + +``` +State : Saya dikenakan biaya dua kali untuk invoice 4411, tolong kembalikan uang saya. +Questions : 3 (cli) +Backend : CPU (12th Gen Intel(R) Core(TM) i7-12700H) +Elapsed : 6.953 s +Inference : 0.732 s +Tokens : 176 in / 0 out + +department : billing + billing=1.0000 technical=0.0000 sales=0.0000 other=0.0000 +urgency : 1.7297 (immediate) + 0=0.0306 1=0.2091 2=0.7603 +refund : 0.7751 + act_probability=1.0000 +``` + +### Bagian atas + +| Baris | Arti | +|---|---| +| `State` | State yang dikirim (dipotong 120 karakter kalau panjang) | +| `Questions` | Berapa pertanyaan dijawab, dan mode yang dipakai (`cli` atau `server`) | +| `Backend` | Backend dan perangkat yang dipakai. Hanya muncul di mode CLI | +| `Elapsed` | Total waktu wall-clock, **termasuk load bobot model** | +| `Inference` | Waktu forward pass saja. Hanya muncul di mode CLI | +| `Tokens` | Jumlah token input yang diproses (output selalu 0, karena model ini tidak generatif) | + +> **Kenapa `Elapsed` jauh lebih besar dari `Inference`?** +> Model ini tidak generatif, jadi forward pass-nya cuma ~0,6 detik. Sisanya (~5 detik) adalah +> waktu baca bobot 644MB dari disk. Ini normal. Kalau kamu pemakai berulang, pakai +> [mode server](#6-mode-server-lebih-cepat). + +### Bagian jawaban + +Nama di kiri adalah **ID pertanyaan** yang kamu tentukan sendiri. Formatnya berbeda per tipe: + +**`choice`** — jawaban berupa salah satu label. Baris kedua menampilkan probabilitas tiap opsi, +diurutkan dari yang tertinggi: + +``` +department : billing + billing=1.0000 technical=0.0000 sales=0.0000 other=0.0000 +``` + +**`score`** — jawaban berupa angka desimal (rata-rata tertimbang dari distribusi), diikuti label +level yang paling dekat dalam kurung. Baris kedua menampilkan probabilitas tiap level **dalam +urutan legend** (tidak diurutkan): + +``` +urgency : 1.7297 (immediate) + 0=0.0306 1=0.2091 2=0.7603 +``` + +**`noul`** — jawaban berupa peluang pernyataan itu benar (0.0 = pasti salah, 1.0 = pasti benar). +Tipe ini tidak punya objek `probabilities`, jadi baris kedua menampilkan +`act_probability`, yaitu peluang bahwa kasus ini perlu di-*escalate* ke manusia: + +``` +refund : 0.7751 + act_probability=1.0000 +``` + +--- + +## 4. Tiga tipe pertanyaan + +| Tipe | Isi `criteria` | Output | Kapan pakai | +|---|---|---|---| +| `choice` | objek label → deskripsi | label terpilih + peluang tiap label | klasifikasi ke kategori | +| `score` | **array** berurutan | angka desimal + label level | intensitas, prioritas, sentimen | +| `noul` | tidak ada | peluang pernyataan `true` | deteksi ya/tidak | + +Contoh lengkap ketiga tipe (ini isi preset `triage` di `config/classify.py`): + +```json +{ + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "invoices, payments, refunds", + "technical": "bugs, outages, system errors", + "sales": "pricing, plans, new contracts", + "other": "everything else" + } + }, + "urgency": { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": ["not urgent", "soon", "immediate"] + }, + "refund": { + "type": "noul", + "instructions": "Does the customer ask for money back?" + } +} +``` + +### Aturan penting + +- **`choice` jangan lebih dari ~20 opsi.** Semua opsi berbagi satu budget token yang tetap, jadi + kalau label terlalu banyak, tiap label cuma kebagian sedikit token dan akurasinya jatuh tajam. +- **`score` pakai array, bukan objek.** Urutan array itu penting — itulah yang menentukan level + 0, 1, 2, dan seterusnya. +- **`instructions` ditulis dalam bahasa Inggris** walau isinya bisa bahasa apa pun. Instruksi + bagian kepala (256 token) dan state diletakkan terpisah oleh `[SEP]` dalam model, jadi + instruksi tidak ikut mengotori bahasa input. +- **Nama kolom output bebas kamu pilih.** Kolom di kiri output itu key yang kamu pakai di + objek `questions`, bukan dari `instructions`. + +### Multibahasa + +Checkpoint-nya memang varian multilingual (100+ bahasa), jadi input non-Inggris diharapkan +berfungsi: + +```bash +$ ./usage-classify.sh "二重に請求されました。返金をお願いします。" +refund : 0.9976 + +$ ./usage-classify.sh "双重收费了,请退款。" +refund : 0.9892 +``` + +--- + +## 5. Pertanyaan sendiri + +Preset `triage` cuma comprise 3 pertanyaan default. Untuk kasus kamu, bikin file JSON: + +```bash +cat > my_questions.json <<'EOF' +{ + "bahasa": { + "type": "choice", + "instructions": "What language is this text written in?", + "criteria": { + "id": "Indonesian", + "en": "English", + "fr": "French", + "zh": "Chinese" + } + }, + "komplain": { + "type": "noul", + "instructions": "Does the customer complain or express dissatisfaction?" + }, + "sentimen": { + "type": "score", + "instructions": "How positive is the tone?", + "criteria": ["very negative", "neutral", "positive", "very positive"] + } +} +EOF + +./usage-classify.sh --questions my_questions.json "produk ini error terus sejak kemarin" +``` + +File-nya boleh berisi **objek questions langsung** (seperti di atas), **atau** objek request +lengkap yang punya key `"questions"`. Runner mendeteksi keduanya. + +Kamu juga bisa simpan pertanyaan sebagai preset permanen di `config/classify.py` supaya bisa +dipakai cuma dengan `--preset nama`: + +```python +PRESETS = { + "triage": { ... }, # bawaan + "bahasa": { ... }, # preset kamu +} +DEFAULT_PRESET = "triage" +``` + +```bash +./usage-classify.sh --preset bahasa "produk ini error terus sejak kemarin" +``` + +--- + +## 6. Mode server (lebih cepat) + +Tiap kali runner dipanggil dalam mode CLI, ia me-*spawn* `laya-cli` dan **load ulang bobot +644MB dari disk**. Itu yang bikin total ~5–7 detik, padahal inferensinya cuma 0,6 detik. + +Kalau kamu classifies berulang, biarkan model tetap *resident* di memory lewat HTTP server bawaan +laya.cpp: + +```bash +third_party/laya.cpp/build-cpu/bin/laya-cli --server --port 8080 \ + --model models/convaiinnovations/laya --variant multilingual --cpu & +``` + +Lalu arahkan runner ke sana: + +```bash +LAYA_URL=http://127.0.0.1:8080 ./usage-classify.sh "teks kamu" +``` + +Hasilnya **~0,6 detik bukan ~6 detik** — sekitar 8x lebih cepat, dengan jawaban yang identik +(bisa diverifikasi langsung: bandingkan output CLI dan server untuk input yang sama). + +### Urutan prioritas URL + +1. `--url` di CLI +2. environment variable `LAYA_URL` +3. `LAYA_URL` di `config/classify.py` + +### Kalau server mati + +Runner **otomatis fallback** ke mode CLI, dengan informasi di stderr: + +``` +laya.cpp server at http://127.0.0.1:8080 unavailable (Connection refused). Falling back to the CLI. +``` + +Jadi tidak ada risiko kalau lupa nyalakan server atau server-nya crash. + +### Endpoint lain yang tersedia + +Server laya.cpp juga menyediakan `POST /predict` (bisa batch beberapa request sekaligus, dan +mengembalikan envelope CLI lengkap termasuk `elapsed_ms`), `GET /health`, dan `GET /v1/models`. +Runner memakai `/v1/systemone` yang bentuk balasannya lebih ringkas. + +--- + +## 7. Output JSON + +```bash +./usage-classify.sh "teks" --json +``` + +Menghasilkan envelope mentah laya.cpp dengan probabilitas presisi penuh (bukan 4 desimal) — +berguna kalau mau pipe ke program lain: + +```json +{ + "results": [ + { + "model": "laya-rl-agent", + "answers": { + "refund": { + "type": "noul", + "confidence": 0.7751, + "action": { "act_probability": 1.0 }, + "noul": 0.7751 + } + }, + "usage": { "input_tokens": 176, "output_tokens": 0 } + } + ], + "elapsed_ms": 732.14, + "backend": "CPU", + "device": "12th Gen Intel(R) Core(TM) i7-12700H" +} +``` + +Bentuk tiap tipe jawaban: + +| Tipe | Field yang ada | +|---|---| +| `choice` | `type`, `choice`, `probabilities`, `confidence`, `action.act_probability` | +| `score` | `type`, `score`, `legend`, `probabilities`, `confidence`, `action.act_probability` | +| `noul` | `type`, `noul`, `confidence`, `action.act_probability` | + +--- + +## 8. Semua opsi CLI + +``` +usage: classify_runner.py [-h] [--state-file STATE_FILE] [--preset PRESET] + [--questions QUESTIONS] [--url URL] [--json] + [state] + +positional arguments: + state State yang akan diklasifikasi (teks biasa atau JSON). + Hilangkan kalau mau baca dari --state-file. + +options: + --state-file STATE_FILE File yang berisi state (teks biasa atau JSON) + --preset PRESET Nama preset pertanyaan dari config/classify.py + --questions QUESTIONS File JSON berisi objek questions, mengoverride --preset + --url URL Base URL server laya.cpp, mengoverride LAYA_URL dan config + --json Cetak envelope respons mentah sebagai JSON + -h, --help Tampilkan bantuan +``` + +> **Catatan:** `--state` **bukan** sebuah flag — state diberikan sebagai argumen posisi. Writer +> `--state` akan menghasilkan error, bukan diam-diam diperlakukan sebagai nama file. + +Exit code: + +| Kode | Arti | +|---|---| +| `0` | Berhasil | +| `1` | Gagal — preset tidak dikenal, file tidak ada, state kosong, atau inference error | + +Contoh pemakaian error yang umum: + +```bash +$ ./usage-classify.sh --preset salah "teks" +Unknown preset: salah. Available: triage + +$ ./usage-classify.sh +Usage: ./usage-classify.sh [--preset ] [--questions ] [--json] +``` + +--- + +## 9. Konfigurasi + +Semua default ada di `config/classify.py`: + +```python +LAYACPP = /third_party/laya.cpp +CLI = LAYACPP / "build-cpu" / "bin" / "laya-cli" + +MODEL_ROOT = MODEL_DIR / "convaiinnovations" / "laya" +VARIANT = "multilingual" +CHECKPOINT = MODEL_ROOT / VARIANT / "model.safetensors" + +BACKEND = "cpu" # cpu | cuda | vulkan | coreml +ALLOW_TRUNCATION = False +LAYA_URL = "" # "" = spawn CLI per jalan +TIMEOUT = 120 + +PRESETS = { "triage": {...} } +DEFAULT_PRESET = "triage" +``` + +| Variabel | Fungsi | +|---|---| +| `CLI` | Lokasi binary `laya-cli` hasil build | +| `MODEL_ROOT` | Direktori induk checkpoint | +| `VARIANT` | Varian checkpoint: `multilingual`, `english`, atau `typed-decisions` | +| `BACKEND` | Backend komputasi. `cpu` = pure CPU, tanpa GPU sama sekali | +| `ALLOW_TRUNCATION` | Kalau `True`, input kepanjangan dipotong diam-diam. Default: request ditolak | +| `LAYA_URL` | Fallback URL server kalau env dan flag tidak di-set | +| `PRESETS` | Kumpulan pertanyaan siap pakai | +| `DEFAULT_PRESET` | Preset yang dipakai kalau `--preset` tidak diberikan | + +### Ganti ke GPU nanti + +Ubah `BACKEND` jadi `"cuda"` atau `"vulkan"`. Tapi kamu harus pakai binary laya.cpp yang dikompilasi +dengan backend tersebut — build CPU-only (yang sekarang) tidak bisa. Detail di +[README.md](README.md#build-layacpp-cpu-only). + +--- + +## 10. Setup dari nol + +Di mesin ini **sudah selesai**. Kalau perlu diulang (misal di mesin lain): + +```bash +./build-laya.sh +``` + +Script-nya **tidak butuh `sudo`** sama sekali: + +- `cmake` dan `ninja` diambil dari pip wheel +- `nlohmann/json` diekstrak ke prefix lokal `third_party/nlohmann-install` +- satu-satunya kebutuhan sistem adalah `libicu-dev` + +Script aman dijalankan berulang — setiap langkah dilewati kalau output-nya sudah ada. + +Yang dilakukan script: + +1. Clone `laya.cpp` beserta submodulnya (`ggml`, `cpp-httplib`) ke `third_party/laya.cpp` +2. Pasang `cmake` (versi 3.x) dan `ninja` lewat pip kalau belum ada +3. Ekstrak `nlohmann/json` ke prefix lokal +4. Build CPU-only ke `third_party/laya.cpp/build-cpu` + +### Tiga hal yang wajib + +| Flag | Kenapa wajib | +|---|---| +| `-DLAYA_CUDA=OFF` | Default-nya **ON**, dan konfigurasi akan gagal mencari `nvcc` | +| `cmake` versi 3.x | ggml yang di-pin laya.cpp deklarasikan `cmake_minimum_required(3.14...3.28)` dan **tidak bisa dikonfigurasi di CMake 4** | +| `--cpu` saat runtime | Default `laya-cli` adalah backend CUDA, jadi build CPU-only akan gagal start tanpa flag ini | + +### Kenapa harus build dari sumber? + +Upstream laya.cpp **tidak menyediakan** binary prebuilt khusus CPU — rilis Linux hanya CUDA 12, +CUDA 13, dan Vulkan. Kalau di mesin dengan GPU NVIDIA, kamu bisa unduh binary prebuilt dan +menghemat langkah build, tapi build CPU-only ini nol dependensi GPU. + +### Struktur checkpoint + +Checkpoint ada di `models/convaiinnovations/laya/multilingual/`. + +> **Penting:** `laya-cli` menambahkan nama varian ke path `--model` untuk semua varian non-English. +> Jadi checkpoint multilingual **harus** berada di `/multilingual/`. +> +> Ada salinan bobot yang identik di `models/convaiinnovations/laya-multilingual/`, tapi layout itu +> **tidak bisa** dipakai laya.cpp — direktori itu untuk implementasi referensi Python. + +--- + +## 11. Catatan penting + +Dua keterbatasan dari model card upstream yang belum diatasi runner ini. Keduanya penting +sebelum kamu mempercayai angkanya: + +### 1. Probabilitasnya belum dikalibrasi + +Model ini di-ship **uncalibrated** dengan `temperature = [1.0, 1.0, 1.0]` tanpa bucket per +jumlah opsi. Akibatnya dia **sistematis over-confident** — rata-rata confidence 0,75–0,83 +padahal akurasinya jauh lebih rendah. + +Jadi nilai `confidence=1.0` di output **belum tentu akurat**. + +Untuk membaikkannya, refit satu temperatur per kombinasi (tipe pertanyaan, jumlah opsi) di data +held-out milikmu sendiri. Dari model card, ini memindahkan mean ECE dari **0,314 → 0,106**. +Lakukan ini sebelum memperlakukan probabilitas sebagai ambang keputusan. + +### 2. Tipe `noul` bisa under-report `true` + +Pada input yang jelas positif, satu pengukuran menempatkan `P(true)` di kisaran 0,5 sementara kasus +negatifnya benar-benar dekat 0. Kalau jawaban `noul` terasa lemah, **cek ulang** dengan versi dua +opsi: + +```json +{ + "cek": { + "type": "choice", + "instructions": "Does the text ask for a refund?", + "criteria": { + "A": "no, it does not ask for a refund", + "B": "yes, it asks for a refund" + } + } +} +``` + +### 3. Bahasa dengan sumber daya rendah masih lemah + +SWE, Tamil, Amharic masih berada di rentang 0,110–0,250. Check akurasi di bahasa-bahasa yang memang +kamu pakai, jangan diasumsikan seragam. + +### 4. Model multilingual lebih lemah untuk teks Inggris + +Makro accuracy 0,619 (multilingual) vs 0,684 (checkpoint Inggris). Kalau mostly English, routing +ke checkpoint `laya` (Inggris) lebih akurat daripada memaksakan pakai multilingual. + +--- + +## Ringkasan perintah + +```bash +# Klasifikasi teks +./usage-classify.sh "teks kamu" + +# Dari file +./usage-classify.sh --state-file ticket.txt + +# Pertanyaan sendiri +./usage-classify.sh --questions my_questions.json "teks kamu" + +# Output JSON untuk di-pipe +./usage-classify.sh "teks kamu" --json + +# Mode cepat: nyalakan server sekali +third_party/laya.cpp/build-cpu/bin/laya-cli --server --port 8080 \ + --model models/convaiinnovations/laya --variant multilingual --cpu & +LAYA_URL=http://127.0.0.1:8080 ./usage-classify.sh "teks kamu" + +# Setup (sekali saja, tanpa sudo) +./build-laya.sh +``` diff --git a/README.md b/README.md index 46bbbdb..a9a7016 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,8 @@ # STT Runner -Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, Text-to-Speech with ZipVoice (zero-shot voice cloning), and LLM/embedding inference with llama.cpp (Granite-4.2 + nomic-embed-text-v1.5). +Speech-to-Text transcription using sherpa-onnx + Qwen3-ASR, Text-to-Speech with ZipVoice (zero-shot voice cloning), LLM/embedding inference with llama.cpp (Granite-4.2 + nomic-embed-text-v1.5), and text classification with Laya typed decisions on the CPU-only laya.cpp runtime. + +> **Panduan lengkap classification runner (bahasa Indonesia): [CLASSIFY.md](CLASSIFY.md)** ## Installation @@ -39,6 +41,47 @@ python embed_runner.py "text to embed" "another text" python embed_runner.py "What is TSNE?" --prefix "search_query: " ``` +### Text Classification (laya.cpp, CPU) + +```bash +./build-laya.sh # once, no sudo needed +./usage-classify.sh "Saya dikenakan biaya dua kali, tolong refund." +./usage-classify.sh --state-file ticket.txt --preset triage +./usage-classify.sh --questions my_questions.json "text" --json +``` + +Laya is a decision model, not a single-label classifier: you supply **typed questions** and each +one declares its own answer space, so nothing has to be retrained. Three types are supported — +`choice` (picks a labelled option), `score` (returns a level from an ordered list) and `noul` +(returns a probability that a statement holds). The default `triage` preset in `config/classify.py` +asks three questions; `--questions FILE` replaces it with your own. The file may hold a bare +questions object or a full laya.cpp request object. A state starting with `{` or `[` is passed +through as a structure, so `{"subject": ..., "body": ...}` works too. + +The checkpoint is the 322M multilingual model, so non-English input is expected to work: + +``` +$ ./usage-classify.sh "二重に請求されました。返金をお願いします。" +refund : 0.9976 +``` + +Each run spawns `laya-cli` and reloads the 644MB weights, which dominates the wall clock +(~5.5s total, of which ~0.6s is inference). For repeated calls keep the checkpoint resident and +point the runner at the HTTP server — same answers, ~8x faster end to end: + +```bash +third_party/laya.cpp/build-cpu/bin/laya-cli --server --port 8080 \ + --model models/convaiinnovations/laya --variant multilingual --cpu & +LAYA_URL=http://127.0.0.1:8080 ./usage-classify.sh "..." +``` + +`LAYA_URL` is read from the environment, `--url` overrides it, and `LAYA_URL` in `config/classify.py` +is the fallback. If the server cannot be reached the runner says so and falls back to the CLI. + +Two caveats from the upstream model card, worth knowing before you trust the numbers: the +probabilities ship **uncalibrated** and systematically over-confident, and `noul` can under-report +a clear `true`. Fit your own thresholds on held-out data before treating a probability as a decision. + ## Configuration Model paths and inference parameters are hardcoded in `config/`: @@ -48,9 +91,53 @@ Model paths and inference parameters are hardcoded in `config/`: - `config/tts.py` — TTS model paths, `REFERENCE_AUDIO`, `REFERENCE_TEXT`, `OUTPUT_FILE`, `NUM_THREADS`, `PROVIDER`, `NUM_STEPS` - `config/llm.py` — Granite-4.2 model path, `N_CTX`, `N_THREADS`, `N_GPU_LAYERS`, `MAX_TOKENS`, `TEMPERATURE`, `TOP_P`, `TOP_K`, `SYSTEM_PROMPT`, `CHAT_TEMPLATE` - `config/embed.py` — nomic-embed-text-v1.5 model path, `N_CTX`, `N_THREADS`, `PREFIX_QUERY`, `PREFIX_DOCUMENT`, `DEFAULT_PREFIX` +- `config/classify.py` — laya.cpp binary path, `MODEL_ROOT`/`VARIANT`, `BACKEND`, `LAYA_URL`, `PRESETS`, `DEFAULT_PRESET` `LANGUAGE` defaults to `""` (all languages / auto-detect). Passing `--language` on the CLI overrides it. +## Build laya.cpp (CPU-only) + +`classify_runner.py` drives the native `laya-cli` binary, so there is no Python inference +dependency — `requirements.txt` is unchanged, and in particular PyTorch is not needed. + +```bash +./build-laya.sh +``` + +The script needs no `sudo`: `cmake` and `ninja` come from pip wheels, `nlohmann/json` is unpacked +into `third_party/nlohmann-install`, and the only system requirement is `libicu-dev` +(`apt install libicu-dev` if it is missing). Re-running it skips finished steps. The manual +equivalent: + +```bash +git clone --recursive --shallow-submodules https://github.com/lkarlslund/laya.cpp third_party/laya.cpp +.venv/bin/pip install "cmake==3.31.*" "ninja<2" +export PATH="$PWD/.venv/bin:$PATH" +cmake -S third_party/laya.cpp -B third_party/laya.cpp/build-cpu -G Ninja \ + -DCMAKE_BUILD_TYPE=Release \ + -DLAYA_CUDA=OFF -DLAYA_VULKAN=OFF -DLAYA_COREML=OFF -DBUILD_TESTING=OFF \ + -DCMAKE_PREFIX_PATH="$PWD/third_party/nlohmann-install" +cmake --build third_party/laya.cpp/build-cpu --parallel "$(nproc)" +``` + +Three flags matter: + +- **`-DLAYA_CUDA=OFF` is mandatory** — `LAYA_CUDA` defaults to `ON` and configuration fails looking for `nvcc`. +- **`cmake` must stay on 3.x** — laya.cpp's pinned ggml declares `cmake_minimum_required(VERSION 3.14...3.28)` and does not configure on CMake 4. +- **`--cpu` is always passed at runtime** — `laya-cli` defaults to the CUDA backend, so a CPU-only build fails at startup without it. + +Upstream ships no CPU-only prebuilt binary (Linux releases are CUDA 12, CUDA 13 and Vulkan only), +which is why this is built from source. On an NVIDIA box you can swap `BACKEND` in +`config/classify.py` to `cuda` and download a prebuilt `laya` instead, keeping the build step out of +the loop entirely. + +`models/convaiinnovations/laya/multilingual/` holds the checkpoint in the layout the CLI expects. +Note that `laya-cli` appends the variant to `--model` for every non-English variant, so the +multilingual checkpoint must be at `/multilingual/`. `config/classify.py` sets +`MODEL_ROOT` to the parent directory. A separate copy of the same weights sits at +`models/convaiinnovations/laya-multilingual/`, which this layout does **not** address — the +standalone HuggingFace repo directory is for the Python reference implementation. + ## Download Model (Qwen3-ASR 1.7B int8) ```bash diff --git a/build-laya.sh b/build-laya.sh new file mode 100755 index 0000000..e6b2f3f --- /dev/null +++ b/build-laya.sh @@ -0,0 +1,72 @@ +#!/usr/bin/env bash +# Build the CPU-only laya.cpp runtime used by classify_runner.py. +# +# Safe to re-run: every step is skipped when its output already exists. +# Nothing here needs sudo. cmake/ninja come from pip wheels and +# nlohmann/json is unpacked into a local prefix, so the only system +# requirement is libicu-dev (already present on Debian/Ubuntu). +set -euo pipefail + +BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +LAYACPP="$BASE_DIR/third_party/laya.cpp" +PREFIX="$BASE_DIR/third_party/nlohmann-install" +BUILD="$LAYACPP/build-cpu" +PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}" + +if [ ! -x "$PYTHON" ]; then + echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2 + exit 1 +fi + +# cmake and ninja are build tools, not runtime deps, so they are installed here +# rather than added to requirements.txt. Pinned below 4.x: laya.cpp's pinned ggml +# declares cmake_minimum_required(VERSION 3.14...3.28) and does not configure on 4.x. +if ! "$BASE_DIR/.venv/bin/cmake" --version 2>/dev/null | grep -qE '^cmake version 3\.'; then + echo "==> Installing cmake (<4) and ninja" + "$PYTHON" -m pip install --quiet "cmake==3.31.*" "ninja<2" +fi +export PATH="$BASE_DIR/.venv/bin:$PATH" + +if [ ! -d "$LAYACPP/.git" ]; then + echo "==> Cloning laya.cpp" + git clone --recursive --shallow-submodules https://github.com/lkarlslund/laya.cpp "$LAYACPP" +fi + +# laya.cpp needs the nlohmann_json CMake package. Debian ships it as +# nlohmann-json3-dev, but a local prefix avoids the sudo requirement. +if [ ! -f "$PREFIX/share/cmake/nlohmann_json/nlohmann_jsonConfig.cmake" ]; then + echo "==> Unpacking nlohmann/json" + tarball="$(mktemp -t json-XXXXXX.tar.xz)" + trap 'rm -f "$tarball"' EXIT + curl -fsSL "https://github.com/nlohmann/json/releases/download/v3.11.3/json.tar.xz" -o "$tarball" + mkdir -p "$BASE_DIR/third_party/nlohmann-prefix" + tar xJf "$tarball" -C "$BASE_DIR/third_party/nlohmann-prefix" --strip-components=1 + # JSON_BuildTests=OFF because the release tarball ships no tests/ directory. + cmake -S "$BASE_DIR/third_party/nlohmann-prefix" -B "$BASE_DIR/third_party/nlohmann-build" -G Ninja \ + -DCMAKE_BUILD_TYPE=Release -DJSON_BuildTests=OFF -DCMAKE_INSTALL_PREFIX="$PREFIX" > /dev/null + cmake --install "$BASE_DIR/third_party/nlohmann-build" > /dev/null +fi + +if [ ! -x "$BUILD/bin/laya-cli" ]; then + # LAYA_CUDA defaults to ON and would require nvcc, so it must be turned off + # explicitly. BUILD_TESTING=OFF skips test binaries that want GPU fixtures. + echo "==> Building laya-cli (CPU only)" + cmake -S "$LAYACPP" -B "$BUILD" -G Ninja \ + -DCMAKE_BUILD_TYPE=Release \ + -DLAYA_CUDA=OFF -DLAYA_VULKAN=OFF -DLAYA_COREML=OFF \ + -DBUILD_TESTING=OFF \ + -DCMAKE_PREFIX_PATH="$PREFIX" > /dev/null + cmake --build "$BUILD" --parallel "$(nproc)" +fi + +echo +echo "Built: $BUILD/bin/laya-cli" +"$BUILD/bin/laya-cli" --help > /dev/null && echo "Smoke check OK." +echo +echo "Try it:" +echo " ./usage-classify.sh \"Saya dikenakan biaya dua kali, tolong refund.\"" +echo +echo "For repeat calls, keep the checkpoint resident (about 8x faster):" +echo " setsid $BUILD/bin/laya-cli --server --port 8080 \\" +echo " --model $BASE_DIR/models/convaiinnovations/laya --variant multilingual --cpu &" +echo " LAYA_URL=http://127.0.0.1:8080 ./usage-classify.sh \"...\"" diff --git a/classify_runner.py b/classify_runner.py new file mode 100644 index 0000000..11508c4 --- /dev/null +++ b/classify_runner.py @@ -0,0 +1,177 @@ +import sys +import json +import os +import time +import subprocess +import urllib.request +from pathlib import Path +from core import classify_room as ap +from config.classify import ( + MODEL_ROOT, VARIANT, CLI, CHECKPOINT, BACKEND, + ALLOW_TRUNCATION, LAYA_URL, TIMEOUT, PRESETS, DEFAULT_PRESET, +) + +def load_state(args): + if args.state_file is not None: + path = Path(args.state_file) + if not path.is_file(): + print(f"State file not found: {path}", file=sys.stderr) + return None + raw = path.read_text(encoding="utf-8") + elif args.state is not None: + raw = args.state + else: + print("No state given. Pass it as an argument or with --state-file.", file=sys.stderr) + return None + + raw = raw.strip() + if raw[:1] in ("{", "["): # JSON object/array state is forwarded as a structure + try: + return json.loads(raw) + except json.JSONDecodeError as e: + print(f"State looks like JSON but does not parse: {e}", file=sys.stderr) + return None + return raw + +def load_questions(args): + if args.questions is not None: + path = Path(args.questions) + if not path.is_file(): + print(f"Questions file not found: {path}", file=sys.stderr) + return None + try: + data = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError as e: + print(f"Questions file is not valid JSON: {e}", file=sys.stderr) + return None + # Accept a bare questions map or a full laya.cpp request object. + data = data.get("questions", data) if isinstance(data, dict) else None + if not isinstance(data, dict) or not data: + print("Questions file must hold a non-empty questions object.", file=sys.stderr) + return None + return data + + name = args.preset if args.preset is not None else DEFAULT_PRESET + if name not in PRESETS: + print(f"Unknown preset: {name}. Available: {', '.join(PRESETS)}", file=sys.stderr) + return None + return PRESETS[name] + +def cli_predict(state, questions): + """Spawn laya-cli for a single request. Weights are reloaded on every call.""" + if not CLI.is_file(): + raise FileNotFoundError(f"laya-cli not found: {CLI}. Build it first (see README).") + + cmd = [str(CLI), "--model", str(MODEL_ROOT), "--variant", VARIANT, "--" + BACKEND] + if ALLOW_TRUNCATION: + cmd.append("--allow-truncation") + + request = json.dumps({"state": state, "questions": questions}, ensure_ascii=False) + proc = subprocess.run(cmd, input=request, capture_output=True, text=True) + if proc.returncode != 0: + raise RuntimeError(proc.stderr.strip() or "laya-cli exited with an error") + + out = proc.stdout.strip() + if not out: + raise RuntimeError("laya-cli produced no output") + return json.loads(out.splitlines()[-1]) # later lines win; "Ready:" goes to stderr + +def server_predict(url, state, questions): + """POST /v1/systemone so the checkpoint stays resident across runs.""" + payload = json.dumps({"state": state, "questions": questions}, ensure_ascii=False).encode("utf-8") + req = urllib.request.Request( + url.rstrip("/") + "/v1/systemone", + data=payload, + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: + return json.load(resp) + +def predict(state, questions, url): + if url: + try: + return server_predict(url, state, questions), "server" + except Exception as e: + print(f"laya.cpp server at {url} unavailable ({e}). Falling back to the CLI.", file=sys.stderr) + return cli_predict(state, questions), "cli" + +def answer_value(answer): + kind = answer.get("type") + if kind == "choice": + return str(answer.get("choice", "")) + if kind == "score": + score = float(answer.get("score", 0.0)) + label = answer.get("legend", {}).get(str(int(round(score)))) + return f"{score:.4f}" + (f" ({label})" if label else "") + return f"{float(answer.get('noul', 0.0)):.4f}" + +def detail_line(answer): + probs = answer.get("probabilities") + if probs: + # Choice options read best by confidence; score levels stay in legend order. + items = probs.items() if answer.get("type") == "score" else sorted( + probs.items(), key=lambda kv: -float(kv[1])) + return " ".join(f"{k}={float(v):.4f}" for k, v in items) + act = answer.get("action", {}).get("act_probability") + return f"act_probability={float(act):.4f}" if act is not None else None + +def classify_run(args): + if not CHECKPOINT.is_file(): + print(f"Model not found: {CHECKPOINT}. Download it first (see README).", file=sys.stderr) + sys.exit(1) + + state = load_state(args) + questions = load_questions(args) + if state is None or questions is None: + sys.exit(1) + + url = args.url if args.url is not None else os.environ.get("LAYA_URL", LAYA_URL) + + start = time.time() + try: + envelope, mode = predict(state, questions, url) + except Exception as e: + print(f"Inference failed: {e}", file=sys.stderr) + sys.exit(1) + elapsed = time.time() - start + + if args.json: + print(json.dumps(envelope, ensure_ascii=False, indent=2)) + return + + results = envelope.get("results") + if results is not None and not results: + print("Inference failed: laya.cpp returned no results", file=sys.stderr) + sys.exit(1) + + result = results[0] if results else envelope + answers = result.get("answers", {}) + usage = result.get("usage", {}) + width = max((len(q) for q in answers), default=0) + + preview = state if isinstance(state, str) else json.dumps(state, ensure_ascii=False) + if len(preview) > 120: + preview = preview[:120] + "..." + + print() + print(f"State : {preview}") + print(f"Questions : {len(answers)} ({mode})") + if envelope.get("backend"): + print(f"Backend : {envelope['backend']} ({envelope.get('device', '')})") + print(f"Elapsed : {elapsed:.3f} s") + if "elapsed_ms" in envelope: + print(f"Inference : {envelope['elapsed_ms'] / 1000:.3f} s") + if usage: + print(f"Tokens : {usage.get('input_tokens', 0)} in / {usage.get('output_tokens', 0)} out") + print() + + for qid, answer in answers.items(): + detail = detail_line(answer) + print(f"{qid.ljust(width)} : {answer_value(answer)}") + if detail: + print(f"{' ' * width} {detail}") + + print() + +if __name__ == "__main__": + classify_run( ap.parser.parse_args() ) diff --git a/config/classify.py b/config/classify.py new file mode 100644 index 0000000..0445c6e --- /dev/null +++ b/config/classify.py @@ -0,0 +1,53 @@ +from pathlib import Path +from config.model import MODEL_DIR + +# Native laya.cpp runtime, built CPU-only from source. See README "Build laya.cpp". +LAYACPP = Path(__file__).resolve().parent.parent / "third_party" / "laya.cpp" +CLI = LAYACPP / "build-cpu" / "bin" / "laya-cli" + +# laya.cpp joins MODEL_ROOT/VARIANT for every non-english variant, so the +# multilingual checkpoint has to live in /multilingual/. +MODEL_ROOT = MODEL_DIR / "convaiinnovations" / "laya" +VARIANT = "multilingual" +CHECKPOINT = MODEL_ROOT / VARIANT / "model.safetensors" + +# cpu | cuda | vulkan | coreml. laya-cli defaults to cuda, so this is always +# passed explicitly; a CPU-only build fails at startup without --cpu. +BACKEND = "cpu" + +# Off by default: requests over the token budget are rejected instead of shortened. +ALLOW_TRUNCATION = False + +# Set this, or export LAYA_URL, to POST /v1/systemone at a running +# `laya-cli --server` instead of spawning the CLI (which reloads weights per run). +LAYA_URL = "" +TIMEOUT = 120 + +# Typed questions. Laya is a decision model, not a single-label classifier: +# each question declares its own answer space, so nothing here needs retraining. +# Keep `choice` questions under ~20 options -- they share one fixed token budget. +PRESETS = { + "triage": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "invoices, payments, refunds", + "technical": "bugs, outages, system errors", + "sales": "pricing, plans, new contracts", + "other": "everything else", + }, + }, + "urgency": { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": ["not urgent", "soon", "immediate"], + }, + "refund": { + "type": "noul", + "instructions": "Does the customer ask for money back?", + }, + }, +} + +DEFAULT_PRESET = "triage" diff --git a/core/classify_room.py b/core/classify_room.py new file mode 100644 index 0000000..9fdf016 --- /dev/null +++ b/core/classify_room.py @@ -0,0 +1,11 @@ +import argparse + +# allow_abbrev=False: "--state" is a prefix of "--state-file", and abbreviation +# matching would otherwise silently swallow it into the wrong option. +parser = argparse.ArgumentParser(allow_abbrev=False) +parser.add_argument("state", nargs="?", default=None, help="State to classify (plain text or JSON). Omit to read --state-file") +parser.add_argument("--state-file", type=str, default=None, help="File holding the state (plain text or JSON)") +parser.add_argument("--preset", type=str, default=None, help="Named question set from config/classify.py") +parser.add_argument("--questions", type=str, default=None, help="JSON file with a questions object, overrides --preset") +parser.add_argument("--url", type=str, default=None, help="laya.cpp server base URL, overrides LAYA_URL and config") +parser.add_argument("--json", action="store_true", help="Print the raw response envelope as JSON") diff --git a/usage-classify.sh b/usage-classify.sh new file mode 100755 index 0000000..178e063 --- /dev/null +++ b/usage-classify.sh @@ -0,0 +1,20 @@ +#!/usr/bin/env bash +set -euo pipefail + +BASE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PYTHON="${PYTHON:-$BASE_DIR/.venv/bin/python}" + +if [ ! -x "$PYTHON" ]; then + echo "Venv not found. Create it with: python3 -m venv .venv && .venv/bin/pip install -r requirements.txt" >&2 + exit 1 +fi + +if [ $# -lt 1 ]; then + echo "Usage: $0 [--preset ] [--questions ] [--json]" >&2 + echo " Classify text with Laya typed decisions (laya.cpp, CPU)." >&2 + echo " Set LAYA_URL to reuse a running 'laya-cli --server'." >&2 + echo " Presets and model paths are set in config/classify.py." >&2 + exit 1 +fi + +"$PYTHON" "$BASE_DIR/classify_runner.py" "$@"