ai-experiment/config/classify.py

54 lines
2.0 KiB
Python

from pathlib import Path
from config.model import MODEL_DIR
# Native laya.cpp runtime, built CPU-only from source. See README "Build laya.cpp".
LAYACPP = Path(__file__).resolve().parent.parent / "third_party" / "laya.cpp"
CLI = LAYACPP / "build-cpu" / "bin" / "laya-cli"
# laya.cpp joins MODEL_ROOT/VARIANT for every non-english variant, so the
# multilingual checkpoint has to live in <MODEL_ROOT>/multilingual/.
MODEL_ROOT = MODEL_DIR / "convaiinnovations" / "laya"
VARIANT = "multilingual"
CHECKPOINT = MODEL_ROOT / VARIANT / "model.safetensors"
# cpu | cuda | vulkan | coreml. laya-cli defaults to cuda, so this is always
# passed explicitly; a CPU-only build fails at startup without --cpu.
BACKEND = "cpu"
# Off by default: requests over the token budget are rejected instead of shortened.
ALLOW_TRUNCATION = False
# Set this, or export LAYA_URL, to POST /v1/systemone at a running
# `laya-cli --server` instead of spawning the CLI (which reloads weights per run).
LAYA_URL = ""
TIMEOUT = 120
# Typed questions. Laya is a decision model, not a single-label classifier:
# each question declares its own answer space, so nothing here needs retraining.
# Keep `choice` questions under ~20 options -- they share one fixed token budget.
PRESETS = {
"triage": {
"department": {
"type": "choice",
"instructions": "Which team should handle this?",
"criteria": {
"billing": "invoices, payments, refunds",
"technical": "bugs, outages, system errors",
"sales": "pricing, plans, new contracts",
"other": "everything else",
},
},
"urgency": {
"type": "score",
"instructions": "How urgent is the request?",
"criteria": ["not urgent", "soon", "immediate"],
},
"refund": {
"type": "noul",
"instructions": "Does the customer ask for money back?",
},
},
}
DEFAULT_PRESET = "triage"