54 lines
2.0 KiB
Python
54 lines
2.0 KiB
Python
from pathlib import Path
|
|
from config.model import MODEL_DIR
|
|
|
|
# Native laya.cpp runtime, built CPU-only from source. See README "Build laya.cpp".
|
|
LAYACPP = Path(__file__).resolve().parent.parent / "third_party" / "laya.cpp"
|
|
CLI = LAYACPP / "build-cpu" / "bin" / "laya-cli"
|
|
|
|
# laya.cpp joins MODEL_ROOT/VARIANT for every non-english variant, so the
|
|
# multilingual checkpoint has to live in <MODEL_ROOT>/multilingual/.
|
|
MODEL_ROOT = MODEL_DIR / "convaiinnovations" / "laya"
|
|
VARIANT = "multilingual"
|
|
CHECKPOINT = MODEL_ROOT / VARIANT / "model.safetensors"
|
|
|
|
# cpu | cuda | vulkan | coreml. laya-cli defaults to cuda, so this is always
|
|
# passed explicitly; a CPU-only build fails at startup without --cpu.
|
|
BACKEND = "cpu"
|
|
|
|
# Off by default: requests over the token budget are rejected instead of shortened.
|
|
ALLOW_TRUNCATION = False
|
|
|
|
# Set this, or export LAYA_URL, to POST /v1/systemone at a running
|
|
# `laya-cli --server` instead of spawning the CLI (which reloads weights per run).
|
|
LAYA_URL = ""
|
|
TIMEOUT = 120
|
|
|
|
# Typed questions. Laya is a decision model, not a single-label classifier:
|
|
# each question declares its own answer space, so nothing here needs retraining.
|
|
# Keep `choice` questions under ~20 options -- they share one fixed token budget.
|
|
PRESETS = {
|
|
"triage": {
|
|
"department": {
|
|
"type": "choice",
|
|
"instructions": "Which team should handle this?",
|
|
"criteria": {
|
|
"billing": "invoices, payments, refunds",
|
|
"technical": "bugs, outages, system errors",
|
|
"sales": "pricing, plans, new contracts",
|
|
"other": "everything else",
|
|
},
|
|
},
|
|
"urgency": {
|
|
"type": "score",
|
|
"instructions": "How urgent is the request?",
|
|
"criteria": ["not urgent", "soon", "immediate"],
|
|
},
|
|
"refund": {
|
|
"type": "noul",
|
|
"instructions": "Does the customer ask for money back?",
|
|
},
|
|
},
|
|
}
|
|
|
|
DEFAULT_PRESET = "triage"
|