from pathlib import Path from config.model import MODEL_DIR # Native laya.cpp runtime, built CPU-only from source. See README "Build laya.cpp". LAYACPP = Path(__file__).resolve().parent.parent / "third_party" / "laya.cpp" CLI = LAYACPP / "build-cpu" / "bin" / "laya-cli" # laya.cpp joins MODEL_ROOT/VARIANT for every non-english variant, so the # multilingual checkpoint has to live in /multilingual/. MODEL_ROOT = MODEL_DIR / "convaiinnovations" / "laya" VARIANT = "multilingual" CHECKPOINT = MODEL_ROOT / VARIANT / "model.safetensors" # cpu | cuda | vulkan | coreml. laya-cli defaults to cuda, so this is always # passed explicitly; a CPU-only build fails at startup without --cpu. BACKEND = "cpu" # Off by default: requests over the token budget are rejected instead of shortened. ALLOW_TRUNCATION = False # Set this, or export LAYA_URL, to POST /v1/systemone at a running # `laya-cli --server` instead of spawning the CLI (which reloads weights per run). LAYA_URL = "" TIMEOUT = 120 # Typed questions. Laya is a decision model, not a single-label classifier: # each question declares its own answer space, so nothing here needs retraining. # Keep `choice` questions under ~20 options -- they share one fixed token budget. PRESETS = { "triage": { "department": { "type": "choice", "instructions": "Which team should handle this?", "criteria": { "billing": "invoices, payments, refunds", "technical": "bugs, outages, system errors", "sales": "pricing, plans, new contracts", "other": "everything else", }, }, "urgency": { "type": "score", "instructions": "How urgent is the request?", "criteria": ["not urgent", "soon", "immediate"], }, "refund": { "type": "noul", "instructions": "Does the customer ask for money back?", }, }, } DEFAULT_PRESET = "triage"