ai-experiment/config/llm.py

19 lines
591 B
Python

from config.model import MODEL_DIR
MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf"
N_THREADS = 10
N_CTX = 8192 # Context window (tokens)
N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU
MAX_TOKENS = 1024 # Max tokens per reply
STREAM = True # Stream tokens as they are generated
TEMPERATURE = 0.7
TOP_P = 0.9
TOP_K = 40
# System prompt for the chat session.
SYSTEM_PROMPT = "You are a helpful, concise assistant."
# Default chat template. Leave empty to use the template embedded in the GGUF.
CHAT_TEMPLATE = ""