from config.model import MODEL_DIR MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf" N_THREADS = 10 N_CTX = 8192 # Context window (tokens) N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU MAX_TOKENS = 1024 # Max tokens per reply STREAM = True # Stream tokens as they are generated TEMPERATURE = 0.7 TOP_P = 0.9 TOP_K = 40 # System prompt for the chat session. SYSTEM_PROMPT = "You are a helpful, concise assistant." # Default chat template. Leave empty to use the template embedded in the GGUF. CHAT_TEMPLATE = ""