19 lines
591 B
Python
19 lines
591 B
Python
from config.model import MODEL_DIR
|
|
|
|
MODEL = MODEL_DIR / "granite-4.2" / "granite-4.2-8b-Q4_K_M.gguf"
|
|
|
|
N_THREADS = 10
|
|
N_CTX = 8192 # Context window (tokens)
|
|
N_GPU_LAYERS = 0 # 0 = full CPU, >0 offload N layers to GPU
|
|
MAX_TOKENS = 1024 # Max tokens per reply
|
|
STREAM = True # Stream tokens as they are generated
|
|
|
|
TEMPERATURE = 0.7
|
|
TOP_P = 0.9
|
|
TOP_K = 40
|
|
|
|
# System prompt for the chat session.
|
|
SYSTEM_PROMPT = "You are a helpful, concise assistant."
|
|
|
|
# Default chat template. Leave empty to use the template embedded in the GGUF.
|
|
CHAT_TEMPLATE = "" |