llama-server parameters

255 parameters
ctrlK

Common params

82
--help-h--usage

print usage and exit

--version

show version and build info

--cache-list-cl

show list of models in cache

--completion-bash

print source-able bash completion script for llama.cpp

--threads-tN

number of CPU threads to use during generation

env LLAMA_ARG_THREADSdefault -1
--threads-batch-tbN

number of threads to use during batch and prompt processing

default same as --threads
--cpu-mask-CM

CPU affinity mask: arbitrarily long hex. Complements cpu-range

default ""
--cpu-range-Crlo-hi

range of CPUs for affinity. Complements --cpu-mask

--cpu-strict<0|1>

use strict CPU placement

default 0
--prioN

set process/thread priority : low(-1), normal(0), medium(1), high(2), realtime(3)

default 0
--poll<0...100>

use polling level to wait for work (0 - no polling, default: 50)

--cpu-mask-batch-CbM

CPU affinity mask: arbitrarily long hex. Complements cpu-range-batch

default same as --cpu-mask
--cpu-range-batch-Crblo-hi

ranges of CPUs for affinity. Complements --cpu-mask-batch

--cpu-strict-batch<0|1>

use strict CPU placement

default same as --cpu-strict
--prio-batchN

set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime

default 0
--poll-batch<0|1>

use polling to wait for work

default same as --poll
--ctx-size-cN

size of the prompt context

env LLAMA_ARG_CTX_SIZEdefault 0, 0 = loaded from model
--predict-n--n-predictN

number of tokens to predict

env LLAMA_ARG_N_PREDICTdefault -1, -1 = infinity
--batch-size-bN

logical maximum batch size

env LLAMA_ARG_BATCHdefault 2048
--ubatch-size-ubN

physical maximum batch size

env LLAMA_ARG_UBATCHdefault 512
--keepN

number of tokens to keep from the initial prompt

default 0, -1 = all
--swa-full

use full-size SWA cache

env LLAMA_ARG_SWA_FULLdefault false
--flash-attn-fa[on|off|auto]

set Flash Attention use ('on', 'off', or 'auto', default: 'auto')

env LLAMA_ARG_FLASH_ATTN
--perf--no-perf

whether to enable internal libllama performance timings

env LLAMA_ARG_PERFdefault false
--escape-e--no-escape

whether to process escapes sequences (\n, \r, \t, \', \", \\)

default true
--rope-scaling{none,linear,yarn}

RoPE frequency scaling method, defaults to linear unless specified by the model

env LLAMA_ARG_ROPE_SCALING_TYPE
--rope-scaleN

RoPE context scaling factor, expands context by a factor of N

env LLAMA_ARG_ROPE_SCALE
--rope-freq-baseN

RoPE base frequency, used by NTK-aware scaling

env LLAMA_ARG_ROPE_FREQ_BASEdefault loaded from model
--rope-freq-scaleN

RoPE frequency scaling factor, expands context by a factor of 1/N

env LLAMA_ARG_ROPE_FREQ_SCALE
--yarn-orig-ctxN

YaRN: original context size of model

env LLAMA_ARG_YARN_ORIG_CTXdefault 0 = model training context size
--yarn-ext-factorN

YaRN: extrapolation mix factor

env LLAMA_ARG_YARN_EXT_FACTORdefault -1.00, 0.0 = full interpolation
--yarn-attn-factorN

YaRN: scale sqrt(t) or attention magnitude

env LLAMA_ARG_YARN_ATTN_FACTORdefault -1.00
--yarn-beta-slowN

YaRN: high correction dim or alpha

env LLAMA_ARG_YARN_BETA_SLOWdefault -1.00
--yarn-beta-fastN

YaRN: low correction dim or beta

env LLAMA_ARG_YARN_BETA_FASTdefault -1.00
--kv-offload-kvo-nkvo--no-kv-offload

whether to enable KV cache offloading

env LLAMA_ARG_KV_OFFLOADdefault enabled
--repack-nr--no-repack

whether to enable weight repacking

env LLAMA_ARG_REPACKdefault enabled
--no-host

bypass host buffer allowing extra buffers to be used

env LLAMA_ARG_NO_HOST
--cache-type-k-ctkTYPE

KV cache data type for K

  • allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1
env LLAMA_ARG_CACHE_TYPE_Kdefault f16
--cache-type-v-ctvTYPE

KV cache data type for V

  • allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1
env LLAMA_ARG_CACHE_TYPE_Vdefault f16
--defrag-thold-dtNdeprecated

KV cache defragmentation threshold (DEPRECATED)

env LLAMA_ARG_DEFRAG_THOLD
--rpcSERVERS

comma-separated list of RPC servers (host:port)

env LLAMA_ARG_RPC
--load-mode-lmMODE

model loading mode

  • - auto: mmap, unless a device does not support it
  • - none: no special loading mode
  • - mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)
  • - mlock: force system to keep model in RAM rather than swapping or compressing
  • - mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing
  • - dio: use DirectIO if available
env LLAMA_ARG_LOAD_MODEdefault auto
--lazy-mode-lzmMODE

on-demand reading of certain tensors, for example per-layer embeddings

  • - on: read the rows of such tensors from disk on demand instead of keeping them resident (requires mmap)
  • - auto: on, but only for tensors larger than 4 GiB
  • - off: always keep them resident
env LLAMA_ARG_LAZY_MODEdefault auto
--numaTYPE

attempt optimizations that help on some NUMA systems

  • - distribute: spread execution evenly over all nodes
  • - isolate: only spawn threads on CPUs on the node that execution started on
  • - numactl: use the CPU map provided by numactl
  • if run without this previously, it is recommended to drop the system page cache before using this
  • see https://github.com/ggml-org/llama.cpp/issues/1437
env LLAMA_ARG_NUMA
--device-dev<dev1,dev2,..>

comma-separated list of devices to use for offloading (none = don't offload)

  • use --list-devices to see a list of available devices
env LLAMA_ARG_DEVICE
--list-devices

print list of available devices and exit

--override-tensor-ot<tensor name pattern>=<buffer type>,...

override tensor buffer type

env LLAMA_ARG_OVERRIDE_TENSOR
--cpu-moe-cmoe

keep all Mixture of Experts (MoE) weights in the CPU

env LLAMA_ARG_CPU_MOE
--n-cpu-moe-ncmoeN

keep the Mixture of Experts (MoE) weights of the first N layers in the CPU

env LLAMA_ARG_N_CPU_MOE
--n-cpu-ffn-ncffnN

keep the dense FFN weights of the first N layers in the CPU

  • (dense models; for MoE expert weights use --n-cpu-moe)
env LLAMA_ARG_N_CPU_FFN
--gpu-layers-ngl--n-gpu-layersN

max. number of layers to store in VRAM, either an exact number, 'auto', or 'all'

env LLAMA_ARG_N_GPU_LAYERSdefault auto
--split-mode-sm{none,layer,row,tensor}

how to split the model across multiple GPUs, one of:

  • - none: use one GPU only
  • - layer (default): split layers and KV across GPUs (pipelined)
  • - row: split weight across GPUs by rows (parallelized)
  • - tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)
env LLAMA_ARG_SPLIT_MODE
--tensor-split-tsN0,N1,N2,...

fraction of the model to offload to each GPU, comma-separated list of proportions, e.g. 3,1

env LLAMA_ARG_TENSOR_SPLIT
--main-gpu-mgINDEX

the GPU to use for the model (with split-mode = none), or for intermediate results and KV (with split-mode = row)

env LLAMA_ARG_MAIN_GPUdefault 0
--fit-fit[on|off]

whether to adjust unset arguments to fit in device memory ('on' or 'off', default: 'on')

env LLAMA_ARG_FIT
--fit-target-fittMiB0,MiB1,MiB2,...

target margin per device for --fit, comma-separated list of values, single value is broadcast across all devices, default: 1024

env LLAMA_ARG_FIT_TARGET
--fit-ctx-fitcN

minimum ctx size that can be set by --fit option, default: 4096

env LLAMA_ARG_FIT_CTX
--check-tensors

check model tensor data for invalid values

default false
--override-kvKEY=TYPE:VALUE,...

advanced option to override model metadata by key. to specify multiple overrides, either use comma-separated values.

  • types: int, float, bool, str. example: --override-kv tokenizer.ggml.add_bos_token=bool:false,tokenizer.ggml.add_eos_token=bool:false
--op-offload--no-op-offload

whether to offload host tensor operations to device

default true
--loraFNAME

path to LoRA adapter (use comma-separated values to load multiple adapters)

--lora-scaledFNAME:SCALE,...

path to LoRA adapter with user defined scaling (format: FNAME:SCALE,...)

  • note: use comma-separated values
--control-vectorFNAME

add a control vector

  • note: use comma-separated values to add multiple control vectors
--control-vector-scaledFNAME:SCALE,...

add a control vector with user defined scaling SCALE

  • note: use comma-separated values (format: FNAME:SCALE,...)
--control-vector-layer-rangeSTART END

layer range to apply the control vector(s) to, start and end inclusive

--model-mFNAME

model path to load

env LLAMA_ARG_MODEL
--model-url-muMODEL_URL

model download url

env LLAMA_ARG_MODEL_URLdefault unused
--docker-repo-dr[<repo>/]<model>[:quant]

Docker Hub model repository. repo is optional, default to ai/. quant is optional, default to :latest.

  • example: gemma3
env LLAMA_ARG_DOCKER_REPOdefault unused
--hf-repo-hf-hfr<user>/<model>[:quant]

Hugging Face model repository; quant is optional, case-insensitive, default to Q4_K_M, or falls back to the first file in the repo if Q4_K_M doesn't exist.

  • mmproj is also downloaded automatically if available. to disable, add --no-mmproj
  • example: ggml-org/GLM-4.7-Flash-GGUF:Q4_K_M
env LLAMA_ARG_HF_REPOdefault unused
--hf-file-hffFILE

Hugging Face model file. If specified, it will override the quant in --hf-repo

env LLAMA_ARG_HF_FILEdefault unused
--hf-token-hftTOKEN

Hugging Face access token

env HF_TOKENdefault value from HF_TOKEN environment variable
--log-disable

Log disable

--log-fileFNAME

Log to file

env LLAMA_ARG_LOG_FILE
--log-jsonl--no-log-jsonl

Log as JSONL (one JSON object per line) to stdout, this also disables colored logging

env LLAMA_ARG_LOG_JSONLdefault disabled
--log-colors[on|off|auto]

Set colored logging ('on', 'off', or 'auto', default: 'auto')

  • 'auto' enables colors when output is to a terminal
env LLAMA_ARG_LOG_COLORS
--verbose-v--log-verbose

Set verbosity level to infinity (i.e. log all messages, useful for debugging)

--offline

Offline mode: forces use of cache, prevents network access

env LLAMA_ARG_OFFLINE
--verbosity-lv--log-verbosityN

Set the verbosity threshold. Messages with a higher verbosity will be ignored. Values:

  • - 0: generic output
  • - 1: error
  • - 2: warning
  • - 3: info
  • - 4: trace (more info)
  • - 5: debug
env LLAMA_ARG_LOG_VERBOSITYdefault 3
--log-prefix--no-log-prefix

Enable prefix in log messages

env LLAMA_ARG_LOG_PREFIX
--log-timestamps--no-log-timestamps

Enable timestamps in log messages

env LLAMA_ARG_LOG_TIMESTAMPS
--spec-draft-type-k-ctkd--cache-type-k-draftTYPE

KV cache data type for K for the draft model

  • allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1
env LLAMA_ARG_SPEC_DRAFT_CACHE_TYPE_Kdefault f16
--spec-draft-type-v-ctvd--cache-type-v-draftTYPE

KV cache data type for V for the draft model

  • allowed values: f32, f16, bf16, q8_0, q4_0, q4_1, iq4_nl, q5_0, q5_1
env LLAMA_ARG_SPEC_DRAFT_CACHE_TYPE_Vdefault f16

Sampling params

34
--samplersSAMPLERS

samplers that will be used for generation in the order, separated by ';'

default penalties;dry;top_n_sigma;top_k;typ_p;top_p;min_p;xtc;temperature
--seed-sSEED

RNG seed

default -1, use random seed for -1
--sampler-seq--sampling-seqSEQUENCE

simplified sequence for samplers that will be used

default edskypmxt
--ignore-eos

ignore end of stream token and continue generating (implies --logit-bias EOS-inf)

--temp--temperatureN

temperature

env LLAMA_ARG_TEMPERATUREdefault 0.80
--top-kN

top-k sampling

env LLAMA_ARG_TOP_Kdefault 40, 0 = disabled
--top-pN

top-p sampling

env LLAMA_ARG_TOP_Pdefault 0.95, 1.0 = disabled
--min-pN

min-p sampling

env LLAMA_ARG_MIN_Pdefault 0.05, 0.0 = disabled
--top-nsigma--top-n-sigmaN

top-n-sigma sampling

default -1.00, -1.0 = disabled
--xtc-probabilityN

xtc probability

default 0.00, 0.0 = disabled
--xtc-thresholdN

xtc threshold

default 0.10, 1.0 = disabled
--typical--typical-pN

locally typical sampling, parameter p

default 1.00, 1.0 = disabled
--repeat-last-nN

last n tokens to consider for penalize

default 64, 0 = disabled
--repeat-penaltyN

penalize repeat sequence of tokens

env LLAMA_ARG_REPEAT_PENALTYdefault 1.00, 1.0 = disabled
--presence-penaltyN

repeat alpha presence penalty

env LLAMA_ARG_PRESENCE_PENALTYdefault 0.00, 0.0 = disabled
--frequency-penaltyN

repeat alpha frequency penalty

env LLAMA_ARG_FREQUENCY_PENALTYdefault 0.00, 0.0 = disabled
--dry-multiplierN

set DRY sampling multiplier

default 0.00, 0.0 = disabled
--dry-baseN

set DRY sampling base value

default 1.75
--dry-allowed-lengthN

set allowed length for DRY sampling

default 2
--dry-penalty-last-nN

set DRY penalty for the last n tokens

default 64, 0 = disable
--dry-sequence-breakerSTRING

add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers

--adaptive-targetN

adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled)

default -1.00
--adaptive-decayN

adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.

  • (valid range 0.0 to 0.99) (default: 0.90)
--dynatemp-rangeN

dynamic temperature range

default 0.00, 0.0 = disabled
--dynatemp-expN

dynamic temperature exponent

default 1.00
--mirostatN

use Mirostat sampling.

  • Top K, Nucleus and Locally Typical samplers are ignored if used.
default 0, 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0
--mirostat-lrN

Mirostat learning rate, parameter eta

default 0.10
--mirostat-entN

Mirostat target entropy, parameter tau

default 5.00
--logit-bias-lTOKEN_ID(+/-)BIAS

modifies the likelihood of token appearing in the completion,

  • i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',
  • or `--logit-bias 15043-1` to decrease likelihood of token ' Hello'
--grammarGRAMMAR

BNF-like grammar to constrain generations (see samples in grammars/ dir)

--grammar-fileFNAME

file to read grammar from

--json-schema-jSCHEMA

JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object

--json-schema-file-jfFILE

File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object

--backend-sampling-bs

enable backend sampling (experimental)

env LLAMA_ARG_BACKEND_SAMPLINGdefault disabled

Server-specific params

139
--lookup-cache-static-lcsFNAME

path to static lookup cache to use for lookup decoding (not updated by generation)

--lookup-cache-dynamic-lcdFNAME

path to dynamic lookup cache to use for lookup decoding (updated by generation)

--kv-unified-per-slotN

context limit per parallel slot.

  • when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N
env LLAMA_ARG_KV_UNIFIED_PER_SLOTdefault unset, behavior unchanged
--ctx-checkpoints-ctxcp--swa-checkpointsN

max number of context checkpoints to create per slot

env LLAMA_ARG_CTX_CHECKPOINTSdefault 32
--checkpoint-min-step-cmsN

minimum spacing between context checkpoints in tokens

env LLAMA_ARG_CHECKPOINT_MIN_SPACING_NTdefault 8192, 0 = no minimum
--cache-ram-cramN

set the maximum cache size in MiB

env LLAMA_ARG_CACHE_RAMdefault 8192, -1 - no limit, 0 - disable
--kv-unified-kvu-no-kvu--no-kv-unified

use single unified KV buffer shared across all sequences

env LLAMA_ARG_KV_UNIFIEDdefault enabled if number of slots is auto
--cache-idle-slots--no-cache-idle-slots

save idle slots to the prompt cache on new task, and clear them when using unified KV

env LLAMA_ARG_CACHE_IDLE_SLOTSdefault enabled, requires cache-ram
--context-shift--no-context-shift

whether to use context shift on infinite text generation

env LLAMA_ARG_CONTEXT_SHIFTdefault disabled
--reverse-prompt-rPROMPT

halt generation at PROMPT, return control in interactive mode

--special-sp

special tokens output enabled

default false
--warmup--no-warmup

whether to perform warmup with an empty run

default enabled
--spm-infill

use Suffix/Prefix/Middle pattern for infill (instead of Prefix/Suffix/Middle) as some models prefer this.

default disabled
--pooling{none,mean,cls,last,rank}

pooling type for embeddings, use model default if unspecified

env LLAMA_ARG_POOLING
--parallel-npN

number of server slots

env LLAMA_ARG_N_PARALLELdefault -1, -1 = auto
--cont-batching-cb-nocb--no-cont-batching

whether to enable continuous batching (a.k.a dynamic batching)

env LLAMA_ARG_CONT_BATCHINGdefault enabled
--mmproj-mmFILE

path to a multimodal projector file. see tools/mtmd/README.md

  • note: if -hf is used, this argument can be omitted
env LLAMA_ARG_MMPROJ
--mmproj-url-mmuURL

URL to a multimodal projector file. see tools/mtmd/README.md

env LLAMA_ARG_MMPROJ_URL
--mmproj-auto--no-mmproj--no-mmproj-auto

whether to use multimodal projector file (if available), useful when using -hf

env LLAMA_ARG_MMPROJ_AUTOdefault enabled
--mmproj-offload--no-mmproj-offload

whether to enable GPU offloading for multimodal projector

env LLAMA_ARG_MMPROJ_OFFLOADdefault enabled
--mmproj-device-mmdevDEVICE

device to use for multimodal projector (none = don't offload, default: follows --device)

  • use --list-devices to see a list of available devices
env MTMD_BACKEND_DEVICE
--image-min-tokensN

minimum number of tokens each image can take, only used by vision models with dynamic resolution

env LLAMA_ARG_IMAGE_MIN_TOKENSdefault read from model
--image-max-tokensN

maximum number of tokens each image can take, only used by vision models with dynamic resolution

env LLAMA_ARG_IMAGE_MAX_TOKENSdefault read from model
--mtmd-batch-max-tokensN

maximum number of image tokens per batch when encoding images

env LLAMA_ARG_MTMD_BATCH_MAX_TOKENSdefault 1024
--video-fpsN

target video frame rate

env LLAMA_ARG_VIDEO_FPSdefault 4.0
--video-timestamp-intervalN

interval in milliseconds between text timestamps

env LLAMA_ARG_VIDEO_TIMESTAMP_INTERVALdefault 5000
--video-ffmpeg-dirDIR

path to the directory containing ffmpeg and ffprobe

env LLAMA_ARG_VIDEO_FFMPEG_DIRdefault search in PATH
--alias-aSTRING

set model name aliases, comma-separated (to be used by API)

env LLAMA_ARG_ALIAS
--tagsSTRING

set model tags, comma-separated (informational, not used for routing)

env LLAMA_ARG_TAGS
--embd-normalizeN

normalisation for embeddings (-1=none, 0=max absolute int16, 1=taxicab, 2=euclidean, >2=p-norm)

default 2
--hostHOST

IP addresses to listen on, comma-separated, or UNIX socket paths ending in .sock; with multiple TCP addresses, :: binds IPv6 only; overlapping addresses result in undefined behavior

env LLAMA_ARG_HOSTdefault 127.0.0.1
--portPORT

port to listen

env LLAMA_ARG_PORTdefault 8080
--reuse-port

allow multiple sockets to bind to the same port

env LLAMA_ARG_REUSE_PORTdefault disabled
--pathPATH

path to serve static files from

env LLAMA_ARG_STATIC_PATHdefault
--cors-originsORIGINS

comma-separated list of allowed origins for CORS

  • if set to special value 'localhost', reflect the Origin header only if it is localhost
env LLAMA_ARG_CORS_ORIGINSdefault *
--cors-methodsMETHODS

comma-separated list of allowed methods for CORS

env LLAMA_ARG_CORS_METHODSdefault GET, POST, DELETE, OPTIONS
--cors-headersHEADERS

comma-separated list of allowed headers for CORS

env LLAMA_ARG_CORS_HEADERSdefault *
--cors-credentials--no-cors-credentials

whether to allow credentials for CORS

  • note: if this is enabled and --cors-origins is set to * (default), the Origin header will be echoed back, and credentials will always be allowed
env LLAMA_ARG_CORS_CREDENTIALSdefault enabled
--api-prefixPREFIX

prefix path the server serves from, without the trailing slash

env LLAMA_ARG_API_PREFIXdefault
--ui-config--webui-configJSON

JSON that provides default UI settings (overrides UI defaults)

env LLAMA_ARG_UI_CONFIG
--ui-config-file--webui-config-filePATH

JSON file that provides default UI settings (overrides UI defaults)

env LLAMA_ARG_UI_CONFIG_FILE
--ui-mcp-proxy--webui-mcp-proxy--no-ui-mcp-proxy--no-webui-mcp-proxy

experimental: whether to enable MCP CORS proxy - do not enable in untrusted environments

env LLAMA_ARG_UI_MCP_PROXYdefault disabled
--toolsTOOL1,TOOL2,...

experimental: whether to enable built-in tools for AI agents - do not enable in untrusted environments

  • specify "all" to enable all tools
  • available tools: read_file, file_glob_search, grep_search, exec_shell_command, write_file, edit_file, get_info
  • note: for security reasons, this will limit --cors-origins to localhost by default
env LLAMA_ARG_TOOLSdefault no tools
--tools-runtimeOPTION

experimental: run tools in a separate runtime environment

  • available options:
  • 'docker:<image>', 'podman:<image>': spin up a new container and reuse it for all invocations, clean up on server exit
  • 'docker-container:<id>', 'podman-container:<id>': use an existing container by ID, won't stop on server exit
  • 'ssh:<target>': run tools on a remote POSIX host over SSH, key-based auth and a trusted host key are required
env LLAMA_ARG_TOOLS_RUNTIMEdefault none, use host environment
--mcp-servers-configPATH

experimental: path to JSON file with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments

  • note: for security reasons, this will limit --cors-origins to localhost by default
env LLAMA_ARG_MCP_SERVERS_CONFIGdefault none
--mcp-servers-jsonJSON

experimental: inline JSON with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments

  • note: for security reasons, this will limit --cors-origins to localhost by default
env LLAMA_ARG_MCP_SERVERS_JSONdefault none
--agent-ag-no-ag--no-agent

whether to enable CORS proxy and all built-in tools - do not enable in untrusted environments

  • note: for security reasons, this will limit --cors-origins to localhost by default
env LLAMA_ARG_AGENTdefault disabled
--ui--webui--no-ui--no-webui

whether to enable the Web UI

env LLAMA_ARG_UIdefault enabled
--embedding--embeddings

restrict to only support embedding use case; use only with dedicated embedding models

env LLAMA_ARG_EMBEDDINGSdefault disabled
--rerank--reranking

enable reranking endpoint on server

env LLAMA_ARG_RERANKINGdefault disabled
--api-keyKEY

API key to use for authentication, multiple keys can be provided as a comma-separated list

env LLAMA_API_KEYdefault none
--api-key-fileFNAME

path to file containing API keys, one per line; lines starting with a hash are treated as comments

env LLAMA_ARG_API_KEY_FILEdefault none
--ssl-key-fileFNAME

path to file a PEM-encoded SSL private key

env LLAMA_ARG_SSL_KEY_FILE
--ssl-cert-fileFNAME

path to file a PEM-encoded SSL certificate

env LLAMA_ARG_SSL_CERT_FILE
--chat-template-kwargsSTRING

sets additional params for the json template parser, must be a valid json object string, e.g. '{"key1":"value1","key2":"value2"}'

env LLAMA_ARG_CHAT_TEMPLATE_KWARGS
--timeout-toN

server read/write timeout in seconds

env LLAMA_ARG_TIMEOUTdefault 3600
--sse-ping-intervalN

server SSE ping interval in seconds (-1 = disabled, default: 30)

env LLAMA_ARG_SSE_PING_INTERVAL
--threads-httpN

number of threads used to process HTTP requests

env LLAMA_ARG_THREADS_HTTPdefault -1
--cache-prompt--no-cache-prompt

whether to enable prompt caching

env LLAMA_ARG_CACHE_PROMPTdefault enabled
--cache-reuseN

min chunk size to attempt reusing from the cache via KV shifting, requires prompt caching to be enabled

env LLAMA_ARG_CACHE_REUSEdefault 0
--metrics

enable prometheus compatible metrics endpoint

env LLAMA_ARG_ENDPOINT_METRICSdefault disabled
--props

enable changing global properties via POST /props

env LLAMA_ARG_ENDPOINT_PROPSdefault disabled
--slots--no-slots

expose slots monitoring endpoint

env LLAMA_ARG_ENDPOINT_SLOTSdefault enabled
--slot-save-pathPATH

path to save slot kv cache

default disabled
--media-pathPATH

directory for loading local media files; files can be accessed via file:// URLs using relative paths

default disabled
--models-dirPATH

directory containing models for the router server

env LLAMA_ARG_MODELS_DIRdefault disabled
--models-presetPATH

path to INI file containing model presets for the router server

env LLAMA_ARG_MODELS_PRESETdefault disabled
--models-maxN

for router server, maximum number of models to load simultaneously

env LLAMA_ARG_MODELS_MAXdefault 4, 0 = unlimited
--models-autoload--no-models-autoload

for router server, whether to automatically load models

env LLAMA_ARG_MODELS_AUTOLOADdefault enabled
--jinja--no-jinja

whether to use jinja template engine for chat

env LLAMA_ARG_JINJAdefault enabled
--reasoning-formatFORMAT

controls whether thought tags are allowed and/or extracted from the response, and in which format they're returned; one of:

  • - none: leaves thoughts unparsed in `message.content`
  • - deepseek: puts thoughts in `message.reasoning_content`
  • - deepseek-legacy: keeps `<think>` tags in `message.content` while also populating `message.reasoning_content`
env LLAMA_ARG_THINKdefault auto
--reasoning-rea[on|off|auto]

Use reasoning/thinking in the chat ('on', 'off', or 'auto', default: 'auto' (detect from template))

env LLAMA_ARG_REASONING
--reasoning-effortLEVEL

reasoning effort level given to the chat template: 'default' to keep the template default,

  • or a level such as 'minimal', 'low', 'medium', 'high', 'xhigh' or 'max' (default: default)
env LLAMA_ARG_REASONING_EFFORT
--reasoning-budgetN

token budget for thinking: -1 for unrestricted, 0 for immediate end, N>0 for token budget

env LLAMA_ARG_THINK_BUDGETdefault -1
--reasoning-budget-messageMESSAGE

message injected before the end-of-thinking tag when reasoning budget is exhausted

env LLAMA_ARG_THINK_BUDGET_MESSAGEdefault none
--reasoning-preserve--no-reasoning-preserve

preserve reasoning trace in the full history, not just the last assistant message

  • compatible with certain templates having 'supports_preserve_reasoning' capability
  • example: https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking
env LLAMA_ARG_REASONING_PRESERVEdefault enabled
--chat-templateJINJA_TEMPLATE

set custom jinja chat template

  • if suffix/prefix are specified, template will be disabled
  • only commonly used templates are accepted (unless --jinja is set before this flag):
  • list of built-in templates:
  • bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr
env LLAMA_ARG_CHAT_TEMPLATEdefault template taken from model's metadata
--chat-template-fileJINJA_TEMPLATE_FILE

set custom jinja chat template file

  • if suffix/prefix are specified, template will be disabled
  • only commonly used templates are accepted (unless --jinja is set before this flag):
  • list of built-in templates:
  • bailing, bailing-think, bailing2, chatglm3, chatglm4, chatml, command-r, deepseek, deepseek-ocr, deepseek2, deepseek3, exaone-moe, exaone3, exaone4, falcon3, gemma, gigachat, glmedge, gpt-oss, granite, granite-4.0, granite-4.1, grok-2, hunyuan-dense, hunyuan-moe, hunyuan-vl, kimi-k2, llama2, llama2-sys, llama2-sys-bos, llama2-sys-strip, llama3, llama4, megrez, minicpm, mistral-v1, mistral-v3, mistral-v3-tekken, mistral-v7, mistral-v7-tekken, monarch, openchat, orion, pangu-embedded, phi3, phi4, rwkv-world, seed_oss, smolvlm, solar-open, vicuna, vicuna-orca, yandex, zephyr
env LLAMA_ARG_CHAT_TEMPLATE_FILEdefault template taken from model's metadata
--skip-chat-parsing--no-skip-chat-parsing

force a pure content parser, even if a Jinja template is specified; model will output everything in the content section, including any reasoning and/or tool calls

env LLAMA_ARG_SKIP_CHAT_PARSINGdefault disabled
--prefill-assistant--no-prefill-assistant

whether to prefill the assistant's response if the last message is an assistant message

  • when this flag is set, if the last message is an assistant message then it will be treated as a full message and not prefilled
env LLAMA_ARG_PREFILL_ASSISTANTdefault prefill enabled
--slot-prompt-similarity-spsSIMILARITY

how much the prompt of a request must match the prompt of a slot in order to use that slot

default 0.10, 0.0 = disabled
--lora-init-without-apply

load LoRA adapters without applying them (apply later via POST /lora-adapters)

default disabled
--sleep-idle-secondsSECONDS

number of seconds of idleness after which the server will sleep

default -1; -1 = disabled
--log-prompts-dirPATH

Log prompts to directory (auto-created if not present; only used for debugging, default: disabled)

--spec-draft-hf-hfd-hfrd--hf-repo-draft<user>/<model>[:quant]

Same as --hf-repo, but for the draft model

env LLAMA_ARG_SPEC_DRAFT_HF_REPOdefault unused
--spec-draft-threads-td--threads-draftN

number of threads to use during generation

default same as --threads
--spec-draft-threads-batch-tbd--threads-batch-draftN

number of threads to use during batch and prompt processing

default same as --threads-draft
--spec-draft-cpu-mask-Cd--cpu-mask-draftM

Draft model CPU affinity mask. Complements cpu-range-draft

default same as --cpu-mask
--spec-draft-cpu-range-Crd--cpu-range-draftlo-hi

Ranges of CPUs for affinity. Complements --cpu-mask-draft

--spec-draft-cpu-strict--cpu-strict-draft<0|1>

Use strict CPU placement for draft model

default same as --cpu-strict
--spec-draft-prio--prio-draftN

set draft process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime

default 0
--spec-draft-poll--poll-draft<0|1>

Use polling to wait for draft model work

default same as --poll
--spec-draft-cpu-mask-batch-Cbd--cpu-mask-batch-draftM

Draft model CPU affinity mask. Complements cpu-range-draft

default same as --cpu-mask
--spec-draft-cpu-strict-batch--cpu-strict-batch-draft<0|1>

Use strict CPU placement for draft model

default --cpu-strict-draft
--spec-draft-prio-batch--prio-batch-draftN

set draft process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime

default 0
--spec-draft-poll-batch--poll-batch-draft<0|1>

Use polling to wait for draft model work

default --poll-draft
--spec-draft-override-tensor-otd--override-tensor-draft<tensor name pattern>=<buffer type>,...

override tensor buffer type for draft model

--spec-draft-cpu-moe-cmoed--cpu-moe-draft

keep all Mixture of Experts (MoE) weights in the CPU for the draft model

env LLAMA_ARG_SPEC_DRAFT_CPU_MOE
--spec-draft-n-cpu-moe--spec-draft-ncmoe-ncmoed--n-cpu-moe-draftN

keep the Mixture of Experts (MoE) weights of the first N layers in the CPU for the draft model

env LLAMA_ARG_SPEC_DRAFT_N_CPU_MOE
--spec-draft-n-maxN

number of tokens to draft for speculative decoding

env LLAMA_ARG_SPEC_DRAFT_N_MAXdefault 3
--spec-draft-n-minN

minimum number of draft tokens to use for speculative decoding

env LLAMA_ARG_SPEC_DRAFT_N_MINdefault 0
--spec-synth-lenL

target mean synthetic acceptance length, including the target token (benchmarking only)

env LLAMA_ARG_SPEC_SYNTH_LEN
--spec-synth-ratesP0,P1,...

comma-separated unconditional per-position synthetic acceptance probabilities (benchmarking only)

env LLAMA_ARG_SPEC_SYNTH_RATES
--spec-draft-p-split--draft-p-splitP

speculative decoding split probability

env LLAMA_ARG_SPEC_DRAFT_P_SPLITdefault 0.10
--spec-draft-p-min--draft-p-minP

minimum speculative decoding probability (greedy)

env LLAMA_ARG_SPEC_DRAFT_P_MINdefault 0.00
--spec-draft-backend-sampling--no-spec-draft-backend-sampling

offload draft sampling to the backend

env LLAMA_ARG_SPEC_DRAFT_BACKEND_SAMPLINGdefault enabled
--spec-draft-device-devd--device-draft<dev1,dev2,..>

comma-separated list of devices to use for offloading the draft model (none = don't offload, default: follows --device)

  • use --list-devices to see a list of available devices
--spec-draft-ngl-ngld--gpu-layers-draft--n-gpu-layers-draftN

max. number of draft model layers to store in VRAM, either an exact number, 'auto', or 'all'

env LLAMA_ARG_N_GPU_LAYERS_DRAFTdefault auto
--spec-draft-model-md--model-draftFNAME

draft model for speculative decoding

env LLAMA_ARG_SPEC_DRAFT_MODELdefault unused
--spec-typenone,draft-simple,draft-eagle3,draft-mtp,draft-dflash,draft-dspark,ngram-simple,ngram-map-k,ngram-map-k4v,ngram-mod,ngram-cache

comma-separated list of types of speculative decoding to use

env LLAMA_ARG_SPEC_TYPEdefault none
--spec-ngram-mod-n-minN

minimum number of ngram tokens to use for ngram-based speculative decoding

default 48
--spec-ngram-mod-n-maxN

maximum number of ngram tokens to use for ngram-based speculative decoding

default 64
--spec-ngram-mod-n-matchN

ngram-mod lookup length

default 24
--spec-ngram-simple-size-nN

ngram size N for ngram-simple speculative decoding, length of lookup n-gram

default 12
--spec-ngram-simple-size-mN

ngram size M for ngram-simple speculative decoding, length of draft m-gram

default 48
--spec-ngram-simple-min-hitsN

minimum hits for ngram-simple speculative decoding

default 1
--spec-ngram-map-k-size-nN

ngram size N for ngram-map-k speculative decoding, length of lookup n-gram

default 12
--spec-ngram-map-k-size-mN

ngram size M for ngram-map-k speculative decoding, length of draft m-gram

default 48
--spec-ngram-map-k-min-hitsN

minimum hits for ngram-map-k speculative decoding

default 1
--spec-ngram-map-k4v-size-nN

ngram size N for ngram-map-k4v speculative decoding, length of lookup n-gram

default 12
--spec-ngram-map-k4v-size-mN

ngram size M for ngram-map-k4v speculative decoding, length of draft m-gram

default 48
--spec-ngram-map-k4v-min-hitsN

minimum hits for ngram-map-k4v speculative decoding

default 1
--draft--draft-n--draft-maxNdeprecated

the argument has been removed. use --spec-draft-n-max or --spec-ngram-mod-n-max

env LLAMA_ARG_DRAFT_MAX
--draft-min--draft-n-minNdeprecated

the argument has been removed. use --spec-draft-n-min or --spec-ngram-mod-n-min

env LLAMA_ARG_DRAFT_MIN
--spec-ngram-size-nNdeprecated

the argument has been removed. use the respective --spec-ngram-*-size-n or --spec-ngram-mod-n-match

--spec-ngram-size-mNdeprecated

the argument has been removed. use the respective --spec-ngram-*-size-m

--spec-ngram-min-hitsNdeprecated

the argument has been removed. use the respective --spec-ngram-*-min-hits

--embd-gemma-default

use default EmbeddingGemma model (note: can download weights from the internet)

--fim-qwen-1.5b-default

use default Qwen 2.5 Coder 1.5B (note: can download weights from the internet)

--fim-qwen-3b-default

use default Qwen 2.5 Coder 3B (note: can download weights from the internet)

--fim-qwen-7b-default

use default Qwen 2.5 Coder 7B (note: can download weights from the internet)

--fim-qwen-7b-spec

use Qwen 2.5 Coder 7B + 0.5B draft for speculative decoding (note: can download weights from the internet)

--fim-qwen-14b-spec

use Qwen 2.5 Coder 14B + 0.5B draft for speculative decoding (note: can download weights from the internet)

--fim-qwen-30b-default

use default Qwen 3 Coder 30B A3B Instruct (note: can download weights from the internet)

--gpt-oss-20b-default

use gpt-oss-20b (note: can download weights from the internet)

--gpt-oss-120b-default

use gpt-oss-120b (note: can download weights from the internet)

--vision-gemma-4b-default

use Gemma 3 4B QAT (note: can download weights from the internet)

--vision-gemma-12b-default

use Gemma 3 12B QAT (note: can download weights from the internet)

--spec-default

enable default speculative decoding config