Guest User

ExLlamaV3 config

a guest
Oct 6th, 2026
90
0
Never
Not a member of Pastebin yet? Sign Up, it unlocks many cool features!
text 17.98 KB | None | 0 0
  1. ENV VARS:
  2.  
  3. - NVIDIA_VISIBLE_DEVICES=all
  4. - EXL3_MOE_PINNED_ARENA=0
  5. - EXL3_HOST_MEM_RESERVE_MB=0
  6. - EXL3_AUTOSPLIT_MARGIN_MB=0
  7. - EXL3_AUTOSPLIT_WORSTCASE=1
  8. #- EXL3_MOE_CPU_WSLOT_MB=384
  9. #- EXL3_MOE_CPU_STAGE_THREADS=8
  10. # Optional exllamav3 tuning knobs (see moe-inference skill):
  11. # - EXL3_MOE_CPU_SLOTS=8
  12. # - EXL3_MOE_CPU_THREADS=6
  13. # - EXL3_MOE_STREAM_T=16
  14. # - EXL3_MOE_ARENA_HUGEPAGE=1
  15.  
  16.  
  17.  
  18. TABBYAPI config.yaml:
  19.  
  20. # Sample YAML file for configuration.
  21. # Comment and uncomment values as needed.
  22. # Every value has a default within the application.
  23. # This file serves to be a drop in for config.yml
  24.  
  25. # Unless specified in the comments, DO NOT put these options in quotes!
  26. # You can use https://www.yamllint.com/ if you want to check your YAML formatting.
  27.  
  28. # Options for networking
  29. network:
  30. # The IP to host on (default: 127.0.0.1).
  31. # Use 0.0.0.0 to expose on all network adapters.
  32. host: 0.0.0.0
  33.  
  34. # The port to host on (default: 5000).
  35. port: 5000
  36.  
  37. # Disable HTTP token authentication with requests.
  38. # WARNING: This will make your instance vulnerable!
  39. # Only turn this on if nothing but trusted local clients can reach the API.
  40. # Note that web pages open in a browser on this machine also count as local
  41. # callers; restrict allowed_origins below if you disable auth.
  42. disable_auth: true
  43.  
  44. # Origins allowed to call the API from a browser (default: ["*"]).
  45. # This is a CORS allowlist, not authentication: "*" lets any site open in your
  46. # browser send requests to this instance, which matters most when disable_auth
  47. # is on. Restrict to your own frontends (e.g. ["http://localhost:8000"]) or use
  48. # an empty list [] to block all browser (cross-origin) callers.
  49. allowed_origins: ["*"]
  50.  
  51. # Disable fetching external content in response to requests,such as images from URLs.
  52. disable_fetch_requests: false
  53.  
  54. # Send tracebacks over the API (default: False).
  55. # NOTE: Only enable this for debug purposes.
  56. send_tracebacks: false
  57.  
  58. # Select API servers to enable (default: ["OAI"]).
  59. # Possible values: OAI, Kobold.
  60. api_servers: ["OAI"]
  61.  
  62. # Seconds between SSE keep-alive pings on streaming responses (default: 15).
  63. # Pings are SSE comments, ignored by compliant clients, and prevent
  64. # connections from dropping during long prefills. Set to 0 to disable.
  65. sse_ping_interval: 15
  66.  
  67. # Log every HTTP request with client address, method, path and status (default: False).
  68. # Generation requests are already logged in detail; this adds the rest, such as model list and health polls.
  69. access_log: false
  70.  
  71. # Options for logging
  72. logging:
  73. # Enable prompt logging (default: False).
  74. log_prompt: false
  75.  
  76. # Enable generation parameter logging (default: False).
  77. log_generation_params: false
  78.  
  79. # Enable request logging (default: False).
  80. # NOTE: Only use this for debugging!
  81. log_requests: false
  82.  
  83. # Show a live status line below the log with cache usage and in-flight jobs (default: True).
  84. # Only shown on an interactive terminal.
  85. log_live_status: true
  86.  
  87. # Prefix console log lines with the time of day (default: True).
  88. # The log files under logs/ always carry full timestamps.
  89. log_timestamps: true
  90.  
  91. # Write every /v1/chat/completions request to logs/debug/ as JSON (default: False).
  92. # Also saves the fully templated prompt (the exact text sent to the tokenizer)
  93. # as a .txt file with the same basename.
  94. # PRIVACY WARNING: Enabling this creates a comprehensive request log, including the
  95. # full message history and generation parameters. API keys are redacted, but prompts
  96. # and user-provided content are preserved for bug-report reproduction.
  97. log_chat_completion_requests: false
  98.  
  99. # Options for model overrides and loading
  100. # Please read the comments to understand how arguments are handled
  101. # between initial and API loads
  102. model:
  103. # Directory to look for models (default: models).
  104. # Windows users, do NOT put this path in quotes!
  105. model_dir: models
  106.  
  107. # Allow direct loading of models from a completion or chat completion request (default: False).
  108. # This method of loading is strict by default.
  109. # Enable dummy models to add exceptions for invalid model names.
  110. inline_model_loading: false
  111.  
  112. # Sends dummy model names when the models endpoint is queried. (default: False)
  113. # Enable this if the client is looking for specific OAI models.
  114. use_dummy_models: false
  115.  
  116. # A list of fake model names that are sent via the /v1/models endpoint. (default: ["gpt-3.5-turbo"])
  117. # Also used as bypasses for strict mode if inline_model_loading is true.
  118. dummy_model_names: ["gpt-3.5-turbo"]
  119.  
  120. # An initial model to load.
  121. # Make sure the model is located in the model directory!
  122. # REQUIRED: This must be filled out to load a model on startup.
  123. model_name: GLM-5.3-Flash-Turboderp-EXL3-4bpw
  124.  
  125. # Names of args to use as a fallback for API load requests (default: []).
  126. # For example, if you always want cache_mode to be Q4 instead of on the inital model load, add "cache_mode" to this array.
  127. # Example: ['max_seq_len', 'cache_mode'].
  128. use_as_default: []
  129.  
  130. # Backend to use for this model (auto-detect if not specified)
  131. # Options: exllamav3
  132. #backend:
  133.  
  134. # Max sequence length (default: min(max_position_embeddings, cache_size)).
  135. # Set to -1 to fetch from the model's config.json
  136. max_seq_len: -1
  137.  
  138. # Size of the key/value cache to allocate, in tokens (default: 4096).
  139. # Must be a multiple of 256.
  140. cache_size: 262144
  141.  
  142. # Enable different cache modes for VRAM savings (default: FP16).
  143. # Specify the pair k_bits,v_bits where k_bits and v_bits are integers from 2-8 (i.e. 8,8).
  144. # The legacy values 'FP16', 'Q8', 'Q6', 'Q4' are also accepted.
  145. cache_mode: FP16
  146.  
  147. # Load model with tensor parallelism.
  148. # Falls back to autosplit if GPU split isn't provided.
  149. # This ignores the gpu_split_auto value.
  150. tensor_parallel: false
  151.  
  152. # Sets a backend type for tensor parallelism. (default: native).
  153. # Options: native, nccl
  154. # Native is recommended for PCIe GPUs
  155. # NCCL is recommended for NVLink.
  156. tensor_parallel_backend: nccl
  157.  
  158. # Automatically allocate resources to GPUs (default: True).
  159. # Not parsed for single GPU users.
  160. gpu_split_auto: true
  161.  
  162. # Reserve VRAM used for autosplit loading (default: 96 MB on GPU 0).
  163. # Represented as an array of MB per GPU.
  164. autosplit_reserve: [96, 96, 96]
  165.  
  166. # Array of VRAM sizes to split between GPUs, in GB (default: []).
  167. # Used both with and without tensor parallelism.
  168. #gpu_split: [24576, 12288, 12288]
  169.  
  170. # Number of mixture-of-expert layers to offload to CPU inference (default: 0)
  171. # Only affects MoE models. Set a large value such as 999 to offload all layers
  172. # Mutually exclusive with cpu_moe_split_experts.
  173. #cpu_moe_offload_layers: 34
  174.  
  175. # Number of routed experts per MoE layer to offload to CPU inference (default: 0).
  176. # Unlike cpu_moe_offload_layers, this splits every MoE layer instead of offloading whole
  177. # layers: the coldest experts are kept in system RAM and computed on the CPU, overlapping
  178. # each layer's own GPU compute, with dynamic placement keeping hot experts in VRAM.
  179. # Mutually exclusive with cpu_moe_offload_layers; not supported with tensor parallelism.
  180. # GLM 5.3 flash has 288 experts per layer / 232 with env vars
  181. cpu_moe_split_experts: 234
  182.  
  183. # Worker thread count for CPU MoE inference (default: None).
  184. # Applies to both cpu_moe_offload_layers and cpu_moe_split_experts. When unset,
  185. # defers to the EXL3_MOE_CPU_THREADS environment variable, then half the CPU core count.
  186. cpu_moe_threads: 12
  187.  
  188. # Load a model's n-gram embedding table fully into system RAM (default: False).
  189. # Only affects PLE models with n-gram embeddings (e.g. Qwen3.8-Flash-Next). By default
  190. # the table is streamed from disk during inference; loading it into RAM avoids
  191. # per-token disk reads at the cost of tens of GB of system memory.
  192. ngram_ram: false
  193.  
  194. # NOTE: If a model has YaRN rope scaling, it will automatically be enabled by ExLlama.
  195. # rope_scale and rope_alpha settings won't apply in this case.
  196.  
  197. # Rope scale (default: 1.0).
  198. # Same as compress_pos_emb.
  199. # Use if the model was trained on long context with rope.
  200. # Leave blank to pull the value from the model.
  201. #rope_scale: 1.0
  202.  
  203. # Rope alpha (default: None).
  204. # Same as alpha_value. Set to "auto" to auto-calculate.
  205. # Leaving this value blank will either pull from the model or auto-calculate.
  206. #rope_alpha:
  207.  
  208. # Chunk size for prompt ingestion (default: 2048).
  209. # A lower value reduces VRAM usage but decreases ingestion speed.
  210. # NOTE: Effects vary depending on the model.
  211. # An ideal value is between 512 and 4096.
  212. chunk_size: 4096
  213.  
  214. # Use output chunking (default: True)
  215. # Instead of allocating cache space for the entire completion at once, allocate in chunks as needed.
  216. # Used by EXL3 models only.
  217. output_chunking: true
  218.  
  219. # Set the maximum number of generation jobs that can run concurrently
  220. # The default maximum batch size for transformer architectures is 32. Recurrent
  221. # models with linear or sliding attention use more VRAM to support larger batches,
  222. # so the default value is reduced to 4. If you do not require concurrency at all, you
  223. # can reduce it further to minimize VRAM overhead.
  224. max_batch_size: 1
  225.  
  226. # Set the prompt template for this model. (default: None)
  227. # If empty, attempts to look for the model's chat template.
  228. # If a model contains multiple templates in its tokenizer_config.json,
  229. # set prompt_template to the name of the template you want to use.
  230. # NOTE: Only works with chat completion message lists!
  231. #prompt_template:
  232.  
  233. # Enables vision support if the model supports it. (default: False)
  234. vision: true
  235.  
  236. # Keep the vision model's weights in system RAM instead of VRAM (default: False).
  237. # Weights are stored in pinned host memory and streamed to the GPU during inference,
  238. # trading vision speed for VRAM. Only applies when vision is enabled.
  239. vision_offload: true
  240.  
  241. # Default chat template variables (default: {}).
  242. # Merged into the template variables of every chat completion request; values
  243. # sent by the client (template_vars / chat_template_kwargs, or the top-level
  244. # reasoning_effort field) take precedence. Use for model-specific reasoning
  245. # knobs, e.g. {enable_thinking: true} or {reasoning_effort: high}.
  246. template_vars_default: {}
  247.  
  248. # Forced chat template variables (default: {}).
  249. # Like template_vars_default, but these override any values sent by the client.
  250. # Replaces the deprecated force_enable_thinking option, which is still accepted
  251. # as an alias for {enable_thinking: true}.
  252. template_vars_force: {}
  253.  
  254. # Enable reasoning parser (default: False).
  255. # Do NOT enable this if the model is not a reasoning model (e.g. deepseek-r1 series)
  256. reasoning: true
  257.  
  258. # The start token for reasoning content (default: "<think>")
  259. #reasoning_start_token: "<think>"
  260.  
  261. # The end token for reasoning content (default: "</think>")
  262. #reasoning_end_token: "</think>"
  263.  
  264. # Whether generation starts inside a reasoning block (default: auto).
  265. # Options: auto, always, never
  266. # auto guesses by scanning the end of the templated prompt for an unclosed reasoning start token.
  267. start_in_reasoning: auto
  268.  
  269. # Parse tool calls that occur inside reasoning content (default: True).
  270. # If False, tool call tags inside a reasoning block are treated as plain reasoning text.
  271. tool_calls_in_reasoning: true
  272.  
  273. # Default reasoning token budget (default: None).
  274. # When a request's reasoning content exceeds the budget, the server forces the end of
  275. # the reasoning phase by injecting reasoning_budget_message followed by the model's
  276. # end-of-reasoning tokens. 0 ends reasoning as soon as it starts; None or a negative
  277. # value disables the budget. Overridable per request via reasoning_budget_tokens
  278. # (aliases: reasoning_budget, thinking_budget, thinking_token_budget) or
  279. # reasoning.max_tokens. Requires a reasoning format: reasoning tags, Harmony or Muse Glimmer.
  280. reasoning_budget_tokens:
  281.  
  282. # Text injected before the end-of-reasoning tokens when the reasoning budget is
  283. # exhausted (default: no text, only the end-of-reasoning tokens are forced). Also overridable
  284. # per request via reasoning_budget_message.
  285. reasoning_budget_message:
  286.  
  287. # Tool format, e.g. 'qwen3_coder'. See docs for supported formats. If left blank,
  288. # tool calls from the model will not be parsed by the server.
  289. tool_format: glm4_5
  290.  
  291. # Parse responses in the Harmony message format (gpt-oss models).
  292. # Auto-detected from the model's special tokens by default; set to true or false
  293. # to override. Setting 'tool_format: harmony' is equivalent to setting this to true.
  294. # When active, supersedes the reasoning and tool format settings.
  295. harmony: false
  296.  
  297. # Parse responses in the Muse Glimmer message format.
  298. # Auto-detected from the model's special tokens by default; set to true or false
  299. # to override. Setting 'tool_format: muse_glimmer' is equivalent to setting this to true.
  300. # When active, supersedes the reasoning and tool format settings.
  301. muse_glimmer: false
  302.  
  303. # Options for draft models (speculative decoding)
  304. # This will use more VRAM!
  305. draft_model:
  306. # Drafting mode for exllamav3 (default: model).
  307. # Options: model, disabled, mtp, ngram.
  308. # In `model` mode, drafting is disabled if no draft_model_name is provided.
  309. draft_mode: disabled
  310.  
  311. # Directory to look for draft models (default: models)
  312. draft_model_dir: models
  313.  
  314. # An initial draft model to load.
  315. # Ensure the model is in the model directory.
  316. #draft_model_name:
  317.  
  318. # Rope scale for draft models (default: 1.0).
  319. # Same as compress_pos_emb.
  320. # Use if the draft model was trained on long context with rope.
  321. #draft_rope_scale: 1.0
  322.  
  323. # Rope alpha for draft models (default: None).
  324. # Same as alpha_value. Set to "auto" to auto-calculate.
  325. # Leaving this value blank will either pull from the model or auto-calculate.
  326. #draft_rope_alpha:
  327.  
  328. # Cache mode for draft models to save VRAM (default: FP16).
  329. # Specify the pair k_bits,v_bits where k_bits and v_bits are integers from 2-8 (i.e. 8,8).
  330. # The legacy values 'FP16', 'Q8', 'Q6', 'Q4' are also accepted.
  331. draft_cache_mode: Q4
  332.  
  333. # Array of VRAM sizes to split between GPUs, in GB (default: []).
  334. # If this isn't filled in, the draft model is autosplit.
  335. #draft_gpu_split: []
  336.  
  337. # Number of tokens to draft per iteration (default: draft model default)
  338. # Recurrent (linear or sliding attention) models use more VRAM for longer drafts.
  339. # This overhead multiplies with the max batch size, so for models with long drafts
  340. # (e.g. DFlash with 15 tokens by default) shorter drafts may be preferable.
  341. draft_num_tokens: 1
  342.  
  343. # Adjust number of draft tokens dynamically based on observed acceptance rates (default: False)
  344. # Ceiling is given by num_draft_tokens.
  345. #dynamic_draft:
  346.  
  347. # Minimum match length for exllamav3 n-gram drafting (default: 2).
  348. # Only used when draft_mode is ngram.
  349. #ngram_match_min: 2
  350.  
  351. # Options for Sampling
  352. sampling:
  353. # Select a sampler override preset (default: None).
  354. # Find this in the sampler-overrides folder.
  355. # This overrides default fallbacks for sampler values that are passed to the API.
  356. # NOTE: safe_defaults is noob friendly and provides fallbacks for frontends that don't send sampling parameters.
  357. # for frontends that don't send sampling parameters. Leaving this blank means no fallbacks at all.
  358. override_preset: safe_defaults
  359.  
  360. # Options for Loras
  361. lora:
  362. # Directory to look for LoRAs (default: loras).
  363. lora_dir: loras
  364.  
  365. # List of LoRAs to load and associated scaling factors (default scale: 1.0).
  366. # For the YAML file, add each entry as a YAML list:
  367. # - name: lora1
  368. # scaling: 1.0
  369. loras:
  370.  
  371. # Options for embedding models and loading.
  372. # NOTE: Embeddings requires the "extras" feature to be installed
  373. # Install it via "pip install .[extras]"
  374. embeddings:
  375. # Directory to look for embedding models (default: models).
  376. embedding_model_dir: models
  377.  
  378. # Device to load embedding models on (default: cpu).
  379. # Possible values: cpu, auto, cuda.
  380. # NOTE: It's recommended to load embedding models on the CPU.
  381. # If using an AMD GPU, set this value to 'cuda'.
  382. embeddings_device: cpu
  383.  
  384. # An initial embedding model to load on the infinity backend.
  385. embedding_model_name:
  386.  
  387. # Global memory settings
  388. memory:
  389.  
  390. # Max size of recurrent cache in system memory, in MB (default: 4096)
  391. sysmem_recurrent_cache: 4096
  392.  
  393. # Size of system memory second-tier K/V cache, in MB (default: 0)
  394. sysmem_kv_cache: 0
  395.  
  396. # Size of the image embedding cache in system memory, in MB (default: 1024).
  397. # Encoded images are kept so repeated turns of a conversation don't re-run the vision model.
  398. # Images already in use by a request are never evicted; a context whose images exceed the
  399. # budget is cached only partially, with a warning. Only applies when vision is enabled.
  400. sysmem_multimodal_cache: 1024
  401.  
  402. # Use the cudaMallocAsync allocator backend in Torch (default: False).
  403. # When False, the allocator is left to the environment: unless PYTORCH_CUDA_ALLOC_CONF is set,
  404. # ExLlamaV3 enables expandable segments in Torch's native allocator, which performs better
  405. # than cudaMallocAsync. Enable this to force the cudaMallocAsync backend instead.
  406. cuda_malloc_async: false
  407.  
  408. # Options for development and experimentation
  409. developer:
  410. # Skip Exllamav3 version check (default: False).
  411. # WARNING: It's highly recommended to update your dependencies rather than enabling this flag.
  412. unsafe_launch: false
  413.  
  414. # Disable API request streaming (default: False).
  415. disable_request_streaming: false
  416.  
  417. # Set process to use a higher priority.
  418. # For realtime process priority, run as administrator or sudo.
  419. # Otherwise, the priority will be set to high.
  420. realtime_process_priority: false
  421.  
  422. # Enable extremely verbose seqlog logging, requires a running Seq server
  423. seqlog: false
  424.  
  425. # Seq server url:port
  426. seqlog_server_url: http://localhost:5341
  427.  
  428. # Seq server API key (default: None)
  429. seqlog_api_key:
Advertisement
Add Comment
Please, Sign In to add comment