Optionalbatch_scheduler_policy
Optionalenable_chunked_context
enable_chunked_context?: boolean
Optionalkv_cache_free_gpu_mem_fraction
kv_cache_free_gpu_mem_fraction?: number
Optionalkv_cache_host_memory_bytes
kv_cache_host_memory_bytes?: number | null
Optionallora_cache_gpu_memory_fraction
lora_cache_gpu_memory_fraction?: number | null
Optionallora_cache_host_memory_bytes
lora_cache_host_memory_bytes?: number | null
Optionallora_cache_max_adapter_size
lora_cache_max_adapter_size?: number | null
Optionallora_cache_optimal_adapter_size
lora_cache_optimal_adapter_size?: number | null
Optionalrequest_default_max_tokens
request_default_max_tokens?: number | null
Optionalserved_model_name
served_model_name?: string | null
Optionaltotal_token_limit
total_token_limit?: number
Optionalwebserver_default_route
webserver_default_route?:
| "/predict"
| "/v1/embeddings"
| "/rerank"
| "/predict_tokens"
| null