x
This commit is contained in:
+1
-3
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"default_model": "Qwen3.5-35B-A3B-GPTQ-Int4",
|
||||
"default_model": "Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit",
|
||||
"models": {
|
||||
"Meta-Llama-3.1-8B-Instruct": {
|
||||
"ctx": "65536",
|
||||
@@ -106,7 +106,6 @@
|
||||
"max_num_seqs": "64",
|
||||
"max_tokens": "32768",
|
||||
"gpu_util": "0.98",
|
||||
"model_impl": "transformers",
|
||||
"tool_call_parser": "qwen3_xml",
|
||||
"enable_auto_tool_choice": true,
|
||||
"served_model_name": "Qwen3.5-27B-FP8",
|
||||
@@ -119,7 +118,6 @@
|
||||
"max_num_seqs": "64",
|
||||
"max_tokens": "32768",
|
||||
"gpu_util": "0.98",
|
||||
"model_impl": "transformers",
|
||||
"tool_call_parser": "qwen3_xml",
|
||||
"enable_auto_tool_choice": true,
|
||||
"served_model_name": "Qwen3.5-35B-A3B-GPTQ-Int4",
|
||||
|
||||
Reference in New Issue
Block a user