Qwen_local_model
This commit is contained in:
+5
-3
@@ -9,6 +9,7 @@
|
||||
"port": 8001
|
||||
}
|
||||
},
|
||||
"public_model_name": "Qwen_local_model",
|
||||
"api_key": "sk-szcjw",
|
||||
"tensor_parallel_size": 2,
|
||||
"dtype": "bfloat16",
|
||||
@@ -70,9 +71,10 @@
|
||||
"ctx": "65536",
|
||||
"trust_remote": true,
|
||||
"valid_tp": [1, 2],
|
||||
"max_num_seqs": "8",
|
||||
"max_tokens": "8192",
|
||||
"gpu_util": "0.92",
|
||||
"max_num_seqs": "4",
|
||||
"max_tokens": "4096",
|
||||
"gpu_util": "0.94",
|
||||
"enforce_eager": true,
|
||||
"tool_call_parser": "qwen3_xml",
|
||||
"enable_auto_tool_choice": true,
|
||||
"served_model_name": "Qwen3.5-35B-A3B-GPTQ-Int4",
|
||||
|
||||
Reference in New Issue
Block a user