diff --git a/config.json b/config.json index 9cd85c2..e473283 100644 --- a/config.json +++ b/config.json @@ -106,6 +106,7 @@ "max_num_seqs": "64", "max_tokens": "32768", "gpu_util": "0.98", + "model_impl": "transformers", "tool_call_parser": "qwen3_xml", "enable_auto_tool_choice": true, "served_model_name": "Qwen3.5-27B-FP8", @@ -118,6 +119,7 @@ "max_num_seqs": "64", "max_tokens": "32768", "gpu_util": "0.98", + "model_impl": "transformers", "tool_call_parser": "qwen3_xml", "enable_auto_tool_choice": true, "served_model_name": "Qwen3.5-35B-A3B-GPTQ-Int4", diff --git a/scripts/start_vllm.py b/scripts/start_vllm.py index 2b19660..5d4afbe 100644 --- a/scripts/start_vllm.py +++ b/scripts/start_vllm.py @@ -144,6 +144,11 @@ def launch_model(model_id, config, model_path, gpu_count): "--gpu-memory-utilization", str(gpu_util), "--dtype", "auto" ] + + model_impl = config.get("model_impl") or os.getenv("VLLM_MODEL_IMPL") + if model_impl: + cmd.extend(["--model-impl", str(model_impl)]) + log(f"Added model implementation: {model_impl}") if config.get("trust_remote"): cmd.append("--trust-remote-code")