diff --git a/Dockerfile b/Dockerfile index 7ea9f77..812d256 100644 --- a/Dockerfile +++ b/Dockerfile @@ -4,9 +4,6 @@ FROM docker.1ms.run/kyuz0/vllm-therock-gfx1201:latest ENV USE_DEFAULT_MODEL=true ENV LOCAL_MODEL_DIR=/opt/model -# Ensure latest Transformers model definitions are available -RUN /opt/venv/bin/pip install --no-cache-dir --upgrade "git+https://github.com/huggingface/transformers.git" - # Create necessary directories RUN mkdir -p /opt/script /opt/model /config diff --git a/config.json b/config.json index e473283..894d53b 100644 --- a/config.json +++ b/config.json @@ -1,5 +1,5 @@ { - "default_model": "Qwen3.5-35B-A3B-GPTQ-Int4", + "default_model": "Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit", "models": { "Meta-Llama-3.1-8B-Instruct": { "ctx": "65536", @@ -106,7 +106,6 @@ "max_num_seqs": "64", "max_tokens": "32768", "gpu_util": "0.98", - "model_impl": "transformers", "tool_call_parser": "qwen3_xml", "enable_auto_tool_choice": true, "served_model_name": "Qwen3.5-27B-FP8", @@ -119,7 +118,6 @@ "max_num_seqs": "64", "max_tokens": "32768", "gpu_util": "0.98", - "model_impl": "transformers", "tool_call_parser": "qwen3_xml", "enable_auto_tool_choice": true, "served_model_name": "Qwen3.5-35B-A3B-GPTQ-Int4", diff --git a/scripts/start_vllm.py b/scripts/start_vllm.py index 5d4afbe..2b19660 100644 --- a/scripts/start_vllm.py +++ b/scripts/start_vllm.py @@ -144,11 +144,6 @@ def launch_model(model_id, config, model_path, gpu_count): "--gpu-memory-utilization", str(gpu_util), "--dtype", "auto" ] - - model_impl = config.get("model_impl") or os.getenv("VLLM_MODEL_IMPL") - if model_impl: - cmd.extend(["--model-impl", str(model_impl)]) - log(f"Added model implementation: {model_impl}") if config.get("trust_remote"): cmd.append("--trust-remote-code")