x
This commit is contained in:
@@ -4,6 +4,9 @@ FROM docker.1ms.run/kyuz0/vllm-therock-gfx1201:latest
|
|||||||
ENV USE_DEFAULT_MODEL=true
|
ENV USE_DEFAULT_MODEL=true
|
||||||
ENV LOCAL_MODEL_DIR=/opt/model
|
ENV LOCAL_MODEL_DIR=/opt/model
|
||||||
|
|
||||||
|
# Ensure latest Transformers model definitions are available
|
||||||
|
RUN /opt/venv/bin/pip install --no-cache-dir --upgrade "git+https://github.com/huggingface/transformers.git"
|
||||||
|
|
||||||
# Create necessary directories
|
# Create necessary directories
|
||||||
RUN mkdir -p /opt/script /opt/model /config
|
RUN mkdir -p /opt/script /opt/model /config
|
||||||
|
|
||||||
|
|||||||
+12
@@ -99,6 +99,18 @@
|
|||||||
"served_model_name": "GLM-4.7-Flash-AWQ",
|
"served_model_name": "GLM-4.7-Flash-AWQ",
|
||||||
"hf_model_id": "THUDM/GLM-4.7-Flash-AWQ"
|
"hf_model_id": "THUDM/GLM-4.7-Flash-AWQ"
|
||||||
},
|
},
|
||||||
|
"Qwen3.5-27B-FP8": {
|
||||||
|
"ctx": "32768",
|
||||||
|
"trust_remote": true,
|
||||||
|
"valid_tp": [1, 2],
|
||||||
|
"max_num_seqs": "64",
|
||||||
|
"max_tokens": "32768",
|
||||||
|
"gpu_util": "0.98",
|
||||||
|
"tool_call_parser": "qwen3_xml",
|
||||||
|
"enable_auto_tool_choice": true,
|
||||||
|
"served_model_name": "Qwen3.5-27B-FP8",
|
||||||
|
"hf_model_id": "RedHatAI/Qwen3.5-27B-FP8-dynamic"
|
||||||
|
},
|
||||||
"Qwen3.5-35B-A3B-GPTQ-Int4": {
|
"Qwen3.5-35B-A3B-GPTQ-Int4": {
|
||||||
"ctx": "32768",
|
"ctx": "32768",
|
||||||
"trust_remote": true,
|
"trust_remote": true,
|
||||||
|
|||||||
Reference in New Issue
Block a user