x
This commit is contained in:
+29
-39
@@ -1,33 +1,23 @@
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
from app.model_catalog import load_catalog, resolve_model_profile, resolve_runtime_settings
|
||||
from app.model_catalog import load_app_config
|
||||
|
||||
|
||||
def build_command() -> list[str]:
|
||||
config_file = "config.json"
|
||||
catalog = load_catalog(config_file)
|
||||
runtime = resolve_runtime_settings(catalog)
|
||||
_, updates, env_vars = resolve_model_profile(
|
||||
content=catalog,
|
||||
requested_model=runtime["model_key"],
|
||||
requested_tp=runtime["tensor_parallel_size"],
|
||||
)
|
||||
for key, value in env_vars.items():
|
||||
os.environ[key] = value
|
||||
host = str(runtime["openai_host"])
|
||||
port = str(runtime["openai_port"])
|
||||
public_model_name = str(runtime["public_model_name"]).strip()
|
||||
default_enable_thinking = bool(runtime["default_enable_thinking"])
|
||||
reasoning_enabled = bool(runtime["reasoning_enabled"])
|
||||
api_key = runtime["api_key"] or ""
|
||||
dtype = str(updates["dtype"] or runtime["dtype"])
|
||||
quantization = str(updates["quantization"] or "").strip()
|
||||
model_impl = str(updates["model_impl"] or "").strip()
|
||||
reasoning_parser = str(updates["reasoning_parser"] or "").strip()
|
||||
revision = runtime["revision"] or ""
|
||||
config = load_app_config("config.json")
|
||||
host = str(config["openai_host"])
|
||||
port = str(config["openai_port"])
|
||||
public_model_name = str(config["public_model_name"]).strip()
|
||||
default_enable_thinking = bool(config["default_enable_thinking"])
|
||||
reasoning_enabled = bool(config["reasoning_enabled"])
|
||||
api_key = config.get("api_key") or ""
|
||||
dtype = str(config.get("dtype", "bfloat16"))
|
||||
quantization = str(config.get("quantization", "")).strip()
|
||||
model_impl = str(config.get("model_impl", "")).strip()
|
||||
reasoning_parser = str(config.get("reasoning_parser", "")).strip()
|
||||
revision = config.get("revision", "") or ""
|
||||
cmd = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
@@ -37,28 +27,28 @@ def build_command() -> list[str]:
|
||||
"--port",
|
||||
port,
|
||||
"--model",
|
||||
str(updates["model_name"]),
|
||||
str(config["model_name"]),
|
||||
"--served-model-name",
|
||||
public_model_name or str(updates["served_model_name"]),
|
||||
public_model_name or str(config["served_model_name"]),
|
||||
"--tensor-parallel-size",
|
||||
str(updates["tensor_parallel_size"]),
|
||||
str(config["tensor_parallel_size"]),
|
||||
"--max-model-len",
|
||||
str(updates["max_model_len"]),
|
||||
str(config["max_model_len"]),
|
||||
"--gpu-memory-utilization",
|
||||
str(updates["gpu_memory_utilization"]),
|
||||
str(config["gpu_memory_utilization"]),
|
||||
"--max-num-seqs",
|
||||
str(updates["max_num_seqs"]),
|
||||
str(config["max_num_seqs"]),
|
||||
"--dtype",
|
||||
dtype,
|
||||
]
|
||||
if updates["trust_remote_code"]:
|
||||
if config["trust_remote_code"]:
|
||||
cmd.append("--trust-remote-code")
|
||||
if updates["enforce_eager"]:
|
||||
if config["enforce_eager"]:
|
||||
cmd.append("--enforce-eager")
|
||||
if updates["enable_auto_tool_choice"]:
|
||||
if config["enable_auto_tool_choice"]:
|
||||
cmd.append("--enable-auto-tool-choice")
|
||||
if updates["tool_call_parser"]:
|
||||
cmd.extend(["--tool-call-parser", str(updates["tool_call_parser"])])
|
||||
if config["tool_call_parser"]:
|
||||
cmd.extend(["--tool-call-parser", str(config["tool_call_parser"])])
|
||||
cmd.extend(
|
||||
[
|
||||
"--default-chat-template-kwargs",
|
||||
@@ -75,12 +65,12 @@ def build_command() -> list[str]:
|
||||
cmd.extend(["--revision", revision])
|
||||
if api_key:
|
||||
cmd.extend(["--api-key", api_key])
|
||||
if updates.get("kv_cache_dtype"):
|
||||
cmd.extend(["--kv-cache-dtype", str(updates["kv_cache_dtype"])])
|
||||
if updates.get("enable_prefix_caching"):
|
||||
if config.get("kv_cache_dtype"):
|
||||
cmd.extend(["--kv-cache-dtype", str(config["kv_cache_dtype"])])
|
||||
if config.get("enable_prefix_caching"):
|
||||
cmd.append("--enable-prefix-caching")
|
||||
if updates.get("max_num_batched_tokens", 0) > 0:
|
||||
cmd.extend(["--max-num-batched-tokens", str(updates["max_num_batched_tokens"])])
|
||||
if config.get("max_num_batched_tokens", 0) > 0:
|
||||
cmd.extend(["--max-num-batched-tokens", str(config["max_num_batched_tokens"])])
|
||||
if updates.get("language_model_only"):
|
||||
cmd.append("--language-model-only")
|
||||
speculative_method = str(updates.get("speculative_method") or "").strip()
|
||||
|
||||
Reference in New Issue
Block a user